From 11d181fd21d979377647eea377b7ac1857599426 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 10:05:32 +0900 Subject: [PATCH 001/216] test(mir): lock structure noninferiority evidence contract --- .../test_structure_noninferiority_policy.py | 240 ++++++++++++++++++ 1 file changed, 240 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_noninferiority_policy.py diff --git a/services/analysis-engine/tests/test_structure_noninferiority_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_policy.py new file mode 100644 index 000000000..1231d864b --- /dev/null +++ b/services/analysis-engine/tests/test_structure_noninferiority_policy.py @@ -0,0 +1,240 @@ +"""Tests for preregistered structure-feature noninferiority evidence.""" + +from __future__ import annotations + +import copy +import math +from types import ModuleType + +import pytest +from conftest import load_module + + +def _validator() -> ModuleType: + """Load the repository-owned structure experiment validator.""" + return load_module( + "scripts/research/validate_structure_noninferiority.py", + "validate_structure_noninferiority", + ) + + +def _registration() -> dict[str, object]: + """Return a valid synthetic registration fixture for policy tests only.""" + return { + "schema_version": 1, + "experiment_id": "structure-chroma-stft-vs-cqt-v1", + "hypothesis": { + "baseline_feature": "chroma_cqt", + "candidate_feature": "chroma_stft", + }, + "metrics": { + "boundary_f_0_5": { + "implementation": "mir_eval.segment.detection", + "window_seconds": 0.5, + "noninferiority_margin": 0.02, + }, + "boundary_f_3_0": { + "implementation": "mir_eval.segment.detection", + "window_seconds": 3.0, + "noninferiority_margin": 0.02, + }, + "functional_label_accuracy": { + "implementation": "mirex2025.frame_level_accuracy", + "noninferiority_margin": 0.02, + }, + "repetition_pairwise_f": { + "implementation": "mir_eval.segment.pairwise", + "frame_size_seconds": 0.1, + "noninferiority_margin": 0.02, + }, + "p95_latency_ratio": { + "maximum_candidate_ratio": 0.8, + }, + }, + "corpus": [ + { + "track_id": "licensed-track-001", + "audio_sha256": "a" * 64, + "annotation_sha256": "b" * 64, + "rights_basis": "evaluation-and-redistribution grant 2026-09-01", + "rights_cleared": True, + "source_uri": "https://example.invalid/licensed-track-001", + }, + { + "track_id": "licensed-track-002", + "audio_sha256": "c" * 64, + "annotation_sha256": "d" * 64, + "rights_basis": "private evaluation license 2026-09-01", + "rights_cleared": True, + "source_uri": "urn:bandscope:private-benchmark:licensed-track-002", + }, + ], + "runtime": { + "source_commit": "e" * 40, + "uv_lock_sha256": "f" * 64, + "python_version": "3.12.11", + "librosa_version": "0.11.0", + "numpy_version": "2.3.3", + "sample_rate_hz": 44100, + "channels": 1, + "host_profile": "registered-cpu-host-v1", + }, + } + + +def _result(registration_sha256: str) -> dict[str, object]: + """Return a result fixture whose paired intervals satisfy the registration.""" + return { + "schema_version": 1, + "experiment_id": "structure-chroma-stft-vs-cqt-v1", + "registration_sha256": registration_sha256, + "corpus_track_ids": ["licensed-track-001", "licensed-track-002"], + "aggregate": { + "baseline": { + "boundary_f_0_5": 0.70, + "boundary_f_3_0": 0.78, + "functional_label_accuracy": 0.72, + "repetition_pairwise_f": 0.74, + "p50_latency_seconds": 4.0, + "p95_latency_seconds": 6.0, + "peak_rss_mib": 650.0, + }, + "candidate": { + "boundary_f_0_5": 0.695, + "boundary_f_3_0": 0.775, + "functional_label_accuracy": 0.715, + "repetition_pairwise_f": 0.738, + "p50_latency_seconds": 1.8, + "p95_latency_seconds": 4.2, + "peak_rss_mib": 590.0, + }, + }, + "paired_delta_ci95": { + "boundary_f_0_5": [-0.015, 0.005], + "boundary_f_3_0": [-0.014, 0.004], + "functional_label_accuracy": [-0.018, 0.003], + "repetition_pairwise_f": [-0.012, 0.006], + }, + "p95_latency_ratio_ci95": [0.66, 0.76], + "failed_tracks": [], + "claim_boundary": ( + "Applies only to the registered rights-cleared corpus and exact runtime identity." + ), + } + + +def test_registration_digest_is_canonical_and_accepts_required_evidence() -> None: + """Equivalent key order must not change the frozen preregistration identity.""" + validator = _validator() + registration = _registration() + + validator.validate_registration(registration) + digest = validator.registration_digest(registration) + reordered = dict(reversed(list(registration.items()))) + + assert len(digest) == 64 + assert digest == validator.registration_digest(reordered) + + +@pytest.mark.parametrize( + ("mutation", "message"), + [ + (lambda value: value["corpus"][0].update(rights_cleared=False), "rights_cleared"), + (lambda value: value["corpus"][0].update(audio_sha256="abcd"), "audio_sha256"), + ( + lambda value: value["corpus"].append(copy.deepcopy(value["corpus"][0])), + "duplicate track_id", + ), + ], +) +def test_registration_rejects_unverifiable_real_audio( + mutation: object, + message: str, +) -> None: + """Rights, content identity, and unique track identity are mandatory evidence.""" + validator = _validator() + registration = _registration() + + mutation(registration) # type: ignore[operator] + + with pytest.raises(ValueError, match=message): + validator.validate_registration(registration) + + +def test_registration_rejects_incomplete_or_post_hoc_decision_contract() -> None: + """Every predeclared quality and latency criterion must be present and bounded.""" + validator = _validator() + registration = _registration() + metrics = registration["metrics"] + assert isinstance(metrics, dict) + del metrics["boundary_f_3_0"] + + with pytest.raises(ValueError, match="boundary_f_3_0"): + validator.validate_registration(registration) + + registration = _registration() + metrics = registration["metrics"] + assert isinstance(metrics, dict) + latency = metrics["p95_latency_ratio"] + assert isinstance(latency, dict) + latency["maximum_candidate_ratio"] = 1.0 + + with pytest.raises(ValueError, match="maximum_candidate_ratio"): + validator.validate_registration(registration) + + +def test_result_passes_only_when_paired_uncertainty_meets_frozen_contract() -> None: + """Decision uses paired CI bounds rather than aggregate point estimates alone.""" + validator = _validator() + registration = _registration() + digest = validator.registration_digest(registration) + + decision = validator.evaluate_result(registration, _result(digest)) + + assert decision["passed"] is True + assert decision["failed_requirements"] == [] + + +def test_result_fails_quality_or_latency_when_ci_crosses_registered_boundary() -> None: + """A favorable point estimate cannot hide a noninferiority or latency CI miss.""" + validator = _validator() + registration = _registration() + digest = validator.registration_digest(registration) + result = _result(digest) + deltas = result["paired_delta_ci95"] + assert isinstance(deltas, dict) + deltas["boundary_f_0_5"] = [-0.021, 0.004] + result["p95_latency_ratio_ci95"] = [0.70, 0.81] + + decision = validator.evaluate_result(registration, result) + + assert decision["passed"] is False + assert decision["failed_requirements"] == [ + "boundary_f_0_5 paired CI lower bound -0.021000 is below -0.020000", + "p95 latency ratio CI upper bound 0.810000 exceeds 0.800000", + ] + + +def test_result_is_bound_to_registration_corpus_and_finite_measurements() -> None: + """Results cannot drift thresholds, corpus identity, or numerical validity.""" + validator = _validator() + registration = _registration() + digest = validator.registration_digest(registration) + + wrong_digest = _result("0" * 64) + with pytest.raises(ValueError, match="registration_sha256"): + validator.evaluate_result(registration, wrong_digest) + + wrong_corpus = _result(digest) + wrong_corpus["corpus_track_ids"] = ["licensed-track-001"] + with pytest.raises(ValueError, match="corpus_track_ids"): + validator.evaluate_result(registration, wrong_corpus) + + nonfinite = _result(digest) + aggregate = nonfinite["aggregate"] + assert isinstance(aggregate, dict) + candidate = aggregate["candidate"] + assert isinstance(candidate, dict) + candidate["p95_latency_seconds"] = math.inf + with pytest.raises(ValueError, match="finite"): + validator.evaluate_result(registration, nonfinite) From bbfa248cb9e626948057c850ebb4beb85afe2bf2 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 10:06:35 +0900 Subject: [PATCH 002/216] feat(mir): validate preregistered structure noninferiority evidence --- .../validate_structure_noninferiority.py | 437 ++++++++++++++++++ 1 file changed, 437 insertions(+) create mode 100644 scripts/research/validate_structure_noninferiority.py diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py new file mode 100644 index 000000000..9a0ae76b1 --- /dev/null +++ b/scripts/research/validate_structure_noninferiority.py @@ -0,0 +1,437 @@ +#!/usr/bin/env python3 +"""Validate preregistered structure-feature noninferiority evidence. + +This module deliberately does not compute MIR metrics. It binds a reviewed +registration to result receipts produced by the recognized evaluation pipeline, +then evaluates the preregistered confidence-interval decision rules. That keeps +metric implementation authority with MIREX/mir_eval-compatible tooling while +preventing thresholds, corpus identity, or runtime identity from drifting after +results are observed. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +import re +from collections.abc import Mapping, Sequence +from pathlib import Path +from typing import Any + +SCHEMA_VERSION = 1 +_SHA256_RE = re.compile(r"^[0-9a-fA-F]{64}$") +_COMMIT_RE = re.compile(r"^[0-9a-fA-F]{40}$") +_WINDOWS_ABSOLUTE_RE = re.compile(r"^[A-Za-z]:[\\/]") + +_QUALITY_METRICS: dict[str, dict[str, float | str]] = { + "boundary_f_0_5": { + "implementation": "mir_eval.segment.detection", + "window_seconds": 0.5, + }, + "boundary_f_3_0": { + "implementation": "mir_eval.segment.detection", + "window_seconds": 3.0, + }, + "functional_label_accuracy": { + "implementation": "mirex2025.frame_level_accuracy", + }, + "repetition_pairwise_f": { + "implementation": "mir_eval.segment.pairwise", + "frame_size_seconds": 0.1, + }, +} +_LATENCY_METRIC = "p95_latency_ratio" +_REPORT_METRICS = ( + "p50_latency_seconds", + "p95_latency_seconds", + "peak_rss_mib", +) + + +def _mapping(value: object, field: str) -> Mapping[str, Any]: + """Return ``value`` as a mapping or raise a field-specific validation error.""" + if not isinstance(value, Mapping): + raise ValueError(f"{field} must be an object") + return value + + +def _sequence(value: object, field: str) -> Sequence[Any]: + """Return a non-string sequence or raise a field-specific validation error.""" + if isinstance(value, (str, bytes)) or not isinstance(value, Sequence): + raise ValueError(f"{field} must be an array") + return value + + +def _nonempty_text(value: object, field: str) -> str: + """Return stripped non-empty text for a required textual field.""" + if not isinstance(value, str) or not value.strip(): + raise ValueError(f"{field} must be non-empty text") + return value.strip() + + +def _finite_number(value: object, field: str) -> float: + """Return a finite real number while rejecting booleans and NaN/Inf.""" + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise ValueError(f"{field} must be a finite number") + number = float(value) + if not math.isfinite(number): + raise ValueError(f"{field} must be finite") + return number + + +def _score(value: object, field: str) -> float: + """Return a finite score constrained to the inclusive 0..1 interval.""" + number = _finite_number(value, field) + if not 0.0 <= number <= 1.0: + raise ValueError(f"{field} must be between 0 and 1") + return number + + +def _sha256(value: object, field: str) -> str: + """Return a normalized SHA-256 hex digest.""" + text = _nonempty_text(value, field) + if _SHA256_RE.fullmatch(text) is None: + raise ValueError(f"{field} must be a 64-character SHA-256 hex digest") + return text.lower() + + +def _commit(value: object, field: str) -> str: + """Return a normalized full Git commit SHA.""" + text = _nonempty_text(value, field) + if _COMMIT_RE.fullmatch(text) is None: + raise ValueError(f"{field} must be a full 40-character Git commit SHA") + return text.lower() + + +def _reject_local_path(value: object, field: str) -> str: + """Reject local filesystem authorities from corpus provenance receipts.""" + text = _nonempty_text(value, field) + lowered = text.casefold() + if ( + text.startswith(("/", "\\\\")) + or _WINDOWS_ABSOLUTE_RE.match(text) is not None + or lowered.startswith("file:") + ): + raise ValueError(f"{field} must be a provenance URI, not a local filesystem path") + return text + + +def _validate_schema_version(value: object, field: str) -> None: + """Require the one currently supported evidence schema version.""" + if isinstance(value, bool) or value != SCHEMA_VERSION: + raise ValueError(f"{field} must equal {SCHEMA_VERSION}") + + +def _validate_metrics(metrics_value: object) -> None: + """Validate the complete preregistered metric and decision contract.""" + metrics = _mapping(metrics_value, "metrics") + expected = set(_QUALITY_METRICS) | {_LATENCY_METRIC} + actual = set(metrics) + missing = sorted(expected - actual) + extra = sorted(actual - expected) + if missing: + raise ValueError(f"metrics missing required metric: {missing[0]}") + if extra: + raise ValueError(f"metrics contains unregistered metric: {extra[0]}") + + for metric_name, expected_config in _QUALITY_METRICS.items(): + config = _mapping(metrics[metric_name], f"metrics.{metric_name}") + implementation = _nonempty_text( + config.get("implementation"), + f"metrics.{metric_name}.implementation", + ) + if implementation != expected_config["implementation"]: + raise ValueError( + f"metrics.{metric_name}.implementation must equal " + f"{expected_config['implementation']}" + ) + for config_name, expected_value in expected_config.items(): + if config_name == "implementation": + continue + actual_value = _finite_number( + config.get(config_name), + f"metrics.{metric_name}.{config_name}", + ) + if not math.isclose(actual_value, float(expected_value), rel_tol=0.0, abs_tol=1e-12): + raise ValueError( + f"metrics.{metric_name}.{config_name} must equal {expected_value}" + ) + margin = _finite_number( + config.get("noninferiority_margin"), + f"metrics.{metric_name}.noninferiority_margin", + ) + if not 0.0 <= margin < 1.0: + raise ValueError( + f"metrics.{metric_name}.noninferiority_margin must be in [0, 1)" + ) + + latency = _mapping(metrics[_LATENCY_METRIC], f"metrics.{_LATENCY_METRIC}") + maximum_ratio = _finite_number( + latency.get("maximum_candidate_ratio"), + f"metrics.{_LATENCY_METRIC}.maximum_candidate_ratio", + ) + if not 0.0 < maximum_ratio < 1.0: + raise ValueError( + "metrics.p95_latency_ratio.maximum_candidate_ratio must be greater than 0 " + "and less than 1" + ) + + +def _validate_corpus(corpus_value: object) -> list[str]: + """Validate rights-cleared real-audio identities and return ordered track IDs.""" + corpus = _sequence(corpus_value, "corpus") + if len(corpus) < 2: + raise ValueError("corpus must contain at least two rights-cleared real-audio tracks") + + seen: set[str] = set() + track_ids: list[str] = [] + for index, raw_track in enumerate(corpus): + field = f"corpus[{index}]" + track = _mapping(raw_track, field) + track_id = _nonempty_text(track.get("track_id"), f"{field}.track_id") + if track_id in seen: + raise ValueError(f"duplicate track_id: {track_id}") + seen.add(track_id) + track_ids.append(track_id) + _sha256(track.get("audio_sha256"), f"{field}.audio_sha256") + _sha256(track.get("annotation_sha256"), f"{field}.annotation_sha256") + _nonempty_text(track.get("rights_basis"), f"{field}.rights_basis") + if track.get("rights_cleared") is not True: + raise ValueError(f"{field}.rights_cleared must be true") + _reject_local_path(track.get("source_uri"), f"{field}.source_uri") + return track_ids + + +def _validate_runtime(runtime_value: object) -> None: + """Validate exact runtime identity needed to reproduce paired measurements.""" + runtime = _mapping(runtime_value, "runtime") + _commit(runtime.get("source_commit"), "runtime.source_commit") + _sha256(runtime.get("uv_lock_sha256"), "runtime.uv_lock_sha256") + for field in ("python_version", "librosa_version", "numpy_version", "host_profile"): + _nonempty_text(runtime.get(field), f"runtime.{field}") + + sample_rate = _finite_number(runtime.get("sample_rate_hz"), "runtime.sample_rate_hz") + if not 1.0 <= sample_rate <= 384000.0 or not sample_rate.is_integer(): + raise ValueError("runtime.sample_rate_hz must be an integer in 1..384000") + channels = runtime.get("channels") + if isinstance(channels, bool) or channels != 1: + raise ValueError("runtime.channels must equal 1 for the registered segmentation input") + + +def validate_registration(registration_value: object) -> None: + """Validate a frozen STFT-vs-CQT structure experiment registration. + + The validator intentionally requires content hashes, rights evidence, exact + runtime identity, recognized metric implementations, and explicit margins. + It does not decide what the margins should be; that scientific/product + choice must be reviewed before any corpus result is inspected. + """ + registration = _mapping(registration_value, "registration") + _validate_schema_version(registration.get("schema_version"), "schema_version") + _nonempty_text(registration.get("experiment_id"), "experiment_id") + + hypothesis = _mapping(registration.get("hypothesis"), "hypothesis") + baseline = _nonempty_text(hypothesis.get("baseline_feature"), "hypothesis.baseline_feature") + candidate = _nonempty_text( + hypothesis.get("candidate_feature"), + "hypothesis.candidate_feature", + ) + if baseline != "chroma_cqt": + raise ValueError("hypothesis.baseline_feature must equal chroma_cqt") + if candidate != "chroma_stft": + raise ValueError("hypothesis.candidate_feature must equal chroma_stft") + + _validate_metrics(registration.get("metrics")) + _validate_corpus(registration.get("corpus")) + _validate_runtime(registration.get("runtime")) + + +def registration_digest(registration_value: object) -> str: + """Return the canonical SHA-256 identity of a valid preregistration.""" + validate_registration(registration_value) + canonical = json.dumps( + registration_value, + allow_nan=False, + ensure_ascii=False, + separators=(",", ":"), + sort_keys=True, + ).encode("utf-8") + return hashlib.sha256(canonical).hexdigest() + + +def _confidence_interval(value: object, field: str) -> tuple[float, float]: + """Validate and return an ordered two-sided confidence interval.""" + interval = _sequence(value, field) + if len(interval) != 2: + raise ValueError(f"{field} must contain exactly two bounds") + lower = _finite_number(interval[0], f"{field}[0]") + upper = _finite_number(interval[1], f"{field}[1]") + if lower > upper: + raise ValueError(f"{field} lower bound must not exceed upper bound") + return lower, upper + + +def _validate_aggregate_side(side_value: object, field: str) -> dict[str, float]: + """Validate one aggregate baseline/candidate result side.""" + side = _mapping(side_value, field) + normalized: dict[str, float] = {} + for metric_name in _QUALITY_METRICS: + normalized[metric_name] = _score(side.get(metric_name), f"{field}.{metric_name}") + for metric_name in _REPORT_METRICS: + value = _finite_number(side.get(metric_name), f"{field}.{metric_name}") + if value < 0.0: + raise ValueError(f"{field}.{metric_name} must be non-negative") + normalized[metric_name] = value + if normalized["p95_latency_seconds"] < normalized["p50_latency_seconds"]: + raise ValueError(f"{field}.p95_latency_seconds must be >= p50_latency_seconds") + return normalized + + +def _validate_result_identity( + registration: Mapping[str, Any], + result: Mapping[str, Any], +) -> list[str]: + """Bind a result receipt to the frozen registration and exact corpus order.""" + _validate_schema_version(result.get("schema_version"), "result.schema_version") + experiment_id = _nonempty_text(result.get("experiment_id"), "result.experiment_id") + if experiment_id != registration["experiment_id"]: + raise ValueError("result.experiment_id does not match registration") + + expected_digest = registration_digest(registration) + actual_digest = _sha256(result.get("registration_sha256"), "result.registration_sha256") + if actual_digest != expected_digest: + raise ValueError("result.registration_sha256 does not match the frozen registration") + + expected_track_ids = _validate_corpus(registration["corpus"]) + actual_track_values = _sequence(result.get("corpus_track_ids"), "result.corpus_track_ids") + actual_track_ids = [ + _nonempty_text(value, f"result.corpus_track_ids[{index}]") + for index, value in enumerate(actual_track_values) + ] + if actual_track_ids != expected_track_ids: + raise ValueError("result.corpus_track_ids must exactly match registration corpus order") + return expected_track_ids + + +def evaluate_result( + registration_value: object, + result_value: object, +) -> dict[str, object]: + """Validate a result receipt and evaluate preregistered CI-based decisions.""" + validate_registration(registration_value) + registration = _mapping(registration_value, "registration") + result = _mapping(result_value, "result") + track_ids = _validate_result_identity(registration, result) + + aggregate = _mapping(result.get("aggregate"), "result.aggregate") + baseline = _validate_aggregate_side(aggregate.get("baseline"), "result.aggregate.baseline") + candidate = _validate_aggregate_side( + aggregate.get("candidate"), + "result.aggregate.candidate", + ) + + raw_intervals = _mapping(result.get("paired_delta_ci95"), "result.paired_delta_ci95") + if set(raw_intervals) != set(_QUALITY_METRICS): + raise ValueError("result.paired_delta_ci95 must contain exactly the registered quality metrics") + + intervals: dict[str, tuple[float, float]] = {} + for metric_name in _QUALITY_METRICS: + interval = _confidence_interval( + raw_intervals[metric_name], + f"result.paired_delta_ci95.{metric_name}", + ) + point_delta = candidate[metric_name] - baseline[metric_name] + if not interval[0] <= point_delta <= interval[1]: + raise ValueError( + f"result.paired_delta_ci95.{metric_name} must contain the aggregate point delta" + ) + intervals[metric_name] = interval + + latency_interval = _confidence_interval( + result.get("p95_latency_ratio_ci95"), + "result.p95_latency_ratio_ci95", + ) + if latency_interval[0] <= 0.0: + raise ValueError("result.p95_latency_ratio_ci95 bounds must be greater than 0") + baseline_p95 = baseline["p95_latency_seconds"] + if baseline_p95 <= 0.0: + raise ValueError("result.aggregate.baseline.p95_latency_seconds must be greater than 0") + latency_ratio = candidate["p95_latency_seconds"] / baseline_p95 + if not latency_interval[0] <= latency_ratio <= latency_interval[1]: + raise ValueError("result.p95_latency_ratio_ci95 must contain the aggregate p95 ratio") + + failed_tracks_raw = _sequence(result.get("failed_tracks"), "result.failed_tracks") + failed_tracks = [ + _nonempty_text(value, f"result.failed_tracks[{index}]") + for index, value in enumerate(failed_tracks_raw) + ] + if len(set(failed_tracks)) != len(failed_tracks): + raise ValueError("result.failed_tracks must not contain duplicates") + unknown_failed = sorted(set(failed_tracks) - set(track_ids)) + if unknown_failed: + raise ValueError(f"result.failed_tracks contains unknown track_id: {unknown_failed[0]}") + _nonempty_text(result.get("claim_boundary"), "result.claim_boundary") + + metrics = _mapping(registration["metrics"], "metrics") + failed_requirements: list[str] = [] + for metric_name in _QUALITY_METRICS: + metric_config = _mapping(metrics[metric_name], f"metrics.{metric_name}") + margin = _finite_number( + metric_config.get("noninferiority_margin"), + f"metrics.{metric_name}.noninferiority_margin", + ) + lower = intervals[metric_name][0] + if lower < -margin: + failed_requirements.append( + f"{metric_name} paired CI lower bound {lower:.6f} is below -{margin:.6f}" + ) + + latency_config = _mapping(metrics[_LATENCY_METRIC], f"metrics.{_LATENCY_METRIC}") + maximum_ratio = _finite_number( + latency_config.get("maximum_candidate_ratio"), + f"metrics.{_LATENCY_METRIC}.maximum_candidate_ratio", + ) + latency_upper = latency_interval[1] + if latency_upper > maximum_ratio: + failed_requirements.append( + "p95 latency ratio CI upper bound " + f"{latency_upper:.6f} exceeds {maximum_ratio:.6f}" + ) + + return { + "passed": not failed_requirements, + "failed_requirements": failed_requirements, + "registration_sha256": registration_digest(registration), + "failed_tracks": failed_tracks, + } + + +def _load_json(path: Path) -> object: + """Load one UTF-8 JSON evidence file.""" + with path.open("r", encoding="utf-8") as handle: + return json.load(handle) + + +def main(argv: Sequence[str] | None = None) -> int: + """Validate a registration and optionally evaluate one bound result receipt.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("registration", type=Path) + parser.add_argument("result", type=Path, nargs="?") + args = parser.parse_args(argv) + + registration = _load_json(args.registration) + digest = registration_digest(registration) + if args.result is None: + print(json.dumps({"registration_sha256": digest}, sort_keys=True)) + return 0 + + result = _load_json(args.result) + decision = evaluate_result(registration, result) + print(json.dumps(decision, sort_keys=True)) + return 0 if decision["passed"] is True else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) From ff268eb57115fb0d8f6d3c2a946d0186ed96c626 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 10:07:36 +0900 Subject: [PATCH 003/216] test(mir): require per-track structure evidence and deviations --- .../test_structure_noninferiority_policy.py | 159 +++++++++++++----- 1 file changed, 117 insertions(+), 42 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_policy.py index 1231d864b..2d2355994 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_policy.py @@ -82,32 +82,74 @@ def _registration() -> dict[str, object]: } +def _measurement( + *, + boundary_f_0_5: float, + boundary_f_3_0: float, + functional_label_accuracy: float, + repetition_pairwise_f: float, + p50_latency_seconds: float, + p95_latency_seconds: float, + peak_rss_mib: float, +) -> dict[str, float]: + """Return one complete track/aggregate measurement record.""" + return { + "boundary_precision_0_5": min(boundary_f_0_5 + 0.02, 1.0), + "boundary_recall_0_5": max(boundary_f_0_5 - 0.02, 0.0), + "boundary_f_0_5": boundary_f_0_5, + "boundary_precision_3_0": min(boundary_f_3_0 + 0.02, 1.0), + "boundary_recall_3_0": max(boundary_f_3_0 - 0.02, 0.0), + "boundary_f_3_0": boundary_f_3_0, + "reference_to_estimate_median_deviation_seconds": 0.18, + "estimate_to_reference_median_deviation_seconds": 0.21, + "functional_label_accuracy": functional_label_accuracy, + "repetition_pairwise_f": repetition_pairwise_f, + "p50_latency_seconds": p50_latency_seconds, + "p95_latency_seconds": p95_latency_seconds, + "peak_rss_mib": peak_rss_mib, + } + + def _result(registration_sha256: str) -> dict[str, object]: """Return a result fixture whose paired intervals satisfy the registration.""" + baseline = _measurement( + boundary_f_0_5=0.70, + boundary_f_3_0=0.78, + functional_label_accuracy=0.72, + repetition_pairwise_f=0.74, + p50_latency_seconds=4.0, + p95_latency_seconds=6.0, + peak_rss_mib=650.0, + ) + candidate = _measurement( + boundary_f_0_5=0.695, + boundary_f_3_0=0.775, + functional_label_accuracy=0.715, + repetition_pairwise_f=0.738, + p50_latency_seconds=1.8, + p95_latency_seconds=4.2, + peak_rss_mib=590.0, + ) return { "schema_version": 1, "experiment_id": "structure-chroma-stft-vs-cqt-v1", "registration_sha256": registration_sha256, "corpus_track_ids": ["licensed-track-001", "licensed-track-002"], - "aggregate": { - "baseline": { - "boundary_f_0_5": 0.70, - "boundary_f_3_0": 0.78, - "functional_label_accuracy": 0.72, - "repetition_pairwise_f": 0.74, - "p50_latency_seconds": 4.0, - "p95_latency_seconds": 6.0, - "peak_rss_mib": 650.0, + "tracks": [ + { + "track_id": "licensed-track-001", + "baseline": copy.deepcopy(baseline), + "candidate": copy.deepcopy(candidate), }, - "candidate": { - "boundary_f_0_5": 0.695, - "boundary_f_3_0": 0.775, - "functional_label_accuracy": 0.715, - "repetition_pairwise_f": 0.738, - "p50_latency_seconds": 1.8, - "p95_latency_seconds": 4.2, - "peak_rss_mib": 590.0, + { + "track_id": "licensed-track-002", + "baseline": copy.deepcopy(baseline), + "candidate": copy.deepcopy(candidate), }, + ], + "aggregate": { + "baseline": baseline, + "candidate": candidate, }, "paired_delta_ci95": { "boundary_f_0_5": [-0.015, 0.005], @@ -123,6 +165,21 @@ def _result(registration_sha256: str) -> dict[str, object]: } +def _corpus(registration: dict[str, object]) -> list[dict[str, object]]: + """Return the typed corpus fixture from a registration.""" + value = registration["corpus"] + assert isinstance(value, list) + assert all(isinstance(track, dict) for track in value) + return value # type: ignore[return-value] + + +def _metrics(registration: dict[str, object]) -> dict[str, object]: + """Return the typed metric registry from a registration.""" + value = registration["metrics"] + assert isinstance(value, dict) + return value + + def test_registration_digest_is_canonical_and_accepts_required_evidence() -> None: """Equivalent key order must not change the frozen preregistration identity.""" validator = _validator() @@ -136,46 +193,37 @@ def test_registration_digest_is_canonical_and_accepts_required_evidence() -> Non assert digest == validator.registration_digest(reordered) -@pytest.mark.parametrize( - ("mutation", "message"), - [ - (lambda value: value["corpus"][0].update(rights_cleared=False), "rights_cleared"), - (lambda value: value["corpus"][0].update(audio_sha256="abcd"), "audio_sha256"), - ( - lambda value: value["corpus"].append(copy.deepcopy(value["corpus"][0])), - "duplicate track_id", - ), - ], -) -def test_registration_rejects_unverifiable_real_audio( - mutation: object, - message: str, -) -> None: +def test_registration_rejects_unverifiable_real_audio() -> None: """Rights, content identity, and unique track identity are mandatory evidence.""" validator = _validator() - registration = _registration() - mutation(registration) # type: ignore[operator] + no_rights = _registration() + _corpus(no_rights)[0]["rights_cleared"] = False + with pytest.raises(ValueError, match="rights_cleared"): + validator.validate_registration(no_rights) - with pytest.raises(ValueError, match=message): - validator.validate_registration(registration) + bad_hash = _registration() + _corpus(bad_hash)[0]["audio_sha256"] = "abcd" + with pytest.raises(ValueError, match="audio_sha256"): + validator.validate_registration(bad_hash) + + duplicate = _registration() + _corpus(duplicate).append(copy.deepcopy(_corpus(duplicate)[0])) + with pytest.raises(ValueError, match="duplicate track_id"): + validator.validate_registration(duplicate) def test_registration_rejects_incomplete_or_post_hoc_decision_contract() -> None: """Every predeclared quality and latency criterion must be present and bounded.""" validator = _validator() registration = _registration() - metrics = registration["metrics"] - assert isinstance(metrics, dict) - del metrics["boundary_f_3_0"] + del _metrics(registration)["boundary_f_3_0"] with pytest.raises(ValueError, match="boundary_f_3_0"): validator.validate_registration(registration) registration = _registration() - metrics = registration["metrics"] - assert isinstance(metrics, dict) - latency = metrics["p95_latency_ratio"] + latency = _metrics(registration)["p95_latency_ratio"] assert isinstance(latency, dict) latency["maximum_candidate_ratio"] = 1.0 @@ -215,6 +263,33 @@ def test_result_fails_quality_or_latency_when_ci_crosses_registered_boundary() - ] +def test_result_requires_track_level_metrics_and_boundary_deviations() -> None: + """Aggregate success cannot replace per-track recognized structure evidence.""" + validator = _validator() + registration = _registration() + digest = validator.registration_digest(registration) + + missing_track_metric = _result(digest) + tracks = missing_track_metric["tracks"] + assert isinstance(tracks, list) + first_track = tracks[0] + assert isinstance(first_track, dict) + baseline = first_track["baseline"] + assert isinstance(baseline, dict) + del baseline["reference_to_estimate_median_deviation_seconds"] + with pytest.raises(ValueError, match="reference_to_estimate_median_deviation_seconds"): + validator.evaluate_result(registration, missing_track_metric) + + missing_aggregate_metric = _result(digest) + aggregate = missing_aggregate_metric["aggregate"] + assert isinstance(aggregate, dict) + candidate = aggregate["candidate"] + assert isinstance(candidate, dict) + del candidate["estimate_to_reference_median_deviation_seconds"] + with pytest.raises(ValueError, match="estimate_to_reference_median_deviation_seconds"): + validator.evaluate_result(registration, missing_aggregate_metric) + + def test_result_is_bound_to_registration_corpus_and_finite_measurements() -> None: """Results cannot drift thresholds, corpus identity, or numerical validity.""" validator = _validator() From 7c2d0ad1eadc650ccc5ff24488ad47877ff63327 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 10:09:03 +0900 Subject: [PATCH 004/216] feat(mir): bind per-track structure evidence to preregistration --- .../validate_structure_noninferiority.py | 224 +++++++++++++----- 1 file changed, 167 insertions(+), 57 deletions(-) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index 9a0ae76b1..2dca9e640 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -1,12 +1,11 @@ #!/usr/bin/env python3 """Validate preregistered structure-feature noninferiority evidence. -This module deliberately does not compute MIR metrics. It binds a reviewed -registration to result receipts produced by the recognized evaluation pipeline, -then evaluates the preregistered confidence-interval decision rules. That keeps -metric implementation authority with MIREX/mir_eval-compatible tooling while -preventing thresholds, corpus identity, or runtime identity from drifting after -results are observed. +This module does not compute MIR metrics. It binds a reviewed registration to +result receipts produced by the recognized evaluation pipeline, then evaluates +the preregistered confidence-interval decision rules. Metric implementation +authority remains with MIREX/mir_eval-compatible tooling while thresholds, +corpus identity, and runtime identity are protected from post-result drift. """ from __future__ import annotations @@ -43,7 +42,15 @@ }, } _LATENCY_METRIC = "p95_latency_ratio" -_REPORT_METRICS = ( +_REPORT_SCORE_METRICS = ( + "boundary_precision_0_5", + "boundary_recall_0_5", + "boundary_precision_3_0", + "boundary_recall_3_0", +) +_REPORT_NONNEGATIVE_METRICS = ( + "reference_to_estimate_median_deviation_seconds", + "estimate_to_reference_median_deviation_seconds", "p50_latency_seconds", "p95_latency_seconds", "peak_rss_mib", @@ -51,21 +58,21 @@ def _mapping(value: object, field: str) -> Mapping[str, Any]: - """Return ``value`` as a mapping or raise a field-specific validation error.""" + """Return ``value`` as a mapping or raise a field-specific error.""" if not isinstance(value, Mapping): raise ValueError(f"{field} must be an object") return value def _sequence(value: object, field: str) -> Sequence[Any]: - """Return a non-string sequence or raise a field-specific validation error.""" + """Return a non-string sequence or raise a field-specific error.""" if isinstance(value, (str, bytes)) or not isinstance(value, Sequence): raise ValueError(f"{field} must be an array") return value def _nonempty_text(value: object, field: str) -> str: - """Return stripped non-empty text for a required textual field.""" + """Return stripped non-empty text for a required field.""" if not isinstance(value, str) or not value.strip(): raise ValueError(f"{field} must be non-empty text") return value.strip() @@ -114,7 +121,9 @@ def _reject_local_path(value: object, field: str) -> str: or _WINDOWS_ABSOLUTE_RE.match(text) is not None or lowered.startswith("file:") ): - raise ValueError(f"{field} must be a provenance URI, not a local filesystem path") + raise ValueError( + f"{field} must be a provenance URI, not a local filesystem path" + ) return text @@ -154,9 +163,16 @@ def _validate_metrics(metrics_value: object) -> None: config.get(config_name), f"metrics.{metric_name}.{config_name}", ) - if not math.isclose(actual_value, float(expected_value), rel_tol=0.0, abs_tol=1e-12): + expected_number = float(expected_value) + if not math.isclose( + actual_value, + expected_number, + rel_tol=0.0, + abs_tol=1e-12, + ): raise ValueError( - f"metrics.{metric_name}.{config_name} must equal {expected_value}" + f"metrics.{metric_name}.{config_name} must equal " + f"{expected_value}" ) margin = _finite_number( config.get("noninferiority_margin"), @@ -174,16 +190,18 @@ def _validate_metrics(metrics_value: object) -> None: ) if not 0.0 < maximum_ratio < 1.0: raise ValueError( - "metrics.p95_latency_ratio.maximum_candidate_ratio must be greater than 0 " - "and less than 1" + "metrics.p95_latency_ratio.maximum_candidate_ratio must be greater " + "than 0 and less than 1" ) def _validate_corpus(corpus_value: object) -> list[str]: - """Validate rights-cleared real-audio identities and return ordered track IDs.""" + """Validate rights-cleared real-audio identities and return ordered IDs.""" corpus = _sequence(corpus_value, "corpus") if len(corpus) < 2: - raise ValueError("corpus must contain at least two rights-cleared real-audio tracks") + raise ValueError( + "corpus must contain at least two rights-cleared real-audio tracks" + ) seen: set[str] = set() track_ids: list[str] = [] @@ -205,35 +223,51 @@ def _validate_corpus(corpus_value: object) -> list[str]: def _validate_runtime(runtime_value: object) -> None: - """Validate exact runtime identity needed to reproduce paired measurements.""" + """Validate exact runtime identity for reproducible paired measurements.""" runtime = _mapping(runtime_value, "runtime") _commit(runtime.get("source_commit"), "runtime.source_commit") _sha256(runtime.get("uv_lock_sha256"), "runtime.uv_lock_sha256") - for field in ("python_version", "librosa_version", "numpy_version", "host_profile"): + for field in ( + "python_version", + "librosa_version", + "numpy_version", + "host_profile", + ): _nonempty_text(runtime.get(field), f"runtime.{field}") - sample_rate = _finite_number(runtime.get("sample_rate_hz"), "runtime.sample_rate_hz") + sample_rate = _finite_number( + runtime.get("sample_rate_hz"), + "runtime.sample_rate_hz", + ) if not 1.0 <= sample_rate <= 384000.0 or not sample_rate.is_integer(): raise ValueError("runtime.sample_rate_hz must be an integer in 1..384000") channels = runtime.get("channels") if isinstance(channels, bool) or channels != 1: - raise ValueError("runtime.channels must equal 1 for the registered segmentation input") + raise ValueError( + "runtime.channels must equal 1 for the registered segmentation input" + ) def validate_registration(registration_value: object) -> None: """Validate a frozen STFT-vs-CQT structure experiment registration. - The validator intentionally requires content hashes, rights evidence, exact - runtime identity, recognized metric implementations, and explicit margins. - It does not decide what the margins should be; that scientific/product - choice must be reviewed before any corpus result is inspected. + This function requires content hashes, rights evidence, exact runtime + identity, recognized metric implementations, and explicit margins. It does + not decide what those margins should be; that scientific/product choice must + be reviewed before any corpus result is inspected. """ registration = _mapping(registration_value, "registration") - _validate_schema_version(registration.get("schema_version"), "schema_version") + _validate_schema_version( + registration.get("schema_version"), + "schema_version", + ) _nonempty_text(registration.get("experiment_id"), "experiment_id") hypothesis = _mapping(registration.get("hypothesis"), "hypothesis") - baseline = _nonempty_text(hypothesis.get("baseline_feature"), "hypothesis.baseline_feature") + baseline = _nonempty_text( + hypothesis.get("baseline_feature"), + "hypothesis.baseline_feature", + ) candidate = _nonempty_text( hypothesis.get("candidate_feature"), "hypothesis.candidate_feature", @@ -273,19 +307,32 @@ def _confidence_interval(value: object, field: str) -> tuple[float, float]: return lower, upper -def _validate_aggregate_side(side_value: object, field: str) -> dict[str, float]: - """Validate one aggregate baseline/candidate result side.""" +def _validate_measurement_side( + side_value: object, + field: str, +) -> dict[str, float]: + """Validate one per-track or aggregate baseline/candidate measurement.""" side = _mapping(side_value, field) normalized: dict[str, float] = {} for metric_name in _QUALITY_METRICS: - normalized[metric_name] = _score(side.get(metric_name), f"{field}.{metric_name}") - for metric_name in _REPORT_METRICS: + normalized[metric_name] = _score( + side.get(metric_name), + f"{field}.{metric_name}", + ) + for metric_name in _REPORT_SCORE_METRICS: + normalized[metric_name] = _score( + side.get(metric_name), + f"{field}.{metric_name}", + ) + for metric_name in _REPORT_NONNEGATIVE_METRICS: value = _finite_number(side.get(metric_name), f"{field}.{metric_name}") if value < 0.0: raise ValueError(f"{field}.{metric_name} must be non-negative") normalized[metric_name] = value if normalized["p95_latency_seconds"] < normalized["p50_latency_seconds"]: - raise ValueError(f"{field}.p95_latency_seconds must be >= p50_latency_seconds") + raise ValueError( + f"{field}.p95_latency_seconds must be >= p50_latency_seconds" + ) return normalized @@ -293,48 +340,95 @@ def _validate_result_identity( registration: Mapping[str, Any], result: Mapping[str, Any], ) -> list[str]: - """Bind a result receipt to the frozen registration and exact corpus order.""" - _validate_schema_version(result.get("schema_version"), "result.schema_version") - experiment_id = _nonempty_text(result.get("experiment_id"), "result.experiment_id") + """Bind a result receipt to the frozen registration and corpus order.""" + _validate_schema_version( + result.get("schema_version"), + "result.schema_version", + ) + experiment_id = _nonempty_text( + result.get("experiment_id"), + "result.experiment_id", + ) if experiment_id != registration["experiment_id"]: raise ValueError("result.experiment_id does not match registration") expected_digest = registration_digest(registration) - actual_digest = _sha256(result.get("registration_sha256"), "result.registration_sha256") + actual_digest = _sha256( + result.get("registration_sha256"), + "result.registration_sha256", + ) if actual_digest != expected_digest: - raise ValueError("result.registration_sha256 does not match the frozen registration") + raise ValueError( + "result.registration_sha256 does not match the frozen registration" + ) expected_track_ids = _validate_corpus(registration["corpus"]) - actual_track_values = _sequence(result.get("corpus_track_ids"), "result.corpus_track_ids") + actual_values = _sequence( + result.get("corpus_track_ids"), + "result.corpus_track_ids", + ) actual_track_ids = [ _nonempty_text(value, f"result.corpus_track_ids[{index}]") - for index, value in enumerate(actual_track_values) + for index, value in enumerate(actual_values) ] if actual_track_ids != expected_track_ids: - raise ValueError("result.corpus_track_ids must exactly match registration corpus order") + raise ValueError( + "result.corpus_track_ids must exactly match registration corpus order" + ) return expected_track_ids +def _validate_track_measurements( + result: Mapping[str, Any], + expected_track_ids: list[str], +) -> None: + """Require one complete baseline/candidate receipt for every corpus track.""" + tracks = _sequence(result.get("tracks"), "result.tracks") + if len(tracks) != len(expected_track_ids): + raise ValueError("result.tracks must contain exactly one receipt per corpus track") + + actual_track_ids: list[str] = [] + for index, raw_track in enumerate(tracks): + field = f"result.tracks[{index}]" + track = _mapping(raw_track, field) + track_id = _nonempty_text(track.get("track_id"), f"{field}.track_id") + actual_track_ids.append(track_id) + _validate_measurement_side(track.get("baseline"), f"{field}.baseline") + _validate_measurement_side(track.get("candidate"), f"{field}.candidate") + if actual_track_ids != expected_track_ids: + raise ValueError("result.tracks must preserve the registered corpus order") + + def evaluate_result( registration_value: object, result_value: object, ) -> dict[str, object]: - """Validate a result receipt and evaluate preregistered CI-based decisions.""" + """Validate a result receipt and evaluate preregistered CI decisions.""" validate_registration(registration_value) registration = _mapping(registration_value, "registration") result = _mapping(result_value, "result") track_ids = _validate_result_identity(registration, result) + _validate_track_measurements(result, track_ids) aggregate = _mapping(result.get("aggregate"), "result.aggregate") - baseline = _validate_aggregate_side(aggregate.get("baseline"), "result.aggregate.baseline") - candidate = _validate_aggregate_side( + baseline = _validate_measurement_side( + aggregate.get("baseline"), + "result.aggregate.baseline", + ) + candidate = _validate_measurement_side( aggregate.get("candidate"), "result.aggregate.candidate", ) - raw_intervals = _mapping(result.get("paired_delta_ci95"), "result.paired_delta_ci95") + raw_intervals = _mapping( + result.get("paired_delta_ci95"), + "result.paired_delta_ci95", + ) if set(raw_intervals) != set(_QUALITY_METRICS): - raise ValueError("result.paired_delta_ci95 must contain exactly the registered quality metrics") + raise ValueError( + "result.paired_delta_ci95 must contain exactly the registered " + "quality metrics" + ) intervals: dict[str, tuple[float, float]] = {} for metric_name in _QUALITY_METRICS: @@ -345,7 +439,8 @@ def evaluate_result( point_delta = candidate[metric_name] - baseline[metric_name] if not interval[0] <= point_delta <= interval[1]: raise ValueError( - f"result.paired_delta_ci95.{metric_name} must contain the aggregate point delta" + f"result.paired_delta_ci95.{metric_name} must contain the " + "aggregate point delta" ) intervals[metric_name] = interval @@ -354,41 +449,56 @@ def evaluate_result( "result.p95_latency_ratio_ci95", ) if latency_interval[0] <= 0.0: - raise ValueError("result.p95_latency_ratio_ci95 bounds must be greater than 0") + raise ValueError( + "result.p95_latency_ratio_ci95 bounds must be greater than 0" + ) baseline_p95 = baseline["p95_latency_seconds"] if baseline_p95 <= 0.0: - raise ValueError("result.aggregate.baseline.p95_latency_seconds must be greater than 0") + raise ValueError( + "result.aggregate.baseline.p95_latency_seconds must be greater than 0" + ) latency_ratio = candidate["p95_latency_seconds"] / baseline_p95 if not latency_interval[0] <= latency_ratio <= latency_interval[1]: - raise ValueError("result.p95_latency_ratio_ci95 must contain the aggregate p95 ratio") + raise ValueError( + "result.p95_latency_ratio_ci95 must contain the aggregate p95 ratio" + ) - failed_tracks_raw = _sequence(result.get("failed_tracks"), "result.failed_tracks") + failed_values = _sequence( + result.get("failed_tracks"), + "result.failed_tracks", + ) failed_tracks = [ _nonempty_text(value, f"result.failed_tracks[{index}]") - for index, value in enumerate(failed_tracks_raw) + for index, value in enumerate(failed_values) ] if len(set(failed_tracks)) != len(failed_tracks): raise ValueError("result.failed_tracks must not contain duplicates") unknown_failed = sorted(set(failed_tracks) - set(track_ids)) if unknown_failed: - raise ValueError(f"result.failed_tracks contains unknown track_id: {unknown_failed[0]}") + raise ValueError( + f"result.failed_tracks contains unknown track_id: {unknown_failed[0]}" + ) _nonempty_text(result.get("claim_boundary"), "result.claim_boundary") metrics = _mapping(registration["metrics"], "metrics") failed_requirements: list[str] = [] for metric_name in _QUALITY_METRICS: - metric_config = _mapping(metrics[metric_name], f"metrics.{metric_name}") + config = _mapping(metrics[metric_name], f"metrics.{metric_name}") margin = _finite_number( - metric_config.get("noninferiority_margin"), + config.get("noninferiority_margin"), f"metrics.{metric_name}.noninferiority_margin", ) lower = intervals[metric_name][0] if lower < -margin: failed_requirements.append( - f"{metric_name} paired CI lower bound {lower:.6f} is below -{margin:.6f}" + f"{metric_name} paired CI lower bound {lower:.6f} " + f"is below -{margin:.6f}" ) - latency_config = _mapping(metrics[_LATENCY_METRIC], f"metrics.{_LATENCY_METRIC}") + latency_config = _mapping( + metrics[_LATENCY_METRIC], + f"metrics.{_LATENCY_METRIC}", + ) maximum_ratio = _finite_number( latency_config.get("maximum_candidate_ratio"), f"metrics.{_LATENCY_METRIC}.maximum_candidate_ratio", @@ -415,7 +525,7 @@ def _load_json(path: Path) -> object: def main(argv: Sequence[str] | None = None) -> int: - """Validate a registration and optionally evaluate one bound result receipt.""" + """Validate a registration and optionally evaluate one bound result.""" parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("registration", type=Path) parser.add_argument("result", type=Path, nargs="?") From 0d32e179c311bdd38f7449ba2d9b3256caff85f1 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 10:09:49 +0900 Subject: [PATCH 005/216] docs(mir): doctor structure feature noninferiority gate --- .../structure-feature-noninferiority.md | 97 +++++++++++++++++++ 1 file changed, 97 insertions(+) create mode 100644 docs/doctoring/structure-feature-noninferiority.md diff --git a/docs/doctoring/structure-feature-noninferiority.md b/docs/doctoring/structure-feature-noninferiority.md new file mode 100644 index 000000000..1d3aeff9f --- /dev/null +++ b/docs/doctoring/structure-feature-noninferiority.md @@ -0,0 +1,97 @@ +# Structure feature noninferiority gate + +Status: Proposed +Owner: Signal-MIR Analysis +Tracking: #1225 +Production baseline: `chroma_cqt` in `sections/segmenter.py` + +## Decision + +The earlier `chroma_cqt` → `chroma_stft` optimization is not a production change until a preregistered, rights-cleared real-audio experiment shows that the candidate is noninferior on rehearsal-relevant structure quality and materially faster on the same corpus and runtime identity. + +The repository now owns a result-admission boundary in `scripts/research/validate_structure_noninferiority.py`. The validator does **not** calculate MIR metrics and does not choose scientific margins. It freezes the reviewed experiment contract, binds a result receipt to that contract by canonical SHA-256, requires per-track and aggregate evidence, and applies the registered confidence-interval decision rules. Metric computation remains a separate scientific measurement step. + +This keeps a performance result from silently changing the corpus, thresholds, feature identities, or runtime after the measurements are visible. + +## Registered inputs + +A registration is valid only when it records all of the following before the result is evaluated: + +- baseline `chroma_cqt` and candidate `chroma_stft`; +- a rights-cleared real-audio corpus with stable track IDs, audio SHA-256, annotation SHA-256, rights basis, and provenance URI; +- exact source commit and `uv.lock` identity plus Python, librosa, NumPy, sample rate, channel count, and host profile; +- the complete metric implementation contract and noninferiority/speed thresholds. + +The validator rejects local filesystem paths as provenance authorities. A benchmark may remain private when licensing requires that, but the receipt must identify the licensed material without leaking the local path that happened to hold it. + +The current minimum of two tracks is only a technical guard against treating one timing sample as a corpus. It is **not** a scientific sample-size claim. Corpus breadth, genre/instrumentation coverage, annotation quality, and a defensible power/uncertainty plan remain part of the experiment review before a production switch can be accepted. + +## Metrics + +BandScope uses MIREX 2025 Music Structure Analysis as the functional-structure reference point. The registered quality gate requires: + +| Evidence | Registered implementation / convention | Decision use | +| --- | --- | --- | +| Boundary precision/recall/F at 0.5 s | `mir_eval.segment.detection`, 0.5 s window | F noninferiority | +| Boundary precision/recall/F at 3.0 s | `mir_eval.segment.detection`, 3.0 s window | F noninferiority | +| Boundary median deviation, both directions | `mir_eval.segment.deviation` convention | Report per track and aggregate | +| Functional-label accuracy | MIREX 2025 frame-level ACC convention | Noninferiority | +| Repetition/group consistency | `mir_eval.segment.pairwise`, 0.1 s frame size | F noninferiority | +| Latency | paired p50/p95 on the registered host | p95 ratio superiority | +| Peak memory | peak RSS on the registered host | Report per track and aggregate | + +MIREX 2025 evaluates functional structure with frame-level label accuracy plus boundary hit-rate F measures at 0.5 s and 3.0 s. `mir_eval.segment.detection` uses one-to-one boundary matching within the selected tolerance; `mir_eval.segment.deviation` reports the median nearest-boundary deviations in both directions; `mir_eval.segment.pairwise` measures structural grouping agreement. These are different questions and must not be collapsed into one score. + +The published MIREX 2025 result table reports the MusicFM baseline at ACC 0.705, HR.5 0.644, and HR3 0.710. Those values are useful external context, **not** BandScope's noninferiority margins. A margin is a product/scientific decision about acceptable loss relative to BandScope's own current baseline on the registered corpus. It must be fixed before examining candidate results. + +## Decision rule + +For each quality metric `m`, let `Δm = candidate - baseline`. The result passes that quality criterion only when the lower bound of the registered paired 95% confidence interval satisfies: + +`lower_CI95(Δm) >= -noninferiority_margin(m)` + +For p95 latency, let `r = candidate_p95 / baseline_p95`. The result passes the speed criterion only when the **upper** bound of the paired 95% interval satisfies: + +`upper_CI95(r) <= maximum_candidate_ratio` + +The validator also requires the aggregate point delta/ratio to lie inside the supplied interval. Favorable point estimates therefore cannot hide an uncertainty interval that crosses a preregistered boundary. + +The repository does not currently contain approved numeric margins. Unit-test values are synthetic policy fixtures only and must never be cited as production acceptance thresholds. + +## Result receipt + +A result receipt must contain the exact registration digest, exact corpus order, one complete baseline/candidate measurement pair per registered track, aggregate measurements, paired 95% intervals for every gated quality metric, a paired p95-latency-ratio interval, failed-track IDs, and a claim boundary. + +The receipt is rejected when a track is omitted/reordered, a required metric is absent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A passing receipt means only that the preregistered decision rule passed for the registered corpus and runtime. It does not generalize automatically to other genres, codecs, sample rates, machines, or annotation regimes. + +## Reproducibility sequence + +1. Review the rights basis, corpus composition, feature hypothesis, metric contract, host profile, and numeric margins **before** running the candidate. +2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain its canonical SHA-256. +3. Run baseline and candidate on the same decoded track identities and host profile. Record per-track recognized MIR metrics, p50/p95 latency, peak RSS, failures, and paired uncertainty. +4. Put the registration digest in the result receipt and run `python scripts/research/validate_structure_noninferiority.py `. +5. Preserve the registration, result receipt, corpus/annotation hashes, exact source commit, lock hash, and measurement procedure together. A production feature switch requires this evidence plus normal code review and protected-head checks. + +No step authorizes committing licensed audio to Git. Rights-cleared means BandScope has the necessary evaluation right; redistribution is a separate permission. + +## Security Notes + +- Audio and annotation files are untrusted inputs to the future experiment runner. This validator reads JSON evidence only and does not open audio, execute subprocesses, make network requests, or follow paths from the registration. +- `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute paths and `file:` URIs are rejected to avoid turning transient workstation paths into provenance authority or leaking them into review artifacts. +- Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt. +- Invalid/non-finite measurements, corpus drift, registration drift, missing per-track evidence, and post-hoc metric additions fail closed. +- The validator does not claim that SHA-256 proves licensing, annotation validity, or scientific adequacy. Rights basis and scientific review remain separate gates. + +## References + +Buisson, M., McFee, B., Essid, S., & Crayencour, H. C. (2024). Self-supervised learning of multi-level audio representations for music segmentation. *IEEE/ACM Transactions on Audio, Speech, and Language Processing, 32*, 2141–2152. https://doi.org/10.1109/TASLP.2024.3379894 + +Kim, T., & Nam, J. (2023). All-in-one metrical and functional structure analysis with neighborhood attentions on demixed audio. In *2023 IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA)* (pp. 1–5). IEEE. https://doi.org/10.1109/WASPAA58266.2023.10248148 + +MIREX. (2025). *Music Structure Analysis*. https://music-ir.org/mirex/wiki/2025:Music_Structure_Analysis + +MIREX. (2025). *Music Structure Analysis Results*. https://music-ir.org/mirex/wiki/2025:Music_Structure_Analysis_Results + +mir_eval contributors. (n.d.). *mir_eval.segment: Structural segmentation evaluation*. https://github.com/mir-evaluation/mir_eval/blob/main/mir_eval/segment.py + +Wang, J.-C., Hung, Y.-N., & Smith, J. B. L. (2022). To catch a chorus, verse, intro, or anything else: Analyzing a song with structural functions. In *ICASSP 2022—2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)* (pp. 416–420). IEEE. https://arxiv.org/abs/2205.14700 From f25ba43d3a0154be123833fe0a950881b17e6c59 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 10:12:12 +0900 Subject: [PATCH 006/216] docs(mir): place noninferiority record under traceability --- .../mir/structure-feature-noninferiority.md | 97 +++++++++++++++++++ 1 file changed, 97 insertions(+) create mode 100644 docs/traceability/mir/structure-feature-noninferiority.md diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md new file mode 100644 index 000000000..97fbe28f3 --- /dev/null +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -0,0 +1,97 @@ +# Structure feature noninferiority gate + +Status: Proposed +Owner: Signal-MIR Analysis +Tracking: #1225 +Production baseline: `chroma_cqt` in `sections/segmenter.py` + +## Decision + +The earlier `chroma_cqt` → `chroma_stft` optimization is not a production change until a preregistered, rights-cleared real-audio experiment shows that the candidate is noninferior on rehearsal-relevant structure quality and materially faster on the same corpus and runtime identity. + +The repository now owns a result-admission boundary in `scripts/research/validate_structure_noninferiority.py`. The validator does **not** calculate MIR metrics and does not choose scientific margins. It freezes the reviewed experiment contract, binds a result receipt to that contract by canonical SHA-256, requires per-track and aggregate evidence, and applies the registered confidence-interval decision rules. Metric computation remains a separate scientific measurement step. + +This keeps a performance result from silently changing the corpus, thresholds, feature identities, or runtime after the measurements are visible. + +## Registered inputs + +A registration is valid only when it records all of the following before the result is evaluated: + +- baseline `chroma_cqt` and candidate `chroma_stft`; +- a rights-cleared real-audio corpus with stable track IDs, audio SHA-256, annotation SHA-256, rights basis, and provenance URI; +- exact source commit and `uv.lock` identity plus Python, librosa, NumPy, sample rate, channel count, and host profile; +- the complete metric implementation contract and noninferiority/speed thresholds. + +The validator rejects local filesystem paths as provenance authorities. A benchmark may remain private when licensing requires that, but the receipt must identify the licensed material without leaking the local path that happened to hold it. + +The current minimum of two tracks is only a technical guard against treating one timing sample as a corpus. It is **not** a scientific sample-size claim. Corpus breadth, genre/instrumentation coverage, annotation quality, and a defensible power/uncertainty plan remain part of the experiment review before a production switch can be accepted. + +## Metrics + +BandScope uses MIREX 2025 Music Structure Analysis as the functional-structure reference point. The registered quality gate requires: + +| Evidence | Registered implementation / convention | Decision use | +| --- | --- | --- | +| Boundary precision/recall/F at 0.5 s | `mir_eval.segment.detection`, 0.5 s window | F noninferiority | +| Boundary precision/recall/F at 3.0 s | `mir_eval.segment.detection`, 3.0 s window | F noninferiority | +| Boundary median deviation, both directions | `mir_eval.segment.deviation` convention | Report per track and aggregate | +| Functional-label accuracy | MIREX 2025 frame-level ACC convention | Noninferiority | +| Repetition/group consistency | `mir_eval.segment.pairwise`, 0.1 s frame size | F noninferiority | +| Latency | paired p50/p95 on the registered host | p95 ratio superiority | +| Peak memory | peak RSS on the registered host | Report per track and aggregate | + +MIREX 2025 evaluates functional structure with frame-level label accuracy plus boundary hit-rate F measures at 0.5 s and 3.0 s. `mir_eval.segment.detection` uses one-to-one boundary matching within the selected tolerance; `mir_eval.segment.deviation` reports the median nearest-boundary deviations in both directions; `mir_eval.segment.pairwise` measures structural grouping agreement. These are different questions and must not be collapsed into one score. + +The published MIREX 2025 result table reports the MusicFM baseline at ACC 0.705, HR.5 0.644, and HR3 0.710. Those values are useful external context, **not** BandScope's noninferiority margins. A margin is a product/scientific decision about acceptable loss relative to BandScope's own current baseline on the registered corpus. It must be fixed before examining candidate results. + +## Decision rule + +For each quality metric `m`, let `Δm = candidate - baseline`. The result passes that quality criterion only when the lower bound of the registered paired 95% confidence interval satisfies: + +`lower_CI95(Δm) >= -noninferiority_margin(m)` + +For p95 latency, let `r = candidate_p95 / baseline_p95`. The result passes the speed criterion only when the **upper** bound of the paired 95% interval satisfies: + +`upper_CI95(r) <= maximum_candidate_ratio` + +The validator also requires the aggregate point delta/ratio to lie inside the supplied interval. Favorable point estimates therefore cannot hide an uncertainty interval that crosses a preregistered boundary. + +The repository does not currently contain approved numeric margins. Unit-test values are synthetic policy fixtures only and must never be cited as production acceptance thresholds. + +## Result receipt + +A result receipt must contain the exact registration digest, exact corpus order, one complete baseline/candidate measurement pair per registered track, aggregate measurements, paired 95% intervals for every gated quality metric, a paired p95-latency-ratio interval, failed-track IDs, and a claim boundary. + +The receipt is rejected when a track is omitted/reordered, a required metric is absent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A passing receipt means only that the preregistered decision rule passed for the registered corpus and runtime. It does not generalize automatically to other genres, codecs, sample rates, machines, or annotation regimes. + +## Reproducibility sequence + +1. Review the rights basis, corpus composition, feature hypothesis, metric contract, host profile, and numeric margins **before** running the candidate. +2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain its canonical SHA-256. +3. Run baseline and candidate on the same decoded track identities and host profile. Record per-track recognized MIR metrics, p50/p95 latency, peak RSS, failures, and paired uncertainty. +4. Put the registration digest in the result receipt and run `python scripts/research/validate_structure_noninferiority.py `. +5. Preserve the registration, result receipt, corpus/annotation hashes, exact source commit, lock hash, and measurement procedure together. A production feature switch requires this evidence plus normal code review and protected-head checks. + +No step authorizes committing licensed audio to Git. Rights-cleared means BandScope has the necessary evaluation right; redistribution is a separate permission. + +## Security Notes + +- Audio and annotation files are untrusted inputs to the future experiment runner. This validator reads JSON evidence only and does not open audio, execute subprocesses, make network requests, or follow paths from the registration. +- `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute paths and `file:` URIs are rejected to avoid turning transient workstation paths into provenance authority or leaking them into review artifacts. +- Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt. +- Invalid/non-finite measurements, corpus drift, registration drift, missing per-track evidence, and post-hoc metric additions fail closed. +- The validator does not claim that SHA-256 proves licensing, annotation validity, or scientific adequacy. Rights basis and scientific review remain separate gates. + +## References + +Buisson, M., McFee, B., Essid, S., & Crayencour, H. C. (2024). Self-supervised learning of multi-level audio representations for music segmentation. *IEEE/ACM Transactions on Audio, Speech, and Language Processing, 32*, 2141–2152. https://doi.org/10.1109/TASLP.2024.3379894 + +Kim, T., & Nam, J. (2023). All-in-one metrical and functional structure analysis with neighborhood attentions on demixed audio. In *2023 IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA)* (pp. 1–5). IEEE. https://doi.org/10.1109/WASPAA58266.2023.10248148 + +MIREX. (2025). *Music Structure Analysis*. https://music-ir.org/mirex/wiki/2025:Music_Structure_Analysis + +MIREX. (2025). *Music Structure Analysis Results*. https://music-ir.org/mirex/wiki/2025:Music_Structure_Analysis_Results + +mir_eval contributors. (n.d.). *mir_eval.segment: Structural segmentation evaluation*. https://github.com/mir-evaluation/mir_eval/blob/main/mir_eval/segment.py + +Wang, J.-C., Hung, Y.-N., & Smith, J. B. L. (2022). To catch a chorus, verse, intro, or anything else: Analyzing a song with structural functions. In *ICASSP 2022—2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)* (pp. 416–420). IEEE. https://arxiv.org/abs/2205.14700 \ No newline at end of file From 6a51021a71917d1a9f43924615b9f9b297454ee7 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 10:12:17 +0900 Subject: [PATCH 007/216] docs(mir): remove duplicate doctoring record --- .../structure-feature-noninferiority.md | 97 ------------------- 1 file changed, 97 deletions(-) delete mode 100644 docs/doctoring/structure-feature-noninferiority.md diff --git a/docs/doctoring/structure-feature-noninferiority.md b/docs/doctoring/structure-feature-noninferiority.md deleted file mode 100644 index 1d3aeff9f..000000000 --- a/docs/doctoring/structure-feature-noninferiority.md +++ /dev/null @@ -1,97 +0,0 @@ -# Structure feature noninferiority gate - -Status: Proposed -Owner: Signal-MIR Analysis -Tracking: #1225 -Production baseline: `chroma_cqt` in `sections/segmenter.py` - -## Decision - -The earlier `chroma_cqt` → `chroma_stft` optimization is not a production change until a preregistered, rights-cleared real-audio experiment shows that the candidate is noninferior on rehearsal-relevant structure quality and materially faster on the same corpus and runtime identity. - -The repository now owns a result-admission boundary in `scripts/research/validate_structure_noninferiority.py`. The validator does **not** calculate MIR metrics and does not choose scientific margins. It freezes the reviewed experiment contract, binds a result receipt to that contract by canonical SHA-256, requires per-track and aggregate evidence, and applies the registered confidence-interval decision rules. Metric computation remains a separate scientific measurement step. - -This keeps a performance result from silently changing the corpus, thresholds, feature identities, or runtime after the measurements are visible. - -## Registered inputs - -A registration is valid only when it records all of the following before the result is evaluated: - -- baseline `chroma_cqt` and candidate `chroma_stft`; -- a rights-cleared real-audio corpus with stable track IDs, audio SHA-256, annotation SHA-256, rights basis, and provenance URI; -- exact source commit and `uv.lock` identity plus Python, librosa, NumPy, sample rate, channel count, and host profile; -- the complete metric implementation contract and noninferiority/speed thresholds. - -The validator rejects local filesystem paths as provenance authorities. A benchmark may remain private when licensing requires that, but the receipt must identify the licensed material without leaking the local path that happened to hold it. - -The current minimum of two tracks is only a technical guard against treating one timing sample as a corpus. It is **not** a scientific sample-size claim. Corpus breadth, genre/instrumentation coverage, annotation quality, and a defensible power/uncertainty plan remain part of the experiment review before a production switch can be accepted. - -## Metrics - -BandScope uses MIREX 2025 Music Structure Analysis as the functional-structure reference point. The registered quality gate requires: - -| Evidence | Registered implementation / convention | Decision use | -| --- | --- | --- | -| Boundary precision/recall/F at 0.5 s | `mir_eval.segment.detection`, 0.5 s window | F noninferiority | -| Boundary precision/recall/F at 3.0 s | `mir_eval.segment.detection`, 3.0 s window | F noninferiority | -| Boundary median deviation, both directions | `mir_eval.segment.deviation` convention | Report per track and aggregate | -| Functional-label accuracy | MIREX 2025 frame-level ACC convention | Noninferiority | -| Repetition/group consistency | `mir_eval.segment.pairwise`, 0.1 s frame size | F noninferiority | -| Latency | paired p50/p95 on the registered host | p95 ratio superiority | -| Peak memory | peak RSS on the registered host | Report per track and aggregate | - -MIREX 2025 evaluates functional structure with frame-level label accuracy plus boundary hit-rate F measures at 0.5 s and 3.0 s. `mir_eval.segment.detection` uses one-to-one boundary matching within the selected tolerance; `mir_eval.segment.deviation` reports the median nearest-boundary deviations in both directions; `mir_eval.segment.pairwise` measures structural grouping agreement. These are different questions and must not be collapsed into one score. - -The published MIREX 2025 result table reports the MusicFM baseline at ACC 0.705, HR.5 0.644, and HR3 0.710. Those values are useful external context, **not** BandScope's noninferiority margins. A margin is a product/scientific decision about acceptable loss relative to BandScope's own current baseline on the registered corpus. It must be fixed before examining candidate results. - -## Decision rule - -For each quality metric `m`, let `Δm = candidate - baseline`. The result passes that quality criterion only when the lower bound of the registered paired 95% confidence interval satisfies: - -`lower_CI95(Δm) >= -noninferiority_margin(m)` - -For p95 latency, let `r = candidate_p95 / baseline_p95`. The result passes the speed criterion only when the **upper** bound of the paired 95% interval satisfies: - -`upper_CI95(r) <= maximum_candidate_ratio` - -The validator also requires the aggregate point delta/ratio to lie inside the supplied interval. Favorable point estimates therefore cannot hide an uncertainty interval that crosses a preregistered boundary. - -The repository does not currently contain approved numeric margins. Unit-test values are synthetic policy fixtures only and must never be cited as production acceptance thresholds. - -## Result receipt - -A result receipt must contain the exact registration digest, exact corpus order, one complete baseline/candidate measurement pair per registered track, aggregate measurements, paired 95% intervals for every gated quality metric, a paired p95-latency-ratio interval, failed-track IDs, and a claim boundary. - -The receipt is rejected when a track is omitted/reordered, a required metric is absent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A passing receipt means only that the preregistered decision rule passed for the registered corpus and runtime. It does not generalize automatically to other genres, codecs, sample rates, machines, or annotation regimes. - -## Reproducibility sequence - -1. Review the rights basis, corpus composition, feature hypothesis, metric contract, host profile, and numeric margins **before** running the candidate. -2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain its canonical SHA-256. -3. Run baseline and candidate on the same decoded track identities and host profile. Record per-track recognized MIR metrics, p50/p95 latency, peak RSS, failures, and paired uncertainty. -4. Put the registration digest in the result receipt and run `python scripts/research/validate_structure_noninferiority.py `. -5. Preserve the registration, result receipt, corpus/annotation hashes, exact source commit, lock hash, and measurement procedure together. A production feature switch requires this evidence plus normal code review and protected-head checks. - -No step authorizes committing licensed audio to Git. Rights-cleared means BandScope has the necessary evaluation right; redistribution is a separate permission. - -## Security Notes - -- Audio and annotation files are untrusted inputs to the future experiment runner. This validator reads JSON evidence only and does not open audio, execute subprocesses, make network requests, or follow paths from the registration. -- `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute paths and `file:` URIs are rejected to avoid turning transient workstation paths into provenance authority or leaking them into review artifacts. -- Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt. -- Invalid/non-finite measurements, corpus drift, registration drift, missing per-track evidence, and post-hoc metric additions fail closed. -- The validator does not claim that SHA-256 proves licensing, annotation validity, or scientific adequacy. Rights basis and scientific review remain separate gates. - -## References - -Buisson, M., McFee, B., Essid, S., & Crayencour, H. C. (2024). Self-supervised learning of multi-level audio representations for music segmentation. *IEEE/ACM Transactions on Audio, Speech, and Language Processing, 32*, 2141–2152. https://doi.org/10.1109/TASLP.2024.3379894 - -Kim, T., & Nam, J. (2023). All-in-one metrical and functional structure analysis with neighborhood attentions on demixed audio. In *2023 IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA)* (pp. 1–5). IEEE. https://doi.org/10.1109/WASPAA58266.2023.10248148 - -MIREX. (2025). *Music Structure Analysis*. https://music-ir.org/mirex/wiki/2025:Music_Structure_Analysis - -MIREX. (2025). *Music Structure Analysis Results*. https://music-ir.org/mirex/wiki/2025:Music_Structure_Analysis_Results - -mir_eval contributors. (n.d.). *mir_eval.segment: Structural segmentation evaluation*. https://github.com/mir-evaluation/mir_eval/blob/main/mir_eval/segment.py - -Wang, J.-C., Hung, Y.-N., & Smith, J. B. L. (2022). To catch a chorus, verse, intro, or anything else: Analyzing a song with structural functions. In *ICASSP 2022—2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)* (pp. 416–420). IEEE. https://arxiv.org/abs/2205.14700 From 677b30939ec3a7d270994e9e9a76c111fcb2e108 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 10:15:18 +0900 Subject: [PATCH 008/216] test(mir): freeze uncertainty plan and metric triplets --- .../test_structure_noninferiority_policy.py | 80 +++++++++++++++---- 1 file changed, 65 insertions(+), 15 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_policy.py index 2d2355994..644cf7d94 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_policy.py @@ -51,6 +51,12 @@ def _registration() -> dict[str, object]: "maximum_candidate_ratio": 0.8, }, }, + "uncertainty": { + "procedure_id": "paired-track-bootstrap-v1", + "confidence_level": 0.95, + "resamples": 10000, + "random_seed": 20260917, + }, "corpus": [ { "track_id": "licensed-track-001", @@ -94,15 +100,17 @@ def _measurement( ) -> dict[str, float]: """Return one complete track/aggregate measurement record.""" return { - "boundary_precision_0_5": min(boundary_f_0_5 + 0.02, 1.0), - "boundary_recall_0_5": max(boundary_f_0_5 - 0.02, 0.0), + "boundary_precision_0_5": boundary_f_0_5, + "boundary_recall_0_5": boundary_f_0_5, "boundary_f_0_5": boundary_f_0_5, - "boundary_precision_3_0": min(boundary_f_3_0 + 0.02, 1.0), - "boundary_recall_3_0": max(boundary_f_3_0 - 0.02, 0.0), + "boundary_precision_3_0": boundary_f_3_0, + "boundary_recall_3_0": boundary_f_3_0, "boundary_f_3_0": boundary_f_3_0, "reference_to_estimate_median_deviation_seconds": 0.18, "estimate_to_reference_median_deviation_seconds": 0.21, "functional_label_accuracy": functional_label_accuracy, + "repetition_pairwise_precision": repetition_pairwise_f, + "repetition_pairwise_recall": repetition_pairwise_f, "repetition_pairwise_f": repetition_pairwise_f, "p50_latency_seconds": p50_latency_seconds, "p95_latency_seconds": p95_latency_seconds, @@ -110,7 +118,14 @@ def _measurement( } -def _result(registration_sha256: str) -> dict[str, object]: +def _uncertainty(registration: dict[str, object]) -> dict[str, object]: + """Return the typed uncertainty plan from a registration.""" + value = registration["uncertainty"] + assert isinstance(value, dict) + return value + + +def _result(registration: dict[str, object], registration_sha256: str) -> dict[str, object]: """Return a result fixture whose paired intervals satisfy the registration.""" baseline = _measurement( boundary_f_0_5=0.70, @@ -134,6 +149,7 @@ def _result(registration_sha256: str) -> dict[str, object]: "schema_version": 1, "experiment_id": "structure-chroma-stft-vs-cqt-v1", "registration_sha256": registration_sha256, + "uncertainty": copy.deepcopy(_uncertainty(registration)), "corpus_track_ids": ["licensed-track-001", "licensed-track-002"], "tracks": [ { @@ -214,7 +230,7 @@ def test_registration_rejects_unverifiable_real_audio() -> None: def test_registration_rejects_incomplete_or_post_hoc_decision_contract() -> None: - """Every predeclared quality and latency criterion must be present and bounded.""" + """Metrics, margins, and uncertainty procedure must be frozen before results.""" validator = _validator() registration = _registration() del _metrics(registration)["boundary_f_3_0"] @@ -226,10 +242,19 @@ def test_registration_rejects_incomplete_or_post_hoc_decision_contract() -> None latency = _metrics(registration)["p95_latency_ratio"] assert isinstance(latency, dict) latency["maximum_candidate_ratio"] = 1.0 - with pytest.raises(ValueError, match="maximum_candidate_ratio"): validator.validate_registration(registration) + registration = _registration() + del _uncertainty(registration)["random_seed"] + with pytest.raises(ValueError, match="random_seed"): + validator.validate_registration(registration) + + registration = _registration() + _uncertainty(registration)["confidence_level"] = 0.90 + with pytest.raises(ValueError, match="confidence_level"): + validator.validate_registration(registration) + def test_result_passes_only_when_paired_uncertainty_meets_frozen_contract() -> None: """Decision uses paired CI bounds rather than aggregate point estimates alone.""" @@ -237,7 +262,7 @@ def test_result_passes_only_when_paired_uncertainty_meets_frozen_contract() -> N registration = _registration() digest = validator.registration_digest(registration) - decision = validator.evaluate_result(registration, _result(digest)) + decision = validator.evaluate_result(registration, _result(registration, digest)) assert decision["passed"] is True assert decision["failed_requirements"] == [] @@ -248,7 +273,7 @@ def test_result_fails_quality_or_latency_when_ci_crosses_registered_boundary() - validator = _validator() registration = _registration() digest = validator.registration_digest(registration) - result = _result(digest) + result = _result(registration, digest) deltas = result["paired_delta_ci95"] assert isinstance(deltas, dict) deltas["boundary_f_0_5"] = [-0.021, 0.004] @@ -269,7 +294,7 @@ def test_result_requires_track_level_metrics_and_boundary_deviations() -> None: registration = _registration() digest = validator.registration_digest(registration) - missing_track_metric = _result(digest) + missing_track_metric = _result(registration, digest) tracks = missing_track_metric["tracks"] assert isinstance(tracks, list) first_track = tracks[0] @@ -280,7 +305,7 @@ def test_result_requires_track_level_metrics_and_boundary_deviations() -> None: with pytest.raises(ValueError, match="reference_to_estimate_median_deviation_seconds"): validator.evaluate_result(registration, missing_track_metric) - missing_aggregate_metric = _result(digest) + missing_aggregate_metric = _result(registration, digest) aggregate = missing_aggregate_metric["aggregate"] assert isinstance(aggregate, dict) candidate = aggregate["candidate"] @@ -290,22 +315,47 @@ def test_result_requires_track_level_metrics_and_boundary_deviations() -> None: validator.evaluate_result(registration, missing_aggregate_metric) +def test_result_rejects_inconsistent_precision_recall_f_triplets() -> None: + """Receipt F values must agree with recognized metric precision and recall.""" + validator = _validator() + registration = _registration() + digest = validator.registration_digest(registration) + result = _result(registration, digest) + tracks = result["tracks"] + assert isinstance(tracks, list) + first_track = tracks[0] + assert isinstance(first_track, dict) + candidate = first_track["candidate"] + assert isinstance(candidate, dict) + candidate["boundary_f_0_5"] = 0.9 + + with pytest.raises(ValueError, match="boundary_f_0_5 must equal the harmonic mean"): + validator.evaluate_result(registration, result) + + def test_result_is_bound_to_registration_corpus_and_finite_measurements() -> None: - """Results cannot drift thresholds, corpus identity, or numerical validity.""" + """Results cannot drift registration, corpus, uncertainty, or numeric validity.""" validator = _validator() registration = _registration() digest = validator.registration_digest(registration) - wrong_digest = _result("0" * 64) + wrong_digest = _result(registration, "0" * 64) with pytest.raises(ValueError, match="registration_sha256"): validator.evaluate_result(registration, wrong_digest) - wrong_corpus = _result(digest) + wrong_corpus = _result(registration, digest) wrong_corpus["corpus_track_ids"] = ["licensed-track-001"] with pytest.raises(ValueError, match="corpus_track_ids"): validator.evaluate_result(registration, wrong_corpus) - nonfinite = _result(digest) + wrong_uncertainty = _result(registration, digest) + uncertainty = wrong_uncertainty["uncertainty"] + assert isinstance(uncertainty, dict) + uncertainty["random_seed"] = 1 + with pytest.raises(ValueError, match="uncertainty"): + validator.evaluate_result(registration, wrong_uncertainty) + + nonfinite = _result(registration, digest) aggregate = nonfinite["aggregate"] assert isinstance(aggregate, dict) candidate = aggregate["candidate"] From c1641796a8cf106ed54e2e1c981ba6dd2e83e4f1 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 10:16:26 +0900 Subject: [PATCH 009/216] test(mir): bound machine-readable evidence admission --- .../test_structure_noninferiority_policy.py | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_policy.py index 644cf7d94..5108444c7 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_policy.py @@ -4,6 +4,7 @@ import copy import math +from pathlib import Path from types import ModuleType import pytest @@ -363,3 +364,20 @@ def test_result_is_bound_to_registration_corpus_and_finite_measurements() -> Non candidate["p95_latency_seconds"] = math.inf with pytest.raises(ValueError, match="finite"): validator.evaluate_result(registration, nonfinite) + + +def test_machine_readable_evidence_is_bounded_and_duplicate_rejecting( + tmp_path: Path, +) -> None: + """Evidence files reject ambiguous objects and unbounded JSON before evaluation.""" + validator = _validator() + duplicate = tmp_path / "duplicate.json" + duplicate.write_text('{"schema_version": 1, "schema_version": 1}', encoding="utf-8") + + with pytest.raises(ValueError, match="duplicate JSON key: schema_version"): + validator._load_json(duplicate) + + oversized = tmp_path / "oversized.json" + oversized.write_bytes(b" " * (validator.MAX_EVIDENCE_BYTES + 1)) + with pytest.raises(ValueError, match="exceeds"): + validator._load_json(oversized) From edfdcefc2b48fa5c8cbdda3815d743b66b71a4a7 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 10:17:35 +0900 Subject: [PATCH 010/216] feat(mir): bind uncertainty plan and recognized metric receipts --- .../validate_structure_noninferiority.py | 186 +++++++++++++++++- 1 file changed, 178 insertions(+), 8 deletions(-) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index 2dca9e640..364606138 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -5,7 +5,8 @@ result receipts produced by the recognized evaluation pipeline, then evaluates the preregistered confidence-interval decision rules. Metric implementation authority remains with MIREX/mir_eval-compatible tooling while thresholds, -corpus identity, and runtime identity are protected from post-result drift. +corpus identity, uncertainty procedure, and runtime identity are protected from +post-result drift. """ from __future__ import annotations @@ -14,12 +15,15 @@ import hashlib import json import math +import os import re +import stat from collections.abc import Mapping, Sequence from pathlib import Path from typing import Any SCHEMA_VERSION = 1 +MAX_EVIDENCE_BYTES = 2 * 1024 * 1024 _SHA256_RE = re.compile(r"^[0-9a-fA-F]{64}$") _COMMIT_RE = re.compile(r"^[0-9a-fA-F]{40}$") _WINDOWS_ABSOLUTE_RE = re.compile(r"^[A-Za-z]:[\\/]") @@ -47,6 +51,8 @@ "boundary_recall_0_5", "boundary_precision_3_0", "boundary_recall_3_0", + "repetition_pairwise_precision", + "repetition_pairwise_recall", ) _REPORT_NONNEGATIVE_METRICS = ( "reference_to_estimate_median_deviation_seconds", @@ -88,6 +94,21 @@ def _finite_number(value: object, field: str) -> float: return number +def _bounded_integer( + value: object, + field: str, + *, + minimum: int, + maximum: int, +) -> int: + """Return a bounded integer while rejecting bools and fractional values.""" + if isinstance(value, bool) or not isinstance(value, int): + raise ValueError(f"{field} must be an integer") + if not minimum <= value <= maximum: + raise ValueError(f"{field} must be in {minimum}..{maximum}") + return value + + def _score(value: object, field: str) -> float: """Return a finite score constrained to the inclusive 0..1 interval.""" number = _finite_number(value, field) @@ -195,6 +216,56 @@ def _validate_metrics(metrics_value: object) -> None: ) +def _validate_uncertainty(uncertainty_value: object, field: str) -> dict[str, object]: + """Validate and normalize the preregistered paired-uncertainty procedure.""" + uncertainty = _mapping(uncertainty_value, field) + required = { + "procedure_id", + "confidence_level", + "resamples", + "random_seed", + } + missing = sorted(required - set(uncertainty)) + extra = sorted(set(uncertainty) - required) + if missing: + raise ValueError(f"{field} missing required field: {missing[0]}") + if extra: + raise ValueError(f"{field} contains unregistered field: {extra[0]}") + + procedure_id = _nonempty_text( + uncertainty.get("procedure_id"), + f"{field}.procedure_id", + ) + if len(procedure_id) > 128: + raise ValueError(f"{field}.procedure_id must be at most 128 characters") + + confidence_level = _finite_number( + uncertainty.get("confidence_level"), + f"{field}.confidence_level", + ) + if not math.isclose(confidence_level, 0.95, rel_tol=0.0, abs_tol=1e-12): + raise ValueError(f"{field}.confidence_level must equal 0.95") + + resamples = _bounded_integer( + uncertainty.get("resamples"), + f"{field}.resamples", + minimum=1000, + maximum=10_000_000, + ) + random_seed = _bounded_integer( + uncertainty.get("random_seed"), + f"{field}.random_seed", + minimum=0, + maximum=(2**32) - 1, + ) + return { + "procedure_id": procedure_id, + "confidence_level": confidence_level, + "resamples": resamples, + "random_seed": random_seed, + } + + def _validate_corpus(corpus_value: object) -> list[str]: """Validate rights-cleared real-audio identities and return ordered IDs.""" corpus = _sequence(corpus_value, "corpus") @@ -252,9 +323,9 @@ def validate_registration(registration_value: object) -> None: """Validate a frozen STFT-vs-CQT structure experiment registration. This function requires content hashes, rights evidence, exact runtime - identity, recognized metric implementations, and explicit margins. It does - not decide what those margins should be; that scientific/product choice must - be reviewed before any corpus result is inspected. + identity, recognized metric implementations, explicit margins, and a paired + uncertainty procedure. Scientific/product choices must be reviewed before + any corpus result is inspected. """ registration = _mapping(registration_value, "registration") _validate_schema_version( @@ -278,6 +349,7 @@ def validate_registration(registration_value: object) -> None: raise ValueError("hypothesis.candidate_feature must equal chroma_stft") _validate_metrics(registration.get("metrics")) + _validate_uncertainty(registration.get("uncertainty"), "uncertainty") _validate_corpus(registration.get("corpus")) _validate_runtime(registration.get("runtime")) @@ -307,6 +379,27 @@ def _confidence_interval(value: object, field: str) -> tuple[float, float]: return lower, upper +def _validate_f_triplet( + normalized: Mapping[str, float], + *, + precision_name: str, + recall_name: str, + f_name: str, + field: str, +) -> None: + """Require reported F to equal the harmonic mean of precision and recall.""" + precision = normalized[precision_name] + recall = normalized[recall_name] + denominator = precision + recall + expected = 0.0 if denominator == 0.0 else (2.0 * precision * recall) / denominator + actual = normalized[f_name] + if not math.isclose(actual, expected, rel_tol=1e-9, abs_tol=1e-12): + raise ValueError( + f"{field}.{f_name} must equal the harmonic mean of " + f"{precision_name} and {recall_name}" + ) + + def _validate_measurement_side( side_value: object, field: str, @@ -329,6 +422,29 @@ def _validate_measurement_side( if value < 0.0: raise ValueError(f"{field}.{metric_name} must be non-negative") normalized[metric_name] = value + + _validate_f_triplet( + normalized, + precision_name="boundary_precision_0_5", + recall_name="boundary_recall_0_5", + f_name="boundary_f_0_5", + field=field, + ) + _validate_f_triplet( + normalized, + precision_name="boundary_precision_3_0", + recall_name="boundary_recall_3_0", + f_name="boundary_f_3_0", + field=field, + ) + _validate_f_triplet( + normalized, + precision_name="repetition_pairwise_precision", + recall_name="repetition_pairwise_recall", + f_name="repetition_pairwise_f", + field=field, + ) + if normalized["p95_latency_seconds"] < normalized["p50_latency_seconds"]: raise ValueError( f"{field}.p95_latency_seconds must be >= p50_latency_seconds" @@ -362,6 +478,19 @@ def _validate_result_identity( "result.registration_sha256 does not match the frozen registration" ) + registered_uncertainty = _validate_uncertainty( + registration.get("uncertainty"), + "uncertainty", + ) + result_uncertainty = _validate_uncertainty( + result.get("uncertainty"), + "result.uncertainty", + ) + if result_uncertainty != registered_uncertainty: + raise ValueError( + "result.uncertainty must exactly match the preregistered uncertainty plan" + ) + expected_track_ids = _validate_corpus(registration["corpus"]) actual_values = _sequence( result.get("corpus_track_ids"), @@ -385,7 +514,9 @@ def _validate_track_measurements( """Require one complete baseline/candidate receipt for every corpus track.""" tracks = _sequence(result.get("tracks"), "result.tracks") if len(tracks) != len(expected_track_ids): - raise ValueError("result.tracks must contain exactly one receipt per corpus track") + raise ValueError( + "result.tracks must contain exactly one receipt per corpus track" + ) actual_track_ids: list[str] = [] for index, raw_track in enumerate(tracks): @@ -518,10 +649,49 @@ def evaluate_result( } +def _reject_duplicate_object_pairs(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + """Reject duplicate JSON object keys instead of accepting last-key-wins.""" + result: dict[str, Any] = {} + for key, value in pairs: + if key in result: + raise ValueError(f"duplicate JSON key: {key}") + result[key] = value + return result + + +def _reject_json_constant(value: str) -> None: + """Reject non-standard JSON NaN/Infinity constants at the parser boundary.""" + raise ValueError(f"invalid JSON constant: {value}") + + def _load_json(path: Path) -> object: - """Load one UTF-8 JSON evidence file.""" - with path.open("r", encoding="utf-8") as handle: - return json.load(handle) + """Load one bounded regular UTF-8 JSON evidence file from one descriptor.""" + with path.open("rb") as handle: + descriptor_stat = os.fstat(handle.fileno()) + if not stat.S_ISREG(descriptor_stat.st_mode): + raise ValueError(f"evidence path is not a regular file: {path.name}") + if descriptor_stat.st_size > MAX_EVIDENCE_BYTES: + raise ValueError( + f"evidence file exceeds {MAX_EVIDENCE_BYTES} bytes: {path.name}" + ) + payload = handle.read(MAX_EVIDENCE_BYTES + 1) + + if len(payload) > MAX_EVIDENCE_BYTES: + raise ValueError( + f"evidence file exceeds {MAX_EVIDENCE_BYTES} bytes: {path.name}" + ) + try: + text = payload.decode("utf-8") + except UnicodeDecodeError as exc: + raise ValueError(f"evidence file is not valid UTF-8: {path.name}") from exc + try: + return json.loads( + text, + object_pairs_hook=_reject_duplicate_object_pairs, + parse_constant=_reject_json_constant, + ) + except json.JSONDecodeError as exc: + raise ValueError(f"evidence file is not valid JSON: {path.name}") from exc def main(argv: Sequence[str] | None = None) -> int: From ddb88a081d1418ce7d2562d54a119c8e1fc212aa Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 10:19:20 +0900 Subject: [PATCH 011/216] docs(mir): bind uncertainty and evidence admission --- .../mir/structure-feature-noninferiority.md | 40 ++++++++++++------- 1 file changed, 26 insertions(+), 14 deletions(-) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index 97fbe28f3..3405d7930 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -9,9 +9,9 @@ Production baseline: `chroma_cqt` in `sections/segmenter.py` The earlier `chroma_cqt` → `chroma_stft` optimization is not a production change until a preregistered, rights-cleared real-audio experiment shows that the candidate is noninferior on rehearsal-relevant structure quality and materially faster on the same corpus and runtime identity. -The repository now owns a result-admission boundary in `scripts/research/validate_structure_noninferiority.py`. The validator does **not** calculate MIR metrics and does not choose scientific margins. It freezes the reviewed experiment contract, binds a result receipt to that contract by canonical SHA-256, requires per-track and aggregate evidence, and applies the registered confidence-interval decision rules. Metric computation remains a separate scientific measurement step. +The repository now owns a result-admission boundary in `scripts/research/validate_structure_noninferiority.py`. The validator does **not** calculate MIR metrics and does not choose scientific margins or an uncertainty procedure. It freezes the reviewed experiment contract, binds a result receipt to that contract by canonical SHA-256, requires per-track and aggregate evidence, and applies the registered confidence-interval decision rules. Metric computation and uncertainty estimation remain separate scientific measurement steps. -This keeps a performance result from silently changing the corpus, thresholds, feature identities, or runtime after the measurements are visible. +This keeps a performance result from silently changing the corpus, thresholds, feature identities, uncertainty procedure, or runtime after the measurements are visible. ## Registered inputs @@ -20,7 +20,10 @@ A registration is valid only when it records all of the following before the res - baseline `chroma_cqt` and candidate `chroma_stft`; - a rights-cleared real-audio corpus with stable track IDs, audio SHA-256, annotation SHA-256, rights basis, and provenance URI; - exact source commit and `uv.lock` identity plus Python, librosa, NumPy, sample rate, channel count, and host profile; -- the complete metric implementation contract and noninferiority/speed thresholds. +- the complete metric implementation contract and noninferiority/speed thresholds; +- the paired-uncertainty procedure identity, confidence level, resample count, and random seed. + +The validator requires a 95% confidence level because the current result schema is explicitly `ci95`; it does not prescribe which scientifically defensible paired procedure must produce that interval. The procedure identifier, resample count, and seed are part of the preregistration digest so they cannot be changed after results are seen. `paired-track-bootstrap-v1` and the numeric values used in unit tests are policy fixtures only, not an approved BandScope production analysis plan. The validator rejects local filesystem paths as provenance authorities. A benchmark may remain private when licensing requires that, but the receipt must identify the licensed material without leaking the local path that happened to hold it. @@ -36,12 +39,14 @@ BandScope uses MIREX 2025 Music Structure Analysis as the functional-structure r | Boundary precision/recall/F at 3.0 s | `mir_eval.segment.detection`, 3.0 s window | F noninferiority | | Boundary median deviation, both directions | `mir_eval.segment.deviation` convention | Report per track and aggregate | | Functional-label accuracy | MIREX 2025 frame-level ACC convention | Noninferiority | -| Repetition/group consistency | `mir_eval.segment.pairwise`, 0.1 s frame size | F noninferiority | +| Repetition/group consistency precision/recall/F | `mir_eval.segment.pairwise`, 0.1 s frame size | F noninferiority | | Latency | paired p50/p95 on the registered host | p95 ratio superiority | | Peak memory | peak RSS on the registered host | Report per track and aggregate | MIREX 2025 evaluates functional structure with frame-level label accuracy plus boundary hit-rate F measures at 0.5 s and 3.0 s. `mir_eval.segment.detection` uses one-to-one boundary matching within the selected tolerance; `mir_eval.segment.deviation` reports the median nearest-boundary deviations in both directions; `mir_eval.segment.pairwise` measures structural grouping agreement. These are different questions and must not be collapsed into one score. +Result receipts must carry the reported precision and recall alongside F for the boundary and repetition measures. The admission validator recomputes the harmonic mean and rejects an internally inconsistent P/R/F triplet. This does not replace the recognized metric implementation; it prevents a malformed receipt from claiming a metric value that its own reported components cannot support. + The published MIREX 2025 result table reports the MusicFM baseline at ACC 0.705, HR.5 0.644, and HR3 0.710. Those values are useful external context, **not** BandScope's noninferiority margins. A margin is a product/scientific decision about acceptable loss relative to BandScope's own current baseline on the registered corpus. It must be fixed before examining candidate results. ## Decision rule @@ -54,33 +59,40 @@ For p95 latency, let `r = candidate_p95 / baseline_p95`. The result passes the s `upper_CI95(r) <= maximum_candidate_ratio` -The validator also requires the aggregate point delta/ratio to lie inside the supplied interval. Favorable point estimates therefore cannot hide an uncertainty interval that crosses a preregistered boundary. +The result receipt must repeat the preregistered uncertainty plan exactly. The validator also requires the aggregate point delta/ratio to lie inside the supplied interval. Favorable point estimates therefore cannot hide an uncertainty interval that crosses a preregistered boundary, and the uncertainty method cannot be swapped after the result is known. -The repository does not currently contain approved numeric margins. Unit-test values are synthetic policy fixtures only and must never be cited as production acceptance thresholds. +The repository does not currently contain approved numeric margins or an approved production corpus/uncertainty procedure. Unit-test values are synthetic policy fixtures only and must never be cited as production acceptance thresholds or scientific design decisions. ## Result receipt -A result receipt must contain the exact registration digest, exact corpus order, one complete baseline/candidate measurement pair per registered track, aggregate measurements, paired 95% intervals for every gated quality metric, a paired p95-latency-ratio interval, failed-track IDs, and a claim boundary. +A result receipt must contain the exact registration digest, exact uncertainty plan, exact corpus order, one complete baseline/candidate measurement pair per registered track, aggregate measurements, paired 95% intervals for every gated quality metric, a paired p95-latency-ratio interval, failed-track IDs, and a claim boundary. + +The receipt is rejected when a track is omitted/reordered, the uncertainty plan differs, a required metric is absent, a boundary/repetition P/R/F triplet is internally inconsistent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A passing receipt means only that the preregistered decision rule passed for the registered corpus, uncertainty procedure, and runtime. It does not generalize automatically to other genres, codecs, sample rates, machines, annotation regimes, or statistical procedures. + +## Machine-readable evidence admission + +Registration and result JSON are evidence, not trusted configuration. CLI admission is bounded to 2 MiB per file, reads from one already-open regular-file descriptor, requires UTF-8 and standards-compliant finite JSON values, and rejects duplicate object keys instead of accepting last-key-wins semantics. These controls prevent ambiguous evidence identities and bound memory use before scientific validation begins. -The receipt is rejected when a track is omitted/reordered, a required metric is absent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A passing receipt means only that the preregistered decision rule passed for the registered corpus and runtime. It does not generalize automatically to other genres, codecs, sample rates, machines, or annotation regimes. +The JSON `source_uri` value remains provenance metadata only; it is never dereferenced by this validator. Audio and annotation bytes are not opened by the evidence validator and are bound to the registration through SHA-256 identities supplied by the experiment process. ## Reproducibility sequence -1. Review the rights basis, corpus composition, feature hypothesis, metric contract, host profile, and numeric margins **before** running the candidate. +1. Review the rights basis, corpus composition, feature hypothesis, metric contract, host profile, numeric margins, and paired-uncertainty procedure **before** running the candidate. Freeze the procedure identifier, confidence level, resample count, and random seed in the registration. 2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain its canonical SHA-256. -3. Run baseline and candidate on the same decoded track identities and host profile. Record per-track recognized MIR metrics, p50/p95 latency, peak RSS, failures, and paired uncertainty. -4. Put the registration digest in the result receipt and run `python scripts/research/validate_structure_noninferiority.py `. +3. Run baseline and candidate on the same decoded track identities and host profile. Record per-track recognized MIR metrics, p50/p95 latency, peak RSS, failures, and paired uncertainty using exactly the preregistered procedure. +4. Put the registration digest and the identical uncertainty-plan fields in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. 5. Preserve the registration, result receipt, corpus/annotation hashes, exact source commit, lock hash, and measurement procedure together. A production feature switch requires this evidence plus normal code review and protected-head checks. No step authorizes committing licensed audio to Git. Rights-cleared means BandScope has the necessary evaluation right; redistribution is a separate permission. ## Security Notes -- Audio and annotation files are untrusted inputs to the future experiment runner. This validator reads JSON evidence only and does not open audio, execute subprocesses, make network requests, or follow paths from the registration. +- Audio and annotation files are untrusted inputs to the future experiment runner. This validator reads bounded JSON evidence only and does not open audio, execute subprocesses, make network requests, or follow paths from the registration. +- Evidence JSON is limited to 2 MiB, must be a regular file read through one open descriptor, must decode as UTF-8, and rejects duplicate keys plus non-standard `NaN`/`Infinity` constants. - `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute paths and `file:` URIs are rejected to avoid turning transient workstation paths into provenance authority or leaking them into review artifacts. - Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt. -- Invalid/non-finite measurements, corpus drift, registration drift, missing per-track evidence, and post-hoc metric additions fail closed. -- The validator does not claim that SHA-256 proves licensing, annotation validity, or scientific adequacy. Rights basis and scientific review remain separate gates. +- Invalid/non-finite measurements, corpus drift, uncertainty-plan drift, registration drift, missing per-track evidence, inconsistent P/R/F triplets, and post-hoc metric additions fail closed. +- The validator does not claim that SHA-256 proves licensing, annotation validity, scientific adequacy, or that the registered statistical procedure is appropriate. Rights and scientific review remain separate gates. ## References From 4db2191d3ae6c37d710dd6a36d4acd9fe1af21cc Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 11:05:52 +0900 Subject: [PATCH 012/216] test(mir): reject relative local provenance paths --- ...ucture_noninferiority_provenance_policy.py | 46 +++++++++++++++++++ 1 file changed, 46 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_noninferiority_provenance_policy.py diff --git a/services/analysis-engine/tests/test_structure_noninferiority_provenance_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_provenance_policy.py new file mode 100644 index 000000000..719c6df5a --- /dev/null +++ b/services/analysis-engine/tests/test_structure_noninferiority_provenance_policy.py @@ -0,0 +1,46 @@ +"""Focused provenance-URI admission regressions for the MIR evidence gate.""" + +from types import ModuleType + +import pytest +from conftest import load_module + + +def _validator() -> ModuleType: + """Load the repository-owned structure experiment validator.""" + return load_module( + "scripts/research/validate_structure_noninferiority.py", + "validate_structure_noninferiority_provenance", + ) + + +@pytest.mark.parametrize( + "source_uri", + [ + "licensed-track-001.wav", + "corpus/licensed-track-001.wav", + "../private/licensed-track-001.wav", + r"..\private\licensed-track-001.wav", + r"C:private\licensed-track-001.wav", + ], +) +def test_provenance_identity_rejects_relative_or_drive_local_paths(source_uri: str) -> None: + """A local path must not be admitted merely because it is not absolute.""" + validator = _validator() + + with pytest.raises(ValueError, match="provenance URI"): + validator._reject_local_path(source_uri, "corpus[0].source_uri") + + +@pytest.mark.parametrize( + "source_uri", + [ + "https://example.invalid/licensed-track-001", + "urn:bandscope:private-benchmark:licensed-track-002", + ], +) +def test_provenance_identity_accepts_explicit_non_file_uri(source_uri: str) -> None: + """Explicit network or opaque provenance URIs remain admissible metadata.""" + validator = _validator() + + assert validator._reject_local_path(source_uri, "corpus[0].source_uri") == source_uri From 028ee71eeb3a67401afc38a8691e31b7c0ff49d4 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 11:07:25 +0900 Subject: [PATCH 013/216] fix(mir): require explicit non-file provenance URIs --- .../research/validate_structure_noninferiority.py | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index 364606138..e2469f81d 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -26,7 +26,8 @@ MAX_EVIDENCE_BYTES = 2 * 1024 * 1024 _SHA256_RE = re.compile(r"^[0-9a-fA-F]{64}$") _COMMIT_RE = re.compile(r"^[0-9a-fA-F]{40}$") -_WINDOWS_ABSOLUTE_RE = re.compile(r"^[A-Za-z]:[\\/]") +_WINDOWS_DRIVE_RE = re.compile(r"^[A-Za-z]:") +_URI_SCHEME_RE = re.compile(r"^[A-Za-z][A-Za-z0-9+.-]*:") _QUALITY_METRICS: dict[str, dict[str, float | str]] = { "boundary_f_0_5": { @@ -134,13 +135,16 @@ def _commit(value: object, field: str) -> str: def _reject_local_path(value: object, field: str) -> str: - """Reject local filesystem authorities from corpus provenance receipts.""" + """Require an explicit non-file URI for corpus provenance receipts.""" text = _nonempty_text(value, field) lowered = text.casefold() + scheme = _URI_SCHEME_RE.match(text) if ( text.startswith(("/", "\\\\")) - or _WINDOWS_ABSOLUTE_RE.match(text) is not None + or _WINDOWS_DRIVE_RE.match(text) is not None or lowered.startswith("file:") + or scheme is None + or scheme.end() == len(text) ): raise ValueError( f"{field} must be a provenance URI, not a local filesystem path" @@ -714,4 +718,4 @@ def main(argv: Sequence[str] | None = None) -> int: if __name__ == "__main__": - raise SystemExit(main()) + raise SystemExit(main()) \ No newline at end of file From e7490da957f405330cc185b352e03bf2ee1c2d4b Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 11:08:50 +0900 Subject: [PATCH 014/216] docs(mir): record provenance URI admission boundary --- docs/traceability/mir/structure-feature-noninferiority.md | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index 3405d7930..9c8b05c67 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -25,7 +25,7 @@ A registration is valid only when it records all of the following before the res The validator requires a 95% confidence level because the current result schema is explicitly `ci95`; it does not prescribe which scientifically defensible paired procedure must produce that interval. The procedure identifier, resample count, and seed are part of the preregistration digest so they cannot be changed after results are seen. `paired-track-bootstrap-v1` and the numeric values used in unit tests are policy fixtures only, not an approved BandScope production analysis plan. -The validator rejects local filesystem paths as provenance authorities. A benchmark may remain private when licensing requires that, but the receipt must identify the licensed material without leaking the local path that happened to hold it. +The validator requires `source_uri` to be an explicit non-`file:` URI. Absolute, relative, drive-relative, and `file:` filesystem forms are rejected as provenance authorities. A benchmark may remain private when licensing requires that, but the receipt must identify the licensed material without leaking the workstation path that happened to hold it. The current minimum of two tracks is only a technical guard against treating one timing sample as a corpus. It is **not** a scientific sample-size claim. Corpus breadth, genre/instrumentation coverage, annotation quality, and a defensible power/uncertainty plan remain part of the experiment review before a production switch can be accepted. @@ -73,7 +73,7 @@ The receipt is rejected when a track is omitted/reordered, the uncertainty plan Registration and result JSON are evidence, not trusted configuration. CLI admission is bounded to 2 MiB per file, reads from one already-open regular-file descriptor, requires UTF-8 and standards-compliant finite JSON values, and rejects duplicate object keys instead of accepting last-key-wins semantics. These controls prevent ambiguous evidence identities and bound memory use before scientific validation begins. -The JSON `source_uri` value remains provenance metadata only; it is never dereferenced by this validator. Audio and annotation bytes are not opened by the evidence validator and are bound to the registration through SHA-256 identities supplied by the experiment process. +The JSON `source_uri` value remains provenance metadata only; it is never dereferenced by this validator. It must use an explicit non-file URI scheme; absolute, relative, drive-relative, and `file:` filesystem forms are rejected. Audio and annotation bytes are not opened by the evidence validator and are bound to the registration through SHA-256 identities supplied by the experiment process. ## Reproducibility sequence @@ -89,7 +89,7 @@ No step authorizes committing licensed audio to Git. Rights-cleared means BandSc - Audio and annotation files are untrusted inputs to the future experiment runner. This validator reads bounded JSON evidence only and does not open audio, execute subprocesses, make network requests, or follow paths from the registration. - Evidence JSON is limited to 2 MiB, must be a regular file read through one open descriptor, must decode as UTF-8, and rejects duplicate keys plus non-standard `NaN`/`Infinity` constants. -- `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute paths and `file:` URIs are rejected to avoid turning transient workstation paths into provenance authority or leaking them into review artifacts. +- `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute, relative, drive-relative, and `file:` forms are rejected; an explicit non-file URI scheme is required so transient workstation paths cannot become provenance authority or leak into review artifacts. - Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt. - Invalid/non-finite measurements, corpus drift, uncertainty-plan drift, registration drift, missing per-track evidence, inconsistent P/R/F triplets, and post-hoc metric additions fail closed. - The validator does not claim that SHA-256 proves licensing, annotation validity, scientific adequacy, or that the registered statistical procedure is appropriate. Rights and scientific review remain separate gates. @@ -106,4 +106,4 @@ MIREX. (2025). *Music Structure Analysis Results*. https://music-ir.org/mirex/wi mir_eval contributors. (n.d.). *mir_eval.segment: Structural segmentation evaluation*. https://github.com/mir-evaluation/mir_eval/blob/main/mir_eval/segment.py -Wang, J.-C., Hung, Y.-N., & Smith, J. B. L. (2022). To catch a chorus, verse, intro, or anything else: Analyzing a song with structural functions. In *ICASSP 2022—2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)* (pp. 416–420). IEEE. https://arxiv.org/abs/2205.14700 \ No newline at end of file +Wang, J.-C., Hung, Y.-N., & Smith, J. B. L. (2022). To catch a chorus, verse, intro, or anything else: Analyzing a song with structural functions. In *ICASSP 2022—2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)* (pp. 416–420). IEEE. https://arxiv.org/abs/2205.14700 From 48789605a175c86f321f202ed1161209c8dba217 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 12:06:22 +0900 Subject: [PATCH 015/216] test(mir): reject silent failed-track acceptance --- ...structure_noninferiority_failure_policy.py | 20 +++++++++++++++++++ 1 file changed, 20 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py diff --git a/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py new file mode 100644 index 000000000..0a2c1088e --- /dev/null +++ b/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py @@ -0,0 +1,20 @@ +"""Regression tests for failed-track scientific acceptance policy.""" + +from test_structure_noninferiority_policy import _registration, _result, _validator + + +def test_failed_track_cannot_pass_without_preregistered_exclusion_policy() -> None: + """A favorable aggregate cannot turn an unregistered track failure into PASS.""" + validator = _validator() + registration = _registration() + digest = validator.registration_digest(registration) + result = _result(registration, digest) + result["failed_tracks"] = ["licensed-track-002"] + + decision = validator.evaluate_result(registration, result) + + assert decision["passed"] is False + assert decision["failed_requirements"] == [ + "failed tracks are not permitted without a preregistered exclusion policy: " + "licensed-track-002" + ] From 102b572c767cdea461dee08542a494520e0b9e23 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 12:07:37 +0900 Subject: [PATCH 016/216] fix(mir): fail closed on unregistered track failures --- scripts/research/validate_structure_noninferiority.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index e2469f81d..54fe74756 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -617,6 +617,11 @@ def evaluate_result( metrics = _mapping(registration["metrics"], "metrics") failed_requirements: list[str] = [] + if failed_tracks: + failed_requirements.append( + "failed tracks are not permitted without a preregistered exclusion policy: " + + ", ".join(failed_tracks) + ) for metric_name in _QUALITY_METRICS: config = _mapping(metrics[metric_name], f"metrics.{metric_name}") margin = _finite_number( From 47d09320957097b5347f81396a57e422dad5813a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 12:08:23 +0900 Subject: [PATCH 017/216] docs(mir): make failed-track acceptance fail closed --- .../mir/structure-feature-noninferiority.md | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index 9c8b05c67..853700f56 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -29,6 +29,8 @@ The validator requires `source_uri` to be an explicit non-`file:` URI. Absolute, The current minimum of two tracks is only a technical guard against treating one timing sample as a corpus. It is **not** a scientific sample-size claim. Corpus breadth, genre/instrumentation coverage, annotation quality, and a defensible power/uncertainty plan remain part of the experiment review before a production switch can be accepted. +Schema v1 does not contain a preregistered dropout, exclusion, or missing-track policy. Therefore every registered track must complete both baseline and candidate measurement for an acceptance PASS. A result may record failed track IDs for diagnosis, but any non-empty `failed_tracks` list makes the acceptance decision fail. A future tolerance for failed or excluded tracks requires a reviewed preregistration rule and schema revision rather than post-result omission. + ## Metrics BandScope uses MIREX 2025 Music Structure Analysis as the functional-structure reference point. The registered quality gate requires: @@ -61,13 +63,15 @@ For p95 latency, let `r = candidate_p95 / baseline_p95`. The result passes the s The result receipt must repeat the preregistered uncertainty plan exactly. The validator also requires the aggregate point delta/ratio to lie inside the supplied interval. Favorable point estimates therefore cannot hide an uncertainty interval that crosses a preregistered boundary, and the uncertainty method cannot be swapped after the result is known. +Because schema v1 has no preregistered exclusion policy, a non-empty `failed_tracks` list is an additional failed requirement even when every reported quality and latency interval is favorable. Failed tracks remain visible in the result receipt for diagnosis; they cannot be silently converted into a complete-case PASS after results are known. + The repository does not currently contain approved numeric margins or an approved production corpus/uncertainty procedure. Unit-test values are synthetic policy fixtures only and must never be cited as production acceptance thresholds or scientific design decisions. ## Result receipt A result receipt must contain the exact registration digest, exact uncertainty plan, exact corpus order, one complete baseline/candidate measurement pair per registered track, aggregate measurements, paired 95% intervals for every gated quality metric, a paired p95-latency-ratio interval, failed-track IDs, and a claim boundary. -The receipt is rejected when a track is omitted/reordered, the uncertainty plan differs, a required metric is absent, a boundary/repetition P/R/F triplet is internally inconsistent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A passing receipt means only that the preregistered decision rule passed for the registered corpus, uncertainty procedure, and runtime. It does not generalize automatically to other genres, codecs, sample rates, machines, annotation regimes, or statistical procedures. +The receipt is rejected when a track is omitted/reordered, the uncertainty plan differs, a required metric is absent, a boundary/repetition P/R/F triplet is internally inconsistent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A structurally valid receipt with one or more known failed tracks is retained for diagnosis but evaluates to `passed=false` under schema v1. A passing receipt therefore means the preregistered decision rule passed for the complete registered corpus, uncertainty procedure, and runtime. It does not generalize automatically to other genres, codecs, sample rates, machines, annotation regimes, or statistical procedures. ## Machine-readable evidence admission @@ -77,10 +81,10 @@ The JSON `source_uri` value remains provenance metadata only; it is never derefe ## Reproducibility sequence -1. Review the rights basis, corpus composition, feature hypothesis, metric contract, host profile, numeric margins, and paired-uncertainty procedure **before** running the candidate. Freeze the procedure identifier, confidence level, resample count, and random seed in the registration. +1. Review the rights basis, corpus composition, feature hypothesis, metric contract, host profile, numeric margins, and paired-uncertainty procedure **before** running the candidate. Freeze the procedure identifier, confidence level, resample count, and random seed in the registration. If any failure/exclusion tolerance is scientifically required, define and version that policy before measurement rather than adding it after failures are observed. 2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain its canonical SHA-256. 3. Run baseline and candidate on the same decoded track identities and host profile. Record per-track recognized MIR metrics, p50/p95 latency, peak RSS, failures, and paired uncertainty using exactly the preregistered procedure. -4. Put the registration digest and the identical uncertainty-plan fields in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. +4. Put the registration digest and the identical uncertainty-plan fields in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. Under schema v1, any recorded failed track keeps the diagnostic receipt valid but makes acceptance fail closed. 5. Preserve the registration, result receipt, corpus/annotation hashes, exact source commit, lock hash, and measurement procedure together. A production feature switch requires this evidence plus normal code review and protected-head checks. No step authorizes committing licensed audio to Git. Rights-cleared means BandScope has the necessary evaluation right; redistribution is a separate permission. @@ -91,7 +95,7 @@ No step authorizes committing licensed audio to Git. Rights-cleared means BandSc - Evidence JSON is limited to 2 MiB, must be a regular file read through one open descriptor, must decode as UTF-8, and rejects duplicate keys plus non-standard `NaN`/`Infinity` constants. - `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute, relative, drive-relative, and `file:` forms are rejected; an explicit non-file URI scheme is required so transient workstation paths cannot become provenance authority or leak into review artifacts. - Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt. -- Invalid/non-finite measurements, corpus drift, uncertainty-plan drift, registration drift, missing per-track evidence, inconsistent P/R/F triplets, and post-hoc metric additions fail closed. +- Invalid/non-finite measurements, corpus drift, uncertainty-plan drift, registration drift, missing per-track evidence, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. - The validator does not claim that SHA-256 proves licensing, annotation validity, scientific adequacy, or that the registered statistical procedure is appropriate. Rights and scientific review remain separate gates. ## References From 65f012d15581c38af31a2f299a56d7a95f60819e Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 13:08:09 +0900 Subject: [PATCH 018/216] test(mir): reject post-hoc measurement receipt fields --- ...ure_noninferiority_result_schema_policy.py | 41 +++++++++++++++++++ 1 file changed, 41 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_noninferiority_result_schema_policy.py diff --git a/services/analysis-engine/tests/test_structure_noninferiority_result_schema_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_result_schema_policy.py new file mode 100644 index 000000000..2a9d60492 --- /dev/null +++ b/services/analysis-engine/tests/test_structure_noninferiority_result_schema_policy.py @@ -0,0 +1,41 @@ +"""Regression tests for closed-world structure experiment result receipts.""" + +from __future__ import annotations + +import pytest + +from test_structure_noninferiority_policy import _registration, _result, _validator + + +def test_track_measurement_rejects_unregistered_post_hoc_field() -> None: + """Per-track evidence cannot grow new result fields after preregistration.""" + validator = _validator() + registration = _registration() + digest = validator.registration_digest(registration) + result = _result(registration, digest) + tracks = result["tracks"] + assert isinstance(tracks, list) + first_track = tracks[0] + assert isinstance(first_track, dict) + candidate = first_track["candidate"] + assert isinstance(candidate, dict) + candidate["post_hoc_quality_score"] = 0.99 + + with pytest.raises(ValueError, match="contains unregistered field: post_hoc_quality_score"): + validator.evaluate_result(registration, result) + + +def test_aggregate_measurement_rejects_unregistered_post_hoc_field() -> None: + """Aggregate evidence uses the same closed metric receipt as each track.""" + validator = _validator() + registration = _registration() + digest = validator.registration_digest(registration) + result = _result(registration, digest) + aggregate = result["aggregate"] + assert isinstance(aggregate, dict) + baseline = aggregate["baseline"] + assert isinstance(baseline, dict) + baseline["post_hoc_quality_score"] = 0.99 + + with pytest.raises(ValueError, match="contains unregistered field: post_hoc_quality_score"): + validator.evaluate_result(registration, result) From 8b5b8df27e918fee083cc4ca86e9c1cea485bee1 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 13:10:11 +0900 Subject: [PATCH 019/216] fix(mir): close measurement receipt schema --- .../research/validate_structure_noninferiority.py | 15 ++++++++++++++- 1 file changed, 14 insertions(+), 1 deletion(-) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index 54fe74756..38b5725e9 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -410,6 +410,19 @@ def _validate_measurement_side( ) -> dict[str, float]: """Validate one per-track or aggregate baseline/candidate measurement.""" side = _mapping(side_value, field) + expected_fields = ( + set(_QUALITY_METRICS) + | set(_REPORT_SCORE_METRICS) + | set(_REPORT_NONNEGATIVE_METRICS) + ) + actual_fields = set(side) + missing_fields = sorted(expected_fields - actual_fields) + extra_fields = sorted(actual_fields - expected_fields) + if missing_fields: + raise ValueError(f"{field} missing required field: {missing_fields[0]}") + if extra_fields: + raise ValueError(f"{field} contains unregistered field: {extra_fields[0]}") + normalized: dict[str, float] = {} for metric_name in _QUALITY_METRICS: normalized[metric_name] = _score( @@ -723,4 +736,4 @@ def main(argv: Sequence[str] | None = None) -> int: if __name__ == "__main__": - raise SystemExit(main()) \ No newline at end of file + raise SystemExit(main()) From d7c65fdfbc40a68b066eb092e76fe5e87a5ad0cd Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 13:11:02 +0900 Subject: [PATCH 020/216] docs(mir): bind result measurement field set --- docs/traceability/mir/structure-feature-noninferiority.md | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index 853700f56..e3589bf09 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -71,7 +71,9 @@ The repository does not currently contain approved numeric margins or an approve A result receipt must contain the exact registration digest, exact uncertainty plan, exact corpus order, one complete baseline/candidate measurement pair per registered track, aggregate measurements, paired 95% intervals for every gated quality metric, a paired p95-latency-ratio interval, failed-track IDs, and a claim boundary. -The receipt is rejected when a track is omitted/reordered, the uncertainty plan differs, a required metric is absent, a boundary/repetition P/R/F triplet is internally inconsistent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A structurally valid receipt with one or more known failed tracks is retained for diagnosis but evaluates to `passed=false` under schema v1. A passing receipt therefore means the preregistered decision rule passed for the complete registered corpus, uncertainty procedure, and runtime. It does not generalize automatically to other genres, codecs, sample rates, machines, annotation regimes, or statistical procedures. +Each per-track and aggregate baseline/candidate measurement is a closed-world receipt: it must contain exactly the registered quality fields plus the reporting-only precision/recall, boundary-deviation, latency, and peak-RSS fields. Missing fields and additional post-hoc measurement fields both fail admission. This prevents a result producer from attaching an unreviewed score after candidate results are visible and presenting it as part of the admitted scientific receipt. + +The receipt is rejected when a track is omitted/reordered, the uncertainty plan differs, a required metric is absent, an unregistered measurement field is added, a boundary/repetition P/R/F triplet is internally inconsistent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A structurally valid receipt with one or more known failed tracks is retained for diagnosis but evaluates to `passed=false` under schema v1. A passing receipt therefore means the preregistered decision rule passed for the complete registered corpus, uncertainty procedure, and runtime. It does not generalize automatically to other genres, codecs, sample rates, machines, annotation regimes, or statistical procedures. ## Machine-readable evidence admission @@ -95,7 +97,7 @@ No step authorizes committing licensed audio to Git. Rights-cleared means BandSc - Evidence JSON is limited to 2 MiB, must be a regular file read through one open descriptor, must decode as UTF-8, and rejects duplicate keys plus non-standard `NaN`/`Infinity` constants. - `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute, relative, drive-relative, and `file:` forms are rejected; an explicit non-file URI scheme is required so transient workstation paths cannot become provenance authority or leak into review artifacts. - Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt. -- Invalid/non-finite measurements, corpus drift, uncertainty-plan drift, registration drift, missing per-track evidence, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. +- Invalid/non-finite measurements, corpus drift, uncertainty-plan drift, registration drift, missing per-track evidence, unregistered measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. - The validator does not claim that SHA-256 proves licensing, annotation validity, scientific adequacy, or that the registered statistical procedure is appropriate. Rights and scientific review remain separate gates. ## References From 8c8b60db35caf7eef3173612637a2215ee3acadb Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 14:05:21 +0900 Subject: [PATCH 021/216] test(mir): reject duplicate corpus audio identity --- ...e_noninferiority_corpus_identity_policy.py | 50 +++++++++++++++++++ 1 file changed, 50 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_noninferiority_corpus_identity_policy.py diff --git a/services/analysis-engine/tests/test_structure_noninferiority_corpus_identity_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_corpus_identity_policy.py new file mode 100644 index 000000000..14864548e --- /dev/null +++ b/services/analysis-engine/tests/test_structure_noninferiority_corpus_identity_policy.py @@ -0,0 +1,50 @@ +"""Focused corpus-identity regressions for the MIR evidence gate.""" + +from types import ModuleType + +import pytest +from conftest import load_module + + +def _validator() -> ModuleType: + """Load the repository-owned structure experiment validator.""" + return load_module( + "scripts/research/validate_structure_noninferiority.py", + "validate_structure_noninferiority_corpus_identity", + ) + + +def _track(track_id: str, audio_digest: str, annotation_digest: str) -> dict[str, object]: + """Return one synthetic corpus identity for policy testing only.""" + return { + "track_id": track_id, + "audio_sha256": audio_digest, + "annotation_sha256": annotation_digest, + "rights_basis": "synthetic policy fixture only", + "rights_cleared": True, + "source_uri": f"urn:bandscope:test:{track_id}", + } + + +def test_corpus_rejects_duplicate_audio_content_under_distinct_track_ids() -> None: + """The same decoded source bytes must not be counted as independent tracks.""" + validator = _validator() + duplicate_audio = "a" * 64 + corpus = [ + _track("track-001", duplicate_audio, "b" * 64), + _track("track-002", duplicate_audio, "c" * 64), + ] + + with pytest.raises(ValueError, match="duplicate audio_sha256"): + validator._validate_corpus(corpus) + + +def test_corpus_accepts_distinct_audio_content_identities() -> None: + """Distinct rights-cleared source bytes remain valid corpus members.""" + validator = _validator() + corpus = [ + _track("track-001", "a" * 64, "b" * 64), + _track("track-002", "c" * 64, "d" * 64), + ] + + assert validator._validate_corpus(corpus) == ["track-001", "track-002"] From 71546adca03aa204f279b5b08d0102332f204fd4 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 14:07:23 +0900 Subject: [PATCH 022/216] fix(mir): reject duplicate corpus audio identity --- .../research/validate_structure_noninferiority.py | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index 38b5725e9..bd14f0551 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -278,17 +278,21 @@ def _validate_corpus(corpus_value: object) -> list[str]: "corpus must contain at least two rights-cleared real-audio tracks" ) - seen: set[str] = set() + seen_track_ids: set[str] = set() + seen_audio_sha256: set[str] = set() track_ids: list[str] = [] for index, raw_track in enumerate(corpus): field = f"corpus[{index}]" track = _mapping(raw_track, field) track_id = _nonempty_text(track.get("track_id"), f"{field}.track_id") - if track_id in seen: + if track_id in seen_track_ids: raise ValueError(f"duplicate track_id: {track_id}") - seen.add(track_id) + seen_track_ids.add(track_id) track_ids.append(track_id) - _sha256(track.get("audio_sha256"), f"{field}.audio_sha256") + audio_sha256 = _sha256(track.get("audio_sha256"), f"{field}.audio_sha256") + if audio_sha256 in seen_audio_sha256: + raise ValueError(f"duplicate audio_sha256: {audio_sha256}") + seen_audio_sha256.add(audio_sha256) _sha256(track.get("annotation_sha256"), f"{field}.annotation_sha256") _nonempty_text(track.get("rights_basis"), f"{field}.rights_basis") if track.get("rights_cleared") is not True: From 7a6327b1b5ecff25e2267bccc5f9da59d36b4528 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 14:08:06 +0900 Subject: [PATCH 023/216] docs(mir): record corpus content-identity guard --- .../mir/structure-feature-noninferiority.md | 16 +++++++++------- 1 file changed, 9 insertions(+), 7 deletions(-) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index e3589bf09..941476144 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -18,7 +18,7 @@ This keeps a performance result from silently changing the corpus, thresholds, f A registration is valid only when it records all of the following before the result is evaluated: - baseline `chroma_cqt` and candidate `chroma_stft`; -- a rights-cleared real-audio corpus with stable track IDs, audio SHA-256, annotation SHA-256, rights basis, and provenance URI; +- a rights-cleared real-audio corpus with stable track IDs, content-unique audio SHA-256 identities, annotation SHA-256, rights basis, and provenance URI; - exact source commit and `uv.lock` identity plus Python, librosa, NumPy, sample rate, channel count, and host profile; - the complete metric implementation contract and noninferiority/speed thresholds; - the paired-uncertainty procedure identity, confidence level, resample count, and random seed. @@ -27,7 +27,9 @@ The validator requires a 95% confidence level because the current result schema The validator requires `source_uri` to be an explicit non-`file:` URI. Absolute, relative, drive-relative, and `file:` filesystem forms are rejected as provenance authorities. A benchmark may remain private when licensing requires that, but the receipt must identify the licensed material without leaking the workstation path that happened to hold it. -The current minimum of two tracks is only a technical guard against treating one timing sample as a corpus. It is **not** a scientific sample-size claim. Corpus breadth, genre/instrumentation coverage, annotation quality, and a defensible power/uncertainty plan remain part of the experiment review before a production switch can be accepted. +Track IDs are labels, not independent scientific units by themselves. Schema v1 rejects duplicate `audio_sha256` values across distinct track IDs so the same audio bytes cannot be counted repeatedly as apparent corpus breadth or independent paired observations. A scientifically justified repeated-item or clustered design would require an explicit preregistered dependence model and a schema revision rather than aliasing one recording under several IDs. + +The current minimum of two tracks is only a technical guard against treating one timing sample as a corpus. It is **not** a scientific sample-size claim. Corpus breadth, genre/instrumentation coverage, annotation quality, independence/dependence structure, and a defensible power/uncertainty plan remain part of the experiment review before a production switch can be accepted. Schema v1 does not contain a preregistered dropout, exclusion, or missing-track policy. Therefore every registered track must complete both baseline and candidate measurement for an acceptance PASS. A result may record failed track IDs for diagnosis, but any non-empty `failed_tracks` list makes the acceptance decision fail. A future tolerance for failed or excluded tracks requires a reviewed preregistration rule and schema revision rather than post-result omission. @@ -79,11 +81,11 @@ The receipt is rejected when a track is omitted/reordered, the uncertainty plan Registration and result JSON are evidence, not trusted configuration. CLI admission is bounded to 2 MiB per file, reads from one already-open regular-file descriptor, requires UTF-8 and standards-compliant finite JSON values, and rejects duplicate object keys instead of accepting last-key-wins semantics. These controls prevent ambiguous evidence identities and bound memory use before scientific validation begins. -The JSON `source_uri` value remains provenance metadata only; it is never dereferenced by this validator. It must use an explicit non-file URI scheme; absolute, relative, drive-relative, and `file:` filesystem forms are rejected. Audio and annotation bytes are not opened by the evidence validator and are bound to the registration through SHA-256 identities supplied by the experiment process. +The JSON `source_uri` value remains provenance metadata only; it is never dereferenced by this validator. It must use an explicit non-file URI scheme; absolute, relative, drive-relative, and `file:` filesystem forms are rejected. Audio and annotation bytes are not opened by the evidence validator and are bound to the registration through SHA-256 identities supplied by the experiment process. Audio content identity must also be unique within schema-v1 corpus membership; a second track ID carrying the same `audio_sha256` fails admission. ## Reproducibility sequence -1. Review the rights basis, corpus composition, feature hypothesis, metric contract, host profile, numeric margins, and paired-uncertainty procedure **before** running the candidate. Freeze the procedure identifier, confidence level, resample count, and random seed in the registration. If any failure/exclusion tolerance is scientifically required, define and version that policy before measurement rather than adding it after failures are observed. +1. Review the rights basis, corpus composition, distinct audio-content identities, feature hypothesis, metric contract, host profile, numeric margins, and paired-uncertainty procedure **before** running the candidate. Freeze the procedure identifier, confidence level, resample count, and random seed in the registration. If repeated recordings, clustering, or any failure/exclusion tolerance is scientifically required, define and version that dependence/exclusion policy before measurement rather than aliasing or omitting observations after results are visible. 2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain its canonical SHA-256. 3. Run baseline and candidate on the same decoded track identities and host profile. Record per-track recognized MIR metrics, p50/p95 latency, peak RSS, failures, and paired uncertainty using exactly the preregistered procedure. 4. Put the registration digest and the identical uncertainty-plan fields in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. Under schema v1, any recorded failed track keeps the diagnostic receipt valid but makes acceptance fail closed. @@ -96,9 +98,9 @@ No step authorizes committing licensed audio to Git. Rights-cleared means BandSc - Audio and annotation files are untrusted inputs to the future experiment runner. This validator reads bounded JSON evidence only and does not open audio, execute subprocesses, make network requests, or follow paths from the registration. - Evidence JSON is limited to 2 MiB, must be a regular file read through one open descriptor, must decode as UTF-8, and rejects duplicate keys plus non-standard `NaN`/`Infinity` constants. - `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute, relative, drive-relative, and `file:` forms are rejected; an explicit non-file URI scheme is required so transient workstation paths cannot become provenance authority or leak into review artifacts. -- Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt. -- Invalid/non-finite measurements, corpus drift, uncertainty-plan drift, registration drift, missing per-track evidence, unregistered measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. -- The validator does not claim that SHA-256 proves licensing, annotation validity, scientific adequacy, or that the registered statistical procedure is appropriate. Rights and scientific review remain separate gates. +- Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt; duplicate audio SHA-256 values under different track IDs are rejected so one recording cannot be silently counted multiple times. +- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, uncertainty-plan drift, registration drift, missing per-track evidence, unregistered measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. +- The validator does not claim that SHA-256 proves licensing, annotation validity, scientific adequacy, independence beyond exact-byte uniqueness, or that the registered statistical procedure is appropriate. Rights and scientific review remain separate gates. ## References From f9e2edb453f55b07e1f9b034db553d24063f3ece Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 14:10:06 +0900 Subject: [PATCH 024/216] docs(mir): connect corpus identity guard to MIR de-duplication evidence --- docs/traceability/mir/structure-feature-noninferiority.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index 941476144..2262ea7b9 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -29,6 +29,8 @@ The validator requires `source_uri` to be an explicit non-`file:` URI. Absolute, Track IDs are labels, not independent scientific units by themselves. Schema v1 rejects duplicate `audio_sha256` values across distinct track IDs so the same audio bytes cannot be counted repeatedly as apparent corpus breadth or independent paired observations. A scientifically justified repeated-item or clustered design would require an explicit preregistered dependence model and a schema revision rather than aliasing one recording under several IDs. +This guard is also consistent with recent MIR dataset-quality work. Choi et al. (2025) show that duplicated music items can make evaluation unreliable through leakage and motivate explicit de-duplication in a large music benchmark. Their study is on symbolic MIDI and therefore does **not** establish BandScope's audio-structure independence assumptions; it supports the narrower engineering rule that duplicate content identity must not silently masquerade as distinct evaluation evidence. + The current minimum of two tracks is only a technical guard against treating one timing sample as a corpus. It is **not** a scientific sample-size claim. Corpus breadth, genre/instrumentation coverage, annotation quality, independence/dependence structure, and a defensible power/uncertainty plan remain part of the experiment review before a production switch can be accepted. Schema v1 does not contain a preregistered dropout, exclusion, or missing-track policy. Therefore every registered track must complete both baseline and candidate measurement for an acceptance PASS. A result may record failed track IDs for diagnosis, but any non-empty `failed_tracks` list makes the acceptance decision fail. A future tolerance for failed or excluded tracks requires a reviewed preregistration rule and schema revision rather than post-result omission. @@ -106,6 +108,8 @@ No step authorizes committing licensed audio to Git. Rights-cleared means BandSc Buisson, M., McFee, B., Essid, S., & Crayencour, H. C. (2024). Self-supervised learning of multi-level audio representations for music segmentation. *IEEE/ACM Transactions on Audio, Speech, and Language Processing, 32*, 2141–2152. https://doi.org/10.1109/TASLP.2024.3379894 +Choi, E., Kim, H., Ryu, J., Nam, J., & Jeong, D. (2025). *On the de-duplication of the Lakh MIDI dataset* [Conference paper]. International Society for Music Information Retrieval Conference. https://doi.org/10.5281/zenodo.17811316 + Kim, T., & Nam, J. (2023). All-in-one metrical and functional structure analysis with neighborhood attentions on demixed audio. In *2023 IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA)* (pp. 1–5). IEEE. https://doi.org/10.1109/WASPAA58266.2023.10248148 MIREX. (2025). *Music Structure Analysis*. https://music-ir.org/mirex/wiki/2025:Music_Structure_Analysis From 23698b2572fe4e63b974e3a3f1c800dc27a9d283 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 15:05:17 +0900 Subject: [PATCH 025/216] test(mir): reject aggregate receipts detached from track evidence --- ...nferiority_aggregate_consistency_policy.py | 51 +++++++++++++++++++ 1 file changed, 51 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_noninferiority_aggregate_consistency_policy.py diff --git a/services/analysis-engine/tests/test_structure_noninferiority_aggregate_consistency_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_aggregate_consistency_policy.py new file mode 100644 index 000000000..6cf47869a --- /dev/null +++ b/services/analysis-engine/tests/test_structure_noninferiority_aggregate_consistency_policy.py @@ -0,0 +1,51 @@ +"""Regression tests for aggregate-to-track structure evidence consistency.""" + +from __future__ import annotations + +import pytest + +from test_structure_noninferiority_policy import _registration, _result, _validator + + +def _track_candidate(result: dict[str, object], index: int) -> dict[str, object]: + """Return one candidate measurement object from a synthetic result fixture.""" + tracks = result["tracks"] + assert isinstance(tracks, list) + track = tracks[index] + assert isinstance(track, dict) + candidate = track["candidate"] + assert isinstance(candidate, dict) + return candidate + + +def test_result_rejects_aggregate_quality_outside_track_evidence_range() -> None: + """Aggregate quality cannot claim a value no admitted track can support.""" + validator = _validator() + registration = _registration() + digest = validator.registration_digest(registration) + result = _result(registration, digest) + + for index in range(2): + candidate = _track_candidate(result, index) + candidate["boundary_precision_0_5"] = 0.50 + candidate["boundary_recall_0_5"] = 0.50 + candidate["boundary_f_0_5"] = 0.50 + + with pytest.raises(ValueError, match="aggregate candidate boundary_f_0_5"): + validator.evaluate_result(registration, result) + + +def test_result_rejects_aggregate_latency_ratio_outside_track_evidence_range() -> None: + """Aggregate speedup cannot contradict every admitted paired track ratio.""" + validator = _validator() + registration = _registration() + digest = validator.registration_digest(registration) + result = _result(registration, digest) + + for index in range(2): + candidate = _track_candidate(result, index) + candidate["p50_latency_seconds"] = 7.0 + candidate["p95_latency_seconds"] = 9.0 + + with pytest.raises(ValueError, match="aggregate p95 latency ratio"): + validator.evaluate_result(registration, result) From 595303747a151eb423e829b8501d2b751f1ee9ac Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 15:06:13 +0900 Subject: [PATCH 026/216] revert(mir): keep aggregation procedure scientifically undecided --- ...nferiority_aggregate_consistency_policy.py | 51 ------------------- 1 file changed, 51 deletions(-) delete mode 100644 services/analysis-engine/tests/test_structure_noninferiority_aggregate_consistency_policy.py diff --git a/services/analysis-engine/tests/test_structure_noninferiority_aggregate_consistency_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_aggregate_consistency_policy.py deleted file mode 100644 index 6cf47869a..000000000 --- a/services/analysis-engine/tests/test_structure_noninferiority_aggregate_consistency_policy.py +++ /dev/null @@ -1,51 +0,0 @@ -"""Regression tests for aggregate-to-track structure evidence consistency.""" - -from __future__ import annotations - -import pytest - -from test_structure_noninferiority_policy import _registration, _result, _validator - - -def _track_candidate(result: dict[str, object], index: int) -> dict[str, object]: - """Return one candidate measurement object from a synthetic result fixture.""" - tracks = result["tracks"] - assert isinstance(tracks, list) - track = tracks[index] - assert isinstance(track, dict) - candidate = track["candidate"] - assert isinstance(candidate, dict) - return candidate - - -def test_result_rejects_aggregate_quality_outside_track_evidence_range() -> None: - """Aggregate quality cannot claim a value no admitted track can support.""" - validator = _validator() - registration = _registration() - digest = validator.registration_digest(registration) - result = _result(registration, digest) - - for index in range(2): - candidate = _track_candidate(result, index) - candidate["boundary_precision_0_5"] = 0.50 - candidate["boundary_recall_0_5"] = 0.50 - candidate["boundary_f_0_5"] = 0.50 - - with pytest.raises(ValueError, match="aggregate candidate boundary_f_0_5"): - validator.evaluate_result(registration, result) - - -def test_result_rejects_aggregate_latency_ratio_outside_track_evidence_range() -> None: - """Aggregate speedup cannot contradict every admitted paired track ratio.""" - validator = _validator() - registration = _registration() - digest = validator.registration_digest(registration) - result = _result(registration, digest) - - for index in range(2): - candidate = _track_candidate(result, index) - candidate["p50_latency_seconds"] = 7.0 - candidate["p95_latency_seconds"] = 9.0 - - with pytest.raises(ValueError, match="aggregate p95 latency ratio"): - validator.evaluate_result(registration, result) From c9cb23b2093a7c236013900b3588e50e2b8534b5 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 15:06:22 +0900 Subject: [PATCH 027/216] test(mir): reject unregistered evidence envelope fields --- ...e_noninferiority_envelope_schema_policy.py | 45 +++++++++++++++++++ 1 file changed, 45 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py diff --git a/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py new file mode 100644 index 000000000..9abdec219 --- /dev/null +++ b/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py @@ -0,0 +1,45 @@ +"""Regression tests for closed-world structure evidence envelopes.""" + +from __future__ import annotations + +import pytest + +from test_structure_noninferiority_policy import _registration, _result, _validator + + +def test_registration_rejects_unregistered_top_level_field() -> None: + """Registration cannot silently widen the preregistered evidence contract.""" + validator = _validator() + registration = _registration() + registration["post_hoc_analysis_plan"] = "not part of schema v1" + + with pytest.raises(ValueError, match="registration contains unregistered field"): + validator.validate_registration(registration) + + +def test_result_rejects_unregistered_top_level_field() -> None: + """Post-result claims outside schema v1 must not enter admitted evidence.""" + validator = _validator() + registration = _registration() + digest = validator.registration_digest(registration) + result = _result(registration, digest) + result["post_hoc_p_value"] = 0.001 + + with pytest.raises(ValueError, match="result contains unregistered field"): + validator.evaluate_result(registration, result) + + +def test_track_receipt_rejects_unregistered_control_field() -> None: + """A track cannot carry a post-hoc selection flag beside its measurements.""" + validator = _validator() + registration = _registration() + digest = validator.registration_digest(registration) + result = _result(registration, digest) + tracks = result["tracks"] + assert isinstance(tracks, list) + first_track = tracks[0] + assert isinstance(first_track, dict) + first_track["selected_for_aggregate"] = True + + with pytest.raises(ValueError, match="result.tracks\[0\] contains unregistered field"): + validator.evaluate_result(registration, result) From 515d1898c9eb7052586339a88554ffbaa57c2405 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 15:06:52 +0900 Subject: [PATCH 028/216] test(mir): cover nested result envelope drift --- ...ucture_noninferiority_envelope_schema_policy.py | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py index 9abdec219..c26612ada 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py @@ -43,3 +43,17 @@ def test_track_receipt_rejects_unregistered_control_field() -> None: with pytest.raises(ValueError, match="result.tracks\[0\] contains unregistered field"): validator.evaluate_result(registration, result) + + +def test_aggregate_receipt_rejects_unregistered_control_field() -> None: + """Aggregate control metadata cannot bypass the closed result schema.""" + validator = _validator() + registration = _registration() + digest = validator.registration_digest(registration) + result = _result(registration, digest) + aggregate = result["aggregate"] + assert isinstance(aggregate, dict) + aggregate["selected_tracks"] = ["licensed-track-001"] + + with pytest.raises(ValueError, match="result.aggregate contains unregistered field"): + validator.evaluate_result(registration, result) From 4b2f1f3f2420702732902da382b48c4f5db68237 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 15:07:58 +0900 Subject: [PATCH 029/216] fix(mir): close scientific evidence envelope schema --- .../validate_structure_noninferiority.py | 43 +++++++++++++++++++ 1 file changed, 43 insertions(+) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index bd14f0551..88c0788e7 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -62,6 +62,30 @@ "p95_latency_seconds", "peak_rss_mib", ) +_REGISTRATION_FIELDS = { + "schema_version", + "experiment_id", + "hypothesis", + "metrics", + "uncertainty", + "corpus", + "runtime", +} +_RESULT_FIELDS = { + "schema_version", + "experiment_id", + "registration_sha256", + "uncertainty", + "corpus_track_ids", + "tracks", + "aggregate", + "paired_delta_ci95", + "p95_latency_ratio_ci95", + "failed_tracks", + "claim_boundary", +} +_TRACK_RECEIPT_FIELDS = {"track_id", "baseline", "candidate"} +_AGGREGATE_FIELDS = {"baseline", "candidate"} def _mapping(value: object, field: str) -> Mapping[str, Any]: @@ -71,6 +95,21 @@ def _mapping(value: object, field: str) -> Mapping[str, Any]: return value +def _require_exact_fields( + value: Mapping[str, Any], + expected: set[str], + field: str, +) -> None: + """Require an evidence object to use exactly its schema-v1 field set.""" + actual = set(value) + missing = sorted(expected - actual) + extra = sorted(actual - expected) + if missing: + raise ValueError(f"{field} missing required field: {missing[0]}") + if extra: + raise ValueError(f"{field} contains unregistered field: {extra[0]}") + + def _sequence(value: object, field: str) -> Sequence[Any]: """Return a non-string sequence or raise a field-specific error.""" if isinstance(value, (str, bytes)) or not isinstance(value, Sequence): @@ -336,6 +375,7 @@ def validate_registration(registration_value: object) -> None: any corpus result is inspected. """ registration = _mapping(registration_value, "registration") + _require_exact_fields(registration, _REGISTRATION_FIELDS, "registration") _validate_schema_version( registration.get("schema_version"), "schema_version", @@ -478,6 +518,7 @@ def _validate_result_identity( result: Mapping[str, Any], ) -> list[str]: """Bind a result receipt to the frozen registration and corpus order.""" + _require_exact_fields(result, _RESULT_FIELDS, "result") _validate_schema_version( result.get("schema_version"), "result.schema_version", @@ -543,6 +584,7 @@ def _validate_track_measurements( for index, raw_track in enumerate(tracks): field = f"result.tracks[{index}]" track = _mapping(raw_track, field) + _require_exact_fields(track, _TRACK_RECEIPT_FIELDS, field) track_id = _nonempty_text(track.get("track_id"), f"{field}.track_id") actual_track_ids.append(track_id) _validate_measurement_side(track.get("baseline"), f"{field}.baseline") @@ -563,6 +605,7 @@ def evaluate_result( _validate_track_measurements(result, track_ids) aggregate = _mapping(result.get("aggregate"), "result.aggregate") + _require_exact_fields(aggregate, _AGGREGATE_FIELDS, "result.aggregate") baseline = _validate_measurement_side( aggregate.get("baseline"), "result.aggregate.baseline", From 8a60241b82f744967c497dac47a3a1598d140ac8 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 15:08:55 +0900 Subject: [PATCH 030/216] docs(mir): close evidence envelope and preserve aggregation boundary --- .../mir/structure-feature-noninferiority.md | 18 ++++++++++++------ 1 file changed, 12 insertions(+), 6 deletions(-) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index 2262ea7b9..8a6ac38b3 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -23,6 +23,8 @@ A registration is valid only when it records all of the following before the res - the complete metric implementation contract and noninferiority/speed thresholds; - the paired-uncertainty procedure identity, confidence level, resample count, and random seed. +The schema-v1 registration envelope is closed-world at the top level. Extra fields are not treated as harmless annotations: an unregistered field changes what reviewers may infer was preregistered, so it fails admission instead of silently entering the evidence artifact. + The validator requires a 95% confidence level because the current result schema is explicitly `ci95`; it does not prescribe which scientifically defensible paired procedure must produce that interval. The procedure identifier, resample count, and seed are part of the preregistration digest so they cannot be changed after results are seen. `paired-track-bootstrap-v1` and the numeric values used in unit tests are policy fixtures only, not an approved BandScope production analysis plan. The validator requires `source_uri` to be an explicit non-`file:` URI. Absolute, relative, drive-relative, and `file:` filesystem forms are rejected as provenance authorities. A benchmark may remain private when licensing requires that, but the receipt must identify the licensed material without leaking the workstation path that happened to hold it. @@ -75,9 +77,13 @@ The repository does not currently contain approved numeric margins or an approve A result receipt must contain the exact registration digest, exact uncertainty plan, exact corpus order, one complete baseline/candidate measurement pair per registered track, aggregate measurements, paired 95% intervals for every gated quality metric, a paired p95-latency-ratio interval, failed-track IDs, and a claim boundary. +Schema v1 treats the result envelope, each per-track receipt, and the aggregate envelope as closed-world objects. A top-level post-result field such as an unregistered p-value, a per-track `selected_for_aggregate` flag, or aggregate-side selection metadata is rejected rather than ignored. The same rule already applies inside each baseline/candidate measurement object. This prevents a producer from attaching an unreviewed post-hoc selection or inferential claim to an otherwise accepted receipt and having that field travel with the admitted evidence artifact. + Each per-track and aggregate baseline/candidate measurement is a closed-world receipt: it must contain exactly the registered quality fields plus the reporting-only precision/recall, boundary-deviation, latency, and peak-RSS fields. Missing fields and additional post-hoc measurement fields both fail admission. This prevents a result producer from attaching an unreviewed score after candidate results are visible and presenting it as part of the admitted scientific receipt. -The receipt is rejected when a track is omitted/reordered, the uncertainty plan differs, a required metric is absent, an unregistered measurement field is added, a boundary/repetition P/R/F triplet is internally inconsistent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A structurally valid receipt with one or more known failed tracks is retained for diagnosis but evaluates to `passed=false` under schema v1. A passing receipt therefore means the preregistered decision rule passed for the complete registered corpus, uncertainty procedure, and runtime. It does not generalize automatically to other genres, codecs, sample rates, machines, annotation regimes, or statistical procedures. +The receipt is rejected when a track is omitted/reordered, the uncertainty plan differs, a required metric is absent, an unregistered envelope or measurement field is added, a boundary/repetition P/R/F triplet is internally inconsistent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A structurally valid receipt with one or more known failed tracks is retained for diagnosis but evaluates to `passed=false` under schema v1. A passing receipt therefore means the preregistered decision rule passed for the complete registered corpus, uncertainty procedure, and runtime. It does not generalize automatically to other genres, codecs, sample rates, machines, annotation regimes, or statistical procedures. + +The validator still does not recompute aggregate statistics or confidence intervals from per-track evidence. During review, a min/max range consistency heuristic was briefly encoded as a RED and then removed before production code changed because it would smuggle an unapproved assumption about aggregation into schema v1. Macro means, weighted means, micro-aggregated precision/recall/F and pooled latency summaries do not share one universally valid range relationship. The correct next step is to preregister the aggregation procedure and implement it in the experiment runner, not to infer a statistical method inside an admission validator after the fact. ## Machine-readable evidence admission @@ -87,11 +93,11 @@ The JSON `source_uri` value remains provenance metadata only; it is never derefe ## Reproducibility sequence -1. Review the rights basis, corpus composition, distinct audio-content identities, feature hypothesis, metric contract, host profile, numeric margins, and paired-uncertainty procedure **before** running the candidate. Freeze the procedure identifier, confidence level, resample count, and random seed in the registration. If repeated recordings, clustering, or any failure/exclusion tolerance is scientifically required, define and version that dependence/exclusion policy before measurement rather than aliasing or omitting observations after results are visible. +1. Review the rights basis, corpus composition, distinct audio-content identities, feature hypothesis, metric contract, host profile, numeric margins, aggregation procedure, and paired-uncertainty procedure **before** running the candidate. Freeze the procedure identifier, confidence level, resample count, and random seed in the registration. If repeated recordings, clustering, or any failure/exclusion tolerance is scientifically required, define and version that dependence/exclusion policy before measurement rather than aliasing or omitting observations after results are visible. 2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain its canonical SHA-256. -3. Run baseline and candidate on the same decoded track identities and host profile. Record per-track recognized MIR metrics, p50/p95 latency, peak RSS, failures, and paired uncertainty using exactly the preregistered procedure. +3. Run baseline and candidate on the same decoded track identities and host profile. The experiment runner must compute the approved aggregation and paired uncertainty procedure from the complete per-track evidence rather than accepting caller-authored aggregate numbers as scientific authority. Record per-track recognized MIR metrics, p50/p95 latency, peak RSS, failures, aggregate outputs, and paired uncertainty using exactly the preregistered procedure. 4. Put the registration digest and the identical uncertainty-plan fields in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. Under schema v1, any recorded failed track keeps the diagnostic receipt valid but makes acceptance fail closed. -5. Preserve the registration, result receipt, corpus/annotation hashes, exact source commit, lock hash, and measurement procedure together. A production feature switch requires this evidence plus normal code review and protected-head checks. +5. Preserve the registration, result receipt, corpus/annotation hashes, exact source commit, lock hash, aggregation implementation identity, and uncertainty procedure together. A production feature switch requires this evidence plus normal code review and protected-head checks. No step authorizes committing licensed audio to Git. Rights-cleared means BandScope has the necessary evaluation right; redistribution is a separate permission. @@ -101,8 +107,8 @@ No step authorizes committing licensed audio to Git. Rights-cleared means BandSc - Evidence JSON is limited to 2 MiB, must be a regular file read through one open descriptor, must decode as UTF-8, and rejects duplicate keys plus non-standard `NaN`/`Infinity` constants. - `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute, relative, drive-relative, and `file:` forms are rejected; an explicit non-file URI scheme is required so transient workstation paths cannot become provenance authority or leak into review artifacts. - Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt; duplicate audio SHA-256 values under different track IDs are rejected so one recording cannot be silently counted multiple times. -- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, uncertainty-plan drift, registration drift, missing per-track evidence, unregistered measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. -- The validator does not claim that SHA-256 proves licensing, annotation validity, scientific adequacy, independence beyond exact-byte uniqueness, or that the registered statistical procedure is appropriate. Rights and scientific review remain separate gates. +- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, uncertainty-plan drift, registration drift, missing per-track evidence, unregistered registration/result/track/aggregate/measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. +- The validator does not claim that SHA-256 proves licensing, annotation validity, scientific adequacy, independence beyond exact-byte uniqueness, aggregate derivation, or that the registered statistical procedure is appropriate. Rights and scientific review remain separate gates. ## References From dad6e22fd13831c1763b072efcbeefc36183e2a9 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 15:11:19 +0900 Subject: [PATCH 031/216] test(mir): close nested registration evidence objects --- ...e_noninferiority_envelope_schema_policy.py | 39 ++++++++++++++++++- 1 file changed, 38 insertions(+), 1 deletion(-) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py index c26612ada..1146f1d03 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py @@ -4,7 +4,13 @@ import pytest -from test_structure_noninferiority_policy import _registration, _result, _validator +from test_structure_noninferiority_policy import ( + _corpus, + _metrics, + _registration, + _result, + _validator, +) def test_registration_rejects_unregistered_top_level_field() -> None: @@ -17,6 +23,37 @@ def test_registration_rejects_unregistered_top_level_field() -> None: validator.validate_registration(registration) +def test_registration_rejects_unregistered_nested_contract_fields() -> None: + """Hypothesis, metric, corpus, and runtime objects are closed schema objects.""" + validator = _validator() + + hypothesis_drift = _registration() + hypothesis = hypothesis_drift["hypothesis"] + assert isinstance(hypothesis, dict) + hypothesis["pilot_selected"] = True + with pytest.raises(ValueError, match="hypothesis contains unregistered field"): + validator.validate_registration(hypothesis_drift) + + metric_drift = _registration() + metric = _metrics(metric_drift)["boundary_f_0_5"] + assert isinstance(metric, dict) + metric["post_hoc_weight"] = 2.0 + with pytest.raises(ValueError, match="metrics.boundary_f_0_5 contains unregistered field"): + validator.validate_registration(metric_drift) + + corpus_drift = _registration() + _corpus(corpus_drift)[0]["selected_for_primary_analysis"] = True + with pytest.raises(ValueError, match="corpus\[0\] contains unregistered field"): + validator.validate_registration(corpus_drift) + + runtime_drift = _registration() + runtime = runtime_drift["runtime"] + assert isinstance(runtime, dict) + runtime["cache_state"] = "warm" + with pytest.raises(ValueError, match="runtime contains unregistered field"): + validator.validate_registration(runtime_drift) + + def test_result_rejects_unregistered_top_level_field() -> None: """Post-result claims outside schema v1 must not enter admitted evidence.""" validator = _validator() From a43e2f3b6d20f96298781fc4be4c2b7e31a4a1fb Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 15:12:23 +0900 Subject: [PATCH 032/216] fix(mir): close nested preregistration schema objects --- .../validate_structure_noninferiority.py | 32 +++++++++++++++++++ 1 file changed, 32 insertions(+) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index 88c0788e7..8bdb2c5a8 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -71,6 +71,25 @@ "corpus", "runtime", } +_HYPOTHESIS_FIELDS = {"baseline_feature", "candidate_feature"} +_CORPUS_TRACK_FIELDS = { + "track_id", + "audio_sha256", + "annotation_sha256", + "rights_basis", + "rights_cleared", + "source_uri", +} +_RUNTIME_FIELDS = { + "source_commit", + "uv_lock_sha256", + "python_version", + "librosa_version", + "numpy_version", + "sample_rate_hz", + "channels", + "host_profile", +} _RESULT_FIELDS = { "schema_version", "experiment_id", @@ -211,6 +230,11 @@ def _validate_metrics(metrics_value: object) -> None: for metric_name, expected_config in _QUALITY_METRICS.items(): config = _mapping(metrics[metric_name], f"metrics.{metric_name}") + _require_exact_fields( + config, + set(expected_config) | {"noninferiority_margin"}, + f"metrics.{metric_name}", + ) implementation = _nonempty_text( config.get("implementation"), f"metrics.{metric_name}.implementation", @@ -248,6 +272,11 @@ def _validate_metrics(metrics_value: object) -> None: ) latency = _mapping(metrics[_LATENCY_METRIC], f"metrics.{_LATENCY_METRIC}") + _require_exact_fields( + latency, + {"maximum_candidate_ratio"}, + f"metrics.{_LATENCY_METRIC}", + ) maximum_ratio = _finite_number( latency.get("maximum_candidate_ratio"), f"metrics.{_LATENCY_METRIC}.maximum_candidate_ratio", @@ -323,6 +352,7 @@ def _validate_corpus(corpus_value: object) -> list[str]: for index, raw_track in enumerate(corpus): field = f"corpus[{index}]" track = _mapping(raw_track, field) + _require_exact_fields(track, _CORPUS_TRACK_FIELDS, field) track_id = _nonempty_text(track.get("track_id"), f"{field}.track_id") if track_id in seen_track_ids: raise ValueError(f"duplicate track_id: {track_id}") @@ -343,6 +373,7 @@ def _validate_corpus(corpus_value: object) -> list[str]: def _validate_runtime(runtime_value: object) -> None: """Validate exact runtime identity for reproducible paired measurements.""" runtime = _mapping(runtime_value, "runtime") + _require_exact_fields(runtime, _RUNTIME_FIELDS, "runtime") _commit(runtime.get("source_commit"), "runtime.source_commit") _sha256(runtime.get("uv_lock_sha256"), "runtime.uv_lock_sha256") for field in ( @@ -383,6 +414,7 @@ def validate_registration(registration_value: object) -> None: _nonempty_text(registration.get("experiment_id"), "experiment_id") hypothesis = _mapping(registration.get("hypothesis"), "hypothesis") + _require_exact_fields(hypothesis, _HYPOTHESIS_FIELDS, "hypothesis") baseline = _nonempty_text( hypothesis.get("baseline_feature"), "hypothesis.baseline_feature", From e69ee29b9789ab2099b62ce0459350bb68287c73 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 15:13:13 +0900 Subject: [PATCH 033/216] docs(mir): document nested closed-world preregistration --- docs/traceability/mir/structure-feature-noninferiority.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index 8a6ac38b3..3230991e5 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -23,7 +23,7 @@ A registration is valid only when it records all of the following before the res - the complete metric implementation contract and noninferiority/speed thresholds; - the paired-uncertainty procedure identity, confidence level, resample count, and random seed. -The schema-v1 registration envelope is closed-world at the top level. Extra fields are not treated as harmless annotations: an unregistered field changes what reviewers may infer was preregistered, so it fails admission instead of silently entering the evidence artifact. +Schema v1 is closed-world not only at the registration top level but also for the hypothesis object, every registered metric configuration, every corpus-track object, and the runtime object. Extra fields are not treated as harmless annotations: an unregistered pilot-selection flag, metric weight, corpus-selection marker, or cache/runtime hint changes what reviewers may infer was preregistered, so it fails admission instead of silently entering the evidence artifact. The validator requires a 95% confidence level because the current result schema is explicitly `ci95`; it does not prescribe which scientifically defensible paired procedure must produce that interval. The procedure identifier, resample count, and seed are part of the preregistration digest so they cannot be changed after results are seen. `paired-track-bootstrap-v1` and the numeric values used in unit tests are policy fixtures only, not an approved BandScope production analysis plan. @@ -107,7 +107,7 @@ No step authorizes committing licensed audio to Git. Rights-cleared means BandSc - Evidence JSON is limited to 2 MiB, must be a regular file read through one open descriptor, must decode as UTF-8, and rejects duplicate keys plus non-standard `NaN`/`Infinity` constants. - `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute, relative, drive-relative, and `file:` forms are rejected; an explicit non-file URI scheme is required so transient workstation paths cannot become provenance authority or leak into review artifacts. - Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt; duplicate audio SHA-256 values under different track IDs are rejected so one recording cannot be silently counted multiple times. -- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, uncertainty-plan drift, registration drift, missing per-track evidence, unregistered registration/result/track/aggregate/measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. +- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, uncertainty-plan drift, registration drift, missing per-track evidence, unregistered registration/hypothesis/metric/corpus/runtime/result/track/aggregate/measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. - The validator does not claim that SHA-256 proves licensing, annotation validity, scientific adequacy, independence beyond exact-byte uniqueness, aggregate derivation, or that the registered statistical procedure is appropriate. Rights and scientific review remain separate gates. ## References From ac81bf9eb9f0bd257205f2fbf400afecf8418cd7 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 16:06:09 +0900 Subject: [PATCH 034/216] test(mir): reject post-result claim-boundary expansion --- ...re_noninferiority_envelope_schema_policy.py | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py index 1146f1d03..75c1c2544 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py @@ -94,3 +94,21 @@ def test_aggregate_receipt_rejects_unregistered_control_field() -> None: with pytest.raises(ValueError, match="result.aggregate contains unregistered field"): validator.evaluate_result(registration, result) + + +def test_result_claim_boundary_cannot_expand_beyond_preregistration() -> None: + """Post-result prose cannot widen the scientific claim after evidence is visible.""" + validator = _validator() + registration = _registration() + registration["claim_boundary"] = ( + "Applies only to the registered rights-cleared corpus and exact runtime identity." + ) + digest = validator.registration_digest(registration) + result = _result(registration, digest) + result["claim_boundary"] = "Applies to all music, codecs, machines, and genres." + + with pytest.raises( + ValueError, + match="result.claim_boundary must exactly match the preregistered claim boundary", + ): + validator.evaluate_result(registration, result) From 0c52e51313cc9b575cb72e75f66a73dd1be5c1da Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 16:09:04 +0900 Subject: [PATCH 035/216] test(mir): bind result claim fixture to preregistration --- .../tests/test_structure_noninferiority_policy.py | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_policy.py index 5108444c7..c9523cedc 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_policy.py @@ -86,6 +86,9 @@ def _registration() -> dict[str, object]: "channels": 1, "host_profile": "registered-cpu-host-v1", }, + "claim_boundary": ( + "Applies only to the registered rights-cleared corpus and exact runtime identity." + ), } @@ -146,6 +149,8 @@ def _result(registration: dict[str, object], registration_sha256: str) -> dict[s p95_latency_seconds=4.2, peak_rss_mib=590.0, ) + claim_boundary = registration["claim_boundary"] + assert isinstance(claim_boundary, str) return { "schema_version": 1, "experiment_id": "structure-chroma-stft-vs-cqt-v1", @@ -176,9 +181,7 @@ def _result(registration: dict[str, object], registration_sha256: str) -> dict[s }, "p95_latency_ratio_ci95": [0.66, 0.76], "failed_tracks": [], - "claim_boundary": ( - "Applies only to the registered rights-cleared corpus and exact runtime identity." - ), + "claim_boundary": claim_boundary, } From 5bca92a69dd9235f3ca36247617a7f8886309275 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 16:10:04 +0900 Subject: [PATCH 036/216] fix(mir): freeze claim boundary before structure experiment --- .../validate_structure_noninferiority.py | 16 +++++++++++++++- 1 file changed, 15 insertions(+), 1 deletion(-) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index 8bdb2c5a8..571d2fd48 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -70,6 +70,7 @@ "uncertainty", "corpus", "runtime", + "claim_boundary", } _HYPOTHESIS_FIELDS = {"baseline_feature", "candidate_feature"} _CORPUS_TRACK_FIELDS = { @@ -432,6 +433,7 @@ def validate_registration(registration_value: object) -> None: _validate_uncertainty(registration.get("uncertainty"), "uncertainty") _validate_corpus(registration.get("corpus")) _validate_runtime(registration.get("runtime")) + _nonempty_text(registration.get("claim_boundary"), "claim_boundary") def registration_digest(registration_value: object) -> str: @@ -598,6 +600,19 @@ def _validate_result_identity( raise ValueError( "result.corpus_track_ids must exactly match registration corpus order" ) + + registered_claim_boundary = _nonempty_text( + registration.get("claim_boundary"), + "claim_boundary", + ) + result_claim_boundary = _nonempty_text( + result.get("claim_boundary"), + "result.claim_boundary", + ) + if result_claim_boundary != registered_claim_boundary: + raise ValueError( + "result.claim_boundary must exactly match the preregistered claim boundary" + ) return expected_track_ids @@ -705,7 +720,6 @@ def evaluate_result( raise ValueError( f"result.failed_tracks contains unknown track_id: {unknown_failed[0]}" ) - _nonempty_text(result.get("claim_boundary"), "result.claim_boundary") metrics = _mapping(registration["metrics"], "metrics") failed_requirements: list[str] = [] From 9416304bdf7e40ce6356245a16142671ae964c77 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 16:10:48 +0900 Subject: [PATCH 037/216] docs(mir): bind scientific claim scope to preregistration --- .../mir/structure-feature-noninferiority.md | 23 ++++++++++--------- 1 file changed, 12 insertions(+), 11 deletions(-) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index 3230991e5..9c21ec536 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -9,9 +9,9 @@ Production baseline: `chroma_cqt` in `sections/segmenter.py` The earlier `chroma_cqt` → `chroma_stft` optimization is not a production change until a preregistered, rights-cleared real-audio experiment shows that the candidate is noninferior on rehearsal-relevant structure quality and materially faster on the same corpus and runtime identity. -The repository now owns a result-admission boundary in `scripts/research/validate_structure_noninferiority.py`. The validator does **not** calculate MIR metrics and does not choose scientific margins or an uncertainty procedure. It freezes the reviewed experiment contract, binds a result receipt to that contract by canonical SHA-256, requires per-track and aggregate evidence, and applies the registered confidence-interval decision rules. Metric computation and uncertainty estimation remain separate scientific measurement steps. +The repository now owns a result-admission boundary in `scripts/research/validate_structure_noninferiority.py`. The validator does **not** calculate MIR metrics and does not choose scientific margins or an uncertainty procedure. It freezes the reviewed experiment contract, including its claim boundary, binds a result receipt to that contract by canonical SHA-256, requires per-track and aggregate evidence, and applies the registered confidence-interval decision rules. Metric computation and uncertainty estimation remain separate scientific measurement steps. -This keeps a performance result from silently changing the corpus, thresholds, feature identities, uncertainty procedure, or runtime after the measurements are visible. +This keeps a performance result from silently changing the corpus, thresholds, feature identities, uncertainty procedure, runtime, or permitted interpretation after the measurements are visible. ## Registered inputs @@ -21,9 +21,10 @@ A registration is valid only when it records all of the following before the res - a rights-cleared real-audio corpus with stable track IDs, content-unique audio SHA-256 identities, annotation SHA-256, rights basis, and provenance URI; - exact source commit and `uv.lock` identity plus Python, librosa, NumPy, sample rate, channel count, and host profile; - the complete metric implementation contract and noninferiority/speed thresholds; -- the paired-uncertainty procedure identity, confidence level, resample count, and random seed. +- the paired-uncertainty procedure identity, confidence level, resample count, and random seed; +- a non-empty claim boundary stating the population/runtime scope to which a passing result may be applied. -Schema v1 is closed-world not only at the registration top level but also for the hypothesis object, every registered metric configuration, every corpus-track object, and the runtime object. Extra fields are not treated as harmless annotations: an unregistered pilot-selection flag, metric weight, corpus-selection marker, or cache/runtime hint changes what reviewers may infer was preregistered, so it fails admission instead of silently entering the evidence artifact. +Schema v1 is closed-world not only at the registration top level but also for the hypothesis object, every registered metric configuration, every corpus-track object, and the runtime object. Extra fields are not treated as harmless annotations: an unregistered pilot-selection flag, metric weight, corpus-selection marker, or cache/runtime hint changes what reviewers may infer was preregistered, so it fails admission instead of silently entering the evidence artifact. The claim boundary is itself a required top-level registration field and therefore participates in the canonical registration SHA-256. The validator requires a 95% confidence level because the current result schema is explicitly `ci95`; it does not prescribe which scientifically defensible paired procedure must produce that interval. The procedure identifier, resample count, and seed are part of the preregistration digest so they cannot be changed after results are seen. `paired-track-bootstrap-v1` and the numeric values used in unit tests are policy fixtures only, not an approved BandScope production analysis plan. @@ -67,7 +68,7 @@ For p95 latency, let `r = candidate_p95 / baseline_p95`. The result passes the s `upper_CI95(r) <= maximum_candidate_ratio` -The result receipt must repeat the preregistered uncertainty plan exactly. The validator also requires the aggregate point delta/ratio to lie inside the supplied interval. Favorable point estimates therefore cannot hide an uncertainty interval that crosses a preregistered boundary, and the uncertainty method cannot be swapped after the result is known. +The result receipt must repeat the preregistered uncertainty plan and claim boundary exactly. The validator also requires the aggregate point delta/ratio to lie inside the supplied interval. Favorable point estimates therefore cannot hide an uncertainty interval that crosses a preregistered boundary, the uncertainty method cannot be swapped after the result is known, and a narrow experiment cannot be relabeled as evidence for a wider population or runtime after measurement. Because schema v1 has no preregistered exclusion policy, a non-empty `failed_tracks` list is an additional failed requirement even when every reported quality and latency interval is favorable. Failed tracks remain visible in the result receipt for diagnosis; they cannot be silently converted into a complete-case PASS after results are known. @@ -75,13 +76,13 @@ The repository does not currently contain approved numeric margins or an approve ## Result receipt -A result receipt must contain the exact registration digest, exact uncertainty plan, exact corpus order, one complete baseline/candidate measurement pair per registered track, aggregate measurements, paired 95% intervals for every gated quality metric, a paired p95-latency-ratio interval, failed-track IDs, and a claim boundary. +A result receipt must contain the exact registration digest, exact uncertainty plan, exact corpus order, one complete baseline/candidate measurement pair per registered track, aggregate measurements, paired 95% intervals for every gated quality metric, a paired p95-latency-ratio interval, failed-track IDs, and the exact preregistered claim boundary. Schema v1 treats the result envelope, each per-track receipt, and the aggregate envelope as closed-world objects. A top-level post-result field such as an unregistered p-value, a per-track `selected_for_aggregate` flag, or aggregate-side selection metadata is rejected rather than ignored. The same rule already applies inside each baseline/candidate measurement object. This prevents a producer from attaching an unreviewed post-hoc selection or inferential claim to an otherwise accepted receipt and having that field travel with the admitted evidence artifact. Each per-track and aggregate baseline/candidate measurement is a closed-world receipt: it must contain exactly the registered quality fields plus the reporting-only precision/recall, boundary-deviation, latency, and peak-RSS fields. Missing fields and additional post-hoc measurement fields both fail admission. This prevents a result producer from attaching an unreviewed score after candidate results are visible and presenting it as part of the admitted scientific receipt. -The receipt is rejected when a track is omitted/reordered, the uncertainty plan differs, a required metric is absent, an unregistered envelope or measurement field is added, a boundary/repetition P/R/F triplet is internally inconsistent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A structurally valid receipt with one or more known failed tracks is retained for diagnosis but evaluates to `passed=false` under schema v1. A passing receipt therefore means the preregistered decision rule passed for the complete registered corpus, uncertainty procedure, and runtime. It does not generalize automatically to other genres, codecs, sample rates, machines, annotation regimes, or statistical procedures. +The receipt is rejected when a track is omitted/reordered, the uncertainty plan differs, the claim boundary differs, a required metric is absent, an unregistered envelope or measurement field is added, a boundary/repetition P/R/F triplet is internally inconsistent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A structurally valid receipt with one or more known failed tracks is retained for diagnosis but evaluates to `passed=false` under schema v1. A passing receipt therefore means the preregistered decision rule passed for the complete registered corpus, uncertainty procedure, runtime, and claim boundary. It does not generalize automatically to other genres, codecs, sample rates, machines, annotation regimes, or statistical procedures. The validator still does not recompute aggregate statistics or confidence intervals from per-track evidence. During review, a min/max range consistency heuristic was briefly encoded as a RED and then removed before production code changed because it would smuggle an unapproved assumption about aggregation into schema v1. Macro means, weighted means, micro-aggregated precision/recall/F and pooled latency summaries do not share one universally valid range relationship. The correct next step is to preregister the aggregation procedure and implement it in the experiment runner, not to infer a statistical method inside an admission validator after the fact. @@ -93,11 +94,11 @@ The JSON `source_uri` value remains provenance metadata only; it is never derefe ## Reproducibility sequence -1. Review the rights basis, corpus composition, distinct audio-content identities, feature hypothesis, metric contract, host profile, numeric margins, aggregation procedure, and paired-uncertainty procedure **before** running the candidate. Freeze the procedure identifier, confidence level, resample count, and random seed in the registration. If repeated recordings, clustering, or any failure/exclusion tolerance is scientifically required, define and version that dependence/exclusion policy before measurement rather than aliasing or omitting observations after results are visible. +1. Review the rights basis, corpus composition, distinct audio-content identities, feature hypothesis, metric contract, host profile, numeric margins, aggregation procedure, paired-uncertainty procedure, and claim boundary **before** running the candidate. Freeze the procedure identifier, confidence level, resample count, random seed, and claim boundary in the registration. If repeated recordings, clustering, or any failure/exclusion tolerance is scientifically required, define and version that dependence/exclusion policy before measurement rather than aliasing or omitting observations after results are visible. 2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain its canonical SHA-256. 3. Run baseline and candidate on the same decoded track identities and host profile. The experiment runner must compute the approved aggregation and paired uncertainty procedure from the complete per-track evidence rather than accepting caller-authored aggregate numbers as scientific authority. Record per-track recognized MIR metrics, p50/p95 latency, peak RSS, failures, aggregate outputs, and paired uncertainty using exactly the preregistered procedure. -4. Put the registration digest and the identical uncertainty-plan fields in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. Under schema v1, any recorded failed track keeps the diagnostic receipt valid but makes acceptance fail closed. -5. Preserve the registration, result receipt, corpus/annotation hashes, exact source commit, lock hash, aggregation implementation identity, and uncertainty procedure together. A production feature switch requires this evidence plus normal code review and protected-head checks. +4. Put the registration digest, identical uncertainty-plan fields, and identical claim boundary in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. Under schema v1, any recorded failed track keeps the diagnostic receipt valid but makes acceptance fail closed; claim-boundary drift is rejected before an acceptance decision is produced. +5. Preserve the registration, result receipt, corpus/annotation hashes, exact source commit, lock hash, aggregation implementation identity, uncertainty procedure, and claim boundary together. A production feature switch requires this evidence plus normal code review and protected-head checks. No step authorizes committing licensed audio to Git. Rights-cleared means BandScope has the necessary evaluation right; redistribution is a separate permission. @@ -107,7 +108,7 @@ No step authorizes committing licensed audio to Git. Rights-cleared means BandSc - Evidence JSON is limited to 2 MiB, must be a regular file read through one open descriptor, must decode as UTF-8, and rejects duplicate keys plus non-standard `NaN`/`Infinity` constants. - `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute, relative, drive-relative, and `file:` forms are rejected; an explicit non-file URI scheme is required so transient workstation paths cannot become provenance authority or leak into review artifacts. - Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt; duplicate audio SHA-256 values under different track IDs are rejected so one recording cannot be silently counted multiple times. -- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, uncertainty-plan drift, registration drift, missing per-track evidence, unregistered registration/hypothesis/metric/corpus/runtime/result/track/aggregate/measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. +- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, uncertainty-plan drift, claim-boundary drift, registration drift, missing per-track evidence, unregistered registration/hypothesis/metric/corpus/runtime/result/track/aggregate/measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. - The validator does not claim that SHA-256 proves licensing, annotation validity, scientific adequacy, independence beyond exact-byte uniqueness, aggregate derivation, or that the registered statistical procedure is appropriate. Rights and scientific review remain separate gates. ## References From efbadeb4bb2a96f7daa0883d2e1ffd06c7b0868c Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 17:05:32 +0900 Subject: [PATCH 038/216] test(mir): preserve failed-track receipts without fake metrics --- ...structure_noninferiority_failure_policy.py | 24 +++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py index 0a2c1088e..3d3dbfa02 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py @@ -18,3 +18,27 @@ def test_failed_track_cannot_pass_without_preregistered_exclusion_policy() -> No "failed tracks are not permitted without a preregistered exclusion policy: " "licensed-track-002" ] + + +def test_failed_track_can_be_retained_without_fabricated_measurements() -> None: + """A real measurement failure must be diagnosable without invented MIR scores.""" + validator = _validator() + registration = _registration() + digest = validator.registration_digest(registration) + result = _result(registration, digest) + result["failed_tracks"] = ["licensed-track-002"] + tracks = result["tracks"] + assert isinstance(tracks, list) + failed_track = tracks[1] + assert isinstance(failed_track, dict) + del failed_track["baseline"] + del failed_track["candidate"] + + decision = validator.evaluate_result(registration, result) + + assert decision["passed"] is False + assert decision["failed_tracks"] == ["licensed-track-002"] + assert decision["failed_requirements"] == [ + "failed tracks are not permitted without a preregistered exclusion policy: " + "licensed-track-002" + ] From 2308e13a0ea5d166d9f3064ac5db9cbaeb821e8a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 17:07:22 +0900 Subject: [PATCH 039/216] test(mir): bind missing measurements to explicit track failure --- ...structure_noninferiority_failure_policy.py | 38 ++++++++++--------- 1 file changed, 20 insertions(+), 18 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py index 3d3dbfa02..3740479fb 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py @@ -1,44 +1,46 @@ """Regression tests for failed-track scientific acceptance policy.""" +import pytest + from test_structure_noninferiority_policy import _registration, _result, _validator +def _remove_failed_measurements(result: dict[str, object]) -> None: + """Model a real track whose MIR measurement could not be produced.""" + tracks = result["tracks"] + assert isinstance(tracks, list) + failed_track = tracks[1] + assert isinstance(failed_track, dict) + del failed_track["baseline"] + del failed_track["candidate"] + + def test_failed_track_cannot_pass_without_preregistered_exclusion_policy() -> None: - """A favorable aggregate cannot turn an unregistered track failure into PASS.""" + """A failed track stays diagnostic evidence and cannot become a complete-case PASS.""" validator = _validator() registration = _registration() digest = validator.registration_digest(registration) result = _result(registration, digest) result["failed_tracks"] = ["licensed-track-002"] + _remove_failed_measurements(result) decision = validator.evaluate_result(registration, result) assert decision["passed"] is False + assert decision["failed_tracks"] == ["licensed-track-002"] assert decision["failed_requirements"] == [ "failed tracks are not permitted without a preregistered exclusion policy: " "licensed-track-002" ] -def test_failed_track_can_be_retained_without_fabricated_measurements() -> None: - """A real measurement failure must be diagnosable without invented MIR scores.""" +def test_missing_measurements_require_explicit_failed_track() -> None: + """Omitted metrics must fail closed unless the same track is declared failed.""" validator = _validator() registration = _registration() digest = validator.registration_digest(registration) result = _result(registration, digest) - result["failed_tracks"] = ["licensed-track-002"] - tracks = result["tracks"] - assert isinstance(tracks, list) - failed_track = tracks[1] - assert isinstance(failed_track, dict) - del failed_track["baseline"] - del failed_track["candidate"] - - decision = validator.evaluate_result(registration, result) + _remove_failed_measurements(result) - assert decision["passed"] is False - assert decision["failed_tracks"] == ["licensed-track-002"] - assert decision["failed_requirements"] == [ - "failed tracks are not permitted without a preregistered exclusion policy: " - "licensed-track-002" - ] + with pytest.raises(ValueError, match="baseline"): + validator.evaluate_result(registration, result) From 7c8dc8a281624484c4a65f2ebb4d98772192b74f Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 17:08:30 +0900 Subject: [PATCH 040/216] fix(mir): admit failed tracks without fabricated measurements --- .../validate_structure_noninferiority.py | 51 ++++++++++++------- 1 file changed, 32 insertions(+), 19 deletions(-) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index 571d2fd48..2f96db026 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -105,6 +105,7 @@ "claim_boundary", } _TRACK_RECEIPT_FIELDS = {"track_id", "baseline", "candidate"} +_FAILED_TRACK_RECEIPT_FIELDS = {"track_id"} _AGGREGATE_FIELDS = {"baseline", "candidate"} @@ -616,11 +617,35 @@ def _validate_result_identity( return expected_track_ids +def _validate_failed_tracks( + result: Mapping[str, Any], + expected_track_ids: list[str], +) -> list[str]: + """Validate explicit measurement failures against the registered corpus.""" + failed_values = _sequence( + result.get("failed_tracks"), + "result.failed_tracks", + ) + failed_tracks = [ + _nonempty_text(value, f"result.failed_tracks[{index}]") + for index, value in enumerate(failed_values) + ] + if len(set(failed_tracks)) != len(failed_tracks): + raise ValueError("result.failed_tracks must not contain duplicates") + unknown_failed = sorted(set(failed_tracks) - set(expected_track_ids)) + if unknown_failed: + raise ValueError( + f"result.failed_tracks contains unknown track_id: {unknown_failed[0]}" + ) + return failed_tracks + + def _validate_track_measurements( result: Mapping[str, Any], expected_track_ids: list[str], + failed_track_ids: set[str], ) -> None: - """Require one complete baseline/candidate receipt for every corpus track.""" + """Require one explicit success-or-failure receipt for every corpus track.""" tracks = _sequence(result.get("tracks"), "result.tracks") if len(tracks) != len(expected_track_ids): raise ValueError( @@ -631,9 +656,12 @@ def _validate_track_measurements( for index, raw_track in enumerate(tracks): field = f"result.tracks[{index}]" track = _mapping(raw_track, field) - _require_exact_fields(track, _TRACK_RECEIPT_FIELDS, field) track_id = _nonempty_text(track.get("track_id"), f"{field}.track_id") actual_track_ids.append(track_id) + if track_id in failed_track_ids: + _require_exact_fields(track, _FAILED_TRACK_RECEIPT_FIELDS, field) + continue + _require_exact_fields(track, _TRACK_RECEIPT_FIELDS, field) _validate_measurement_side(track.get("baseline"), f"{field}.baseline") _validate_measurement_side(track.get("candidate"), f"{field}.candidate") if actual_track_ids != expected_track_ids: @@ -649,7 +677,8 @@ def evaluate_result( registration = _mapping(registration_value, "registration") result = _mapping(result_value, "result") track_ids = _validate_result_identity(registration, result) - _validate_track_measurements(result, track_ids) + failed_tracks = _validate_failed_tracks(result, track_ids) + _validate_track_measurements(result, track_ids, set(failed_tracks)) aggregate = _mapping(result.get("aggregate"), "result.aggregate") _require_exact_fields(aggregate, _AGGREGATE_FIELDS, "result.aggregate") @@ -705,22 +734,6 @@ def evaluate_result( "result.p95_latency_ratio_ci95 must contain the aggregate p95 ratio" ) - failed_values = _sequence( - result.get("failed_tracks"), - "result.failed_tracks", - ) - failed_tracks = [ - _nonempty_text(value, f"result.failed_tracks[{index}]") - for index, value in enumerate(failed_values) - ] - if len(set(failed_tracks)) != len(failed_tracks): - raise ValueError("result.failed_tracks must not contain duplicates") - unknown_failed = sorted(set(failed_tracks) - set(track_ids)) - if unknown_failed: - raise ValueError( - f"result.failed_tracks contains unknown track_id: {unknown_failed[0]}" - ) - metrics = _mapping(registration["metrics"], "metrics") failed_requirements: list[str] = [] if failed_tracks: From 4e58ae73ae6c4d898c7204a8bb404fb0fd574559 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 17:09:19 +0900 Subject: [PATCH 041/216] docs(mir): preserve real measurement failures without fake scores --- .../mir/structure-feature-noninferiority.md | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index 9c21ec536..00704e8f8 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -36,7 +36,7 @@ This guard is also consistent with recent MIR dataset-quality work. Choi et al. The current minimum of two tracks is only a technical guard against treating one timing sample as a corpus. It is **not** a scientific sample-size claim. Corpus breadth, genre/instrumentation coverage, annotation quality, independence/dependence structure, and a defensible power/uncertainty plan remain part of the experiment review before a production switch can be accepted. -Schema v1 does not contain a preregistered dropout, exclusion, or missing-track policy. Therefore every registered track must complete both baseline and candidate measurement for an acceptance PASS. A result may record failed track IDs for diagnosis, but any non-empty `failed_tracks` list makes the acceptance decision fail. A future tolerance for failed or excluded tracks requires a reviewed preregistration rule and schema revision rather than post-result omission. +Schema v1 does not contain a preregistered dropout, exclusion, or missing-track policy. Therefore every registered track must complete both baseline and candidate measurement for an acceptance PASS. A measurement failure is still evidence: its track ID remains in registered corpus order, appears in `failed_tracks`, and uses a closed failed-track receipt containing only `track_id` rather than invented MIR or latency values. Any non-empty `failed_tracks` list makes the acceptance decision fail. Conversely, omitting baseline/candidate measurements without declaring the same track failed is rejected. A future tolerance for failed or excluded tracks requires a reviewed preregistration rule and schema revision rather than post-result omission. ## Metrics @@ -70,19 +70,19 @@ For p95 latency, let `r = candidate_p95 / baseline_p95`. The result passes the s The result receipt must repeat the preregistered uncertainty plan and claim boundary exactly. The validator also requires the aggregate point delta/ratio to lie inside the supplied interval. Favorable point estimates therefore cannot hide an uncertainty interval that crosses a preregistered boundary, the uncertainty method cannot be swapped after the result is known, and a narrow experiment cannot be relabeled as evidence for a wider population or runtime after measurement. -Because schema v1 has no preregistered exclusion policy, a non-empty `failed_tracks` list is an additional failed requirement even when every reported quality and latency interval is favorable. Failed tracks remain visible in the result receipt for diagnosis; they cannot be silently converted into a complete-case PASS after results are known. +Because schema v1 has no preregistered exclusion policy, a non-empty `failed_tracks` list is an additional failed requirement even when every reported quality and latency interval is favorable. Failed tracks remain visible without fabricated measurements; they cannot be silently converted into a complete-case PASS after results are known. The repository does not currently contain approved numeric margins or an approved production corpus/uncertainty procedure. Unit-test values are synthetic policy fixtures only and must never be cited as production acceptance thresholds or scientific design decisions. ## Result receipt -A result receipt must contain the exact registration digest, exact uncertainty plan, exact corpus order, one complete baseline/candidate measurement pair per registered track, aggregate measurements, paired 95% intervals for every gated quality metric, a paired p95-latency-ratio interval, failed-track IDs, and the exact preregistered claim boundary. +A result receipt must contain the exact registration digest, exact uncertainty plan, exact corpus order, one ordered track receipt per registered track, aggregate measurements, paired 95% intervals for every gated quality metric, a paired p95-latency-ratio interval, failed-track IDs, and the exact preregistered claim boundary. A successful track receipt contains `track_id`, `baseline`, and `candidate`; a track listed in `failed_tracks` contains only `track_id`, preserving the failure without manufacturing scores that were never measured. Schema v1 treats the result envelope, each per-track receipt, and the aggregate envelope as closed-world objects. A top-level post-result field such as an unregistered p-value, a per-track `selected_for_aggregate` flag, or aggregate-side selection metadata is rejected rather than ignored. The same rule already applies inside each baseline/candidate measurement object. This prevents a producer from attaching an unreviewed post-hoc selection or inferential claim to an otherwise accepted receipt and having that field travel with the admitted evidence artifact. -Each per-track and aggregate baseline/candidate measurement is a closed-world receipt: it must contain exactly the registered quality fields plus the reporting-only precision/recall, boundary-deviation, latency, and peak-RSS fields. Missing fields and additional post-hoc measurement fields both fail admission. This prevents a result producer from attaching an unreviewed score after candidate results are visible and presenting it as part of the admitted scientific receipt. +Each successful per-track and aggregate baseline/candidate measurement is a closed-world receipt: it must contain exactly the registered quality fields plus the reporting-only precision/recall, boundary-deviation, latency, and peak-RSS fields. Missing fields and additional post-hoc measurement fields both fail admission. A declared failed track is the only per-track exception and is instead closed to exactly `track_id`; attaching partial or invented measurement fields to that failed receipt is rejected. This prevents a result producer from filling a real measurement failure with synthetic values merely to satisfy the evidence schema. -The receipt is rejected when a track is omitted/reordered, the uncertainty plan differs, the claim boundary differs, a required metric is absent, an unregistered envelope or measurement field is added, a boundary/repetition P/R/F triplet is internally inconsistent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A structurally valid receipt with one or more known failed tracks is retained for diagnosis but evaluates to `passed=false` under schema v1. A passing receipt therefore means the preregistered decision rule passed for the complete registered corpus, uncertainty procedure, runtime, and claim boundary. It does not generalize automatically to other genres, codecs, sample rates, machines, annotation regimes, or statistical procedures. +The receipt is rejected when a track is omitted/reordered, missing measurements are not paired with an explicit failed-track declaration, a failed receipt carries measurement fields, the uncertainty plan differs, the claim boundary differs, a required metric is absent, an unregistered envelope or measurement field is added, a boundary/repetition P/R/F triplet is internally inconsistent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A structurally valid receipt with one or more known failed tracks is retained for diagnosis but evaluates to `passed=false` under schema v1. A passing receipt therefore means the preregistered decision rule passed for the complete registered corpus, uncertainty procedure, runtime, and claim boundary. It does not generalize automatically to other genres, codecs, sample rates, machines, annotation regimes, or statistical procedures. The validator still does not recompute aggregate statistics or confidence intervals from per-track evidence. During review, a min/max range consistency heuristic was briefly encoded as a RED and then removed before production code changed because it would smuggle an unapproved assumption about aggregation into schema v1. Macro means, weighted means, micro-aggregated precision/recall/F and pooled latency summaries do not share one universally valid range relationship. The correct next step is to preregister the aggregation procedure and implement it in the experiment runner, not to infer a statistical method inside an admission validator after the fact. @@ -96,8 +96,8 @@ The JSON `source_uri` value remains provenance metadata only; it is never derefe 1. Review the rights basis, corpus composition, distinct audio-content identities, feature hypothesis, metric contract, host profile, numeric margins, aggregation procedure, paired-uncertainty procedure, and claim boundary **before** running the candidate. Freeze the procedure identifier, confidence level, resample count, random seed, and claim boundary in the registration. If repeated recordings, clustering, or any failure/exclusion tolerance is scientifically required, define and version that dependence/exclusion policy before measurement rather than aliasing or omitting observations after results are visible. 2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain its canonical SHA-256. -3. Run baseline and candidate on the same decoded track identities and host profile. The experiment runner must compute the approved aggregation and paired uncertainty procedure from the complete per-track evidence rather than accepting caller-authored aggregate numbers as scientific authority. Record per-track recognized MIR metrics, p50/p95 latency, peak RSS, failures, aggregate outputs, and paired uncertainty using exactly the preregistered procedure. -4. Put the registration digest, identical uncertainty-plan fields, and identical claim boundary in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. Under schema v1, any recorded failed track keeps the diagnostic receipt valid but makes acceptance fail closed; claim-boundary drift is rejected before an acceptance decision is produced. +3. Run baseline and candidate on the same decoded track identities and host profile. The experiment runner must compute the approved aggregation and paired uncertainty procedure from the complete measured-track evidence rather than accepting caller-authored aggregate numbers as scientific authority. Record recognized MIR metrics, p50/p95 latency, and peak RSS for successful tracks. If a registered track cannot produce a baseline/candidate measurement, preserve its ordered `track_id` receipt and add that exact ID to `failed_tracks`; do not invent metric values to make the receipt structurally complete. Record aggregate outputs and paired uncertainty using exactly the preregistered procedure, with any failure forcing schema-v1 acceptance to fail. +4. Put the registration digest, identical uncertainty-plan fields, and identical claim boundary in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. Under schema v1, any recorded failed track keeps the diagnostic receipt valid but makes acceptance fail closed; undeclared missing measurements and claim-boundary drift are rejected before an acceptance decision is produced. 5. Preserve the registration, result receipt, corpus/annotation hashes, exact source commit, lock hash, aggregation implementation identity, uncertainty procedure, and claim boundary together. A production feature switch requires this evidence plus normal code review and protected-head checks. No step authorizes committing licensed audio to Git. Rights-cleared means BandScope has the necessary evaluation right; redistribution is a separate permission. @@ -108,7 +108,7 @@ No step authorizes committing licensed audio to Git. Rights-cleared means BandSc - Evidence JSON is limited to 2 MiB, must be a regular file read through one open descriptor, must decode as UTF-8, and rejects duplicate keys plus non-standard `NaN`/`Infinity` constants. - `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute, relative, drive-relative, and `file:` forms are rejected; an explicit non-file URI scheme is required so transient workstation paths cannot become provenance authority or leak into review artifacts. - Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt; duplicate audio SHA-256 values under different track IDs are rejected so one recording cannot be silently counted multiple times. -- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, uncertainty-plan drift, claim-boundary drift, registration drift, missing per-track evidence, unregistered registration/hypothesis/metric/corpus/runtime/result/track/aggregate/measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. +- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, uncertainty-plan drift, claim-boundary drift, registration drift, undeclared missing measurements, measurement-bearing failed receipts, unregistered registration/hypothesis/metric/corpus/runtime/result/track/aggregate/measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. - The validator does not claim that SHA-256 proves licensing, annotation validity, scientific adequacy, independence beyond exact-byte uniqueness, aggregate derivation, or that the registered statistical procedure is appropriate. Rights and scientific review remain separate gates. ## References From c4ed428bf257515cd4cbb78479dc3f15b38efe8b Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 17:10:49 +0900 Subject: [PATCH 042/216] test(mir): keep failed-track regression formatter-safe --- .../tests/test_structure_noninferiority_failure_policy.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py index 3740479fb..e6f9699ae 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py @@ -16,7 +16,7 @@ def _remove_failed_measurements(result: dict[str, object]) -> None: def test_failed_track_cannot_pass_without_preregistered_exclusion_policy() -> None: - """A failed track stays diagnostic evidence and cannot become a complete-case PASS.""" + """A failed track remains diagnostic evidence and cannot PASS.""" validator = _validator() registration = _registration() digest = validator.registration_digest(registration) From 39eb3da33b9a28cee35034169dfbbfb0d1f1f35c Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 17:12:11 +0900 Subject: [PATCH 043/216] test(mir): reject measurements on failed-track receipts --- .../test_structure_noninferiority_failure_policy.py | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py index e6f9699ae..eaa24e73c 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py @@ -44,3 +44,15 @@ def test_missing_measurements_require_explicit_failed_track() -> None: with pytest.raises(ValueError, match="baseline"): validator.evaluate_result(registration, result) + + +def test_failed_track_cannot_carry_fabricated_measurements() -> None: + """A failed receipt must not retain values that imply successful measurement.""" + validator = _validator() + registration = _registration() + digest = validator.registration_digest(registration) + result = _result(registration, digest) + result["failed_tracks"] = ["licensed-track-002"] + + with pytest.raises(ValueError, match="unregistered field: baseline"): + validator.evaluate_result(registration, result) From 5c6de3a19c2f4d00e74cc301559fdd593baaac3d Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 18:02:04 +0900 Subject: [PATCH 044/216] test(mir): reject summaries after track measurement failure --- ...structure_noninferiority_failure_policy.py | 24 +++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py index eaa24e73c..4bace0c68 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py @@ -15,6 +15,13 @@ def _remove_failed_measurements(result: dict[str, object]) -> None: del failed_track["candidate"] +def _remove_post_failure_summaries(result: dict[str, object]) -> None: + """Avoid inventing complete-case aggregates after a registered track fails.""" + result["aggregate"] = None + result["paired_delta_ci95"] = None + result["p95_latency_ratio_ci95"] = None + + def test_failed_track_cannot_pass_without_preregistered_exclusion_policy() -> None: """A failed track remains diagnostic evidence and cannot PASS.""" validator = _validator() @@ -23,6 +30,7 @@ def test_failed_track_cannot_pass_without_preregistered_exclusion_policy() -> No result = _result(registration, digest) result["failed_tracks"] = ["licensed-track-002"] _remove_failed_measurements(result) + _remove_post_failure_summaries(result) decision = validator.evaluate_result(registration, result) @@ -34,6 +42,22 @@ def test_failed_track_cannot_pass_without_preregistered_exclusion_policy() -> No ] +def test_failed_track_cannot_publish_complete_case_summary() -> None: + """A failed run cannot attach aggregates computed from only surviving tracks.""" + validator = _validator() + registration = _registration() + digest = validator.registration_digest(registration) + result = _result(registration, digest) + result["failed_tracks"] = ["licensed-track-002"] + _remove_failed_measurements(result) + + with pytest.raises( + ValueError, + match="result.aggregate must be null when result.failed_tracks is non-empty", + ): + validator.evaluate_result(registration, result) + + def test_missing_measurements_require_explicit_failed_track() -> None: """Omitted metrics must fail closed unless the same track is declared failed.""" validator = _validator() From 643b09b2a3701b43d633a82807dc658ca55edcae Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 18:03:25 +0900 Subject: [PATCH 045/216] fix(mir): suppress invalid summaries after track failure --- .../validate_structure_noninferiority.py | 25 +++++++++++++++---- 1 file changed, 20 insertions(+), 5 deletions(-) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index 2f96db026..f13bc3311 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -680,6 +680,26 @@ def evaluate_result( failed_tracks = _validate_failed_tracks(result, track_ids) _validate_track_measurements(result, track_ids, set(failed_tracks)) + if failed_tracks: + for field in ( + "aggregate", + "paired_delta_ci95", + "p95_latency_ratio_ci95", + ): + if result.get(field) is not None: + raise ValueError( + f"result.{field} must be null when result.failed_tracks is non-empty" + ) + return { + "passed": False, + "failed_requirements": [ + "failed tracks are not permitted without a preregistered exclusion " + "policy: " + ", ".join(failed_tracks) + ], + "registration_sha256": registration_digest(registration), + "failed_tracks": failed_tracks, + } + aggregate = _mapping(result.get("aggregate"), "result.aggregate") _require_exact_fields(aggregate, _AGGREGATE_FIELDS, "result.aggregate") baseline = _validate_measurement_side( @@ -736,11 +756,6 @@ def evaluate_result( metrics = _mapping(registration["metrics"], "metrics") failed_requirements: list[str] = [] - if failed_tracks: - failed_requirements.append( - "failed tracks are not permitted without a preregistered exclusion policy: " - + ", ".join(failed_tracks) - ) for metric_name in _QUALITY_METRICS: config = _mapping(metrics[metric_name], f"metrics.{metric_name}") margin = _finite_number( From cc5261f973e1283298a9ec6d1ff79c6bd6036905 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 18:04:10 +0900 Subject: [PATCH 046/216] docs(mir): define failed-run null summary contract --- .../mir/structure-feature-noninferiority.md | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index 00704e8f8..2933ac115 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -9,7 +9,7 @@ Production baseline: `chroma_cqt` in `sections/segmenter.py` The earlier `chroma_cqt` → `chroma_stft` optimization is not a production change until a preregistered, rights-cleared real-audio experiment shows that the candidate is noninferior on rehearsal-relevant structure quality and materially faster on the same corpus and runtime identity. -The repository now owns a result-admission boundary in `scripts/research/validate_structure_noninferiority.py`. The validator does **not** calculate MIR metrics and does not choose scientific margins or an uncertainty procedure. It freezes the reviewed experiment contract, including its claim boundary, binds a result receipt to that contract by canonical SHA-256, requires per-track and aggregate evidence, and applies the registered confidence-interval decision rules. Metric computation and uncertainty estimation remain separate scientific measurement steps. +The repository now owns a result-admission boundary in `scripts/research/validate_structure_noninferiority.py`. The validator does **not** calculate MIR metrics and does not choose scientific margins or an uncertainty procedure. It freezes the reviewed experiment contract, including its claim boundary, binds a result receipt to that contract by canonical SHA-256, requires one explicit success-or-failure receipt per registered track, and applies the registered aggregate confidence-interval decision rules only when the complete corpus produced measurements. Metric computation and uncertainty estimation remain separate scientific measurement steps. This keeps a performance result from silently changing the corpus, thresholds, feature identities, uncertainty procedure, runtime, or permitted interpretation after the measurements are visible. @@ -36,7 +36,7 @@ This guard is also consistent with recent MIR dataset-quality work. Choi et al. The current minimum of two tracks is only a technical guard against treating one timing sample as a corpus. It is **not** a scientific sample-size claim. Corpus breadth, genre/instrumentation coverage, annotation quality, independence/dependence structure, and a defensible power/uncertainty plan remain part of the experiment review before a production switch can be accepted. -Schema v1 does not contain a preregistered dropout, exclusion, or missing-track policy. Therefore every registered track must complete both baseline and candidate measurement for an acceptance PASS. A measurement failure is still evidence: its track ID remains in registered corpus order, appears in `failed_tracks`, and uses a closed failed-track receipt containing only `track_id` rather than invented MIR or latency values. Any non-empty `failed_tracks` list makes the acceptance decision fail. Conversely, omitting baseline/candidate measurements without declaring the same track failed is rejected. A future tolerance for failed or excluded tracks requires a reviewed preregistration rule and schema revision rather than post-result omission. +Schema v1 does not contain a preregistered dropout, exclusion, or missing-track policy. Therefore every registered track must complete both baseline and candidate measurement for an acceptance PASS. A measurement failure is still evidence: its track ID remains in registered corpus order, appears in `failed_tracks`, and uses a closed failed-track receipt containing only `track_id` rather than invented MIR or latency values. Any non-empty `failed_tracks` list makes the acceptance decision fail. Because an aggregate or confidence interval computed after such a failure would necessarily summarize an unregistered complete-case subset, schema v1 also requires `aggregate`, `paired_delta_ci95`, and `p95_latency_ratio_ci95` to be JSON `null` whenever `failed_tracks` is non-empty. Conversely, omitting baseline/candidate measurements without declaring the same track failed is rejected. A future tolerance for failed or excluded tracks requires a reviewed preregistration rule and schema revision rather than post-result omission. ## Metrics @@ -68,21 +68,21 @@ For p95 latency, let `r = candidate_p95 / baseline_p95`. The result passes the s `upper_CI95(r) <= maximum_candidate_ratio` -The result receipt must repeat the preregistered uncertainty plan and claim boundary exactly. The validator also requires the aggregate point delta/ratio to lie inside the supplied interval. Favorable point estimates therefore cannot hide an uncertainty interval that crosses a preregistered boundary, the uncertainty method cannot be swapped after the result is known, and a narrow experiment cannot be relabeled as evidence for a wider population or runtime after measurement. +The result receipt must repeat the preregistered uncertainty plan and claim boundary exactly. On a complete run, the validator also requires the aggregate point delta/ratio to lie inside the supplied interval. Favorable point estimates therefore cannot hide an uncertainty interval that crosses a preregistered boundary, the uncertainty method cannot be swapped after the result is known, and a narrow experiment cannot be relabeled as evidence for a wider population or runtime after measurement. -Because schema v1 has no preregistered exclusion policy, a non-empty `failed_tracks` list is an additional failed requirement even when every reported quality and latency interval is favorable. Failed tracks remain visible without fabricated measurements; they cannot be silently converted into a complete-case PASS after results are known. +Because schema v1 has no preregistered exclusion policy, a non-empty `failed_tracks` list fails before aggregate decision rules are evaluated. The failed run preserves track-level diagnostic identity but must carry `null` aggregate and CI summaries, so a complete-case subset cannot masquerade as the preregistered whole-corpus analysis. The repository does not currently contain approved numeric margins or an approved production corpus/uncertainty procedure. Unit-test values are synthetic policy fixtures only and must never be cited as production acceptance thresholds or scientific design decisions. ## Result receipt -A result receipt must contain the exact registration digest, exact uncertainty plan, exact corpus order, one ordered track receipt per registered track, aggregate measurements, paired 95% intervals for every gated quality metric, a paired p95-latency-ratio interval, failed-track IDs, and the exact preregistered claim boundary. A successful track receipt contains `track_id`, `baseline`, and `candidate`; a track listed in `failed_tracks` contains only `track_id`, preserving the failure without manufacturing scores that were never measured. +A result receipt must contain the exact registration digest, exact uncertainty plan, exact corpus order, one ordered track receipt per registered track, failed-track IDs, and the exact preregistered claim boundary. A successful complete run additionally carries aggregate measurements, paired 95% intervals for every gated quality metric, and a paired p95-latency-ratio interval. A successful track receipt contains `track_id`, `baseline`, and `candidate`; a track listed in `failed_tracks` contains only `track_id`, preserving the failure without manufacturing scores that were never measured. When any track fails, the three aggregate/CI fields remain present in the closed top-level schema but their values must be JSON `null`. Schema v1 treats the result envelope, each per-track receipt, and the aggregate envelope as closed-world objects. A top-level post-result field such as an unregistered p-value, a per-track `selected_for_aggregate` flag, or aggregate-side selection metadata is rejected rather than ignored. The same rule already applies inside each baseline/candidate measurement object. This prevents a producer from attaching an unreviewed post-hoc selection or inferential claim to an otherwise accepted receipt and having that field travel with the admitted evidence artifact. Each successful per-track and aggregate baseline/candidate measurement is a closed-world receipt: it must contain exactly the registered quality fields plus the reporting-only precision/recall, boundary-deviation, latency, and peak-RSS fields. Missing fields and additional post-hoc measurement fields both fail admission. A declared failed track is the only per-track exception and is instead closed to exactly `track_id`; attaching partial or invented measurement fields to that failed receipt is rejected. This prevents a result producer from filling a real measurement failure with synthetic values merely to satisfy the evidence schema. -The receipt is rejected when a track is omitted/reordered, missing measurements are not paired with an explicit failed-track declaration, a failed receipt carries measurement fields, the uncertainty plan differs, the claim boundary differs, a required metric is absent, an unregistered envelope or measurement field is added, a boundary/repetition P/R/F triplet is internally inconsistent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A structurally valid receipt with one or more known failed tracks is retained for diagnosis but evaluates to `passed=false` under schema v1. A passing receipt therefore means the preregistered decision rule passed for the complete registered corpus, uncertainty procedure, runtime, and claim boundary. It does not generalize automatically to other genres, codecs, sample rates, machines, annotation regimes, or statistical procedures. +The receipt is rejected when a track is omitted/reordered, missing measurements are not paired with an explicit failed-track declaration, a failed receipt carries measurement fields, a failed run carries non-null aggregate/CI summaries, the uncertainty plan differs, the claim boundary differs, a required metric is absent, an unregistered envelope or measurement field is added, a boundary/repetition P/R/F triplet is internally inconsistent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A structurally valid receipt with one or more known failed tracks is retained for diagnosis with null aggregate summaries and evaluates to `passed=false` under schema v1. A passing receipt therefore means the preregistered decision rule passed for the complete registered corpus, uncertainty procedure, runtime, and claim boundary. It does not generalize automatically to other genres, codecs, sample rates, machines, annotation regimes, or statistical procedures. The validator still does not recompute aggregate statistics or confidence intervals from per-track evidence. During review, a min/max range consistency heuristic was briefly encoded as a RED and then removed before production code changed because it would smuggle an unapproved assumption about aggregation into schema v1. Macro means, weighted means, micro-aggregated precision/recall/F and pooled latency summaries do not share one universally valid range relationship. The correct next step is to preregister the aggregation procedure and implement it in the experiment runner, not to infer a statistical method inside an admission validator after the fact. @@ -96,8 +96,8 @@ The JSON `source_uri` value remains provenance metadata only; it is never derefe 1. Review the rights basis, corpus composition, distinct audio-content identities, feature hypothesis, metric contract, host profile, numeric margins, aggregation procedure, paired-uncertainty procedure, and claim boundary **before** running the candidate. Freeze the procedure identifier, confidence level, resample count, random seed, and claim boundary in the registration. If repeated recordings, clustering, or any failure/exclusion tolerance is scientifically required, define and version that dependence/exclusion policy before measurement rather than aliasing or omitting observations after results are visible. 2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain its canonical SHA-256. -3. Run baseline and candidate on the same decoded track identities and host profile. The experiment runner must compute the approved aggregation and paired uncertainty procedure from the complete measured-track evidence rather than accepting caller-authored aggregate numbers as scientific authority. Record recognized MIR metrics, p50/p95 latency, and peak RSS for successful tracks. If a registered track cannot produce a baseline/candidate measurement, preserve its ordered `track_id` receipt and add that exact ID to `failed_tracks`; do not invent metric values to make the receipt structurally complete. Record aggregate outputs and paired uncertainty using exactly the preregistered procedure, with any failure forcing schema-v1 acceptance to fail. -4. Put the registration digest, identical uncertainty-plan fields, and identical claim boundary in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. Under schema v1, any recorded failed track keeps the diagnostic receipt valid but makes acceptance fail closed; undeclared missing measurements and claim-boundary drift are rejected before an acceptance decision is produced. +3. Run baseline and candidate on the same decoded track identities and host profile. The experiment runner must compute the approved aggregation and paired uncertainty procedure from the complete measured-track evidence rather than accepting caller-authored aggregate numbers as scientific authority. Record recognized MIR metrics, p50/p95 latency, and peak RSS for successful tracks. If a registered track cannot produce a baseline/candidate measurement, preserve its ordered `track_id` receipt, add that exact ID to `failed_tracks`, and set the aggregate plus both CI summary fields to JSON `null`; do not invent metric values or compute a complete-case aggregate from the surviving tracks. Only a complete run records aggregate outputs and paired uncertainty using the preregistered procedure. +4. Put the registration digest, identical uncertainty-plan fields, and identical claim boundary in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. Under schema v1, any recorded failed track keeps the diagnostic receipt valid but makes acceptance fail closed before aggregate decision rules; undeclared missing measurements, non-null post-failure summaries, and claim-boundary drift are rejected. 5. Preserve the registration, result receipt, corpus/annotation hashes, exact source commit, lock hash, aggregation implementation identity, uncertainty procedure, and claim boundary together. A production feature switch requires this evidence plus normal code review and protected-head checks. No step authorizes committing licensed audio to Git. Rights-cleared means BandScope has the necessary evaluation right; redistribution is a separate permission. @@ -108,7 +108,7 @@ No step authorizes committing licensed audio to Git. Rights-cleared means BandSc - Evidence JSON is limited to 2 MiB, must be a regular file read through one open descriptor, must decode as UTF-8, and rejects duplicate keys plus non-standard `NaN`/`Infinity` constants. - `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute, relative, drive-relative, and `file:` forms are rejected; an explicit non-file URI scheme is required so transient workstation paths cannot become provenance authority or leak into review artifacts. - Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt; duplicate audio SHA-256 values under different track IDs are rejected so one recording cannot be silently counted multiple times. -- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, uncertainty-plan drift, claim-boundary drift, registration drift, undeclared missing measurements, measurement-bearing failed receipts, unregistered registration/hypothesis/metric/corpus/runtime/result/track/aggregate/measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. +- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, uncertainty-plan drift, claim-boundary drift, registration drift, undeclared missing measurements, measurement-bearing failed receipts, non-null aggregate/CI summaries after any failed track, unregistered registration/hypothesis/metric/corpus/runtime/result/track/aggregate/measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. - The validator does not claim that SHA-256 proves licensing, annotation validity, scientific adequacy, independence beyond exact-byte uniqueness, aggregate derivation, or that the registered statistical procedure is appropriate. Rights and scientific review remain separate gates. ## References From 240feb8e4711cd5211aeb8c3b23c45ecd36adc99 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 19:03:24 +0900 Subject: [PATCH 047/216] test(mir): expose normalized experiment identity mismatch --- ...ucture_noninferiority_envelope_schema_policy.py | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py index 75c1c2544..5f791fe17 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py @@ -112,3 +112,17 @@ def test_result_claim_boundary_cannot_expand_beyond_preregistration() -> None: match="result.claim_boundary must exactly match the preregistered claim boundary", ): validator.evaluate_result(registration, result) + + +def test_result_experiment_id_matches_normalized_preregistration_identity() -> None: + """Whitespace in a valid registration ID must not make its result impossible.""" + validator = _validator() + registration = _registration() + registration["experiment_id"] = " structure-chroma-stft-vs-cqt-v1 " + digest = validator.registration_digest(registration) + result = _result(registration, digest) + + decision = validator.evaluate_result(registration, result) + + assert decision["passed"] is True + assert decision["registration_sha256"] == digest From e0b897be2fd6d68147344421eeea5f799dd742e1 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 19:05:01 +0900 Subject: [PATCH 048/216] fix(mir): compare normalized preregistered experiment identity --- scripts/research/validate_structure_noninferiority.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index f13bc3311..2fda9f4ff 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -558,11 +558,15 @@ def _validate_result_identity( result.get("schema_version"), "result.schema_version", ) + registered_experiment_id = _nonempty_text( + registration.get("experiment_id"), + "experiment_id", + ) experiment_id = _nonempty_text( result.get("experiment_id"), "result.experiment_id", ) - if experiment_id != registration["experiment_id"]: + if experiment_id != registered_experiment_id: raise ValueError("result.experiment_id does not match registration") expected_digest = registration_digest(registration) From 76053a920b4dac4036e90f6502230a5dd0b5436e Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 19:06:10 +0900 Subject: [PATCH 049/216] docs(mir): record normalized experiment identity binding --- docs/traceability/mir/structure-feature-noninferiority.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index 2933ac115..6653e410c 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -26,6 +26,8 @@ A registration is valid only when it records all of the following before the res Schema v1 is closed-world not only at the registration top level but also for the hypothesis object, every registered metric configuration, every corpus-track object, and the runtime object. Extra fields are not treated as harmless annotations: an unregistered pilot-selection flag, metric weight, corpus-selection marker, or cache/runtime hint changes what reviewers may infer was preregistered, so it fails admission instead of silently entering the evidence artifact. The claim boundary is itself a required top-level registration field and therefore participates in the canonical registration SHA-256. +`experiment_id` is a semantic label, while the registration SHA-256 is the exact evidence identity. The validator therefore compares the registration and result experiment IDs after the same required-text normalization, but hashes the original validated registration representation unchanged. This prevents an otherwise valid registration with surrounding whitespace from becoming impossible to reference while keeping byte-distinct registrations cryptographically distinct. + The validator requires a 95% confidence level because the current result schema is explicitly `ci95`; it does not prescribe which scientifically defensible paired procedure must produce that interval. The procedure identifier, resample count, and seed are part of the preregistration digest so they cannot be changed after results are seen. `paired-track-bootstrap-v1` and the numeric values used in unit tests are policy fixtures only, not an approved BandScope production analysis plan. The validator requires `source_uri` to be an explicit non-`file:` URI. Absolute, relative, drive-relative, and `file:` filesystem forms are rejected as provenance authorities. A benchmark may remain private when licensing requires that, but the receipt must identify the licensed material without leaking the workstation path that happened to hold it. From 83e59f220eba26272840f3f901cdaaf50d86d3c2 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 20:04:34 +0900 Subject: [PATCH 050/216] research(mir): admit local real-audio corpus bytes --- scripts/research/verify_structure_corpus.py | 335 ++++++++++++++++++++ 1 file changed, 335 insertions(+) create mode 100644 scripts/research/verify_structure_corpus.py diff --git a/scripts/research/verify_structure_corpus.py b/scripts/research/verify_structure_corpus.py new file mode 100644 index 000000000..f81bb1ee7 --- /dev/null +++ b/scripts/research/verify_structure_corpus.py @@ -0,0 +1,335 @@ +#!/usr/bin/env python3 +"""Admit local real-audio corpus material for a frozen BandScope MIR experiment. + +The preregistration stores provenance and content digests, never workstation +paths. This tool resolves a local-only manifest, hashes audio and annotation +bytes from opened regular-file descriptors, decodes each audio item through the +registered librosa normalization contract, and emits a path-free receipt with a +SHA-256 identity of the exact mono float32 PCM presented to later analysis. + +It does not calculate MIR metrics, choose thresholds, or make a noninferiority +decision. Synthetic audio is suitable for unit tests only; production receipts +require the rights-cleared real-audio corpus named by the preregistration. +""" + +from __future__ import annotations + +import argparse +import hashlib +import importlib.util +import json +import os +import platform +import stat +import subprocess +from collections.abc import Callable, Mapping, Sequence +from pathlib import Path +from types import ModuleType +from typing import Any + +MANIFEST_SCHEMA_VERSION = 1 +MAX_MANIFEST_BYTES = 2 * 1024 * 1024 +_MANIFEST_FIELDS = {"schema_version", "registration_sha256", "tracks"} +_TRACK_FIELDS = {"track_id", "audio_path", "annotation_path"} +_RUNTIME_IDENTITY_FIELDS = ( + "source_commit", + "uv_lock_sha256", + "python_version", + "librosa_version", + "numpy_version", +) + + +def _load_validator() -> ModuleType: + path = Path(__file__).with_name("validate_structure_noninferiority.py") + spec = importlib.util.spec_from_file_location("structure_noninferiority_validator", path) + if spec is None or spec.loader is None: + raise RuntimeError("could not load structure noninferiority validator") + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def _mapping(value: object, field: str) -> Mapping[str, Any]: + if not isinstance(value, Mapping): + raise ValueError(f"{field} must be an object") + return value + + +def _sequence(value: object, field: str) -> Sequence[Any]: + if isinstance(value, (str, bytes)) or not isinstance(value, Sequence): + raise ValueError(f"{field} must be an array") + return value + + +def _exact_fields(value: Mapping[str, Any], expected: set[str], field: str) -> None: + missing = sorted(expected - set(value)) + extra = sorted(set(value) - expected) + if missing: + raise ValueError(f"{field} missing required field: {missing[0]}") + if extra: + raise ValueError(f"{field} contains unregistered field: {extra[0]}") + + +def _text(value: object, field: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ValueError(f"{field} must be non-empty text") + return value.strip() + + +def _reject_constant(value: str) -> None: + raise ValueError(f"non-standard JSON number is not allowed: {value}") + + +def _unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + result: dict[str, Any] = {} + for key, value in pairs: + if key in result: + raise ValueError(f"duplicate JSON key: {key}") + result[key] = value + return result + + +def _load_json(path: Path) -> Mapping[str, Any]: + try: + size = path.stat().st_size + except OSError as exc: + raise ValueError("manifest JSON could not be stat'ed") from exc + if size > MAX_MANIFEST_BYTES: + raise ValueError(f"manifest JSON exceeds {MAX_MANIFEST_BYTES} bytes") + try: + text = path.read_text(encoding="utf-8") + except (OSError, UnicodeError) as exc: + raise ValueError("manifest JSON must be readable UTF-8") from exc + value = json.loads( + text, + object_pairs_hook=_unique_object, + parse_constant=_reject_constant, + ) + return _mapping(value, str(path)) + + +def _sha256_file_descriptor(fd: int) -> str: + digest = hashlib.sha256() + os.lseek(fd, 0, os.SEEK_SET) + while True: + chunk = os.read(fd, 1024 * 1024) + if not chunk: + break + digest.update(chunk) + os.lseek(fd, 0, os.SEEK_SET) + return digest.hexdigest() + + +def _open_regular_file(path: Path, field: str) -> int: + flags = os.O_RDONLY + if hasattr(os, "O_BINARY"): + flags |= os.O_BINARY + if hasattr(os, "O_NOFOLLOW"): + flags |= os.O_NOFOLLOW + try: + fd = os.open(path, flags) + except OSError as exc: + raise ValueError(f"{field} could not be opened as a regular file") from exc + try: + metadata = os.fstat(fd) + if not stat.S_ISREG(metadata.st_mode): + raise ValueError(f"{field} must reference a regular file") + if not hasattr(os, "O_NOFOLLOW") and path.is_symlink(): + raise ValueError(f"{field} must not be a symbolic link") + return fd + except Exception: + os.close(fd) + raise + + +def _decode_pcm_identity(fd: int, target_sample_rate_hz: int) -> tuple[str, int]: + """Decode from the admitted descriptor and hash canonical mono float32 PCM.""" + import librosa + import numpy as np + + os.lseek(fd, 0, os.SEEK_SET) + duplicate_fd = os.dup(fd) + try: + with os.fdopen(duplicate_fd, "rb", closefd=True) as fileobj: + samples, actual_sample_rate_hz = librosa.load( + fileobj, + sr=target_sample_rate_hz, + mono=True, + ) + except Exception: + try: + os.close(duplicate_fd) + except OSError: + pass + raise + if int(actual_sample_rate_hz) != target_sample_rate_hz: + raise ValueError("decoder did not honor the registered sample rate") + canonical = np.asarray(samples, dtype=" dict[str, object]: + import librosa + import numpy as np + + completed = subprocess.run( + ["git", "rev-parse", "HEAD"], + cwd=repo_root, + check=False, + capture_output=True, + text=True, + timeout=10, + ) + if completed.returncode != 0: + raise RuntimeError("git rev-parse HEAD failed") + lock_path = repo_root / "uv.lock" + with lock_path.open("rb") as lock_file: + lock_digest = hashlib.file_digest(lock_file, "sha256").hexdigest() + return { + "source_commit": completed.stdout.strip(), + "uv_lock_sha256": lock_digest, + "python_version": platform.python_version(), + "librosa_version": str(librosa.__version__), + "numpy_version": str(np.__version__), + } + + +def _validate_runtime_identity( + registration: Mapping[str, Any], runtime_identity: Mapping[str, object] +) -> dict[str, str]: + runtime = _mapping(registration.get("runtime"), "registration.runtime") + normalized: dict[str, str] = {} + for field in _RUNTIME_IDENTITY_FIELDS: + actual = _text(runtime_identity.get(field), f"runtime_identity.{field}") + expected = _text(runtime.get(field), f"registration.runtime.{field}") + if actual != expected: + raise ValueError( + f"runtime identity mismatch for {field}: expected {expected}, got {actual}" + ) + normalized[field] = actual + return normalized + + +def verify_corpus( + registration: Mapping[str, Any], + manifest: Mapping[str, Any], + *, + runtime_identity: Mapping[str, object], + decoder: Callable[[int, int], tuple[str, int]] = _decode_pcm_identity, +) -> dict[str, object]: + """Verify local files and return a path-free decoded-corpus receipt.""" + validator = _load_validator() + validator.validate_registration(registration) + registration_sha256 = validator.registration_digest(registration) + + _exact_fields(manifest, _MANIFEST_FIELDS, "manifest") + if manifest.get("schema_version") != MANIFEST_SCHEMA_VERSION: + raise ValueError(f"manifest.schema_version must equal {MANIFEST_SCHEMA_VERSION}") + manifest_digest = _text(manifest.get("registration_sha256"), "manifest.registration_sha256") + if manifest_digest != registration_sha256: + raise ValueError("manifest.registration_sha256 does not match registration") + normalized_runtime = _validate_runtime_identity(registration, runtime_identity) + + corpus = _sequence(registration.get("corpus"), "registration.corpus") + corpus_by_id: dict[str, Mapping[str, Any]] = {} + corpus_ids: list[str] = [] + for index, raw_track in enumerate(corpus): + track = _mapping(raw_track, f"registration.corpus[{index}]") + track_id = _text(track.get("track_id"), f"registration.corpus[{index}].track_id") + corpus_by_id[track_id] = track + corpus_ids.append(track_id) + + manifest_tracks = _sequence(manifest.get("tracks"), "manifest.tracks") + if len(manifest_tracks) != len(corpus_ids): + raise ValueError("manifest.tracks must contain exactly the registered corpus") + + seen: set[str] = set() + receipt_tracks: list[dict[str, object]] = [] + target_sample_rate_hz = int(_mapping(registration["runtime"], "runtime")["sample_rate_hz"]) + for index, raw_track in enumerate(manifest_tracks): + field = f"manifest.tracks[{index}]" + track = _mapping(raw_track, field) + _exact_fields(track, _TRACK_FIELDS, field) + track_id = _text(track.get("track_id"), f"{field}.track_id") + if track_id in seen: + raise ValueError(f"duplicate manifest track_id: {track_id}") + seen.add(track_id) + if index >= len(corpus_ids) or track_id != corpus_ids[index]: + raise ValueError("manifest track order must exactly match registration corpus order") + registered = corpus_by_id[track_id] + + audio_path = Path(_text(track.get("audio_path"), f"{field}.audio_path")) + annotation_path = Path(_text(track.get("annotation_path"), f"{field}.annotation_path")) + audio_fd = _open_regular_file(audio_path, f"{field}.audio_path") + try: + audio_sha256 = _sha256_file_descriptor(audio_fd) + expected_audio = _text( + registered.get("audio_sha256"), "registered.audio_sha256" + ).lower() + if audio_sha256 != expected_audio: + raise ValueError(f"{field}.audio_path SHA-256 does not match registration") + decoded_pcm_sha256, decoded_frames = decoder(audio_fd, target_sample_rate_hz) + finally: + os.close(audio_fd) + + annotation_fd = _open_regular_file(annotation_path, f"{field}.annotation_path") + try: + annotation_sha256 = _sha256_file_descriptor(annotation_fd) + finally: + os.close(annotation_fd) + expected_annotation = _text( + registered.get("annotation_sha256"), "registered.annotation_sha256" + ).lower() + if annotation_sha256 != expected_annotation: + raise ValueError(f"{field}.annotation_path SHA-256 does not match registration") + + receipt_tracks.append( + { + "track_id": track_id, + "audio_sha256": audio_sha256, + "annotation_sha256": annotation_sha256, + "decoded_pcm_sha256": decoded_pcm_sha256, + "decoded_frames": decoded_frames, + "sample_rate_hz": target_sample_rate_hz, + "channels": 1, + } + ) + + return { + "schema_version": MANIFEST_SCHEMA_VERSION, + "registration_sha256": registration_sha256, + "runtime_identity": normalized_runtime, + "tracks": receipt_tracks, + } + + +def main() -> int: + parser = argparse.ArgumentParser( + description="Admit local real-audio files for a frozen structure experiment" + ) + parser.add_argument("registration", type=Path) + parser.add_argument("manifest", type=Path) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + + validator = _load_validator() + registration = _mapping(validator._load_json(args.registration), "registration") + manifest = _load_json(args.manifest) + repo_root = Path(__file__).resolve().parents[2] + receipt = verify_corpus( + registration, + manifest, + runtime_identity=_current_runtime_identity(repo_root), + ) + args.output.write_text( + json.dumps(receipt, sort_keys=True, separators=(",", ":")) + "\n", + encoding="utf-8", + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) From 6a0c80ff0877fa14d4d2f23bbf15451f0608ac32 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 20:04:54 +0900 Subject: [PATCH 051/216] test(mir): guard real-audio corpus admission --- .../tests/test_structure_corpus_admission.py | 241 ++++++++++++++++++ 1 file changed, 241 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_corpus_admission.py diff --git a/services/analysis-engine/tests/test_structure_corpus_admission.py b/services/analysis-engine/tests/test_structure_corpus_admission.py new file mode 100644 index 000000000..41cb1d1ae --- /dev/null +++ b/services/analysis-engine/tests/test_structure_corpus_admission.py @@ -0,0 +1,241 @@ +"""Tests for local real-audio corpus admission into the MIR experiment lane.""" + +from __future__ import annotations + +import hashlib +import os +from pathlib import Path +from types import ModuleType + +import pytest +from conftest import load_module, make_symlink_or_skip + + +def _admission() -> ModuleType: + return load_module( + "scripts/research/verify_structure_corpus.py", + "verify_structure_corpus", + ) + + +def _registration(audio_hashes: list[str], annotation_hashes: list[str]) -> dict[str, object]: + return { + "schema_version": 1, + "experiment_id": "structure-chroma-stft-vs-cqt-v1", + "hypothesis": { + "baseline_feature": "chroma_cqt", + "candidate_feature": "chroma_stft", + }, + "metrics": { + "boundary_f_0_5": { + "implementation": "mir_eval.segment.detection", + "window_seconds": 0.5, + "noninferiority_margin": 0.02, + }, + "boundary_f_3_0": { + "implementation": "mir_eval.segment.detection", + "window_seconds": 3.0, + "noninferiority_margin": 0.02, + }, + "functional_label_accuracy": { + "implementation": "mirex2025.frame_level_accuracy", + "noninferiority_margin": 0.02, + }, + "repetition_pairwise_f": { + "implementation": "mir_eval.segment.pairwise", + "frame_size_seconds": 0.1, + "noninferiority_margin": 0.02, + }, + "p95_latency_ratio": {"maximum_candidate_ratio": 0.8}, + }, + "uncertainty": { + "procedure_id": "paired-track-bootstrap-v1", + "confidence_level": 0.95, + "resamples": 1000, + "random_seed": 7, + }, + "corpus": [ + { + "track_id": "track-001", + "audio_sha256": audio_hashes[0], + "annotation_sha256": annotation_hashes[0], + "rights_basis": "unit-test fixture only", + "rights_cleared": True, + "source_uri": "urn:bandscope:test:track-001", + }, + { + "track_id": "track-002", + "audio_sha256": audio_hashes[1], + "annotation_sha256": annotation_hashes[1], + "rights_basis": "unit-test fixture only", + "rights_cleared": True, + "source_uri": "urn:bandscope:test:track-002", + }, + ], + "runtime": { + "source_commit": "e" * 40, + "uv_lock_sha256": "f" * 64, + "python_version": "3.12.11", + "librosa_version": "0.11.0", + "numpy_version": "2.3.3", + "sample_rate_hz": 44100, + "channels": 1, + "host_profile": "unit-test-host", + }, + "claim_boundary": "Unit-test fixture only; not production scientific evidence.", + } + + +def _runtime() -> dict[str, object]: + return { + "source_commit": "e" * 40, + "uv_lock_sha256": "f" * 64, + "python_version": "3.12.11", + "librosa_version": "0.11.0", + "numpy_version": "2.3.3", + } + + +def _files(tmp_path: Path) -> tuple[list[Path], list[Path]]: + audio_paths = [tmp_path / "one.wav", tmp_path / "two.wav"] + annotation_paths = [tmp_path / "one.lab", tmp_path / "two.lab"] + audio_paths[0].write_bytes(b"unit-audio-one") + audio_paths[1].write_bytes(b"unit-audio-two") + annotation_paths[0].write_bytes(b"0.0\tverse\n") + annotation_paths[1].write_bytes(b"0.0\tchorus\n") + return audio_paths, annotation_paths + + +def _digest(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _manifest( + admission: ModuleType, + registration: dict[str, object], + audio_paths: list[Path], + annotation_paths: list[Path], +) -> dict[str, object]: + validator = admission._load_validator() + return { + "schema_version": 1, + "registration_sha256": validator.registration_digest(registration), + "tracks": [ + { + "track_id": "track-001", + "audio_path": str(audio_paths[0]), + "annotation_path": str(annotation_paths[0]), + }, + { + "track_id": "track-002", + "audio_path": str(audio_paths[1]), + "annotation_path": str(annotation_paths[1]), + }, + ], + } + + +def test_admission_hashes_actual_files_and_emits_no_local_paths(tmp_path: Path) -> None: + admission = _admission() + audio_paths, annotation_paths = _files(tmp_path) + registration = _registration( + [_digest(path) for path in audio_paths], + [_digest(path) for path in annotation_paths], + ) + manifest = _manifest(admission, registration, audio_paths, annotation_paths) + decoded_digests: list[str] = [] + + def fake_decoder(fd: int, sample_rate_hz: int) -> tuple[str, int]: + assert sample_rate_hz == 44100 + assert os.read(fd, 1) + digest = hashlib.sha256(f"decoded-{len(decoded_digests)}".encode()).hexdigest() + decoded_digests.append(digest) + return digest, 44100 + + receipt = admission.verify_corpus( + registration, + manifest, + runtime_identity=_runtime(), + decoder=fake_decoder, + ) + + tracks = receipt["tracks"] + assert isinstance(tracks, list) + assert [track["track_id"] for track in tracks] == ["track-001", "track-002"] + assert [track["decoded_pcm_sha256"] for track in tracks] == decoded_digests + serialized = repr(receipt) + assert str(tmp_path) not in serialized + assert "audio_path" not in serialized + assert "annotation_path" not in serialized + + +def test_admission_rejects_byte_drift_before_decode(tmp_path: Path) -> None: + admission = _admission() + audio_paths, annotation_paths = _files(tmp_path) + registration = _registration( + [_digest(path) for path in audio_paths], + [_digest(path) for path in annotation_paths], + ) + manifest = _manifest(admission, registration, audio_paths, annotation_paths) + audio_paths[0].write_bytes(b"mutated-after-registration") + + with pytest.raises(ValueError, match="audio_path SHA-256"): + admission.verify_corpus( + registration, + manifest, + runtime_identity=_runtime(), + decoder=lambda _fd, _sr: pytest.fail("decoder must not run on hash mismatch"), + ) + + +def test_admission_rejects_runtime_drift(tmp_path: Path) -> None: + admission = _admission() + audio_paths, annotation_paths = _files(tmp_path) + registration = _registration( + [_digest(path) for path in audio_paths], + [_digest(path) for path in annotation_paths], + ) + manifest = _manifest(admission, registration, audio_paths, annotation_paths) + runtime = _runtime() + runtime["numpy_version"] = "9.9.9" + + with pytest.raises(ValueError, match="runtime identity mismatch for numpy_version"): + admission.verify_corpus( + registration, + manifest, + runtime_identity=runtime, + decoder=lambda _fd, _sr: ("0" * 64, 1), + ) + + +def test_admission_rejects_symlinked_corpus_material(tmp_path: Path) -> None: + admission = _admission() + audio_paths, annotation_paths = _files(tmp_path) + registration = _registration( + [_digest(path) for path in audio_paths], + [_digest(path) for path in annotation_paths], + ) + link = tmp_path / "linked.wav" + make_symlink_or_skip(link, audio_paths[0]) + manifest = _manifest(admission, registration, [link, audio_paths[1]], annotation_paths) + + with pytest.raises(ValueError, match="audio_path"): + admission.verify_corpus( + registration, + manifest, + runtime_identity=_runtime(), + decoder=lambda _fd, _sr: ("0" * 64, 1), + ) + + +def test_manifest_loader_rejects_duplicate_keys_and_nonstandard_numbers(tmp_path: Path) -> None: + admission = _admission() + duplicate = tmp_path / "duplicate.json" + duplicate.write_text('{"schema_version":1,"schema_version":1}', encoding="utf-8") + with pytest.raises(ValueError, match="duplicate JSON key"): + admission._load_json(duplicate) + + nonstandard = tmp_path / "nan.json" + nonstandard.write_text('{"schema_version":NaN}', encoding="utf-8") + with pytest.raises(ValueError, match="non-standard JSON number"): + admission._load_json(nonstandard) From e1de49897c709edb8a5a38a43a4c0b5038d87bce Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 20:05:21 +0900 Subject: [PATCH 052/216] docs(mir): trace local corpus admission evidence --- .../mir/structure-corpus-admission.md | 54 +++++++++++++++++++ 1 file changed, 54 insertions(+) create mode 100644 docs/traceability/mir/structure-corpus-admission.md diff --git a/docs/traceability/mir/structure-corpus-admission.md b/docs/traceability/mir/structure-corpus-admission.md new file mode 100644 index 000000000..b71b8c59f --- /dev/null +++ b/docs/traceability/mir/structure-corpus-admission.md @@ -0,0 +1,54 @@ +# Structure experiment corpus admission + +Status: Proposed +Owner: Signal-MIR Analysis +Tracking: #1225, #1228 +Parent contract: `docs/traceability/mir/structure-feature-noninferiority.md` + +## Problem + +The noninferiority registration names rights-cleared audio and annotation content by SHA-256 and provenance URI, but those declarations alone do not prove that a workstation actually measured the registered bytes. A local manifest can point at the wrong revision, a file can change after the manifest is prepared, or the decoder can normalize different bytes than reviewers believe were admitted. + +The experiment therefore needs a local-only admission step before any CQT/STFT metric or latency measurement. Workstation paths are execution details and must not become scientific provenance or appear in durable receipts. + +## Decision + +`scripts/research/verify_structure_corpus.py` resolves a local manifest only at execution time. For every registered track it: + +- requires the manifest order and track IDs to exactly match the preregistered corpus; +- opens audio and annotation inputs as regular files without following symlinks where the platform provides `O_NOFOLLOW`; +- computes SHA-256 from the opened descriptors and compares those digests with the preregistration before decoding; +- requires the current source commit, `uv.lock`, Python, librosa, and NumPy identities to match the registered runtime; +- decodes the already-admitted audio descriptor through `librosa.load(..., sr=, mono=True)` and computes a canonical little-endian float32 PCM SHA-256; +- emits only registration identity, runtime identity, content digests, decoded PCM digest/frame count, sample rate, channel count, and track ID. Local audio/annotation paths are never copied into the receipt. + +The tool does not calculate MIR metrics, aggregate tracks, estimate uncertainty, or make a noninferiority decision. Those remain separate scientific steps. A passing corpus-admission receipt is therefore necessary evidence for a run, not sufficient evidence for a production representation change. + +## Constraints and rejected alternatives + +Dereferencing `source_uri` was rejected. Provenance URI is evidence metadata and may identify licensed material that cannot be fetched by CI. Network retrieval would also turn a local-first experiment into a mutable external dependency. + +Hashing by pathname and then reopening for decode was rejected because the pathname can change between the two operations. Admission hashes the opened descriptor, rewinds that descriptor, duplicates it, and decodes the same opened file identity. + +Persisting workstation paths in the receipt was rejected because they are neither stable provenance nor purpose-bound evidence and can expose local usernames, mounts, or project layout. + +The decoded PCM digest does not by itself prove that two machines will decode a compressed source identically. It records the exact normalized signal used by one run. The preregistration already binds librosa/NumPy and the source/lock identity; if future production acceptance requires cross-decoder equivalence, the schema must explicitly bind the remaining decoder/backend contract rather than assuming it. + +## Audio normalization authority + +BandScope's registered structure input is mono at the registered sample rate. librosa 0.11.0 documents `load` as producing a floating-point time series, accepting file-like inputs, converting to mono when requested, and resampling to the requested `sr`; its documented default resampler is `soxr_hq`. The admission tool intentionally calls `librosa.load` without overriding `res_type`, so the decoder behavior is tied to the preregistered librosa version rather than duplicated in a second local normalization implementation. + +MIREX 2025 Music Structure Analysis evaluates mono 44.1 kHz WAV input and uses frame-level functional-label accuracy plus boundary retrieval F-measures at 0.5 s and 3.0 s. Corpus admission does not claim those evaluation metrics; it only ensures that later metric computation starts from the registered local bytes and records the exact decoded PCM identity. + +## Test boundary + +Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, and non-standard JSON numbers. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. + +## Remaining scientific work + +A reviewed rights-cleared real corpus and independent annotation manifest are still required. Before candidate results are inspected, the actual aggregation, paired uncertainty, margins, latency threshold, dependence assumptions, claim boundary, and any failure/exclusion rule must be approved. The next runner must consume admitted decoded audio, calculate the recognized MIREX/mir_eval track metrics for both CQT and STFT on the same PCM, and derive aggregate/CI evidence from the preregistered procedure rather than accepting caller-authored summaries. + +## References + +- MIREX. (2025). *Music Structure Analysis*. International Music Information Retrieval Systems Evaluation Laboratory. https://music-ir.org/mirex/wiki/2025%3AMusic_Structure_Analysis +- McFee, B., et al. (2025). *librosa 0.11.0 documentation: Core IO and DSP*. https://librosa.org/doc/0.11.0/core.html From f3abb6489836fd775cf67ae48cdc2c320f64a518 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 20:05:56 +0900 Subject: [PATCH 053/216] test(mir): normalize runtime digest identity --- ...tructure_corpus_runtime_identity_policy.py | 40 +++++++++++++++++++ 1 file changed, 40 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_corpus_runtime_identity_policy.py diff --git a/services/analysis-engine/tests/test_structure_corpus_runtime_identity_policy.py b/services/analysis-engine/tests/test_structure_corpus_runtime_identity_policy.py new file mode 100644 index 000000000..01175b48d --- /dev/null +++ b/services/analysis-engine/tests/test_structure_corpus_runtime_identity_policy.py @@ -0,0 +1,40 @@ +"""Runtime identity policy for real-audio corpus admission.""" + +from __future__ import annotations + +from types import ModuleType + +from conftest import load_module + + +def _admission() -> ModuleType: + return load_module( + "scripts/research/verify_structure_corpus.py", + "verify_structure_corpus_runtime_identity", + ) + + +def test_runtime_digest_identity_is_hex_case_insensitive() -> None: + """Validator-accepted uppercase hex must match lowercase runtime evidence.""" + admission = _admission() + registration = { + "runtime": { + "source_commit": "A" * 40, + "uv_lock_sha256": "B" * 64, + "python_version": "3.12.11", + "librosa_version": "0.11.0", + "numpy_version": "2.3.3", + } + } + runtime_identity = { + "source_commit": "a" * 40, + "uv_lock_sha256": "b" * 64, + "python_version": "3.12.11", + "librosa_version": "0.11.0", + "numpy_version": "2.3.3", + } + + normalized = admission._validate_runtime_identity(registration, runtime_identity) + + assert normalized["source_commit"] == "a" * 40 + assert normalized["uv_lock_sha256"] == "b" * 64 From 174c6d33b44ef8e20e7e923ad638d7489442a460 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 20:06:27 +0900 Subject: [PATCH 054/216] fix(mir): normalize registered runtime digests --- scripts/research/verify_structure_corpus.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/scripts/research/verify_structure_corpus.py b/scripts/research/verify_structure_corpus.py index f81bb1ee7..3f1c2473b 100644 --- a/scripts/research/verify_structure_corpus.py +++ b/scripts/research/verify_structure_corpus.py @@ -38,6 +38,7 @@ "librosa_version", "numpy_version", ) +_HEX_RUNTIME_IDENTITY_FIELDS = {"source_commit", "uv_lock_sha256"} def _load_validator() -> ModuleType: @@ -205,6 +206,9 @@ def _validate_runtime_identity( for field in _RUNTIME_IDENTITY_FIELDS: actual = _text(runtime_identity.get(field), f"runtime_identity.{field}") expected = _text(runtime.get(field), f"registration.runtime.{field}") + if field in _HEX_RUNTIME_IDENTITY_FIELDS: + actual = actual.lower() + expected = expected.lower() if actual != expected: raise ValueError( f"runtime identity mismatch for {field}: expected {expected}, got {actual}" From 3effa5a01b078c08f0a8bc91e3d8d1f37c12a43a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 20:08:12 +0900 Subject: [PATCH 055/216] docs(mir): record canonical runtime hex identity --- docs/traceability/mir/structure-corpus-admission.md | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/docs/traceability/mir/structure-corpus-admission.md b/docs/traceability/mir/structure-corpus-admission.md index b71b8c59f..d29bcd3cd 100644 --- a/docs/traceability/mir/structure-corpus-admission.md +++ b/docs/traceability/mir/structure-corpus-admission.md @@ -19,6 +19,7 @@ The experiment therefore needs a local-only admission step before any CQT/STFT m - opens audio and annotation inputs as regular files without following symlinks where the platform provides `O_NOFOLLOW`; - computes SHA-256 from the opened descriptors and compares those digests with the preregistration before decoding; - requires the current source commit, `uv.lock`, Python, librosa, and NumPy identities to match the registered runtime; +- treats Git commit and SHA-256 values as case-insensitive hexadecimal identities and emits them lowercase, matching the validator contract, while Python/librosa/NumPy version strings remain exact; - decodes the already-admitted audio descriptor through `librosa.load(..., sr=, mono=True)` and computes a canonical little-endian float32 PCM SHA-256; - emits only registration identity, runtime identity, content digests, decoded PCM digest/frame count, sample rate, channel count, and track ID. Local audio/annotation paths are never copied into the receipt. @@ -42,7 +43,7 @@ MIREX 2025 Music Structure Analysis evaluates mono 44.1 kHz WAV input and uses f ## Test boundary -Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, and non-standard JSON numbers. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. +Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, and the validator-compatible case-insensitive identity of Git/SHA-256 hexadecimal fields. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. ## Remaining scientific work From 1f653439a31b27ff905dd2e825a99f79188598d6 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 20:10:04 +0900 Subject: [PATCH 056/216] test(mir): require immutable admitted audio snapshot --- .../test_structure_corpus_snapshot_policy.py | 37 +++++++++++++++++++ 1 file changed, 37 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_corpus_snapshot_policy.py diff --git a/services/analysis-engine/tests/test_structure_corpus_snapshot_policy.py b/services/analysis-engine/tests/test_structure_corpus_snapshot_policy.py new file mode 100644 index 000000000..7d82ad6c4 --- /dev/null +++ b/services/analysis-engine/tests/test_structure_corpus_snapshot_policy.py @@ -0,0 +1,37 @@ +"""Snapshot policy for real-audio corpus admission.""" + +from __future__ import annotations + +import hashlib +import os +from pathlib import Path +from types import ModuleType + +from conftest import load_module + + +def _admission() -> ModuleType: + return load_module( + "scripts/research/verify_structure_corpus.py", + "verify_structure_corpus_snapshot_policy", + ) + + +def test_admitted_audio_snapshot_is_immutable_against_source_mutation(tmp_path: Path) -> None: + """Decode input must be the exact bytes hashed at admission, not a live source inode.""" + admission = _admission() + source = tmp_path / "track.wav" + original = b"registered-audio-bytes" + source.write_bytes(original) + fd = admission._open_regular_file(source, "audio_path") + try: + snapshot, digest = admission._snapshot_and_hash(fd) + try: + source.write_bytes(b"mutated-after-admission") + snapshot.seek(0) + assert snapshot.read() == original + assert digest == hashlib.sha256(original).hexdigest() + finally: + snapshot.close() + finally: + os.close(fd) From 47184b8aa8bff7a0405d30ef79a6274e64e9d483 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 20:10:41 +0900 Subject: [PATCH 057/216] fix(mir): decode immutable admitted audio snapshot --- scripts/research/verify_structure_corpus.py | 50 ++++++++++++++++----- 1 file changed, 39 insertions(+), 11 deletions(-) diff --git a/scripts/research/verify_structure_corpus.py b/scripts/research/verify_structure_corpus.py index 3f1c2473b..fb448b217 100644 --- a/scripts/research/verify_structure_corpus.py +++ b/scripts/research/verify_structure_corpus.py @@ -2,8 +2,9 @@ """Admit local real-audio corpus material for a frozen BandScope MIR experiment. The preregistration stores provenance and content digests, never workstation -paths. This tool resolves a local-only manifest, hashes audio and annotation -bytes from opened regular-file descriptors, decodes each audio item through the +paths. This tool resolves a local-only manifest, snapshots and hashes audio +bytes from opened regular-file descriptors, hashes annotation bytes from their +opened descriptors, decodes each immutable audio snapshot through the registered librosa normalization contract, and emits a path-free receipt with a SHA-256 identity of the exact mono float32 PCM presented to later analysis. @@ -22,10 +23,11 @@ import platform import stat import subprocess +import tempfile from collections.abc import Callable, Mapping, Sequence from pathlib import Path from types import ModuleType -from typing import Any +from typing import Any, BinaryIO MANIFEST_SCHEMA_VERSION = 1 MAX_MANIFEST_BYTES = 2 * 1024 * 1024 @@ -122,6 +124,27 @@ def _sha256_file_descriptor(fd: int) -> str: return digest.hexdigest() +def _snapshot_and_hash(fd: int) -> tuple[BinaryIO, str]: + """Copy one opened source into a process-owned snapshot while hashing it.""" + digest = hashlib.sha256() + snapshot = tempfile.TemporaryFile(mode="w+b") + try: + os.lseek(fd, 0, os.SEEK_SET) + while True: + chunk = os.read(fd, 1024 * 1024) + if not chunk: + break + digest.update(chunk) + snapshot.write(chunk) + snapshot.flush() + snapshot.seek(0) + os.lseek(fd, 0, os.SEEK_SET) + return snapshot, digest.hexdigest() + except Exception: + snapshot.close() + raise + + def _open_regular_file(path: Path, field: str) -> int: flags = os.O_RDONLY if hasattr(os, "O_BINARY"): @@ -145,7 +168,7 @@ def _open_regular_file(path: Path, field: str) -> int: def _decode_pcm_identity(fd: int, target_sample_rate_hz: int) -> tuple[str, int]: - """Decode from the admitted descriptor and hash canonical mono float32 PCM.""" + """Decode from the admitted snapshot and hash canonical mono float32 PCM.""" import librosa import numpy as np @@ -269,13 +292,18 @@ def verify_corpus( annotation_path = Path(_text(track.get("annotation_path"), f"{field}.annotation_path")) audio_fd = _open_regular_file(audio_path, f"{field}.audio_path") try: - audio_sha256 = _sha256_file_descriptor(audio_fd) - expected_audio = _text( - registered.get("audio_sha256"), "registered.audio_sha256" - ).lower() - if audio_sha256 != expected_audio: - raise ValueError(f"{field}.audio_path SHA-256 does not match registration") - decoded_pcm_sha256, decoded_frames = decoder(audio_fd, target_sample_rate_hz) + snapshot, audio_sha256 = _snapshot_and_hash(audio_fd) + try: + expected_audio = _text( + registered.get("audio_sha256"), "registered.audio_sha256" + ).lower() + if audio_sha256 != expected_audio: + raise ValueError(f"{field}.audio_path SHA-256 does not match registration") + decoded_pcm_sha256, decoded_frames = decoder( + snapshot.fileno(), target_sample_rate_hz + ) + finally: + snapshot.close() finally: os.close(audio_fd) From f28764fc20322a1405e730245185cc8dba12f888 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 20:10:59 +0900 Subject: [PATCH 058/216] test(mir): make snapshot mutation check portable --- .../test_structure_corpus_snapshot_policy.py | 16 +++++++--------- 1 file changed, 7 insertions(+), 9 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_corpus_snapshot_policy.py b/services/analysis-engine/tests/test_structure_corpus_snapshot_policy.py index 7d82ad6c4..48ec3563b 100644 --- a/services/analysis-engine/tests/test_structure_corpus_snapshot_policy.py +++ b/services/analysis-engine/tests/test_structure_corpus_snapshot_policy.py @@ -24,14 +24,12 @@ def test_admitted_audio_snapshot_is_immutable_against_source_mutation(tmp_path: original = b"registered-audio-bytes" source.write_bytes(original) fd = admission._open_regular_file(source, "audio_path") + snapshot, digest = admission._snapshot_and_hash(fd) + os.close(fd) try: - snapshot, digest = admission._snapshot_and_hash(fd) - try: - source.write_bytes(b"mutated-after-admission") - snapshot.seek(0) - assert snapshot.read() == original - assert digest == hashlib.sha256(original).hexdigest() - finally: - snapshot.close() + source.write_bytes(b"mutated-after-admission") + snapshot.seek(0) + assert snapshot.read() == original + assert digest == hashlib.sha256(original).hexdigest() finally: - os.close(fd) + snapshot.close() From 647481b19386c649f4c5c33ba06a9a1b48509781 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 20:11:14 +0900 Subject: [PATCH 059/216] docs(mir): bind decode to immutable admitted snapshot --- docs/traceability/mir/structure-corpus-admission.md | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/docs/traceability/mir/structure-corpus-admission.md b/docs/traceability/mir/structure-corpus-admission.md index d29bcd3cd..52e722755 100644 --- a/docs/traceability/mir/structure-corpus-admission.md +++ b/docs/traceability/mir/structure-corpus-admission.md @@ -17,10 +17,11 @@ The experiment therefore needs a local-only admission step before any CQT/STFT m - requires the manifest order and track IDs to exactly match the preregistered corpus; - opens audio and annotation inputs as regular files without following symlinks where the platform provides `O_NOFOLLOW`; -- computes SHA-256 from the opened descriptors and compares those digests with the preregistration before decoding; +- copies the opened audio stream into a process-owned temporary snapshot while computing the SHA-256, then compares that digest with the preregistration before decoding; +- hashes annotation bytes from the opened annotation descriptor and compares them with the registered annotation identity; - requires the current source commit, `uv.lock`, Python, librosa, and NumPy identities to match the registered runtime; - treats Git commit and SHA-256 values as case-insensitive hexadecimal identities and emits them lowercase, matching the validator contract, while Python/librosa/NumPy version strings remain exact; -- decodes the already-admitted audio descriptor through `librosa.load(..., sr=, mono=True)` and computes a canonical little-endian float32 PCM SHA-256; +- decodes the immutable admitted audio snapshot through `librosa.load(..., sr=, mono=True)` and computes a canonical little-endian float32 PCM SHA-256; - emits only registration identity, runtime identity, content digests, decoded PCM digest/frame count, sample rate, channel count, and track ID. Local audio/annotation paths are never copied into the receipt. The tool does not calculate MIR metrics, aggregate tracks, estimate uncertainty, or make a noninferiority decision. Those remain separate scientific steps. A passing corpus-admission receipt is therefore necessary evidence for a run, not sufficient evidence for a production representation change. @@ -29,7 +30,7 @@ The tool does not calculate MIR metrics, aggregate tracks, estimate uncertainty, Dereferencing `source_uri` was rejected. Provenance URI is evidence metadata and may identify licensed material that cannot be fetched by CI. Network retrieval would also turn a local-first experiment into a mutable external dependency. -Hashing by pathname and then reopening for decode was rejected because the pathname can change between the two operations. Admission hashes the opened descriptor, rewinds that descriptor, duplicates it, and decodes the same opened file identity. +Hashing an opened source descriptor and then decoding that still-live source descriptor was rejected after hostile review. A pathname cannot be swapped once the descriptor is open, but another writer can still change the underlying regular-file bytes between the hash and decode. Admission therefore snapshots the source bytes while hashing and decodes only that process-owned snapshot. Source mutation after snapshot creation cannot change the admitted decoder input. Persisting workstation paths in the receipt was rejected because they are neither stable provenance nor purpose-bound evidence and can expose local usernames, mounts, or project layout. @@ -43,7 +44,7 @@ MIREX 2025 Music Structure Analysis evaluates mono 44.1 kHz WAV input and uses f ## Test boundary -Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, and the validator-compatible case-insensitive identity of Git/SHA-256 hexadecimal fields. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. +Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, and source mutation after snapshot admission. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. ## Remaining scientific work From af927bbd429647f5cb655ec488ea6e7f16f8a7a2 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 21:06:50 +0900 Subject: [PATCH 060/216] test(mir): require admitted PCM handoff to runner --- .../tests/test_structure_corpus_admission.py | 52 +++++++++++++++++++ 1 file changed, 52 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_corpus_admission.py b/services/analysis-engine/tests/test_structure_corpus_admission.py index 41cb1d1ae..18bbbfcc8 100644 --- a/services/analysis-engine/tests/test_structure_corpus_admission.py +++ b/services/analysis-engine/tests/test_structure_corpus_admission.py @@ -239,3 +239,55 @@ def test_manifest_loader_rejects_duplicate_keys_and_nonstandard_numbers(tmp_path nonstandard.write_text('{"schema_version":NaN}', encoding="utf-8") with pytest.raises(ValueError, match="non-standard JSON number"): admission._load_json(nonstandard) + + +def test_admission_hands_exact_pcm_and_annotation_snapshot_to_consumer(tmp_path: Path) -> None: + """A runner must consume the admitted signal and annotation, not reopen source paths.""" + admission = _admission() + audio_paths, annotation_paths = _files(tmp_path) + original_annotations = [path.read_bytes() for path in annotation_paths] + registration = _registration( + [_digest(path) for path in audio_paths], + [_digest(path) for path in annotation_paths], + ) + manifest = _manifest(admission, registration, audio_paths, annotation_paths) + pcm_payloads = [memoryview(b"\x00\x00\x00\x00"), memoryview(b"\x00\x00\x80?")] + decoded_index = 0 + consumed: list[tuple[str, bytes, bytes, int]] = [] + + def fake_decoder(fd: int, sample_rate_hz: int) -> tuple[str, int, memoryview]: + nonlocal decoded_index + assert sample_rate_hz == 44100 + assert os.read(fd, 1) + pcm = pcm_payloads[decoded_index] + decoded_index += 1 + return hashlib.sha256(pcm).hexdigest(), 1, pcm + + def consume_track( + track_id: str, + pcm: memoryview, + annotation_snapshot: object, + sample_rate_hz: int, + ) -> None: + index = int(track_id[-1]) - 1 + annotation_paths[index].write_bytes(b"mutated-after-admission") + annotation_snapshot.seek(0) + consumed.append( + (track_id, bytes(pcm), annotation_snapshot.read(), sample_rate_hz) + ) + + receipt = admission.verify_corpus( + registration, + manifest, + runtime_identity=_runtime(), + decoder=fake_decoder, + track_consumer=consume_track, + ) + + assert consumed == [ + ("track-001", bytes(pcm_payloads[0]), original_annotations[0], 44100), + ("track-002", bytes(pcm_payloads[1]), original_annotations[1], 44100), + ] + assert [ + track["decoded_pcm_sha256"] for track in receipt["tracks"] + ] == [hashlib.sha256(payload).hexdigest() for payload in pcm_payloads] From 813dd925ec80d66e23a57c03ff52c143d569f9c7 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 21:07:37 +0900 Subject: [PATCH 061/216] fix(mir): hand admitted PCM directly to experiment consumer --- scripts/research/verify_structure_corpus.py | 70 ++++++++++++++++----- 1 file changed, 53 insertions(+), 17 deletions(-) diff --git a/scripts/research/verify_structure_corpus.py b/scripts/research/verify_structure_corpus.py index fb448b217..5f40f0457 100644 --- a/scripts/research/verify_structure_corpus.py +++ b/scripts/research/verify_structure_corpus.py @@ -8,9 +8,15 @@ registered librosa normalization contract, and emits a path-free receipt with a SHA-256 identity of the exact mono float32 PCM presented to later analysis. -It does not calculate MIR metrics, choose thresholds, or make a noninferiority -decision. Synthetic audio is suitable for unit tests only; production receipts -require the rights-cleared real-audio corpus named by the preregistration. +An in-process track consumer may receive that exact normalized PCM together +with the immutable admitted annotation snapshot. This is the handoff boundary +for a later MIR experiment runner: the runner must not reopen workstation source +paths after admission merely because the durable receipt contains only digests. + +The tool does not calculate MIR metrics, choose thresholds, or make a +noninferiority decision. Synthetic audio is suitable for unit tests only; +production receipts require the rights-cleared real-audio corpus named by the +preregistration. """ from __future__ import annotations @@ -167,8 +173,8 @@ def _open_regular_file(path: Path, field: str) -> int: raise -def _decode_pcm_identity(fd: int, target_sample_rate_hz: int) -> tuple[str, int]: - """Decode from the admitted snapshot and hash canonical mono float32 PCM.""" +def _decode_pcm_identity(fd: int, target_sample_rate_hz: int) -> tuple[str, int, memoryview]: + """Decode the admitted snapshot and expose canonical read-only mono float32 PCM.""" import librosa import numpy as np @@ -192,7 +198,9 @@ def _decode_pcm_identity(fd: int, target_sample_rate_hz: int) -> tuple[str, int] canonical = np.asarray(samples, dtype=" dict[str, object]: @@ -245,9 +253,21 @@ def verify_corpus( manifest: Mapping[str, Any], *, runtime_identity: Mapping[str, object], - decoder: Callable[[int, int], tuple[str, int]] = _decode_pcm_identity, + decoder: Callable[ + [int, int], + tuple[str, int] | tuple[str, int, memoryview], + ] = _decode_pcm_identity, + track_consumer: Callable[[str, memoryview, BinaryIO, int], None] | None = None, ) -> dict[str, object]: - """Verify local files and return a path-free decoded-corpus receipt.""" + """Verify local files and return a path-free decoded-corpus receipt. + + When ``track_consumer`` is supplied, it runs only after both registered + content identities are verified. It receives the exact canonical PCM used + for ``decoded_pcm_sha256`` plus a borrowed immutable-source annotation + snapshot. The callback must finish before this function returns; the + annotation handle is closed immediately afterward and is never persisted in + the receipt. + """ validator = _load_validator() validator.validate_registration(registration) registration_sha256 = validator.registration_digest(registration) @@ -299,9 +319,9 @@ def verify_corpus( ).lower() if audio_sha256 != expected_audio: raise ValueError(f"{field}.audio_path SHA-256 does not match registration") - decoded_pcm_sha256, decoded_frames = decoder( - snapshot.fileno(), target_sample_rate_hz - ) + decoded = decoder(snapshot.fileno(), target_sample_rate_hz) + decoded_pcm_sha256, decoded_frames = decoded[:2] + decoded_pcm = decoded[2] if len(decoded) == 3 else None finally: snapshot.close() finally: @@ -309,14 +329,30 @@ def verify_corpus( annotation_fd = _open_regular_file(annotation_path, f"{field}.annotation_path") try: - annotation_sha256 = _sha256_file_descriptor(annotation_fd) + annotation_snapshot, annotation_sha256 = _snapshot_and_hash(annotation_fd) finally: os.close(annotation_fd) - expected_annotation = _text( - registered.get("annotation_sha256"), "registered.annotation_sha256" - ).lower() - if annotation_sha256 != expected_annotation: - raise ValueError(f"{field}.annotation_path SHA-256 does not match registration") + try: + expected_annotation = _text( + registered.get("annotation_sha256"), "registered.annotation_sha256" + ).lower() + if annotation_sha256 != expected_annotation: + raise ValueError(f"{field}.annotation_path SHA-256 does not match registration") + + if track_consumer is not None: + if decoded_pcm is None: + raise ValueError( + "decoder must expose admitted PCM when track_consumer is configured" + ) + annotation_snapshot.seek(0) + track_consumer( + track_id, + decoded_pcm, + annotation_snapshot, + target_sample_rate_hz, + ) + finally: + annotation_snapshot.close() receipt_tracks.append( { From 359b95d7c9ae0e7c4cb1437b599358074f18cd16 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 21:08:35 +0900 Subject: [PATCH 062/216] docs(mir): bind runner handoff to admitted PCM --- .../mir/structure-corpus-admission.md | 35 +++++++++++++++---- 1 file changed, 28 insertions(+), 7 deletions(-) diff --git a/docs/traceability/mir/structure-corpus-admission.md b/docs/traceability/mir/structure-corpus-admission.md index 52e722755..4a30964d8 100644 --- a/docs/traceability/mir/structure-corpus-admission.md +++ b/docs/traceability/mir/structure-corpus-admission.md @@ -11,6 +11,8 @@ The noninferiority registration names rights-cleared audio and annotation conten The experiment therefore needs a local-only admission step before any CQT/STFT metric or latency measurement. Workstation paths are execution details and must not become scientific provenance or appear in durable receipts. +A digest-only receipt is also not a measurement handoff. Earlier admission code decoded the verified audio snapshot, recorded its PCM digest/frame count, then discarded both the snapshot and decoded signal before returning. A later experiment runner would therefore have had to reopen or re-decode workstation material, making it impossible to prove that the PCM identified by the admission receipt was the exact signal supplied to CQT and STFT. + ## Decision `scripts/research/verify_structure_corpus.py` resolves a local manifest only at execution time. For every registered track it: @@ -18,37 +20,56 @@ The experiment therefore needs a local-only admission step before any CQT/STFT m - requires the manifest order and track IDs to exactly match the preregistered corpus; - opens audio and annotation inputs as regular files without following symlinks where the platform provides `O_NOFOLLOW`; - copies the opened audio stream into a process-owned temporary snapshot while computing the SHA-256, then compares that digest with the preregistration before decoding; -- hashes annotation bytes from the opened annotation descriptor and compares them with the registered annotation identity; +- snapshots and hashes annotation bytes from the opened annotation descriptor and compares them with the registered annotation identity; - requires the current source commit, `uv.lock`, Python, librosa, and NumPy identities to match the registered runtime; - treats Git commit and SHA-256 values as case-insensitive hexadecimal identities and emits them lowercase, matching the validator contract, while Python/librosa/NumPy version strings remain exact; -- decodes the immutable admitted audio snapshot through `librosa.load(..., sr=, mono=True)` and computes a canonical little-endian float32 PCM SHA-256; +- decodes the immutable admitted audio snapshot through `librosa.load(..., sr=, mono=True)`, converts it to canonical little-endian float32, marks that NumPy array read-only, and computes the PCM SHA-256 over the same byte view; +- optionally invokes an in-process `track_consumer` only after both registered audio and annotation identities have passed. The callback receives that exact canonical PCM byte view, the process-owned annotation snapshot, track ID, and sample rate. It receives no workstation path; - emits only registration identity, runtime identity, content digests, decoded PCM digest/frame count, sample rate, channel count, and track ID. Local audio/annotation paths are never copied into the receipt. +The standalone CLI still writes only the path-free receipt. A scientific runner that claims "same admitted PCM" must use the in-process consumer boundary, or a future equivalently strong content-addressed handoff, rather than reading the receipt and reopening source paths afterward. + The tool does not calculate MIR metrics, aggregate tracks, estimate uncertainty, or make a noninferiority decision. Those remain separate scientific steps. A passing corpus-admission receipt is therefore necessary evidence for a run, not sufficient evidence for a production representation change. +## RED -> GREEN lineage + +The initial implementation `83e59f220eba26272840f3f901cdaaf50d86d3c2` hashed an opened audio descriptor and decoded that same live descriptor. Hostile review found that pathname substitution was prevented but same-inode byte mutation between hash and decode was not. + +RED `1f653439a31b27ff905dd2e825a99f79188598d6` required post-admission source mutation not to change decoder input. GREEN `47184b8aa8bff7a0405d30ef79a6274e64e9d483` moved decode onto a process-owned snapshot created while hashing. `f28764fc20322a1405e730245185cc8dba12f888` made the mutation regression portable across Windows and POSIX. + +A second gap remained: the verified normalized PCM was destroyed before any scientific consumer could use it. RED `af927bbd429647f5cb655ec488ea6e7f16f8a7a2` requires a consumer to receive the exact admitted PCM and annotation snapshot, and verifies that mutating the original annotation path after admission cannot change what the consumer reads. GREEN `813dd925ec80d66e23a57c03ff52c143d569f9c7` exposes that in-process handoff while preserving the path-free durable receipt and legacy hash-only verifier use. + +Earlier RED `f3abb6489836fd775cf67ae48cdc2c320f64a518` -> GREEN `174c6d33b44ef8e20e7e923ad638d7489442a460` aligned runtime identity with the evidence validator by comparing Git commit and SHA-256 fields as case-insensitive hexadecimal identities while keeping version strings exact. + ## Constraints and rejected alternatives Dereferencing `source_uri` was rejected. Provenance URI is evidence metadata and may identify licensed material that cannot be fetched by CI. Network retrieval would also turn a local-first experiment into a mutable external dependency. -Hashing an opened source descriptor and then decoding that still-live source descriptor was rejected after hostile review. A pathname cannot be swapped once the descriptor is open, but another writer can still change the underlying regular-file bytes between the hash and decode. Admission therefore snapshots the source bytes while hashing and decodes only that process-owned snapshot. Source mutation after snapshot creation cannot change the admitted decoder input. +Hashing an opened source descriptor and then decoding that still-live source descriptor was rejected. A pathname cannot be swapped once the descriptor is open, but another writer can still change the underlying regular-file bytes between the hash and decode. Admission therefore snapshots the source bytes while hashing and decodes only that process-owned snapshot. + +Treating the durable receipt as the experiment input was rejected. The receipt proves identities but does not contain the admitted signal or annotations. Reopening source paths from a later process would create a second admission event and sever the claim that baseline and candidate consumed the exact PCM whose digest is recorded. The current safe boundary is therefore in-process consumption during admission; a future persistent artifact bundle would need its own atomic, content-addressed and resource-bounded contract before it could replace this boundary. Persisting workstation paths in the receipt was rejected because they are neither stable provenance nor purpose-bound evidence and can expose local usernames, mounts, or project layout. -The decoded PCM digest does not by itself prove that two machines will decode a compressed source identically. It records the exact normalized signal used by one run. The preregistration already binds librosa/NumPy and the source/lock identity; if future production acceptance requires cross-decoder equivalence, the schema must explicitly bind the remaining decoder/backend contract rather than assuming it. +The decoded PCM digest does not by itself prove that two machines will decode a compressed source identically. It records the exact normalized signal used by one admission/consumer run. The preregistration already binds librosa/NumPy and the source/lock identity; if future production acceptance requires cross-decoder equivalence, the schema must explicitly bind the remaining decoder/backend contract rather than assuming it. ## Audio normalization authority BandScope's registered structure input is mono at the registered sample rate. librosa 0.11.0 documents `load` as producing a floating-point time series, accepting file-like inputs, converting to mono when requested, and resampling to the requested `sr`; its documented default resampler is `soxr_hq`. The admission tool intentionally calls `librosa.load` without overriding `res_type`, so the decoder behavior is tied to the preregistered librosa version rather than duplicated in a second local normalization implementation. -MIREX 2025 Music Structure Analysis evaluates mono 44.1 kHz WAV input and uses frame-level functional-label accuracy plus boundary retrieval F-measures at 0.5 s and 3.0 s. Corpus admission does not claim those evaluation metrics; it only ensures that later metric computation starts from the registered local bytes and records the exact decoded PCM identity. +MIREX 2025 Music Structure Analysis evaluates mono 44.1 kHz WAV input and uses frame-level functional-label accuracy plus boundary retrieval F-measures at 0.5 s and 3.0 s. Corpus admission does not claim those evaluation metrics; it only ensures that later metric computation can start from the same registered local bytes and exact decoded PCM identity. ## Test boundary -Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, and source mutation after snapshot admission. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. +Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, source mutation after snapshot admission, and exact PCM/annotation consumer handoff. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. + +The handoff regression deliberately mutates the original annotation file after its process-owned snapshot has been created and requires the consumer to observe the admitted bytes, not the mutated path content. It also requires the consumer's PCM bytes to hash to the digest emitted in the receipt. ## Remaining scientific work -A reviewed rights-cleared real corpus and independent annotation manifest are still required. Before candidate results are inspected, the actual aggregation, paired uncertainty, margins, latency threshold, dependence assumptions, claim boundary, and any failure/exclusion rule must be approved. The next runner must consume admitted decoded audio, calculate the recognized MIREX/mir_eval track metrics for both CQT and STFT on the same PCM, and derive aggregate/CI evidence from the preregistered procedure rather than accepting caller-authored summaries. +A reviewed rights-cleared real corpus and independent annotation manifest are still required. Before candidate results are inspected, the actual aggregation, paired uncertainty, margins, latency threshold, dependence assumptions, claim boundary, and any failure/exclusion rule must be approved. + +The next runner must execute through the in-process admitted-track consumer boundary, interpret the admitted annotation snapshot, run CQT and STFT on the exact same read-only PCM, calculate the recognized MIREX/mir_eval track metrics and latency/memory measurements, and derive aggregate/CI evidence from the preregistered procedure rather than accepting caller-authored summaries. The standalone receipt is evidence for that run, not a license to reopen media after admission. ## References From 12fa14b50ecf694be7775af2be69935eded5d734 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 21:12:27 +0900 Subject: [PATCH 063/216] test(mir): document corpus-admission contracts --- .../analysis-engine/tests/test_structure_corpus_admission.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_corpus_admission.py b/services/analysis-engine/tests/test_structure_corpus_admission.py index 18bbbfcc8..aa81f27dc 100644 --- a/services/analysis-engine/tests/test_structure_corpus_admission.py +++ b/services/analysis-engine/tests/test_structure_corpus_admission.py @@ -136,6 +136,7 @@ def _manifest( def test_admission_hashes_actual_files_and_emits_no_local_paths(tmp_path: Path) -> None: + """Receipt binds actual corpus bytes without persisting workstation paths.""" admission = _admission() audio_paths, annotation_paths = _files(tmp_path) registration = _registration( @@ -170,6 +171,7 @@ def fake_decoder(fd: int, sample_rate_hz: int) -> tuple[str, int]: def test_admission_rejects_byte_drift_before_decode(tmp_path: Path) -> None: + """Changed registered audio must fail before any decoder executes.""" admission = _admission() audio_paths, annotation_paths = _files(tmp_path) registration = _registration( @@ -189,6 +191,7 @@ def test_admission_rejects_byte_drift_before_decode(tmp_path: Path) -> None: def test_admission_rejects_runtime_drift(tmp_path: Path) -> None: + """Runtime drift must fail before corpus measurement begins.""" admission = _admission() audio_paths, annotation_paths = _files(tmp_path) registration = _registration( @@ -209,6 +212,7 @@ def test_admission_rejects_runtime_drift(tmp_path: Path) -> None: def test_admission_rejects_symlinked_corpus_material(tmp_path: Path) -> None: + """Corpus material must not cross the admission boundary through a symlink.""" admission = _admission() audio_paths, annotation_paths = _files(tmp_path) registration = _registration( @@ -229,6 +233,7 @@ def test_admission_rejects_symlinked_corpus_material(tmp_path: Path) -> None: def test_manifest_loader_rejects_duplicate_keys_and_nonstandard_numbers(tmp_path: Path) -> None: + """Manifest JSON must reject ambiguous keys and non-finite numeric constants.""" admission = _admission() duplicate = tmp_path / "duplicate.json" duplicate.write_text('{"schema_version":1,"schema_version":1}', encoding="utf-8") From 692011de8dc73122347a3bf6fa1aafb3320ca67b Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 21:13:03 +0900 Subject: [PATCH 064/216] docs(mir): satisfy admission docstring gate --- scripts/research/verify_structure_corpus.py | 1 + 1 file changed, 1 insertion(+) diff --git a/scripts/research/verify_structure_corpus.py b/scripts/research/verify_structure_corpus.py index 5f40f0457..298b859d6 100644 --- a/scripts/research/verify_structure_corpus.py +++ b/scripts/research/verify_structure_corpus.py @@ -375,6 +375,7 @@ def verify_corpus( def main() -> int: + """Run corpus admission and write one path-free verification receipt.""" parser = argparse.ArgumentParser( description="Admit local real-audio files for a frozen structure experiment" ) From 1e5aa12bd4965cfc709e35a626fa5727397c9a6d Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 22:04:31 +0900 Subject: [PATCH 065/216] test(mir): reject mutation of admitted annotation snapshot --- ...ucture_corpus_consumer_integrity_policy.py | 58 +++++++++++++++++++ 1 file changed, 58 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py diff --git a/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py b/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py new file mode 100644 index 000000000..b87c61e6f --- /dev/null +++ b/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py @@ -0,0 +1,58 @@ +"""Consumer-integrity policy for admitted MIR corpus material.""" + +from __future__ import annotations + +import hashlib +import os +from pathlib import Path +from typing import BinaryIO + +import pytest + +from test_structure_corpus_admission import ( + _admission, + _digest, + _files, + _manifest, + _registration, + _runtime, +) + + +def test_consumer_cannot_mutate_admitted_annotation_without_detection( + tmp_path: Path, +) -> None: + """Receipt admission must fail if a consumer mutates the hashed annotation snapshot.""" + admission = _admission() + audio_paths, annotation_paths = _files(tmp_path) + registration = _registration( + [_digest(path) for path in audio_paths], + [_digest(path) for path in annotation_paths], + ) + manifest = _manifest(admission, registration, audio_paths, annotation_paths) + pcm = memoryview(b"\x00\x00\x00\x00") + + def fake_decoder(fd: int, sample_rate_hz: int) -> tuple[str, int, memoryview]: + assert sample_rate_hz == 44100 + assert os.read(fd, 1) + return hashlib.sha256(pcm).hexdigest(), 1, pcm + + def mutating_consumer( + _track_id: str, + _pcm: memoryview, + annotation_snapshot: BinaryIO, + _sample_rate_hz: int, + ) -> None: + annotation_snapshot.seek(0) + annotation_snapshot.write(b"tampered-after-admission") + annotation_snapshot.truncate() + annotation_snapshot.flush() + + with pytest.raises(ValueError, match="consumer mutated admitted annotation"): + admission.verify_corpus( + registration, + manifest, + runtime_identity=_runtime(), + decoder=fake_decoder, + track_consumer=mutating_consumer, + ) From 61a74d6753f58069f4ff91c910dd237bd21c98fc Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 22:05:33 +0900 Subject: [PATCH 066/216] fix(mir): detect consumer mutation of admitted annotations --- scripts/research/verify_structure_corpus.py | 23 ++++++++++++++------- 1 file changed, 16 insertions(+), 7 deletions(-) diff --git a/scripts/research/verify_structure_corpus.py b/scripts/research/verify_structure_corpus.py index 298b859d6..6ffed7658 100644 --- a/scripts/research/verify_structure_corpus.py +++ b/scripts/research/verify_structure_corpus.py @@ -9,9 +9,10 @@ SHA-256 identity of the exact mono float32 PCM presented to later analysis. An in-process track consumer may receive that exact normalized PCM together -with the immutable admitted annotation snapshot. This is the handoff boundary -for a later MIR experiment runner: the runner must not reopen workstation source -paths after admission merely because the durable receipt contains only digests. +with the admitted annotation snapshot. This is the handoff boundary for a later +MIR experiment runner: the runner must not reopen workstation source paths after +admission merely because the durable receipt contains only digests. Any +consumer-side mutation of the admitted annotation snapshot invalidates the run. The tool does not calculate MIR metrics, choose thresholds, or make a noninferiority decision. Synthetic audio is suitable for unit tests only; @@ -263,10 +264,10 @@ def verify_corpus( When ``track_consumer`` is supplied, it runs only after both registered content identities are verified. It receives the exact canonical PCM used - for ``decoded_pcm_sha256`` plus a borrowed immutable-source annotation - snapshot. The callback must finish before this function returns; the - annotation handle is closed immediately afterward and is never persisted in - the receipt. + for ``decoded_pcm_sha256`` plus a borrowed annotation snapshot. The callback + must finish before this function returns; mutation of that snapshot is + detected and invalidates admission, and the handle is closed immediately + afterward rather than persisted in the receipt. """ validator = _load_validator() validator.validate_registration(registration) @@ -351,6 +352,14 @@ def verify_corpus( annotation_snapshot, target_sample_rate_hz, ) + annotation_snapshot.flush() + if ( + _sha256_file_descriptor(annotation_snapshot.fileno()) + != annotation_sha256 + ): + raise ValueError( + f"{field} consumer mutated admitted annotation snapshot" + ) finally: annotation_snapshot.close() From 383cd57c075abe77e0d9cfed69622c11ddb52319 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 22:06:11 +0900 Subject: [PATCH 067/216] docs(mir): trace consumer mutation fail-closed boundary --- docs/traceability/mir/structure-corpus-admission.md | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/docs/traceability/mir/structure-corpus-admission.md b/docs/traceability/mir/structure-corpus-admission.md index 4a30964d8..2f840a3b2 100644 --- a/docs/traceability/mir/structure-corpus-admission.md +++ b/docs/traceability/mir/structure-corpus-admission.md @@ -24,7 +24,7 @@ A digest-only receipt is also not a measurement handoff. Earlier admission code - requires the current source commit, `uv.lock`, Python, librosa, and NumPy identities to match the registered runtime; - treats Git commit and SHA-256 values as case-insensitive hexadecimal identities and emits them lowercase, matching the validator contract, while Python/librosa/NumPy version strings remain exact; - decodes the immutable admitted audio snapshot through `librosa.load(..., sr=, mono=True)`, converts it to canonical little-endian float32, marks that NumPy array read-only, and computes the PCM SHA-256 over the same byte view; -- optionally invokes an in-process `track_consumer` only after both registered audio and annotation identities have passed. The callback receives that exact canonical PCM byte view, the process-owned annotation snapshot, track ID, and sample rate. It receives no workstation path; +- optionally invokes an in-process `track_consumer` only after both registered audio and annotation identities have passed. The callback receives that exact canonical PCM byte view, the process-owned annotation snapshot, track ID, and sample rate. It receives no workstation path. Because the temporary annotation handle is writable internally, admission re-hashes it after the callback and rejects the run if the consumer changed any admitted annotation byte; - emits only registration identity, runtime identity, content digests, decoded PCM digest/frame count, sample rate, channel count, and track ID. Local audio/annotation paths are never copied into the receipt. The standalone CLI still writes only the path-free receipt. A scientific runner that claims "same admitted PCM" must use the in-process consumer boundary, or a future equivalently strong content-addressed handoff, rather than reading the receipt and reopening source paths afterward. @@ -39,6 +39,8 @@ RED `1f653439a31b27ff905dd2e825a99f79188598d6` required post-admission source mu A second gap remained: the verified normalized PCM was destroyed before any scientific consumer could use it. RED `af927bbd429647f5cb655ec488ea6e7f16f8a7a2` requires a consumer to receive the exact admitted PCM and annotation snapshot, and verifies that mutating the original annotation path after admission cannot change what the consumer reads. GREEN `813dd925ec80d66e23a57c03ff52c143d569f9c7` exposes that in-process handoff while preserving the path-free durable receipt and legacy hash-only verifier use. +That handoff still exposed the process-owned annotation snapshot as a writable `BinaryIO`. The source path could no longer drift, but the runner itself could mutate the hashed snapshot and then compute MIR metrics from bytes that no longer matched `annotation_sha256`. RED `1e5aa12bd4965cfc709e35a626fa5727397c9a6d` requires such a mutation to invalidate admission. GREEN `61a74d6753f58069f4ff91c910dd237bd21c98fc` flushes and re-hashes the annotation snapshot after the consumer returns, failing closed before a receipt is emitted when the admitted bytes changed. + Earlier RED `f3abb6489836fd775cf67ae48cdc2c320f64a518` -> GREEN `174c6d33b44ef8e20e7e923ad638d7489442a460` aligned runtime identity with the evidence validator by comparing Git commit and SHA-256 fields as case-insensitive hexadecimal identities while keeping version strings exact. ## Constraints and rejected alternatives @@ -49,6 +51,8 @@ Hashing an opened source descriptor and then decoding that still-live source des Treating the durable receipt as the experiment input was rejected. The receipt proves identities but does not contain the admitted signal or annotations. Reopening source paths from a later process would create a second admission event and sever the claim that baseline and candidate consumed the exact PCM whose digest is recorded. The current safe boundary is therefore in-process consumption during admission; a future persistent artifact bundle would need its own atomic, content-addressed and resource-bounded contract before it could replace this boundary. +Treating a writable temporary annotation handle as intrinsically immutable was rejected. The current callback API remains file-like for streaming parsers, but the snapshot digest is checked again after callback execution. A future read-only snapshot abstraction may tighten capability exposure further; until then, any consumer mutation invalidates the run and produces no admitted receipt. + Persisting workstation paths in the receipt was rejected because they are neither stable provenance nor purpose-bound evidence and can expose local usernames, mounts, or project layout. The decoded PCM digest does not by itself prove that two machines will decode a compressed source identically. It records the exact normalized signal used by one admission/consumer run. The preregistration already binds librosa/NumPy and the source/lock identity; if future production acceptance requires cross-decoder equivalence, the schema must explicitly bind the remaining decoder/backend contract rather than assuming it. @@ -61,15 +65,15 @@ MIREX 2025 Music Structure Analysis evaluates mono 44.1 kHz WAV input and uses f ## Test boundary -Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, source mutation after snapshot admission, and exact PCM/annotation consumer handoff. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. +Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, source mutation after snapshot admission, exact PCM/annotation consumer handoff, and consumer-side annotation mutation detection. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. -The handoff regression deliberately mutates the original annotation file after its process-owned snapshot has been created and requires the consumer to observe the admitted bytes, not the mutated path content. It also requires the consumer's PCM bytes to hash to the digest emitted in the receipt. +The handoff regression deliberately mutates the original annotation file after its process-owned snapshot has been created and requires the consumer to observe the admitted bytes, not the mutated path content. It also requires the consumer's PCM bytes to hash to the digest emitted in the receipt. A separate hostile-case regression writes directly through the borrowed annotation snapshot and requires admission to fail instead of emitting a receipt for post-hash annotation bytes. ## Remaining scientific work A reviewed rights-cleared real corpus and independent annotation manifest are still required. Before candidate results are inspected, the actual aggregation, paired uncertainty, margins, latency threshold, dependence assumptions, claim boundary, and any failure/exclusion rule must be approved. -The next runner must execute through the in-process admitted-track consumer boundary, interpret the admitted annotation snapshot, run CQT and STFT on the exact same read-only PCM, calculate the recognized MIREX/mir_eval track metrics and latency/memory measurements, and derive aggregate/CI evidence from the preregistered procedure rather than accepting caller-authored summaries. The standalone receipt is evidence for that run, not a license to reopen media after admission. +The next runner must execute through the in-process admitted-track consumer boundary, interpret the admitted annotation snapshot without mutating it, run CQT and STFT on the exact same read-only PCM, calculate the recognized MIREX/mir_eval track metrics and latency/memory measurements, and derive aggregate/CI evidence from the preregistered procedure rather than accepting caller-authored summaries. The standalone receipt is evidence for that run, not a license to reopen media after admission. ## References From 106d4e37e98a76f48f73db1ba1854209de873040 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 22:06:51 +0900 Subject: [PATCH 068/216] test(mir): bind manifest size and bytes to one descriptor --- ...structure_corpus_manifest_loader_policy.py | 39 +++++++++++++++++++ 1 file changed, 39 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_corpus_manifest_loader_policy.py diff --git a/services/analysis-engine/tests/test_structure_corpus_manifest_loader_policy.py b/services/analysis-engine/tests/test_structure_corpus_manifest_loader_policy.py new file mode 100644 index 000000000..7a2435be9 --- /dev/null +++ b/services/analysis-engine/tests/test_structure_corpus_manifest_loader_policy.py @@ -0,0 +1,39 @@ +"""Manifest-loader policy for local MIR corpus admission.""" + +from __future__ import annotations + +from pathlib import Path + +import pytest + +from test_structure_corpus_admission import _admission + + +def test_manifest_loader_uses_one_open_descriptor_after_path_resolution( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + """Size admission and JSON bytes must come from one already-open regular file.""" + admission = _admission() + manifest = tmp_path / "manifest.json" + manifest.write_text( + '{"schema_version":1,"registration_sha256":"abc","tracks":[]}', + encoding="utf-8", + ) + + def reject_path_stat(_self: Path, *_args: object, **_kwargs: object) -> object: + pytest.fail("manifest loader must use fstat on its opened descriptor") + + def reject_path_read_text( + _self: Path, *_args: object, **_kwargs: object + ) -> str: + pytest.fail("manifest loader must not reopen the pathname for JSON bytes") + + monkeypatch.setattr(Path, "stat", reject_path_stat) + monkeypatch.setattr(Path, "read_text", reject_path_read_text) + + loaded = admission._load_json(manifest) + + assert loaded["schema_version"] == 1 + assert loaded["registration_sha256"] == "abc" + assert loaded["tracks"] == [] From ad5daf36a51b3db859ec25475b883c101a89a28e Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 22:07:38 +0900 Subject: [PATCH 069/216] fix(mir): load admission manifest from one descriptor --- scripts/research/verify_structure_corpus.py | 29 ++++++++++++++++----- 1 file changed, 22 insertions(+), 7 deletions(-) diff --git a/scripts/research/verify_structure_corpus.py b/scripts/research/verify_structure_corpus.py index 6ffed7658..a8ac9f1be 100644 --- a/scripts/research/verify_structure_corpus.py +++ b/scripts/research/verify_structure_corpus.py @@ -101,22 +101,37 @@ def _unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]: def _load_json(path: Path) -> Mapping[str, Any]: + """Load one bounded manifest from the same opened regular-file descriptor.""" + fd = _open_regular_file(path, "manifest JSON") try: - size = path.stat().st_size - except OSError as exc: - raise ValueError("manifest JSON could not be stat'ed") from exc - if size > MAX_MANIFEST_BYTES: + metadata = os.fstat(fd) + if metadata.st_size > MAX_MANIFEST_BYTES: + raise ValueError(f"manifest JSON exceeds {MAX_MANIFEST_BYTES} bytes") + os.lseek(fd, 0, os.SEEK_SET) + chunks: list[bytes] = [] + remaining = MAX_MANIFEST_BYTES + 1 + while remaining > 0: + chunk = os.read(fd, min(1024 * 1024, remaining)) + if not chunk: + break + chunks.append(chunk) + remaining -= len(chunk) + payload = b"".join(chunks) + finally: + os.close(fd) + + if len(payload) > MAX_MANIFEST_BYTES: raise ValueError(f"manifest JSON exceeds {MAX_MANIFEST_BYTES} bytes") try: - text = path.read_text(encoding="utf-8") - except (OSError, UnicodeError) as exc: + text = payload.decode("utf-8") + except UnicodeDecodeError as exc: raise ValueError("manifest JSON must be readable UTF-8") from exc value = json.loads( text, object_pairs_hook=_unique_object, parse_constant=_reject_constant, ) - return _mapping(value, str(path)) + return _mapping(value, "manifest JSON") def _sha256_file_descriptor(fd: int) -> str: From 0c4e1b434f3966fe72af7725e47e8b22ff56d52e Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 22:08:12 +0900 Subject: [PATCH 070/216] docs(mir): trace single-descriptor manifest admission --- docs/traceability/mir/structure-corpus-admission.md | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/docs/traceability/mir/structure-corpus-admission.md b/docs/traceability/mir/structure-corpus-admission.md index 2f840a3b2..4434b832f 100644 --- a/docs/traceability/mir/structure-corpus-admission.md +++ b/docs/traceability/mir/structure-corpus-admission.md @@ -15,7 +15,7 @@ A digest-only receipt is also not a measurement handoff. Earlier admission code ## Decision -`scripts/research/verify_structure_corpus.py` resolves a local manifest only at execution time. For every registered track it: +`scripts/research/verify_structure_corpus.py` resolves a local manifest only at execution time. The manifest itself is admitted as bounded UTF-8 JSON from one already-open regular-file descriptor: size metadata comes from `fstat`, JSON bytes are read from that same descriptor with a 2 MiB cap, and pathname-level stat/read reopening is not used after resolution. For every registered track it: - requires the manifest order and track IDs to exactly match the preregistered corpus; - opens audio and annotation inputs as regular files without following symlinks where the platform provides `O_NOFOLLOW`; @@ -41,12 +41,16 @@ A second gap remained: the verified normalized PCM was destroyed before any scie That handoff still exposed the process-owned annotation snapshot as a writable `BinaryIO`. The source path could no longer drift, but the runner itself could mutate the hashed snapshot and then compute MIR metrics from bytes that no longer matched `annotation_sha256`. RED `1e5aa12bd4965cfc709e35a626fa5727397c9a6d` requires such a mutation to invalidate admission. GREEN `61a74d6753f58069f4ff91c910dd237bd21c98fc` flushes and re-hashes the annotation snapshot after the consumer returns, failing closed before a receipt is emitted when the admitted bytes changed. +The manifest loader had a separate pathname TOCTOU: it called `Path.stat()` for the 2 MiB admission decision and then reopened the path with `Path.read_text()`. A rename or replacement between those calls could make the bounded metadata and parsed bytes refer to different files. RED `106d4e37e98a76f48f73db1ba1854209de873040` requires both size and JSON bytes to come from one already-open descriptor without pathname stat/read reopening. GREEN `ad5daf36a51b3db859ec25475b883c101a89a28e` now uses the regular-file admission helper, `fstat`, and a bounded descriptor read, with a second byte-count guard for concurrent growth. + Earlier RED `f3abb6489836fd775cf67ae48cdc2c320f64a518` -> GREEN `174c6d33b44ef8e20e7e923ad638d7489442a460` aligned runtime identity with the evidence validator by comparing Git commit and SHA-256 fields as case-insensitive hexadecimal identities while keeping version strings exact. ## Constraints and rejected alternatives Dereferencing `source_uri` was rejected. Provenance URI is evidence metadata and may identify licensed material that cannot be fetched by CI. Network retrieval would also turn a local-first experiment into a mutable external dependency. +Checking manifest size by pathname and then reopening the pathname for JSON was rejected. Size admission and parsed bytes must describe the same opened regular file. The loader therefore resolves once, uses descriptor metadata, and reads at most the configured limit plus one byte from that descriptor. + Hashing an opened source descriptor and then decoding that still-live source descriptor was rejected. A pathname cannot be swapped once the descriptor is open, but another writer can still change the underlying regular-file bytes between the hash and decode. Admission therefore snapshots the source bytes while hashing and decodes only that process-owned snapshot. Treating the durable receipt as the experiment input was rejected. The receipt proves identities but does not contain the admitted signal or annotations. Reopening source paths from a later process would create a second admission event and sever the claim that baseline and candidate consumed the exact PCM whose digest is recorded. The current safe boundary is therefore in-process consumption during admission; a future persistent artifact bundle would need its own atomic, content-addressed and resource-bounded contract before it could replace this boundary. @@ -65,7 +69,7 @@ MIREX 2025 Music Structure Analysis evaluates mono 44.1 kHz WAV input and uses f ## Test boundary -Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, source mutation after snapshot admission, exact PCM/annotation consumer handoff, and consumer-side annotation mutation detection. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. +Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise bounded single-descriptor manifest loading, hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, source mutation after snapshot admission, exact PCM/annotation consumer handoff, and consumer-side annotation mutation detection. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. The handoff regression deliberately mutates the original annotation file after its process-owned snapshot has been created and requires the consumer to observe the admitted bytes, not the mutated path content. It also requires the consumer's PCM bytes to hash to the digest emitted in the receipt. A separate hostile-case regression writes directly through the borrowed annotation snapshot and requires admission to fail instead of emitting a receipt for post-hash annotation bytes. From 986170737dc3b1a11a04abf47b90a868e93362e7 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 22:10:23 +0900 Subject: [PATCH 071/216] test(mir): reject dirty source identity for real-audio runs --- ...structure_corpus_source_identity_policy.py | 42 +++++++++++++++++++ 1 file changed, 42 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_corpus_source_identity_policy.py diff --git a/services/analysis-engine/tests/test_structure_corpus_source_identity_policy.py b/services/analysis-engine/tests/test_structure_corpus_source_identity_policy.py new file mode 100644 index 000000000..3b19051e4 --- /dev/null +++ b/services/analysis-engine/tests/test_structure_corpus_source_identity_policy.py @@ -0,0 +1,42 @@ +"""Source-identity policy for real-audio MIR corpus admission.""" + +from __future__ import annotations + +import subprocess +from pathlib import Path + +import pytest + +from test_structure_corpus_admission import _admission + + +def _run_git(repo: Path, *args: str) -> None: + """Run a local Git command for the isolated source-identity fixture.""" + subprocess.run( + ["git", *args], + cwd=repo, + check=True, + capture_output=True, + text=True, + timeout=10, + ) + + +def test_runtime_identity_rejects_uncommitted_source_drift(tmp_path: Path) -> None: + """Registered HEAD cannot identify an experiment executed from a dirty worktree.""" + admission = _admission() + repo = tmp_path / "repo" + repo.mkdir() + _run_git(repo, "init") + _run_git(repo, "config", "user.email", "bandscope-test@example.invalid") + _run_git(repo, "config", "user.name", "BandScope Test") + (repo / "uv.lock").write_text("lock-v1\n", encoding="utf-8") + source = repo / "analysis.py" + source.write_text("FEATURE = 'registered'\n", encoding="utf-8") + _run_git(repo, "add", "uv.lock", "analysis.py") + _run_git(repo, "commit", "-m", "fixture") + + source.write_text("FEATURE = 'uncommitted-drift'\n", encoding="utf-8") + + with pytest.raises(RuntimeError, match="working tree must be clean"): + admission._current_runtime_identity(repo) From 4e8f3fa277a411a45f012daa5083f9ea5193cd17 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 22:11:08 +0900 Subject: [PATCH 072/216] fix(mir): require clean registered source worktree --- scripts/research/verify_structure_corpus.py | 22 +++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/scripts/research/verify_structure_corpus.py b/scripts/research/verify_structure_corpus.py index a8ac9f1be..d1e6a2be1 100644 --- a/scripts/research/verify_structure_corpus.py +++ b/scripts/research/verify_structure_corpus.py @@ -233,6 +233,28 @@ def _current_runtime_identity(repo_root: Path) -> dict[str, object]: ) if completed.returncode != 0: raise RuntimeError("git rev-parse HEAD failed") + + status = subprocess.run( + [ + "git", + "status", + "--porcelain=v1", + "--untracked-files=all", + "--ignore-submodules=none", + ], + cwd=repo_root, + check=False, + capture_output=True, + text=True, + timeout=10, + ) + if status.returncode != 0: + raise RuntimeError("git status --porcelain failed") + if status.stdout.strip(): + raise RuntimeError( + "git working tree must be clean for registered source identity" + ) + lock_path = repo_root / "uv.lock" with lock_path.open("rb") as lock_file: lock_digest = hashlib.file_digest(lock_file, "sha256").hexdigest() From 0e5ed41878c2584557d1a94be732b830295f011f Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 22:11:40 +0900 Subject: [PATCH 073/216] docs(mir): bind experiments to clean registered source --- .../mir/structure-corpus-admission.md | 17 +++++++++++++---- 1 file changed, 13 insertions(+), 4 deletions(-) diff --git a/docs/traceability/mir/structure-corpus-admission.md b/docs/traceability/mir/structure-corpus-admission.md index 4434b832f..5c23d3270 100644 --- a/docs/traceability/mir/structure-corpus-admission.md +++ b/docs/traceability/mir/structure-corpus-admission.md @@ -15,13 +15,17 @@ A digest-only receipt is also not a measurement handoff. Earlier admission code ## Decision -`scripts/research/verify_structure_corpus.py` resolves a local manifest only at execution time. The manifest itself is admitted as bounded UTF-8 JSON from one already-open regular-file descriptor: size metadata comes from `fstat`, JSON bytes are read from that same descriptor with a 2 MiB cap, and pathname-level stat/read reopening is not used after resolution. For every registered track it: +`scripts/research/verify_structure_corpus.py` resolves a local manifest only at execution time. The manifest itself is admitted as bounded UTF-8 JSON from one already-open regular-file descriptor: size metadata comes from `fstat`, JSON bytes are read from that same descriptor with a 2 MiB cap, and pathname-level stat/read reopening is not used after resolution. + +Runtime source identity is admissible only from a clean Git worktree. `git rev-parse HEAD` alone is insufficient because tracked, staged, or untracked non-ignored files can alter the code or imports used by a scientific run without changing `HEAD`. Admission therefore rejects any non-empty `git status --porcelain=v1 --untracked-files=all --ignore-submodules=none` before binding the registered source commit and `uv.lock` identity. Local corpus/manifest/output material must live outside the repository or be intentionally ignored if it is not source evidence. + +For every registered track the admission tool: - requires the manifest order and track IDs to exactly match the preregistered corpus; - opens audio and annotation inputs as regular files without following symlinks where the platform provides `O_NOFOLLOW`; - copies the opened audio stream into a process-owned temporary snapshot while computing the SHA-256, then compares that digest with the preregistration before decoding; - snapshots and hashes annotation bytes from the opened annotation descriptor and compares them with the registered annotation identity; -- requires the current source commit, `uv.lock`, Python, librosa, and NumPy identities to match the registered runtime; +- requires the clean current source commit, `uv.lock`, Python, librosa, and NumPy identities to match the registered runtime; - treats Git commit and SHA-256 values as case-insensitive hexadecimal identities and emits them lowercase, matching the validator contract, while Python/librosa/NumPy version strings remain exact; - decodes the immutable admitted audio snapshot through `librosa.load(..., sr=, mono=True)`, converts it to canonical little-endian float32, marks that NumPy array read-only, and computes the PCM SHA-256 over the same byte view; - optionally invokes an in-process `track_consumer` only after both registered audio and annotation identities have passed. The callback receives that exact canonical PCM byte view, the process-owned annotation snapshot, track ID, and sample rate. It receives no workstation path. Because the temporary annotation handle is writable internally, admission re-hashes it after the callback and rejects the run if the consumer changed any admitted annotation byte; @@ -43,6 +47,8 @@ That handoff still exposed the process-owned annotation snapshot as a writable ` The manifest loader had a separate pathname TOCTOU: it called `Path.stat()` for the 2 MiB admission decision and then reopened the path with `Path.read_text()`. A rename or replacement between those calls could make the bounded metadata and parsed bytes refer to different files. RED `106d4e37e98a76f48f73db1ba1854209de873040` requires both size and JSON bytes to come from one already-open descriptor without pathname stat/read reopening. GREEN `ad5daf36a51b3db859ec25475b883c101a89a28e` now uses the regular-file admission helper, `fstat`, and a bounded descriptor read, with a second byte-count guard for concurrent growth. +Source identity then exposed a different reproducibility gap. `git rev-parse HEAD` can report the registered commit while the worktree contains modified or additional executable source. RED `986170737dc3b1a11a04abf47b90a868e93362e7` creates an isolated Git repository, modifies a tracked analysis file after commit, and requires runtime admission to reject it. GREEN `4e8f3fa277a411a45f012daa5083f9ea5193cd17` requires a clean porcelain status before the source commit can enter the receipt. + Earlier RED `f3abb6489836fd775cf67ae48cdc2c320f64a518` -> GREEN `174c6d33b44ef8e20e7e923ad638d7489442a460` aligned runtime identity with the evidence validator by comparing Git commit and SHA-256 fields as case-insensitive hexadecimal identities while keeping version strings exact. ## Constraints and rejected alternatives @@ -51,6 +57,8 @@ Dereferencing `source_uri` was rejected. Provenance URI is evidence metadata and Checking manifest size by pathname and then reopening the pathname for JSON was rejected. Size admission and parsed bytes must describe the same opened regular file. The loader therefore resolves once, uses descriptor metadata, and reads at most the configured limit plus one byte from that descriptor. +Treating `HEAD` as sufficient source evidence while allowing a dirty worktree was rejected. The registered commit must describe the source actually executed. Uncommitted tracked changes, staged changes, and untracked non-ignored files therefore fail admission rather than being silently attributed to the registered commit. + Hashing an opened source descriptor and then decoding that still-live source descriptor was rejected. A pathname cannot be swapped once the descriptor is open, but another writer can still change the underlying regular-file bytes between the hash and decode. Admission therefore snapshots the source bytes while hashing and decodes only that process-owned snapshot. Treating the durable receipt as the experiment input was rejected. The receipt proves identities but does not contain the admitted signal or annotations. Reopening source paths from a later process would create a second admission event and sever the claim that baseline and candidate consumed the exact PCM whose digest is recorded. The current safe boundary is therefore in-process consumption during admission; a future persistent artifact bundle would need its own atomic, content-addressed and resource-bounded contract before it could replace this boundary. @@ -69,7 +77,7 @@ MIREX 2025 Music Structure Analysis evaluates mono 44.1 kHz WAV input and uses f ## Test boundary -Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise bounded single-descriptor manifest loading, hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, source mutation after snapshot admission, exact PCM/annotation consumer handoff, and consumer-side annotation mutation detection. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. +Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise bounded single-descriptor manifest loading, clean-source identity, hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, source mutation after snapshot admission, exact PCM/annotation consumer handoff, and consumer-side annotation mutation detection. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. The handoff regression deliberately mutates the original annotation file after its process-owned snapshot has been created and requires the consumer to observe the admitted bytes, not the mutated path content. It also requires the consumer's PCM bytes to hash to the digest emitted in the receipt. A separate hostile-case regression writes directly through the borrowed annotation snapshot and requires admission to fail instead of emitting a receipt for post-hash annotation bytes. @@ -77,9 +85,10 @@ The handoff regression deliberately mutates the original annotation file after i A reviewed rights-cleared real corpus and independent annotation manifest are still required. Before candidate results are inspected, the actual aggregation, paired uncertainty, margins, latency threshold, dependence assumptions, claim boundary, and any failure/exclusion rule must be approved. -The next runner must execute through the in-process admitted-track consumer boundary, interpret the admitted annotation snapshot without mutating it, run CQT and STFT on the exact same read-only PCM, calculate the recognized MIREX/mir_eval track metrics and latency/memory measurements, and derive aggregate/CI evidence from the preregistered procedure rather than accepting caller-authored summaries. The standalone receipt is evidence for that run, not a license to reopen media after admission. +The next runner must execute from the exact clean registered source/lock identity through the in-process admitted-track consumer boundary, interpret the admitted annotation snapshot without mutating it, run CQT and STFT on the exact same read-only PCM, calculate the recognized MIREX/mir_eval track metrics and latency/memory measurements, and derive aggregate/CI evidence from the preregistered procedure rather than accepting caller-authored summaries. The standalone receipt is evidence for that run, not a license to reopen media after admission. ## References +- MITRE. (2026). *CWE-367: Time-of-check Time-of-use (TOCTOU) Race Condition*. https://cwe.mitre.org/data/definitions/367.html - MIREX. (2025). *Music Structure Analysis*. International Music Information Retrieval Systems Evaluation Laboratory. https://music-ir.org/mirex/wiki/2025%3AMusic_Structure_Analysis - McFee, B., et al. (2025). *librosa 0.11.0 documentation: Core IO and DSP*. https://librosa.org/doc/0.11.0/core.html From 5654096aeb4f8112eef4d613a5189b94ae12ac4a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 22:12:34 +0900 Subject: [PATCH 074/216] test(mir): consolidate runtime source-identity policy --- ...tructure_corpus_runtime_identity_policy.py | 35 +++++++++++++++++++ 1 file changed, 35 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_corpus_runtime_identity_policy.py b/services/analysis-engine/tests/test_structure_corpus_runtime_identity_policy.py index 01175b48d..649ce06e0 100644 --- a/services/analysis-engine/tests/test_structure_corpus_runtime_identity_policy.py +++ b/services/analysis-engine/tests/test_structure_corpus_runtime_identity_policy.py @@ -2,8 +2,11 @@ from __future__ import annotations +import subprocess +from pathlib import Path from types import ModuleType +import pytest from conftest import load_module @@ -14,6 +17,18 @@ def _admission() -> ModuleType: ) +def _run_git(repo: Path, *args: str) -> None: + """Run a local Git command for the isolated source-identity fixture.""" + subprocess.run( + ["git", *args], + cwd=repo, + check=True, + capture_output=True, + text=True, + timeout=10, + ) + + def test_runtime_digest_identity_is_hex_case_insensitive() -> None: """Validator-accepted uppercase hex must match lowercase runtime evidence.""" admission = _admission() @@ -38,3 +53,23 @@ def test_runtime_digest_identity_is_hex_case_insensitive() -> None: assert normalized["source_commit"] == "a" * 40 assert normalized["uv_lock_sha256"] == "b" * 64 + + +def test_runtime_identity_rejects_uncommitted_source_drift(tmp_path: Path) -> None: + """Registered HEAD cannot identify an experiment executed from a dirty worktree.""" + admission = _admission() + repo = tmp_path / "repo" + repo.mkdir() + _run_git(repo, "init") + _run_git(repo, "config", "user.email", "bandscope-test@example.invalid") + _run_git(repo, "config", "user.name", "BandScope Test") + (repo / "uv.lock").write_text("lock-v1\n", encoding="utf-8") + source = repo / "analysis.py" + source.write_text("FEATURE = 'registered'\n", encoding="utf-8") + _run_git(repo, "add", "uv.lock", "analysis.py") + _run_git(repo, "commit", "-m", "fixture") + + source.write_text("FEATURE = 'uncommitted-drift'\n", encoding="utf-8") + + with pytest.raises(RuntimeError, match="working tree must be clean"): + admission._current_runtime_identity(repo) From bc0eb4061f8652d9af9bfac8062c7d72ff6b839c Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 22:12:45 +0900 Subject: [PATCH 075/216] test(mir): consolidate source identity into runtime policy --- ...structure_corpus_source_identity_policy.py | 42 ------------------- 1 file changed, 42 deletions(-) delete mode 100644 services/analysis-engine/tests/test_structure_corpus_source_identity_policy.py diff --git a/services/analysis-engine/tests/test_structure_corpus_source_identity_policy.py b/services/analysis-engine/tests/test_structure_corpus_source_identity_policy.py deleted file mode 100644 index 3b19051e4..000000000 --- a/services/analysis-engine/tests/test_structure_corpus_source_identity_policy.py +++ /dev/null @@ -1,42 +0,0 @@ -"""Source-identity policy for real-audio MIR corpus admission.""" - -from __future__ import annotations - -import subprocess -from pathlib import Path - -import pytest - -from test_structure_corpus_admission import _admission - - -def _run_git(repo: Path, *args: str) -> None: - """Run a local Git command for the isolated source-identity fixture.""" - subprocess.run( - ["git", *args], - cwd=repo, - check=True, - capture_output=True, - text=True, - timeout=10, - ) - - -def test_runtime_identity_rejects_uncommitted_source_drift(tmp_path: Path) -> None: - """Registered HEAD cannot identify an experiment executed from a dirty worktree.""" - admission = _admission() - repo = tmp_path / "repo" - repo.mkdir() - _run_git(repo, "init") - _run_git(repo, "config", "user.email", "bandscope-test@example.invalid") - _run_git(repo, "config", "user.name", "BandScope Test") - (repo / "uv.lock").write_text("lock-v1\n", encoding="utf-8") - source = repo / "analysis.py" - source.write_text("FEATURE = 'registered'\n", encoding="utf-8") - _run_git(repo, "add", "uv.lock", "analysis.py") - _run_git(repo, "commit", "-m", "fixture") - - source.write_text("FEATURE = 'uncommitted-drift'\n", encoding="utf-8") - - with pytest.raises(RuntimeError, match="working tree must be clean"): - admission._current_runtime_identity(repo) From 59cb9dc4c41b4208a0c6e25d9365e924061c98a7 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 23:00:09 +0900 Subject: [PATCH 076/216] test(mir): require immutable admitted annotations --- ...ucture_corpus_consumer_integrity_policy.py | 42 +++++++++++-------- 1 file changed, 24 insertions(+), 18 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py b/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py index b87c61e6f..e159c3512 100644 --- a/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py +++ b/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py @@ -5,7 +5,6 @@ import hashlib import os from pathlib import Path -from typing import BinaryIO import pytest @@ -19,10 +18,10 @@ ) -def test_consumer_cannot_mutate_admitted_annotation_without_detection( +def test_consumer_receives_intrinsically_read_only_annotation_bytes( tmp_path: Path, ) -> None: - """Receipt admission must fail if a consumer mutates the hashed annotation snapshot.""" + """A consumer must not be able to mutate annotation evidence then restore it.""" admission = _admission() audio_paths, annotation_paths = _files(tmp_path) registration = _registration( @@ -31,28 +30,35 @@ def test_consumer_cannot_mutate_admitted_annotation_without_detection( ) manifest = _manifest(admission, registration, audio_paths, annotation_paths) pcm = memoryview(b"\x00\x00\x00\x00") + consumed_annotations: list[bytes] = [] def fake_decoder(fd: int, sample_rate_hz: int) -> tuple[str, int, memoryview]: assert sample_rate_hz == 44100 assert os.read(fd, 1) return hashlib.sha256(pcm).hexdigest(), 1, pcm - def mutating_consumer( + def restoring_attack_consumer( _track_id: str, _pcm: memoryview, - annotation_snapshot: BinaryIO, + annotation_bytes: memoryview, _sample_rate_hz: int, ) -> None: - annotation_snapshot.seek(0) - annotation_snapshot.write(b"tampered-after-admission") - annotation_snapshot.truncate() - annotation_snapshot.flush() - - with pytest.raises(ValueError, match="consumer mutated admitted annotation"): - admission.verify_corpus( - registration, - manifest, - runtime_identity=_runtime(), - decoder=fake_decoder, - track_consumer=mutating_consumer, - ) + assert annotation_bytes.readonly + original = bytes(annotation_bytes) + with pytest.raises(TypeError): + annotation_bytes[0] = (annotation_bytes[0] + 1) % 256 + assert bytes(annotation_bytes) == original + consumed_annotations.append(original) + + receipt = admission.verify_corpus( + registration, + manifest, + runtime_identity=_runtime(), + decoder=fake_decoder, + track_consumer=restoring_attack_consumer, + ) + + assert consumed_annotations == [path.read_bytes() for path in annotation_paths] + assert [track["annotation_sha256"] for track in receipt["tracks"]] == [ + _digest(path) for path in annotation_paths + ] From 62d7d0b601503e5f83b872d8bd3f6ac202da95fc Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 23:01:55 +0900 Subject: [PATCH 077/216] fix(mir): make admitted annotations intrinsically immutable --- scripts/research/verify_structure_corpus.py | 32 +++++++++------------ 1 file changed, 14 insertions(+), 18 deletions(-) diff --git a/scripts/research/verify_structure_corpus.py b/scripts/research/verify_structure_corpus.py index d1e6a2be1..1ac5c8fcf 100644 --- a/scripts/research/verify_structure_corpus.py +++ b/scripts/research/verify_structure_corpus.py @@ -9,10 +9,11 @@ SHA-256 identity of the exact mono float32 PCM presented to later analysis. An in-process track consumer may receive that exact normalized PCM together -with the admitted annotation snapshot. This is the handoff boundary for a later -MIR experiment runner: the runner must not reopen workstation source paths after -admission merely because the durable receipt contains only digests. Any -consumer-side mutation of the admitted annotation snapshot invalidates the run. +with an immutable read-only view of the admitted annotation bytes. This is the +handoff boundary for a later MIR experiment runner: the runner must not reopen +workstation source paths after admission merely because the durable receipt +contains only digests, and it must not be able to mutate annotation evidence +before calculating metrics. The tool does not calculate MIR metrics, choose thresholds, or make a noninferiority decision. Synthetic audio is suitable for unit tests only; @@ -295,16 +296,15 @@ def verify_corpus( [int, int], tuple[str, int] | tuple[str, int, memoryview], ] = _decode_pcm_identity, - track_consumer: Callable[[str, memoryview, BinaryIO, int], None] | None = None, + track_consumer: Callable[[str, memoryview, memoryview, int], None] | None = None, ) -> dict[str, object]: """Verify local files and return a path-free decoded-corpus receipt. When ``track_consumer`` is supplied, it runs only after both registered content identities are verified. It receives the exact canonical PCM used - for ``decoded_pcm_sha256`` plus a borrowed annotation snapshot. The callback - must finish before this function returns; mutation of that snapshot is - detected and invalidates admission, and the handle is closed immediately - afterward rather than persisted in the receipt. + for ``decoded_pcm_sha256`` plus a read-only memory view over immutable + annotation bytes. The callback must finish before this function returns; + neither measurement input can be changed through the consumer boundary. """ validator = _load_validator() validator.validate_registration(registration) @@ -383,20 +383,16 @@ def verify_corpus( "decoder must expose admitted PCM when track_consumer is configured" ) annotation_snapshot.seek(0) + annotation_bytes = annotation_snapshot.read() + annotation_view = memoryview(annotation_bytes) + if not annotation_view.readonly: + raise RuntimeError("annotation handoff must be intrinsically read-only") track_consumer( track_id, decoded_pcm, - annotation_snapshot, + annotation_view, target_sample_rate_hz, ) - annotation_snapshot.flush() - if ( - _sha256_file_descriptor(annotation_snapshot.fileno()) - != annotation_sha256 - ): - raise ValueError( - f"{field} consumer mutated admitted annotation snapshot" - ) finally: annotation_snapshot.close() From 4da0bd97b761dd7d098041daec7a32f7606840f4 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 23:02:25 +0900 Subject: [PATCH 078/216] test(mir): consume immutable annotation views --- .../tests/test_structure_corpus_admission.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_corpus_admission.py b/services/analysis-engine/tests/test_structure_corpus_admission.py index aa81f27dc..b3cdef7c8 100644 --- a/services/analysis-engine/tests/test_structure_corpus_admission.py +++ b/services/analysis-engine/tests/test_structure_corpus_admission.py @@ -247,7 +247,7 @@ def test_manifest_loader_rejects_duplicate_keys_and_nonstandard_numbers(tmp_path def test_admission_hands_exact_pcm_and_annotation_snapshot_to_consumer(tmp_path: Path) -> None: - """A runner must consume the admitted signal and annotation, not reopen source paths.""" + """A runner must consume admitted signal/annotation bytes, not reopen source paths.""" admission = _admission() audio_paths, annotation_paths = _files(tmp_path) original_annotations = [path.read_bytes() for path in annotation_paths] @@ -271,14 +271,14 @@ def fake_decoder(fd: int, sample_rate_hz: int) -> tuple[str, int, memoryview]: def consume_track( track_id: str, pcm: memoryview, - annotation_snapshot: object, + annotation_bytes: memoryview, sample_rate_hz: int, ) -> None: index = int(track_id[-1]) - 1 annotation_paths[index].write_bytes(b"mutated-after-admission") - annotation_snapshot.seek(0) + assert annotation_bytes.readonly consumed.append( - (track_id, bytes(pcm), annotation_snapshot.read(), sample_rate_hz) + (track_id, bytes(pcm), bytes(annotation_bytes), sample_rate_hz) ) receipt = admission.verify_corpus( From 8731ffb78cc810c2c3d16cde95ddf938e8088413 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 23:03:18 +0900 Subject: [PATCH 079/216] docs(mir): record immutable annotation handoff --- .../traceability/mir/structure-corpus-admission.md | 14 ++++++++------ 1 file changed, 8 insertions(+), 6 deletions(-) diff --git a/docs/traceability/mir/structure-corpus-admission.md b/docs/traceability/mir/structure-corpus-admission.md index 5c23d3270..f6413a6c2 100644 --- a/docs/traceability/mir/structure-corpus-admission.md +++ b/docs/traceability/mir/structure-corpus-admission.md @@ -28,7 +28,7 @@ For every registered track the admission tool: - requires the clean current source commit, `uv.lock`, Python, librosa, and NumPy identities to match the registered runtime; - treats Git commit and SHA-256 values as case-insensitive hexadecimal identities and emits them lowercase, matching the validator contract, while Python/librosa/NumPy version strings remain exact; - decodes the immutable admitted audio snapshot through `librosa.load(..., sr=, mono=True)`, converts it to canonical little-endian float32, marks that NumPy array read-only, and computes the PCM SHA-256 over the same byte view; -- optionally invokes an in-process `track_consumer` only after both registered audio and annotation identities have passed. The callback receives that exact canonical PCM byte view, the process-owned annotation snapshot, track ID, and sample rate. It receives no workstation path. Because the temporary annotation handle is writable internally, admission re-hashes it after the callback and rejects the run if the consumer changed any admitted annotation byte; +- optionally invokes an in-process `track_consumer` only after both registered audio and annotation identities have passed. The callback receives that exact canonical PCM byte view and a `memoryview` over immutable `bytes` copied from the admitted annotation snapshot, plus track ID and sample rate. It receives no workstation path and has no write capability over either scientific input; - emits only registration identity, runtime identity, content digests, decoded PCM digest/frame count, sample rate, channel count, and track ID. Local audio/annotation paths are never copied into the receipt. The standalone CLI still writes only the path-free receipt. A scientific runner that claims "same admitted PCM" must use the in-process consumer boundary, or a future equivalently strong content-addressed handoff, rather than reading the receipt and reopening source paths afterward. @@ -43,7 +43,9 @@ RED `1f653439a31b27ff905dd2e825a99f79188598d6` required post-admission source mu A second gap remained: the verified normalized PCM was destroyed before any scientific consumer could use it. RED `af927bbd429647f5cb655ec488ea6e7f16f8a7a2` requires a consumer to receive the exact admitted PCM and annotation snapshot, and verifies that mutating the original annotation path after admission cannot change what the consumer reads. GREEN `813dd925ec80d66e23a57c03ff52c143d569f9c7` exposes that in-process handoff while preserving the path-free durable receipt and legacy hash-only verifier use. -That handoff still exposed the process-owned annotation snapshot as a writable `BinaryIO`. The source path could no longer drift, but the runner itself could mutate the hashed snapshot and then compute MIR metrics from bytes that no longer matched `annotation_sha256`. RED `1e5aa12bd4965cfc709e35a626fa5727397c9a6d` requires such a mutation to invalidate admission. GREEN `61a74d6753f58069f4ff91c910dd237bd21c98fc` flushes and re-hashes the annotation snapshot after the consumer returns, failing closed before a receipt is emitted when the admitted bytes changed. +That handoff initially exposed the process-owned annotation snapshot as a writable `BinaryIO`. RED `1e5aa12bd4965cfc709e35a626fa5727397c9a6d` required persistent mutation to invalidate admission, and GREEN `61a74d6753f58069f4ff91c910dd237bd21c98fc` re-hashed the snapshot after the callback. Hostile review found that this was still not an integrity boundary: a consumer could mutate annotation bytes, calculate metrics from the modified content, then restore the original bytes before returning, causing the post-callback digest to match while the measurements were already contaminated. + +RED `59cb9dc4c41b4208a0c6e25d9365e924061c98a7` therefore requires the consumer-facing annotation object itself to be intrinsically read-only and exercises the mutate/use/restore attack shape. GREEN `62d7d0b601503e5f83b872d8bd3f6ac202da95fc` removes write capability from the callback boundary by passing a `memoryview` over immutable admitted annotation `bytes` instead of the temporary writable handle. `4da0bd97b761dd7d098041daec7a32f7606840f4` updates the end-to-end handoff regression to consume those immutable bytes while still proving that later mutation of the workstation annotation path cannot affect the admitted input. The manifest loader had a separate pathname TOCTOU: it called `Path.stat()` for the 2 MiB admission decision and then reopened the path with `Path.read_text()`. A rename or replacement between those calls could make the bounded metadata and parsed bytes refer to different files. RED `106d4e37e98a76f48f73db1ba1854209de873040` requires both size and JSON bytes to come from one already-open descriptor without pathname stat/read reopening. GREEN `ad5daf36a51b3db859ec25475b883c101a89a28e` now uses the regular-file admission helper, `fstat`, and a bounded descriptor read, with a second byte-count guard for concurrent growth. @@ -63,7 +65,7 @@ Hashing an opened source descriptor and then decoding that still-live source des Treating the durable receipt as the experiment input was rejected. The receipt proves identities but does not contain the admitted signal or annotations. Reopening source paths from a later process would create a second admission event and sever the claim that baseline and candidate consumed the exact PCM whose digest is recorded. The current safe boundary is therefore in-process consumption during admission; a future persistent artifact bundle would need its own atomic, content-addressed and resource-bounded contract before it could replace this boundary. -Treating a writable temporary annotation handle as intrinsically immutable was rejected. The current callback API remains file-like for streaming parsers, but the snapshot digest is checked again after callback execution. A future read-only snapshot abstraction may tighten capability exposure further; until then, any consumer mutation invalidates the run and produces no admitted receipt. +Re-hashing a writable annotation handle only after consumer execution was rejected as insufficient. Integrity-after-return does not prove integrity-during-measurement: a consumer can mutate, use, and restore bytes before the check. The callback therefore receives an immutable `bytes`-backed read-only `memoryview`, so the capability to alter annotation evidence is absent rather than detected after the fact. Persisting workstation paths in the receipt was rejected because they are neither stable provenance nor purpose-bound evidence and can expose local usernames, mounts, or project layout. @@ -77,15 +79,15 @@ MIREX 2025 Music Structure Analysis evaluates mono 44.1 kHz WAV input and uses f ## Test boundary -Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise bounded single-descriptor manifest loading, clean-source identity, hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, source mutation after snapshot admission, exact PCM/annotation consumer handoff, and consumer-side annotation mutation detection. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. +Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise bounded single-descriptor manifest loading, clean-source identity, hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, source mutation after snapshot admission, exact PCM/annotation consumer handoff, and intrinsic annotation immutability. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. -The handoff regression deliberately mutates the original annotation file after its process-owned snapshot has been created and requires the consumer to observe the admitted bytes, not the mutated path content. It also requires the consumer's PCM bytes to hash to the digest emitted in the receipt. A separate hostile-case regression writes directly through the borrowed annotation snapshot and requires admission to fail instead of emitting a receipt for post-hash annotation bytes. +The handoff regression deliberately mutates the original annotation file after its process-owned snapshot has been created and requires the consumer to observe the admitted bytes, not the mutated path content. It also requires the consumer's PCM bytes to hash to the digest emitted in the receipt. The hostile-case regression requires the consumer-facing annotation view to report `readonly` and rejects the earlier mutate/use/restore capability by making writes fail at the boundary instead of relying on a later digest comparison. ## Remaining scientific work A reviewed rights-cleared real corpus and independent annotation manifest are still required. Before candidate results are inspected, the actual aggregation, paired uncertainty, margins, latency threshold, dependence assumptions, claim boundary, and any failure/exclusion rule must be approved. -The next runner must execute from the exact clean registered source/lock identity through the in-process admitted-track consumer boundary, interpret the admitted annotation snapshot without mutating it, run CQT and STFT on the exact same read-only PCM, calculate the recognized MIREX/mir_eval track metrics and latency/memory measurements, and derive aggregate/CI evidence from the preregistered procedure rather than accepting caller-authored summaries. The standalone receipt is evidence for that run, not a license to reopen media after admission. +The next runner must execute from the exact clean registered source/lock identity through the in-process admitted-track consumer boundary, interpret the admitted read-only annotation bytes, run CQT and STFT on the exact same read-only PCM, calculate the recognized MIREX/mir_eval track metrics and latency/memory measurements, and derive aggregate/CI evidence from the preregistered procedure rather than accepting caller-authored summaries. The standalone receipt is evidence for that run, not a license to reopen media after admission. ## References From e6a62f393daf5ca1fb7e44271777dbac6f1956e1 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 23:03:58 +0900 Subject: [PATCH 080/216] test(mir): bind immutable PCM handoff to receipt digest --- ...ucture_corpus_consumer_integrity_policy.py | 74 +++++++++++++++++-- 1 file changed, 68 insertions(+), 6 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py b/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py index e159c3512..cb885080d 100644 --- a/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py +++ b/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py @@ -18,10 +18,7 @@ ) -def test_consumer_receives_intrinsically_read_only_annotation_bytes( - tmp_path: Path, -) -> None: - """A consumer must not be able to mutate annotation evidence then restore it.""" +def _registered_inputs(tmp_path: Path) -> tuple[object, dict[str, object], dict[str, object]]: admission = _admission() audio_paths, annotation_paths = _files(tmp_path) registration = _registration( @@ -29,6 +26,14 @@ def test_consumer_receives_intrinsically_read_only_annotation_bytes( [_digest(path) for path in annotation_paths], ) manifest = _manifest(admission, registration, audio_paths, annotation_paths) + return admission, registration, manifest + + +def test_consumer_receives_intrinsically_read_only_annotation_bytes( + tmp_path: Path, +) -> None: + """A consumer must not be able to mutate annotation evidence then restore it.""" + admission, registration, manifest = _registered_inputs(tmp_path) pcm = memoryview(b"\x00\x00\x00\x00") consumed_annotations: list[bytes] = [] @@ -44,6 +49,7 @@ def restoring_attack_consumer( _sample_rate_hz: int, ) -> None: assert annotation_bytes.readonly + assert isinstance(annotation_bytes.obj, bytes) original = bytes(annotation_bytes) with pytest.raises(TypeError): annotation_bytes[0] = (annotation_bytes[0] + 1) % 256 @@ -58,7 +64,63 @@ def restoring_attack_consumer( track_consumer=restoring_attack_consumer, ) - assert consumed_annotations == [path.read_bytes() for path in annotation_paths] + assert len(consumed_annotations) == 2 assert [track["annotation_sha256"] for track in receipt["tracks"]] == [ - _digest(path) for path in annotation_paths + track["annotation_sha256"] for track in registration["corpus"] ] + + +def test_consumer_receives_immutable_pcm_even_if_decoder_exposes_mutable_memory( + tmp_path: Path, +) -> None: + """Decoder-owned mutable buffers must not cross the scientific handoff boundary.""" + admission, registration, manifest = _registered_inputs(tmp_path) + mutable_pcm = memoryview(bytearray(b"\x00\x00\x00\x00")) + + def fake_decoder(fd: int, sample_rate_hz: int) -> tuple[str, int, memoryview]: + assert sample_rate_hz == 44100 + assert os.read(fd, 1) + return hashlib.sha256(mutable_pcm).hexdigest(), 1, mutable_pcm + + def mutating_consumer( + _track_id: str, + pcm: memoryview, + _annotation_bytes: memoryview, + _sample_rate_hz: int, + ) -> None: + assert pcm.readonly + assert isinstance(pcm.obj, bytes) + with pytest.raises(TypeError): + pcm[0] = 255 + + admission.verify_corpus( + registration, + manifest, + runtime_identity=_runtime(), + decoder=fake_decoder, + track_consumer=mutating_consumer, + ) + + +def test_admission_rejects_decoder_digest_that_does_not_match_handoff_pcm( + tmp_path: Path, +) -> None: + """Receipt identity must be recomputed from the exact PCM handed to measurement.""" + admission, registration, manifest = _registered_inputs(tmp_path) + pcm = memoryview(b"\x00\x00\x00\x00") + + def lying_decoder(fd: int, sample_rate_hz: int) -> tuple[str, int, memoryview]: + assert sample_rate_hz == 44100 + assert os.read(fd, 1) + return "0" * 64, 1, pcm + + with pytest.raises(ValueError, match="decoded PCM SHA-256"): + admission.verify_corpus( + registration, + manifest, + runtime_identity=_runtime(), + decoder=lying_decoder, + track_consumer=lambda *_args: pytest.fail( + "consumer must not run when decoder identity is inconsistent" + ), + ) From 5b0660e49e4d01246cafeea988e2b8c222713be9 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 23:04:45 +0900 Subject: [PATCH 081/216] fix(mir): bind immutable PCM handoff to receipt identity --- scripts/research/verify_structure_corpus.py | 49 ++++++++++++++------- 1 file changed, 32 insertions(+), 17 deletions(-) diff --git a/scripts/research/verify_structure_corpus.py b/scripts/research/verify_structure_corpus.py index 1ac5c8fcf..4516cdf6c 100644 --- a/scripts/research/verify_structure_corpus.py +++ b/scripts/research/verify_structure_corpus.py @@ -12,7 +12,7 @@ with an immutable read-only view of the admitted annotation bytes. This is the handoff boundary for a later MIR experiment runner: the runner must not reopen workstation source paths after admission merely because the durable receipt -contains only digests, and it must not be able to mutate annotation evidence +contains only digests, and it must not be able to mutate measurement inputs before calculating metrics. The tool does not calculate MIR metrics, choose thresholds, or make a @@ -135,18 +135,6 @@ def _load_json(path: Path) -> Mapping[str, Any]: return _mapping(value, "manifest JSON") -def _sha256_file_descriptor(fd: int) -> str: - digest = hashlib.sha256() - os.lseek(fd, 0, os.SEEK_SET) - while True: - chunk = os.read(fd, 1024 * 1024) - if not chunk: - break - digest.update(chunk) - os.lseek(fd, 0, os.SEEK_SET) - return digest.hexdigest() - - def _snapshot_and_hash(fd: int) -> tuple[BinaryIO, str]: """Copy one opened source into a process-owned snapshot while hashing it.""" digest = hashlib.sha256() @@ -220,6 +208,27 @@ def _decode_pcm_identity(fd: int, target_sample_rate_hz: int) -> tuple[str, int, return hashlib.sha256(pcm).hexdigest(), int(canonical.size), pcm +def _bind_decoded_pcm( + decoded_pcm_sha256: object, + decoded_frames: object, + decoded_pcm: memoryview, +) -> tuple[str, int, memoryview]: + """Bind receipt identity to an immutable copy of the exact PCM handoff.""" + pcm_bytes = bytes(decoded_pcm) + pcm_view = memoryview(pcm_bytes) + actual_digest = hashlib.sha256(pcm_view).hexdigest() + claimed_digest = _text(decoded_pcm_sha256, "decoder.decoded_pcm_sha256").lower() + if actual_digest != claimed_digest: + raise ValueError("decoded PCM SHA-256 does not match decoder handoff bytes") + try: + frame_count = int(decoded_frames) + except (TypeError, ValueError) as exc: + raise ValueError("decoded frame count must be an integer") from exc + if frame_count < 1 or len(pcm_view) != frame_count * 4: + raise ValueError("decoded frame count does not match mono float32 PCM bytes") + return actual_digest, frame_count, pcm_view + + def _current_runtime_identity(repo_root: Path) -> dict[str, object]: import librosa import numpy as np @@ -301,10 +310,10 @@ def verify_corpus( """Verify local files and return a path-free decoded-corpus receipt. When ``track_consumer`` is supplied, it runs only after both registered - content identities are verified. It receives the exact canonical PCM used - for ``decoded_pcm_sha256`` plus a read-only memory view over immutable - annotation bytes. The callback must finish before this function returns; - neither measurement input can be changed through the consumer boundary. + content identities are verified. It receives an immutable bytes-backed copy + of the exact canonical PCM whose digest/frame count enter the receipt plus a + read-only view over immutable annotation bytes. Neither measurement input can + be changed through the consumer boundary. """ validator = _load_validator() validator.validate_registration(registration) @@ -360,6 +369,12 @@ def verify_corpus( decoded = decoder(snapshot.fileno(), target_sample_rate_hz) decoded_pcm_sha256, decoded_frames = decoded[:2] decoded_pcm = decoded[2] if len(decoded) == 3 else None + if decoded_pcm is not None: + decoded_pcm_sha256, decoded_frames, decoded_pcm = _bind_decoded_pcm( + decoded_pcm_sha256, + decoded_frames, + decoded_pcm, + ) finally: snapshot.close() finally: From e6f3ba0205976fceaef845bb18ffbeb28b4588af Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 17 Sep 2026 23:05:28 +0900 Subject: [PATCH 082/216] docs(mir): bind PCM receipt identity to immutable handoff --- .../mir/structure-corpus-admission.md | 15 ++++++++++----- 1 file changed, 10 insertions(+), 5 deletions(-) diff --git a/docs/traceability/mir/structure-corpus-admission.md b/docs/traceability/mir/structure-corpus-admission.md index f6413a6c2..acc3d2b80 100644 --- a/docs/traceability/mir/structure-corpus-admission.md +++ b/docs/traceability/mir/structure-corpus-admission.md @@ -27,8 +27,9 @@ For every registered track the admission tool: - snapshots and hashes annotation bytes from the opened annotation descriptor and compares them with the registered annotation identity; - requires the clean current source commit, `uv.lock`, Python, librosa, and NumPy identities to match the registered runtime; - treats Git commit and SHA-256 values as case-insensitive hexadecimal identities and emits them lowercase, matching the validator contract, while Python/librosa/NumPy version strings remain exact; -- decodes the immutable admitted audio snapshot through `librosa.load(..., sr=, mono=True)`, converts it to canonical little-endian float32, marks that NumPy array read-only, and computes the PCM SHA-256 over the same byte view; -- optionally invokes an in-process `track_consumer` only after both registered audio and annotation identities have passed. The callback receives that exact canonical PCM byte view and a `memoryview` over immutable `bytes` copied from the admitted annotation snapshot, plus track ID and sample rate. It receives no workstation path and has no write capability over either scientific input; +- decodes the immutable admitted audio snapshot through `librosa.load(..., sr=, mono=True)`, converts it to canonical little-endian float32, and computes a PCM SHA-256; +- when a decoder exposes PCM bytes for scientific consumption, independently recomputes the digest and frame count from that exact byte sequence, rejects an inconsistent decoder claim, and copies the sequence into immutable `bytes` before the callback boundary; +- optionally invokes an in-process `track_consumer` only after both registered audio and annotation identities have passed. The callback receives a read-only `memoryview` over the immutable PCM bytes and another read-only `memoryview` over immutable admitted annotation bytes, plus track ID and sample rate. It receives no workstation path and has no write capability over either scientific input; - emits only registration identity, runtime identity, content digests, decoded PCM digest/frame count, sample rate, channel count, and track ID. Local audio/annotation paths are never copied into the receipt. The standalone CLI still writes only the path-free receipt. A scientific runner that claims "same admitted PCM" must use the in-process consumer boundary, or a future equivalently strong content-addressed handoff, rather than reading the receipt and reopening source paths afterward. @@ -47,6 +48,8 @@ That handoff initially exposed the process-owned annotation snapshot as a writab RED `59cb9dc4c41b4208a0c6e25d9365e924061c98a7` therefore requires the consumer-facing annotation object itself to be intrinsically read-only and exercises the mutate/use/restore attack shape. GREEN `62d7d0b601503e5f83b872d8bd3f6ac202da95fc` removes write capability from the callback boundary by passing a `memoryview` over immutable admitted annotation `bytes` instead of the temporary writable handle. `4da0bd97b761dd7d098041daec7a32f7606840f4` updates the end-to-end handoff regression to consume those immutable bytes while still proving that later mutation of the workstation annotation path cannot affect the admitted input. +A parallel PCM capability gap remained. The default decoder marked its NumPy array read-only, but the consumer boundary accepted whatever `memoryview` the decoder returned and trusted the decoder-supplied digest. A custom or future decoder could expose a mutable buffer, or report a digest that did not describe the bytes actually handed to CQT/STFT. RED `e6a62f393daf5ca1fb7e44271777dbac6f1956e1` requires the PCM handoff to be immutable even when the decoder owns mutable memory and requires admission to reject a decoder digest that disagrees with the handoff bytes. GREEN `5b0660e49e4d01246cafeea988e2b8c222713be9` recomputes SHA-256 and mono-float32 frame count from the exact exposed PCM, then copies it into immutable `bytes` before any scientific consumer runs. + The manifest loader had a separate pathname TOCTOU: it called `Path.stat()` for the 2 MiB admission decision and then reopened the path with `Path.read_text()`. A rename or replacement between those calls could make the bounded metadata and parsed bytes refer to different files. RED `106d4e37e98a76f48f73db1ba1854209de873040` requires both size and JSON bytes to come from one already-open descriptor without pathname stat/read reopening. GREEN `ad5daf36a51b3db859ec25475b883c101a89a28e` now uses the regular-file admission helper, `fstat`, and a bounded descriptor read, with a second byte-count guard for concurrent growth. Source identity then exposed a different reproducibility gap. `git rev-parse HEAD` can report the registered commit while the worktree contains modified or additional executable source. RED `986170737dc3b1a11a04abf47b90a868e93362e7` creates an isolated Git repository, modifies a tracked analysis file after commit, and requires runtime admission to reject it. GREEN `4e8f3fa277a411a45f012daa5083f9ea5193cd17` requires a clean porcelain status before the source commit can enter the receipt. @@ -67,6 +70,8 @@ Treating the durable receipt as the experiment input was rejected. The receipt p Re-hashing a writable annotation handle only after consumer execution was rejected as insufficient. Integrity-after-return does not prove integrity-during-measurement: a consumer can mutate, use, and restore bytes before the check. The callback therefore receives an immutable `bytes`-backed read-only `memoryview`, so the capability to alter annotation evidence is absent rather than detected after the fact. +Trusting a decoder-supplied PCM digest or a read-only flag on decoder-owned memory was rejected. A scientific receipt must identify the bytes actually given to the measurement code, and the consumer must not be able to recover a mutable backing object. Admission therefore derives the digest/frame count from the exposed byte sequence and crosses the consumer boundary only after copying it into immutable `bytes`. + Persisting workstation paths in the receipt was rejected because they are neither stable provenance nor purpose-bound evidence and can expose local usernames, mounts, or project layout. The decoded PCM digest does not by itself prove that two machines will decode a compressed source identically. It records the exact normalized signal used by one admission/consumer run. The preregistration already binds librosa/NumPy and the source/lock identity; if future production acceptance requires cross-decoder equivalence, the schema must explicitly bind the remaining decoder/backend contract rather than assuming it. @@ -79,15 +84,15 @@ MIREX 2025 Music Structure Analysis evaluates mono 44.1 kHz WAV input and uses f ## Test boundary -Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise bounded single-descriptor manifest loading, clean-source identity, hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, source mutation after snapshot admission, exact PCM/annotation consumer handoff, and intrinsic annotation immutability. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. +Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise bounded single-descriptor manifest loading, clean-source identity, hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, source mutation after snapshot admission, exact PCM/annotation consumer handoff, intrinsic annotation immutability, mutable-decoder PCM confinement, and decoder-digest mismatch rejection. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. -The handoff regression deliberately mutates the original annotation file after its process-owned snapshot has been created and requires the consumer to observe the admitted bytes, not the mutated path content. It also requires the consumer's PCM bytes to hash to the digest emitted in the receipt. The hostile-case regression requires the consumer-facing annotation view to report `readonly` and rejects the earlier mutate/use/restore capability by making writes fail at the boundary instead of relying on a later digest comparison. +The handoff regression deliberately mutates the original annotation file after its process-owned snapshot has been created and requires the consumer to observe the admitted bytes, not the mutated path content. The annotation hostile case makes writes fail at the boundary instead of relying on a later digest comparison. The PCM hostile cases start from mutable decoder-owned memory and from a deliberately false decoder digest; admission must convert the former to immutable bytes and reject the latter before a consumer can run. ## Remaining scientific work A reviewed rights-cleared real corpus and independent annotation manifest are still required. Before candidate results are inspected, the actual aggregation, paired uncertainty, margins, latency threshold, dependence assumptions, claim boundary, and any failure/exclusion rule must be approved. -The next runner must execute from the exact clean registered source/lock identity through the in-process admitted-track consumer boundary, interpret the admitted read-only annotation bytes, run CQT and STFT on the exact same read-only PCM, calculate the recognized MIREX/mir_eval track metrics and latency/memory measurements, and derive aggregate/CI evidence from the preregistered procedure rather than accepting caller-authored summaries. The standalone receipt is evidence for that run, not a license to reopen media after admission. +The next runner must execute from the exact clean registered source/lock identity through the in-process admitted-track consumer boundary, interpret the admitted read-only annotation bytes, run CQT and STFT on the exact same immutable PCM, calculate the recognized MIREX/mir_eval track metrics and latency/memory measurements, and derive aggregate/CI evidence from the preregistered procedure rather than accepting caller-authored summaries. The standalone receipt is evidence for that run, not a license to reopen media after admission. ## References From cd4818ff329a5214a152ec0a485b50177498b2e1 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 00:04:27 +0900 Subject: [PATCH 083/216] test(mir): reject unverifiable decoder receipts --- ...ucture_corpus_consumer_integrity_policy.py | 20 +++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py b/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py index cb885080d..9dc0af1cd 100644 --- a/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py +++ b/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py @@ -124,3 +124,23 @@ def lying_decoder(fd: int, sample_rate_hz: int) -> tuple[str, int, memoryview]: "consumer must not run when decoder identity is inconsistent" ), ) + + +def test_admission_rejects_decoder_without_verifiable_pcm( + tmp_path: Path, +) -> None: + """A receipt must not trust a decoder-claimed digest without the decoded PCM bytes.""" + admission, registration, manifest = _registered_inputs(tmp_path) + + def digest_only_decoder(fd: int, sample_rate_hz: int) -> tuple[str, int]: + assert sample_rate_hz == 44100 + assert os.read(fd, 1) + return hashlib.sha256(b"unexposed-pcm").hexdigest(), 1 + + with pytest.raises(ValueError, match="decoder must expose admitted PCM"): + admission.verify_corpus( + registration, + manifest, + runtime_identity=_runtime(), + decoder=digest_only_decoder, + ) From f6085084721478c4440eb723487e1696e8087a99 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 00:05:15 +0900 Subject: [PATCH 084/216] test(mir): expose PCM in admission fixture decoder --- .../tests/test_structure_corpus_admission.py | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_corpus_admission.py b/services/analysis-engine/tests/test_structure_corpus_admission.py index b3cdef7c8..eb0a457c2 100644 --- a/services/analysis-engine/tests/test_structure_corpus_admission.py +++ b/services/analysis-engine/tests/test_structure_corpus_admission.py @@ -146,12 +146,13 @@ def test_admission_hashes_actual_files_and_emits_no_local_paths(tmp_path: Path) manifest = _manifest(admission, registration, audio_paths, annotation_paths) decoded_digests: list[str] = [] - def fake_decoder(fd: int, sample_rate_hz: int) -> tuple[str, int]: + def fake_decoder(fd: int, sample_rate_hz: int) -> tuple[str, int, memoryview]: assert sample_rate_hz == 44100 assert os.read(fd, 1) - digest = hashlib.sha256(f"decoded-{len(decoded_digests)}".encode()).hexdigest() + pcm = memoryview(bytes([len(decoded_digests), 0, 0, 0])) + digest = hashlib.sha256(pcm).hexdigest() decoded_digests.append(digest) - return digest, 44100 + return digest, 1, pcm receipt = admission.verify_corpus( registration, From 4bd685ad500b53bb0a4d70da35139fd92a2aaa35 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 00:06:13 +0900 Subject: [PATCH 085/216] fix(mir): require verifiable PCM for admission receipts --- scripts/research/verify_structure_corpus.py | 28 +++++++++------------ 1 file changed, 12 insertions(+), 16 deletions(-) diff --git a/scripts/research/verify_structure_corpus.py b/scripts/research/verify_structure_corpus.py index 4516cdf6c..8fc42bd4a 100644 --- a/scripts/research/verify_structure_corpus.py +++ b/scripts/research/verify_structure_corpus.py @@ -301,10 +301,7 @@ def verify_corpus( manifest: Mapping[str, Any], *, runtime_identity: Mapping[str, object], - decoder: Callable[ - [int, int], - tuple[str, int] | tuple[str, int, memoryview], - ] = _decode_pcm_identity, + decoder: Callable[[int, int], tuple[str, int, memoryview]] = _decode_pcm_identity, track_consumer: Callable[[str, memoryview, memoryview, int], None] | None = None, ) -> dict[str, object]: """Verify local files and return a path-free decoded-corpus receipt. @@ -367,14 +364,17 @@ def verify_corpus( if audio_sha256 != expected_audio: raise ValueError(f"{field}.audio_path SHA-256 does not match registration") decoded = decoder(snapshot.fileno(), target_sample_rate_hz) - decoded_pcm_sha256, decoded_frames = decoded[:2] - decoded_pcm = decoded[2] if len(decoded) == 3 else None - if decoded_pcm is not None: - decoded_pcm_sha256, decoded_frames, decoded_pcm = _bind_decoded_pcm( - decoded_pcm_sha256, - decoded_frames, - decoded_pcm, - ) + try: + decoded_pcm_sha256, decoded_frames, decoded_pcm = decoded + except (TypeError, ValueError) as exc: + raise ValueError( + "decoder must expose admitted PCM with digest and frame count" + ) from exc + decoded_pcm_sha256, decoded_frames, decoded_pcm = _bind_decoded_pcm( + decoded_pcm_sha256, + decoded_frames, + decoded_pcm, + ) finally: snapshot.close() finally: @@ -393,10 +393,6 @@ def verify_corpus( raise ValueError(f"{field}.annotation_path SHA-256 does not match registration") if track_consumer is not None: - if decoded_pcm is None: - raise ValueError( - "decoder must expose admitted PCM when track_consumer is configured" - ) annotation_snapshot.seek(0) annotation_bytes = annotation_snapshot.read() annotation_view = memoryview(annotation_bytes) From 205ae80ac88c062904b19d01d2b22192fd72a68a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 00:07:06 +0900 Subject: [PATCH 086/216] docs(mir): close digest-only decoder evidence gap --- docs/traceability/mir/structure-corpus-admission.md | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/docs/traceability/mir/structure-corpus-admission.md b/docs/traceability/mir/structure-corpus-admission.md index acc3d2b80..c765cb120 100644 --- a/docs/traceability/mir/structure-corpus-admission.md +++ b/docs/traceability/mir/structure-corpus-admission.md @@ -28,7 +28,7 @@ For every registered track the admission tool: - requires the clean current source commit, `uv.lock`, Python, librosa, and NumPy identities to match the registered runtime; - treats Git commit and SHA-256 values as case-insensitive hexadecimal identities and emits them lowercase, matching the validator contract, while Python/librosa/NumPy version strings remain exact; - decodes the immutable admitted audio snapshot through `librosa.load(..., sr=, mono=True)`, converts it to canonical little-endian float32, and computes a PCM SHA-256; -- when a decoder exposes PCM bytes for scientific consumption, independently recomputes the digest and frame count from that exact byte sequence, rejects an inconsistent decoder claim, and copies the sequence into immutable `bytes` before the callback boundary; +- requires every decoder implementation to expose the exact decoded PCM bytes together with its claimed digest and frame count. Admission recomputes the digest and mono-float32 frame count from that byte sequence, rejects a digest-only decoder receipt or any inconsistency, and copies the sequence into immutable `bytes` before the callback boundary; - optionally invokes an in-process `track_consumer` only after both registered audio and annotation identities have passed. The callback receives a read-only `memoryview` over the immutable PCM bytes and another read-only `memoryview` over immutable admitted annotation bytes, plus track ID and sample rate. It receives no workstation path and has no write capability over either scientific input; - emits only registration identity, runtime identity, content digests, decoded PCM digest/frame count, sample rate, channel count, and track ID. Local audio/annotation paths are never copied into the receipt. @@ -50,6 +50,8 @@ RED `59cb9dc4c41b4208a0c6e25d9365e924061c98a7` therefore requires the consumer-f A parallel PCM capability gap remained. The default decoder marked its NumPy array read-only, but the consumer boundary accepted whatever `memoryview` the decoder returned and trusted the decoder-supplied digest. A custom or future decoder could expose a mutable buffer, or report a digest that did not describe the bytes actually handed to CQT/STFT. RED `e6a62f393daf5ca1fb7e44271777dbac6f1956e1` requires the PCM handoff to be immutable even when the decoder owns mutable memory and requires admission to reject a decoder digest that disagrees with the handoff bytes. GREEN `5b0660e49e4d01246cafeea988e2b8c222713be9` recomputes SHA-256 and mono-float32 frame count from the exact exposed PCM, then copies it into immutable `bytes` before any scientific consumer runs. +That repair still preserved a digest-only compatibility path: an injected decoder could return only `(decoded_pcm_sha256, decoded_frames)`, in which case admission emitted those caller-claimed values without ever seeing the PCM bytes. That meant a path-free receipt could claim a normalized signal identity that admission had no way to verify. RED `cd4818ff329a5214a152ec0a485b50177498b2e1` requires such a decoder to fail even when no `track_consumer` is configured. Fixture alignment `f6085084721478c4440eb723487e1696e8087a99` makes the successful injected decoder expose its synthetic PCM. GREEN `4bd685ad500b53bb0a4d70da35139fd92a2aaa35` removes the digest-only decoder contract: every admitted receipt now derives its PCM digest and frame count from the exact exposed bytes before it can be emitted. + The manifest loader had a separate pathname TOCTOU: it called `Path.stat()` for the 2 MiB admission decision and then reopened the path with `Path.read_text()`. A rename or replacement between those calls could make the bounded metadata and parsed bytes refer to different files. RED `106d4e37e98a76f48f73db1ba1854209de873040` requires both size and JSON bytes to come from one already-open descriptor without pathname stat/read reopening. GREEN `ad5daf36a51b3db859ec25475b883c101a89a28e` now uses the regular-file admission helper, `fstat`, and a bounded descriptor read, with a second byte-count guard for concurrent growth. Source identity then exposed a different reproducibility gap. `git rev-parse HEAD` can report the registered commit while the worktree contains modified or additional executable source. RED `986170737dc3b1a11a04abf47b90a868e93362e7` creates an isolated Git repository, modifies a tracked analysis file after commit, and requires runtime admission to reject it. GREEN `4e8f3fa277a411a45f012daa5083f9ea5193cd17` requires a clean porcelain status before the source commit can enter the receipt. @@ -72,6 +74,8 @@ Re-hashing a writable annotation handle only after consumer execution was reject Trusting a decoder-supplied PCM digest or a read-only flag on decoder-owned memory was rejected. A scientific receipt must identify the bytes actually given to the measurement code, and the consumer must not be able to recover a mutable backing object. Admission therefore derives the digest/frame count from the exposed byte sequence and crosses the consumer boundary only after copying it into immutable `bytes`. +Retaining a digest-only decoder compatibility path was rejected for the same reason. A decoder-provided SHA-256 string is not evidence of decoded PCM identity unless admission can recompute it from the bytes that were actually produced. Decoder implementations used for scientific admission must therefore expose the PCM byte sequence even when the standalone caller only wants a durable receipt and no measurement callback. + Persisting workstation paths in the receipt was rejected because they are neither stable provenance nor purpose-bound evidence and can expose local usernames, mounts, or project layout. The decoded PCM digest does not by itself prove that two machines will decode a compressed source identically. It records the exact normalized signal used by one admission/consumer run. The preregistration already binds librosa/NumPy and the source/lock identity; if future production acceptance requires cross-decoder equivalence, the schema must explicitly bind the remaining decoder/backend contract rather than assuming it. @@ -84,9 +88,9 @@ MIREX 2025 Music Structure Analysis evaluates mono 44.1 kHz WAV input and uses f ## Test boundary -Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise bounded single-descriptor manifest loading, clean-source identity, hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, source mutation after snapshot admission, exact PCM/annotation consumer handoff, intrinsic annotation immutability, mutable-decoder PCM confinement, and decoder-digest mismatch rejection. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. +Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise bounded single-descriptor manifest loading, clean-source identity, hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, source mutation after snapshot admission, exact PCM/annotation consumer handoff, intrinsic annotation immutability, mutable-decoder PCM confinement, decoder-digest mismatch rejection, and digest-only decoder rejection. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. -The handoff regression deliberately mutates the original annotation file after its process-owned snapshot has been created and requires the consumer to observe the admitted bytes, not the mutated path content. The annotation hostile case makes writes fail at the boundary instead of relying on a later digest comparison. The PCM hostile cases start from mutable decoder-owned memory and from a deliberately false decoder digest; admission must convert the former to immutable bytes and reject the latter before a consumer can run. +The handoff regression deliberately mutates the original annotation file after its process-owned snapshot has been created and requires the consumer to observe the admitted bytes, not the mutated path content. The annotation hostile case makes writes fail at the boundary instead of relying on a later digest comparison. The PCM hostile cases start from mutable decoder-owned memory, from a deliberately false decoder digest, and from a decoder that withholds the PCM bytes entirely; admission must convert exposed PCM to immutable bytes, reject a false digest, and reject an unverifiable digest-only receipt before any scientific evidence is emitted. ## Remaining scientific work From f453e9b8152e36590ed73afbb3d41c8b41e81ed9 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 00:59:57 +0900 Subject: [PATCH 087/216] test(mir): reject non-finite admitted PCM --- ...ucture_corpus_consumer_integrity_policy.py | 27 +++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py b/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py index 9dc0af1cd..6999ebf6f 100644 --- a/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py +++ b/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py @@ -4,6 +4,7 @@ import hashlib import os +import struct from pathlib import Path import pytest @@ -144,3 +145,29 @@ def digest_only_decoder(fd: int, sample_rate_hz: int) -> tuple[str, int]: runtime_identity=_runtime(), decoder=digest_only_decoder, ) + + +@pytest.mark.parametrize("non_finite_sample", [float("nan"), float("inf"), float("-inf")]) +def test_admission_rejects_non_finite_pcm_before_measurement( + tmp_path: Path, + non_finite_sample: float, +) -> None: + """NaN or infinite decoded samples must never enter MIR measurement.""" + admission, registration, manifest = _registered_inputs(tmp_path) + pcm = memoryview(struct.pack(" tuple[str, int, memoryview]: + assert sample_rate_hz == 44100 + assert os.read(fd, 1) + return hashlib.sha256(pcm).hexdigest(), 1, pcm + + with pytest.raises(ValueError, match="finite float32"): + admission.verify_corpus( + registration, + manifest, + runtime_identity=_runtime(), + decoder=non_finite_decoder, + track_consumer=lambda *_args: pytest.fail( + "consumer must not run with non-finite PCM" + ), + ) From c87d9806d960ee6c1200d0be3c9e65a3c893453c Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 01:00:56 +0900 Subject: [PATCH 088/216] fix(mir): reject non-finite decoded PCM --- scripts/research/verify_structure_corpus.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/scripts/research/verify_structure_corpus.py b/scripts/research/verify_structure_corpus.py index 8fc42bd4a..9f83c9350 100644 --- a/scripts/research/verify_structure_corpus.py +++ b/scripts/research/verify_structure_corpus.py @@ -214,6 +214,8 @@ def _bind_decoded_pcm( decoded_pcm: memoryview, ) -> tuple[str, int, memoryview]: """Bind receipt identity to an immutable copy of the exact PCM handoff.""" + import numpy as np + pcm_bytes = bytes(decoded_pcm) pcm_view = memoryview(pcm_bytes) actual_digest = hashlib.sha256(pcm_view).hexdigest() @@ -226,6 +228,9 @@ def _bind_decoded_pcm( raise ValueError("decoded frame count must be an integer") from exc if frame_count < 1 or len(pcm_view) != frame_count * 4: raise ValueError("decoded frame count does not match mono float32 PCM bytes") + samples = np.frombuffer(pcm_bytes, dtype=" int: if __name__ == "__main__": - raise SystemExit(main()) + raise SystemExit(main()) \ No newline at end of file From acf7adf54342a2d29afa85947bd06c22af8b6ac1 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 01:01:46 +0900 Subject: [PATCH 089/216] style(mir): format non-finite PCM regression --- .../tests/test_structure_corpus_consumer_integrity_policy.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py b/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py index 6999ebf6f..be3e6775b 100644 --- a/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py +++ b/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py @@ -147,7 +147,10 @@ def digest_only_decoder(fd: int, sample_rate_hz: int) -> tuple[str, int]: ) -@pytest.mark.parametrize("non_finite_sample", [float("nan"), float("inf"), float("-inf")]) +@pytest.mark.parametrize( + "non_finite_sample", + [float("nan"), float("inf"), float("-inf")], +) def test_admission_rejects_non_finite_pcm_before_measurement( tmp_path: Path, non_finite_sample: float, From ae2cb0806c6862e9a8bee69cfcd0665b70b1dc1f Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 01:02:30 +0900 Subject: [PATCH 090/216] docs(mir): trace non-finite PCM admission boundary --- docs/traceability/mir/structure-corpus-admission.md | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/docs/traceability/mir/structure-corpus-admission.md b/docs/traceability/mir/structure-corpus-admission.md index c765cb120..29fc66734 100644 --- a/docs/traceability/mir/structure-corpus-admission.md +++ b/docs/traceability/mir/structure-corpus-admission.md @@ -28,7 +28,7 @@ For every registered track the admission tool: - requires the clean current source commit, `uv.lock`, Python, librosa, and NumPy identities to match the registered runtime; - treats Git commit and SHA-256 values as case-insensitive hexadecimal identities and emits them lowercase, matching the validator contract, while Python/librosa/NumPy version strings remain exact; - decodes the immutable admitted audio snapshot through `librosa.load(..., sr=, mono=True)`, converts it to canonical little-endian float32, and computes a PCM SHA-256; -- requires every decoder implementation to expose the exact decoded PCM bytes together with its claimed digest and frame count. Admission recomputes the digest and mono-float32 frame count from that byte sequence, rejects a digest-only decoder receipt or any inconsistency, and copies the sequence into immutable `bytes` before the callback boundary; +- requires every decoder implementation to expose the exact decoded PCM bytes together with its claimed digest and frame count. Admission recomputes the digest and mono-float32 frame count from that byte sequence, rejects a digest-only decoder receipt or any inconsistency, rejects NaN and positive/negative infinity in the decoded float32 signal, and copies the sequence into immutable `bytes` before the callback boundary; - optionally invokes an in-process `track_consumer` only after both registered audio and annotation identities have passed. The callback receives a read-only `memoryview` over the immutable PCM bytes and another read-only `memoryview` over immutable admitted annotation bytes, plus track ID and sample rate. It receives no workstation path and has no write capability over either scientific input; - emits only registration identity, runtime identity, content digests, decoded PCM digest/frame count, sample rate, channel count, and track ID. Local audio/annotation paths are never copied into the receipt. @@ -52,6 +52,8 @@ A parallel PCM capability gap remained. The default decoder marked its NumPy arr That repair still preserved a digest-only compatibility path: an injected decoder could return only `(decoded_pcm_sha256, decoded_frames)`, in which case admission emitted those caller-claimed values without ever seeing the PCM bytes. That meant a path-free receipt could claim a normalized signal identity that admission had no way to verify. RED `cd4818ff329a5214a152ec0a485b50177498b2e1` requires such a decoder to fail even when no `track_consumer` is configured. Fixture alignment `f6085084721478c4440eb723487e1696e8087a99` makes the successful injected decoder expose its synthetic PCM. GREEN `4bd685ad500b53bb0a4d70da35139fd92a2aaa35` removes the digest-only decoder contract: every admitted receipt now derives its PCM digest and frame count from the exact exposed bytes before it can be emitted. +A further scientific-validity gap remained after byte identity was bound. Any four-byte-aligned payload can be interpreted as float32, including NaN and positive or negative infinity. Such a signal can have a perfectly matching SHA-256 and frame count while making downstream spectral features or MIR metrics non-finite, so byte integrity alone is not sufficient signal admissibility. RED `f453e9b8152e36590ed73afbb3d41c8b41e81ed9` injects each IEEE-754 non-finite class and requires failure before the measurement callback can run. GREEN `c87d9806d960ee6c1200d0be3c9e65a3c893453c` interprets the exact immutable little-endian float32 handoff and requires `numpy.isfinite(...).all()` before evidence is emitted or measurement begins. `acf7adf54342a2d29afa85947bd06c22af8b6ac1` only normalizes the regression formatting for the repository's Ruff gate. + The manifest loader had a separate pathname TOCTOU: it called `Path.stat()` for the 2 MiB admission decision and then reopened the path with `Path.read_text()`. A rename or replacement between those calls could make the bounded metadata and parsed bytes refer to different files. RED `106d4e37e98a76f48f73db1ba1854209de873040` requires both size and JSON bytes to come from one already-open descriptor without pathname stat/read reopening. GREEN `ad5daf36a51b3db859ec25475b883c101a89a28e` now uses the regular-file admission helper, `fstat`, and a bounded descriptor read, with a second byte-count guard for concurrent growth. Source identity then exposed a different reproducibility gap. `git rev-parse HEAD` can report the registered commit while the worktree contains modified or additional executable source. RED `986170737dc3b1a11a04abf47b90a868e93362e7` creates an isolated Git repository, modifies a tracked analysis file after commit, and requires runtime admission to reject it. GREEN `4e8f3fa277a411a45f012daa5083f9ea5193cd17` requires a clean porcelain status before the source commit can enter the receipt. @@ -76,6 +78,8 @@ Trusting a decoder-supplied PCM digest or a read-only flag on decoder-owned memo Retaining a digest-only decoder compatibility path was rejected for the same reason. A decoder-provided SHA-256 string is not evidence of decoded PCM identity unless admission can recompute it from the bytes that were actually produced. Decoder implementations used for scientific admission must therefore expose the PCM byte sequence even when the standalone caller only wants a durable receipt and no measurement callback. +Treating byte-aligned PCM as scientifically admissible solely because its digest and frame count match was rejected. IEEE-754 float32 includes non-finite values; those bytes are structurally valid floats but are not valid measurement input for this experiment because they can contaminate spectral features and paired metrics. Admission therefore rejects NaN and both infinities before the consumer is invoked. It does not impose clipping or amplitude normalization, which would change the signal rather than validate it. + Persisting workstation paths in the receipt was rejected because they are neither stable provenance nor purpose-bound evidence and can expose local usernames, mounts, or project layout. The decoded PCM digest does not by itself prove that two machines will decode a compressed source identically. It records the exact normalized signal used by one admission/consumer run. The preregistration already binds librosa/NumPy and the source/lock identity; if future production acceptance requires cross-decoder equivalence, the schema must explicitly bind the remaining decoder/backend contract rather than assuming it. @@ -88,15 +92,15 @@ MIREX 2025 Music Structure Analysis evaluates mono 44.1 kHz WAV input and uses f ## Test boundary -Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise bounded single-descriptor manifest loading, clean-source identity, hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, source mutation after snapshot admission, exact PCM/annotation consumer handoff, intrinsic annotation immutability, mutable-decoder PCM confinement, decoder-digest mismatch rejection, and digest-only decoder rejection. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. +Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise bounded single-descriptor manifest loading, clean-source identity, hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, source mutation after snapshot admission, exact PCM/annotation consumer handoff, intrinsic annotation immutability, mutable-decoder PCM confinement, decoder-digest mismatch rejection, digest-only decoder rejection, and NaN/positive-infinity/negative-infinity PCM rejection. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. -The handoff regression deliberately mutates the original annotation file after its process-owned snapshot has been created and requires the consumer to observe the admitted bytes, not the mutated path content. The annotation hostile case makes writes fail at the boundary instead of relying on a later digest comparison. The PCM hostile cases start from mutable decoder-owned memory, from a deliberately false decoder digest, and from a decoder that withholds the PCM bytes entirely; admission must convert exposed PCM to immutable bytes, reject a false digest, and reject an unverifiable digest-only receipt before any scientific evidence is emitted. +The handoff regression deliberately mutates the original annotation file after its process-owned snapshot has been created and requires the consumer to observe the admitted bytes, not the mutated path content. The annotation hostile case makes writes fail at the boundary instead of relying on a later digest comparison. The PCM hostile cases start from mutable decoder-owned memory, from a deliberately false decoder digest, from a decoder that withholds the PCM bytes entirely, and from byte-identical non-finite float32 samples; admission must convert exposed PCM to immutable bytes, reject a false digest, reject an unverifiable digest-only receipt, and reject non-finite measurement input before any scientific evidence or callback execution occurs. ## Remaining scientific work A reviewed rights-cleared real corpus and independent annotation manifest are still required. Before candidate results are inspected, the actual aggregation, paired uncertainty, margins, latency threshold, dependence assumptions, claim boundary, and any failure/exclusion rule must be approved. -The next runner must execute from the exact clean registered source/lock identity through the in-process admitted-track consumer boundary, interpret the admitted read-only annotation bytes, run CQT and STFT on the exact same immutable PCM, calculate the recognized MIREX/mir_eval track metrics and latency/memory measurements, and derive aggregate/CI evidence from the preregistered procedure rather than accepting caller-authored summaries. The standalone receipt is evidence for that run, not a license to reopen media after admission. +The next runner must execute from the exact clean registered source/lock identity through the in-process admitted-track consumer boundary, interpret the admitted read-only annotation bytes, run CQT and STFT on the exact same immutable finite PCM, calculate the recognized MIREX/mir_eval track metrics and latency/memory measurements, and derive aggregate/CI evidence from the preregistered procedure rather than accepting caller-authored summaries. The standalone receipt is evidence for that run, not a license to reopen media after admission. ## References From ea20efdef7a5d1ad2bf44e3bfb8b6bf42005a576 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 02:03:15 +0900 Subject: [PATCH 091/216] test(mir): reject coerced decoded frame counts --- ...ucture_corpus_consumer_integrity_policy.py | 26 +++++++++++++++++++ 1 file changed, 26 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py b/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py index be3e6775b..ab151e974 100644 --- a/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py +++ b/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py @@ -174,3 +174,29 @@ def non_finite_decoder(fd: int, sample_rate_hz: int) -> tuple[str, int, memoryvi "consumer must not run with non-finite PCM" ), ) + + +@pytest.mark.parametrize("invalid_frame_count", [True, "1", 1.0]) +def test_admission_rejects_non_integer_decoder_frame_count( + tmp_path: Path, + invalid_frame_count: object, +) -> None: + """Decoder frame evidence must already be an integer, not a coercible value.""" + admission, registration, manifest = _registered_inputs(tmp_path) + pcm = memoryview(b"\x00\x00\x00\x00") + + def malformed_decoder(fd: int, sample_rate_hz: int) -> tuple[str, object, memoryview]: + assert sample_rate_hz == 44100 + assert os.read(fd, 1) + return hashlib.sha256(pcm).hexdigest(), invalid_frame_count, pcm + + with pytest.raises(ValueError, match="decoded frame count must be an integer"): + admission.verify_corpus( + registration, + manifest, + runtime_identity=_runtime(), + decoder=malformed_decoder, + track_consumer=lambda *_args: pytest.fail( + "consumer must not run with coerced frame-count evidence" + ), + ) From 527fda5a5b0a8992f61ca8fe7401ff796b1ac62e Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 02:04:26 +0900 Subject: [PATCH 092/216] fix(mir): reject coerced decoded frame counts --- scripts/research/verify_structure_corpus.py | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/scripts/research/verify_structure_corpus.py b/scripts/research/verify_structure_corpus.py index 9f83c9350..d4a7f9eea 100644 --- a/scripts/research/verify_structure_corpus.py +++ b/scripts/research/verify_structure_corpus.py @@ -222,10 +222,9 @@ def _bind_decoded_pcm( claimed_digest = _text(decoded_pcm_sha256, "decoder.decoded_pcm_sha256").lower() if actual_digest != claimed_digest: raise ValueError("decoded PCM SHA-256 does not match decoder handoff bytes") - try: - frame_count = int(decoded_frames) - except (TypeError, ValueError) as exc: - raise ValueError("decoded frame count must be an integer") from exc + if isinstance(decoded_frames, bool) or not isinstance(decoded_frames, int): + raise ValueError("decoded frame count must be an integer") + frame_count = decoded_frames if frame_count < 1 or len(pcm_view) != frame_count * 4: raise ValueError("decoded frame count does not match mono float32 PCM bytes") samples = np.frombuffer(pcm_bytes, dtype=" int: if __name__ == "__main__": - raise SystemExit(main()) \ No newline at end of file + raise SystemExit(main()) From 7fac0feb0261fb868c1c6033ab817a56f6eade04 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 02:05:19 +0900 Subject: [PATCH 093/216] docs(mir): trace strict decoder frame evidence --- docs/traceability/mir/structure-corpus-admission.md | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/docs/traceability/mir/structure-corpus-admission.md b/docs/traceability/mir/structure-corpus-admission.md index 29fc66734..982559b01 100644 --- a/docs/traceability/mir/structure-corpus-admission.md +++ b/docs/traceability/mir/structure-corpus-admission.md @@ -28,7 +28,7 @@ For every registered track the admission tool: - requires the clean current source commit, `uv.lock`, Python, librosa, and NumPy identities to match the registered runtime; - treats Git commit and SHA-256 values as case-insensitive hexadecimal identities and emits them lowercase, matching the validator contract, while Python/librosa/NumPy version strings remain exact; - decodes the immutable admitted audio snapshot through `librosa.load(..., sr=, mono=True)`, converts it to canonical little-endian float32, and computes a PCM SHA-256; -- requires every decoder implementation to expose the exact decoded PCM bytes together with its claimed digest and frame count. Admission recomputes the digest and mono-float32 frame count from that byte sequence, rejects a digest-only decoder receipt or any inconsistency, rejects NaN and positive/negative infinity in the decoded float32 signal, and copies the sequence into immutable `bytes` before the callback boundary; +- requires every decoder implementation to expose the exact decoded PCM bytes together with its claimed digest and frame count. Admission recomputes the digest from that byte sequence, requires the decoder frame count to already be a non-boolean integer, verifies the mono-float32 byte length against that count, rejects a digest-only decoder receipt or any inconsistency, rejects NaN and positive/negative infinity in the decoded float32 signal, and copies the sequence into immutable `bytes` before the callback boundary; - optionally invokes an in-process `track_consumer` only after both registered audio and annotation identities have passed. The callback receives a read-only `memoryview` over the immutable PCM bytes and another read-only `memoryview` over immutable admitted annotation bytes, plus track ID and sample rate. It receives no workstation path and has no write capability over either scientific input; - emits only registration identity, runtime identity, content digests, decoded PCM digest/frame count, sample rate, channel count, and track ID. Local audio/annotation paths are never copied into the receipt. @@ -54,6 +54,8 @@ That repair still preserved a digest-only compatibility path: an injected decode A further scientific-validity gap remained after byte identity was bound. Any four-byte-aligned payload can be interpreted as float32, including NaN and positive or negative infinity. Such a signal can have a perfectly matching SHA-256 and frame count while making downstream spectral features or MIR metrics non-finite, so byte integrity alone is not sufficient signal admissibility. RED `f453e9b8152e36590ed73afbb3d41c8b41e81ed9` injects each IEEE-754 non-finite class and requires failure before the measurement callback can run. GREEN `c87d9806d960ee6c1200d0be3c9e65a3c893453c` interprets the exact immutable little-endian float32 handoff and requires `numpy.isfinite(...).all()` before evidence is emitted or measurement begins. `acf7adf54342a2d29afa85947bd06c22af8b6ac1` only normalizes the regression formatting for the repository's Ruff gate. +Frame-count evidence had a separate type-integrity gap. `_bind_decoded_pcm()` coerced the decoder value with `int(...)`, so `True`, `"1"`, or `1.0` could be normalized to the integer `1` and accepted when four PCM bytes were present. That silently rewrote malformed decoder evidence instead of validating the declared contract. RED `ea20efdef7a5d1ad2bf44e3bfb8b6bf42005a576` requires those coercible non-integer values to fail before measurement. GREEN `527fda5a5b0a8992f61ca8fe7401ff796b1ac62e` now requires a non-boolean Python integer and only then checks the exact mono-float32 byte-count relation. + The manifest loader had a separate pathname TOCTOU: it called `Path.stat()` for the 2 MiB admission decision and then reopened the path with `Path.read_text()`. A rename or replacement between those calls could make the bounded metadata and parsed bytes refer to different files. RED `106d4e37e98a76f48f73db1ba1854209de873040` requires both size and JSON bytes to come from one already-open descriptor without pathname stat/read reopening. GREEN `ad5daf36a51b3db859ec25475b883c101a89a28e` now uses the regular-file admission helper, `fstat`, and a bounded descriptor read, with a second byte-count guard for concurrent growth. Source identity then exposed a different reproducibility gap. `git rev-parse HEAD` can report the registered commit while the worktree contains modified or additional executable source. RED `986170737dc3b1a11a04abf47b90a868e93362e7` creates an isolated Git repository, modifies a tracked analysis file after commit, and requires runtime admission to reject it. GREEN `4e8f3fa277a411a45f012daa5083f9ea5193cd17` requires a clean porcelain status before the source commit can enter the receipt. @@ -80,6 +82,8 @@ Retaining a digest-only decoder compatibility path was rejected for the same rea Treating byte-aligned PCM as scientifically admissible solely because its digest and frame count match was rejected. IEEE-754 float32 includes non-finite values; those bytes are structurally valid floats but are not valid measurement input for this experiment because they can contaminate spectral features and paired metrics. Admission therefore rejects NaN and both infinities before the consumer is invoked. It does not impose clipping or amplitude normalization, which would change the signal rather than validate it. +Coercing decoder frame-count evidence was rejected. The decoder contract declares an integer count; accepting booleans, numeric strings, or integral floats and normalizing them with `int(...)` would hide an upstream contract violation in the durable receipt. Admission validates the declared type first and only then verifies that `decoded_frames * 4` exactly equals the immutable mono-float32 byte length. + Persisting workstation paths in the receipt was rejected because they are neither stable provenance nor purpose-bound evidence and can expose local usernames, mounts, or project layout. The decoded PCM digest does not by itself prove that two machines will decode a compressed source identically. It records the exact normalized signal used by one admission/consumer run. The preregistration already binds librosa/NumPy and the source/lock identity; if future production acceptance requires cross-decoder equivalence, the schema must explicitly bind the remaining decoder/backend contract rather than assuming it. @@ -92,9 +96,9 @@ MIREX 2025 Music Structure Analysis evaluates mono 44.1 kHz WAV input and uses f ## Test boundary -Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise bounded single-descriptor manifest loading, clean-source identity, hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, source mutation after snapshot admission, exact PCM/annotation consumer handoff, intrinsic annotation immutability, mutable-decoder PCM confinement, decoder-digest mismatch rejection, digest-only decoder rejection, and NaN/positive-infinity/negative-infinity PCM rejection. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. +Unit tests use tiny synthetic byte fixtures and an injected decoder to exercise bounded single-descriptor manifest loading, clean-source identity, hash drift, runtime drift, symlink rejection, path non-disclosure, duplicate JSON keys, non-standard JSON numbers, case-insensitive Git/SHA-256 identity, source mutation after snapshot admission, exact PCM/annotation consumer handoff, intrinsic annotation immutability, mutable-decoder PCM confinement, decoder-digest mismatch rejection, digest-only decoder rejection, non-integer frame-count rejection, and NaN/positive-infinity/negative-infinity PCM rejection. These fixtures are not production scientific evidence and do not satisfy #1225's rights-cleared real-music corpus requirement. -The handoff regression deliberately mutates the original annotation file after its process-owned snapshot has been created and requires the consumer to observe the admitted bytes, not the mutated path content. The annotation hostile case makes writes fail at the boundary instead of relying on a later digest comparison. The PCM hostile cases start from mutable decoder-owned memory, from a deliberately false decoder digest, from a decoder that withholds the PCM bytes entirely, and from byte-identical non-finite float32 samples; admission must convert exposed PCM to immutable bytes, reject a false digest, reject an unverifiable digest-only receipt, and reject non-finite measurement input before any scientific evidence or callback execution occurs. +The handoff regression deliberately mutates the original annotation file after its process-owned snapshot has been created and requires the consumer to observe the admitted bytes, not the mutated path content. The annotation hostile case makes writes fail at the boundary instead of relying on a later digest comparison. The PCM hostile cases start from mutable decoder-owned memory, from a deliberately false decoder digest, from a decoder that withholds the PCM bytes entirely, from coercible non-integer frame counts, and from byte-identical non-finite float32 samples; admission must convert exposed PCM to immutable bytes, reject a false digest, reject an unverifiable digest-only receipt, reject malformed frame-count evidence without coercion, and reject non-finite measurement input before any scientific evidence or callback execution occurs. ## Remaining scientific work From 394e161b601e61eab11acd319b6196499e251f24 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 03:05:06 +0900 Subject: [PATCH 094/216] test(mir): require atomic receipt publication --- ...t_structure_corpus_receipt_write_policy.py | 58 +++++++++++++++++++ 1 file changed, 58 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_corpus_receipt_write_policy.py diff --git a/services/analysis-engine/tests/test_structure_corpus_receipt_write_policy.py b/services/analysis-engine/tests/test_structure_corpus_receipt_write_policy.py new file mode 100644 index 000000000..d144cf1f6 --- /dev/null +++ b/services/analysis-engine/tests/test_structure_corpus_receipt_write_policy.py @@ -0,0 +1,58 @@ +"""Policy tests for crash-safe corpus-admission receipt persistence.""" + +from __future__ import annotations + +import json +from pathlib import Path +from types import ModuleType + +import pytest +from conftest import load_module, make_symlink_or_skip + + +def _admission() -> ModuleType: + """Load the corpus-admission module under test.""" + return load_module( + "scripts/research/verify_structure_corpus.py", + "verify_structure_corpus_receipt_write_policy", + ) + + +def test_receipt_write_preserves_existing_file_when_replace_fails( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + """A failed publication must not truncate the last complete receipt.""" + admission = _admission() + output = tmp_path / "receipt.json" + previous = '{"previous":true}\n' + output.write_text(previous, encoding="utf-8") + + def fail_replace(_source: object, _destination: object) -> None: + raise OSError("simulated atomic replace failure") + + monkeypatch.setattr(admission.os, "replace", fail_replace) + + with pytest.raises(OSError, match="simulated atomic replace failure"): + admission._write_receipt_atomic(output, {"replacement": True}) + + assert output.read_text(encoding="utf-8") == previous + assert list(tmp_path.glob(f".{output.name}.*.tmp")) == [] + + +def test_receipt_write_replaces_output_symlink_without_following_target( + tmp_path: Path, +) -> None: + """An untrusted output symlink must not redirect receipt bytes to its target.""" + admission = _admission() + protected_target = tmp_path / "protected.txt" + protected_target.write_text("keep-me\n", encoding="utf-8") + output = tmp_path / "receipt.json" + make_symlink_or_skip(output, protected_target) + + receipt = {"schema_version": 1, "registration_sha256": "a" * 64} + admission._write_receipt_atomic(output, receipt) + + assert protected_target.read_text(encoding="utf-8") == "keep-me\n" + assert not output.is_symlink() + assert json.loads(output.read_text(encoding="utf-8")) == receipt From 7941fd6e1d0f40e3fa8d119e1a9c37b6c4a54811 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 03:06:01 +0900 Subject: [PATCH 095/216] fix(mir): publish admission receipts atomically --- scripts/research/verify_structure_corpus.py | 34 +++++++++++++++++---- 1 file changed, 28 insertions(+), 6 deletions(-) diff --git a/scripts/research/verify_structure_corpus.py b/scripts/research/verify_structure_corpus.py index d4a7f9eea..79cb36f1c 100644 --- a/scripts/research/verify_structure_corpus.py +++ b/scripts/research/verify_structure_corpus.py @@ -431,8 +431,33 @@ def verify_corpus( } +def _write_receipt_atomic(path: Path, receipt: Mapping[str, object]) -> None: + """Publish one complete receipt without following or truncating the target path.""" + payload = ( + json.dumps(receipt, sort_keys=True, separators=(",", ":")) + "\n" + ).encode("utf-8") + fd, temporary_name = tempfile.mkstemp( + prefix=f".{path.name}.", + suffix=".tmp", + dir=path.parent, + ) + temporary_path = Path(temporary_name) + try: + with os.fdopen(fd, "wb", closefd=True) as temporary_file: + temporary_file.write(payload) + temporary_file.flush() + os.fsync(temporary_file.fileno()) + os.replace(temporary_path, path) + except Exception: + try: + temporary_path.unlink() + except FileNotFoundError: + pass + raise + + def main() -> int: - """Run corpus admission and write one path-free verification receipt.""" + """Run corpus admission and atomically publish one path-free receipt.""" parser = argparse.ArgumentParser( description="Admit local real-audio files for a frozen structure experiment" ) @@ -450,12 +475,9 @@ def main() -> int: manifest, runtime_identity=_current_runtime_identity(repo_root), ) - args.output.write_text( - json.dumps(receipt, sort_keys=True, separators=(",", ":")) + "\n", - encoding="utf-8", - ) + _write_receipt_atomic(args.output, receipt) return 0 if __name__ == "__main__": - raise SystemExit(main()) + raise SystemExit(main()) \ No newline at end of file From 1e86c3b91de5dcec23e4478e19a8bc05e5211da2 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 03:07:12 +0900 Subject: [PATCH 096/216] docs(mir): trace atomic receipt publication --- .../structure-corpus-receipt-publication.md | 42 +++++++++++++++++++ 1 file changed, 42 insertions(+) create mode 100644 docs/traceability/mir/structure-corpus-receipt-publication.md diff --git a/docs/traceability/mir/structure-corpus-receipt-publication.md b/docs/traceability/mir/structure-corpus-receipt-publication.md new file mode 100644 index 000000000..adb78a842 --- /dev/null +++ b/docs/traceability/mir/structure-corpus-receipt-publication.md @@ -0,0 +1,42 @@ +# Structure corpus receipt publication + +Status: Proposed +Owner: Signal-MIR Analysis +Tracking: #1225, #1228 +Parent: `docs/traceability/mir/structure-corpus-admission.md` + +## Problem + +Corpus admission produces a durable JSON receipt that binds the preregistration, runtime identity, registered audio/annotation content and the exact admitted mono-float32 PCM. Before this change, the CLI published that evidence with `Path.write_text(...)` directly to the caller-supplied `--output` path. + +That write path had two independent failure modes. An interruption or write failure after opening an existing receipt could truncate the last complete evidence document. A caller-supplied output path that was a symbolic link could also redirect the write into the symlink target. Both behaviors conflict with BandScope's storage-boundary rule that local paths are untrusted and with the scientific requirement that a receipt be either the previous complete document or the new complete document, never a partially replaced evidence artifact. + +## Decision + +Receipt publication now serializes the complete canonical JSON payload in memory, creates a restrictive process-owned temporary file in the destination directory with `mkstemp`, writes the whole payload, flushes it, calls `fsync` on that file, closes it, and only then uses `os.replace` to publish it at the requested path. + +Using a temporary file in the same directory keeps the final rename on the same filesystem and gives the operation replacement semantics instead of in-place truncation. Replacing a destination symlink replaces the directory entry itself rather than following the link and writing through to its target. If writing, flushing, syncing, or replacing fails, the temporary file is removed and any previously complete destination remains untouched. + +This is an atomic-publication boundary, not a claim that every supported filesystem persists directory metadata across sudden power loss. The file contents are synced before replacement, but directory-fsync semantics are platform-specific and are not represented as stronger durability evidence here. Project-level crash/power-loss durability remains owned by Project Persistence rather than this research receipt helper. + +## RED → GREEN lineage + +- RED `394e161b601e61eab11acd319b6196499e251f24`: a failed final replacement must preserve the existing complete receipt, remove the abandoned temporary file, and an output symlink must not allow receipt bytes to overwrite its target. +- GREEN `7941fd6e1d0f40e3fa8d119e1a9c37b6c4a54811`: `_write_receipt_atomic` writes and syncs a same-directory temporary file, then publishes with `os.replace`; `main()` no longer writes directly through `Path.write_text`. + +## Security Notes + +- **Untrusted input:** `--output` is caller-controlled filesystem input. The implementation does not dereference a destination symlink for the final write. +- **Trust boundary:** receipt bytes cross from the scientific admission process into the local storage boundary only after complete serialization and file sync. +- **Scope restriction:** the helper writes exactly one explicitly requested receipt path and one same-directory temporary file. It does not add directory scanning or a generic filesystem API. +- **Safe failure:** pre-publication failures leave the previous destination intact and clean up the temporary file. No retry loop or alternate output location widens the write scope. +- **Privacy:** durable receipt contents remain path-free; this change does not add workstation paths, usernames, raw audio or annotation content to the receipt. +- **Test points:** simulated `os.replace` failure preserves the existing receipt; a symlinked output is replaced without modifying its target; normal publication remains valid JSON. + +## Constraints and rejected alternatives + +Writing directly to a temporary file in the system-wide temp directory and then moving it was rejected because cross-filesystem moves can lose atomic replacement semantics. Writing directly to the target and relying on JSON parsing to detect truncation was rejected because detection does not preserve the last complete scientific receipt. Refusing every pre-existing symlink before publication was not used as the sole defense because a path check followed by a later write would reintroduce a check/use race; final `os.replace` semantics remove the write-through capability instead. + +## Remaining scientific boundary + +A crash-safe admission receipt is necessary provenance, not MIR acceptance. #1225 still requires a reviewed rights-cleared real-music corpus and independent annotations, pre-data margins/aggregation/paired uncertainty/dependence/claim boundaries, a recognized MIREX/mir_eval runner that consumes the admitted immutable PCM/annotation handoff, and a paired CQT/STFT execution on the same admitted inputs and runtime identity. From b48340b0f1819083670c2f01f04138099cf70aed Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 03:08:12 +0900 Subject: [PATCH 097/216] style(mir): preserve source newline at EOF --- scripts/research/verify_structure_corpus.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scripts/research/verify_structure_corpus.py b/scripts/research/verify_structure_corpus.py index 79cb36f1c..414674fdc 100644 --- a/scripts/research/verify_structure_corpus.py +++ b/scripts/research/verify_structure_corpus.py @@ -480,4 +480,4 @@ def main() -> int: if __name__ == "__main__": - raise SystemExit(main()) \ No newline at end of file + raise SystemExit(main()) From 2e046ba76a4fa4ca7cab9580b4b1b2b0c65c4cf8 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 03:09:31 +0900 Subject: [PATCH 098/216] docs(mir): record receipt publication follow-up --- docs/traceability/mir/structure-corpus-receipt-publication.md | 1 + 1 file changed, 1 insertion(+) diff --git a/docs/traceability/mir/structure-corpus-receipt-publication.md b/docs/traceability/mir/structure-corpus-receipt-publication.md index adb78a842..3944123b9 100644 --- a/docs/traceability/mir/structure-corpus-receipt-publication.md +++ b/docs/traceability/mir/structure-corpus-receipt-publication.md @@ -23,6 +23,7 @@ This is an atomic-publication boundary, not a claim that every supported filesys - RED `394e161b601e61eab11acd319b6196499e251f24`: a failed final replacement must preserve the existing complete receipt, remove the abandoned temporary file, and an output symlink must not allow receipt bytes to overwrite its target. - GREEN `7941fd6e1d0f40e3fa8d119e1a9c37b6c4a54811`: `_write_receipt_atomic` writes and syncs a same-directory temporary file, then publishes with `os.replace`; `main()` no longer writes directly through `Path.write_text`. +- Gate follow-up `b48340b0f1819083670c2f01f04138099cf70aed`: restores the source file's final newline after the contents-API replacement so repository style/lint semantics remain unchanged; there is no behavioral delta. ## Security Notes From 4275f922fe8d2e1d4e0acb43661ae166fb957936 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 04:06:36 +0900 Subject: [PATCH 099/216] test(mir): reject linked scientific evidence paths --- ...ure_noninferiority_evidence_path_policy.py | 58 +++++++++++++++++++ 1 file changed, 58 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_noninferiority_evidence_path_policy.py diff --git a/services/analysis-engine/tests/test_structure_noninferiority_evidence_path_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_evidence_path_policy.py new file mode 100644 index 000000000..0d2dc83c3 --- /dev/null +++ b/services/analysis-engine/tests/test_structure_noninferiority_evidence_path_policy.py @@ -0,0 +1,58 @@ +"""Tests for structure-experiment evidence path admission.""" + +from __future__ import annotations + +import os +from pathlib import Path +from types import ModuleType + +import pytest +from conftest import load_module + + +def _validator() -> ModuleType: + """Load the repository-owned structure experiment validator.""" + return load_module( + "scripts/research/validate_structure_noninferiority.py", + "validate_structure_noninferiority_evidence_path_policy", + ) + + +def test_evidence_loader_rejects_symlink_when_no_nofollow_flag( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + """Fallback admission must reject a link instead of following its target.""" + validator = _validator() + evidence = tmp_path / "registration.json" + evidence.write_text("{}", encoding="utf-8") + + if hasattr(validator.os, "O_NOFOLLOW"): + monkeypatch.delattr(validator.os, "O_NOFOLLOW") + monkeypatch.setattr(Path, "is_symlink", lambda self: self == evidence) + + with pytest.raises(ValueError, match="symbolic link"): + validator._load_json(evidence) + + +def test_evidence_loader_uses_nofollow_when_platform_exposes_it( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + """No-follow platforms must pass the kernel flag at the open boundary.""" + validator = _validator() + evidence = tmp_path / "result.json" + evidence.write_text("{}", encoding="utf-8") + monkeypatch.setattr(validator.os, "O_NOFOLLOW", 0x40000000, raising=False) + original_open = os.open + observed_flags: list[int] = [] + + def recording_open(path: object, flags: int, *args: object, **kwargs: object) -> int: + observed_flags.append(flags) + return original_open(path, flags & ~0x40000000, *args, **kwargs) + + monkeypatch.setattr(validator.os, "open", recording_open) + + assert validator._load_json(evidence) == {} + assert observed_flags + assert observed_flags[0] & 0x40000000 From 8f68c11b85d31cbf04ea21d5bc9a62ac67aa9c8d Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 04:07:44 +0900 Subject: [PATCH 100/216] fix(mir): refuse linked evidence inputs --- .../validate_structure_noninferiority.py | 18 ++++++++++++++++-- 1 file changed, 16 insertions(+), 2 deletions(-) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index 2fda9f4ff..7b91e4dda 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -812,11 +812,25 @@ def _reject_json_constant(value: str) -> None: def _load_json(path: Path) -> object: - """Load one bounded regular UTF-8 JSON evidence file from one descriptor.""" - with path.open("rb") as handle: + """Load one bounded regular non-link UTF-8 JSON file from one descriptor.""" + flags = os.O_RDONLY + if hasattr(os, "O_BINARY"): + flags |= os.O_BINARY + if hasattr(os, "O_NOFOLLOW"): + flags |= os.O_NOFOLLOW + try: + fd = os.open(path, flags) + except OSError as exc: + raise ValueError( + f"evidence path could not be opened as a regular non-link file: {path.name}" + ) from exc + + with os.fdopen(fd, "rb", closefd=True) as handle: descriptor_stat = os.fstat(handle.fileno()) if not stat.S_ISREG(descriptor_stat.st_mode): raise ValueError(f"evidence path is not a regular file: {path.name}") + if not hasattr(os, "O_NOFOLLOW") and path.is_symlink(): + raise ValueError(f"evidence path must not be a symbolic link: {path.name}") if descriptor_stat.st_size > MAX_EVIDENCE_BYTES: raise ValueError( f"evidence file exceeds {MAX_EVIDENCE_BYTES} bytes: {path.name}" From b62bc821d6260ffb1baa06c6ef3e15eeefe8871a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 04:08:14 +0900 Subject: [PATCH 101/216] docs(mir): trace evidence no-follow admission --- .../mir/structure-evidence-path-admission.md | 40 +++++++++++++++++++ 1 file changed, 40 insertions(+) create mode 100644 docs/traceability/mir/structure-evidence-path-admission.md diff --git a/docs/traceability/mir/structure-evidence-path-admission.md b/docs/traceability/mir/structure-evidence-path-admission.md new file mode 100644 index 000000000..d0240f7c1 --- /dev/null +++ b/docs/traceability/mir/structure-evidence-path-admission.md @@ -0,0 +1,40 @@ +# Structure evidence path admission + +Status: Proposed +Owner: Signal-MIR Analysis +Tracking: #1225, #1228 + +## Problem + +The structure noninferiority validator treated registration/result JSON as bounded untrusted evidence, but opened caller-selected paths with `Path.open("rb")`. A symbolic link therefore resolved to its target before the regular-file `fstat` check. The parser still read one descriptor and enforced the 2 MiB bound, but the path boundary did not prove that the admitted filesystem object was the object named by the caller rather than a linked target. + +This is a narrower finding than corpus audio/annotation admission. The corpus tool already requests `O_NOFOLLOW` where available and has a fallback symbolic-link rejection. The result/registration validator did not have the equivalent boundary. + +## Decision + +Registration and result evidence paths are admitted as regular, non-link files. + +- Build the open flags from `O_RDONLY`, optional `O_BINARY`, and `O_NOFOLLOW` when the platform exposes it. +- Open once with `os.open` and perform `fstat`, size validation, bounded read, UTF-8 decode, duplicate-key rejection, and JSON parsing from that same descriptor. +- On platforms without `O_NOFOLLOW`, reject `Path.is_symlink()` after the descriptor is opened and before evidence bytes are consumed. This fallback closes ordinary linked-path admission but is not claimed as a kernel-atomic no-follow primitive. +- Do not resolve/canonicalize and then reopen the path: that would create a separate check/use pathname window. +- Do not broaden the validator into generic filesystem policy or corpus resource admission; this boundary owns only registration/result evidence files. + +MITRE classifies link following before file access as CWE-59 because an attacker-controlled filename can identify an unintended linked resource. The kernel no-follow flag is preferred when available because the rejection occurs at open rather than by trusting a pathname check performed earlier. + +## RED → GREEN evidence + +- RED `4275f922fe8d2e1d4e0acb43661ae166fb957936`: a platform without `O_NOFOLLOW` must reject an evidence path classified as a symbolic link; a platform exposing `O_NOFOLLOW` must pass that flag at the open boundary. +- GREEN `8f68c11b85d31cbf04ea21d5bc9a62ac67aa9c8d`: `_load_json` now uses `os.open`, optional `O_NOFOLLOW`, same-descriptor `fstat`, bounded read, and fallback link rejection. + +The tests deliberately exercise both branches without depending on developer-mode or elevated symlink privileges on Windows CI. + +## Security and claim boundary + +This repair prevents the validator from intentionally following a symbolic-link evidence path. It does not claim protection against every filesystem namespace attack, hard-link policy, privileged mount manipulation, or a hostile kernel/filesystem. Existing evidence constraints remain unchanged: regular file, at most 2 MiB, UTF-8 JSON, no duplicate object keys, no non-standard NaN/Infinity constants, and closed-world scientific schemas. + +## References + +MITRE. (2026). *CWE-59: Improper Link Resolution Before File Access ('Link Following')*. Common Weakness Enumeration, version 4.20. https://cwe.mitre.org/data/definitions/59.html + +Python Software Foundation. (2026). *os — Miscellaneous operating system interfaces*. Python 3 documentation. `os.O_NOFOLLOW` is used when exposed by the host platform. From 6f19315d0092004499fea7e581e1f00cc6d021c1 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 14:01:44 +0900 Subject: [PATCH 102/216] test(mir): freeze functional-label evaluation semantics --- ...ucture_functional_label_contract_policy.py | 60 +++++++++++++++++++ 1 file changed, 60 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_functional_label_contract_policy.py diff --git a/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py b/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py new file mode 100644 index 000000000..35d614c43 --- /dev/null +++ b/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py @@ -0,0 +1,60 @@ +"""Preregistration policy for functional-label ACC interpretation semantics.""" + +from __future__ import annotations + +import copy + +import pytest + +from test_structure_noninferiority_policy import _metrics, _registration, _validator + + +_FUNCTIONAL_ACC_CONTRACT = { + "frame_size_seconds": 0.1, + "annotation_contract_version": 1.0, + "label_mapping_contract_version": 1.0, +} + + +def _functional_metric(registration: dict[str, object]) -> dict[str, object]: + """Return the functional-label metric configuration from one registration.""" + metric = _metrics(registration)["functional_label_accuracy"] + assert isinstance(metric, dict) + return metric + + +def test_functional_label_accuracy_freezes_frame_and_annotation_semantics() -> None: + """ACC cannot be preregistered without one exact frame/parser/mapping contract.""" + validator = _validator() + registration = _registration() + _functional_metric(registration).update(_FUNCTIONAL_ACC_CONTRACT) + + validator.validate_registration(registration) + + for field in _FUNCTIONAL_ACC_CONTRACT: + missing = copy.deepcopy(registration) + del _functional_metric(missing)[field] + with pytest.raises(ValueError, match=field): + validator.validate_registration(missing) + + +def test_functional_label_accuracy_rejects_post_hoc_contract_drift() -> None: + """Frame resolution and interpretation versions are fixed before results.""" + validator = _validator() + registration = _registration() + _functional_metric(registration).update(_FUNCTIONAL_ACC_CONTRACT) + + drifted_frame = copy.deepcopy(registration) + _functional_metric(drifted_frame)["frame_size_seconds"] = 0.01 + with pytest.raises(ValueError, match="frame_size_seconds"): + validator.validate_registration(drifted_frame) + + drifted_parser = copy.deepcopy(registration) + _functional_metric(drifted_parser)["annotation_contract_version"] = 2.0 + with pytest.raises(ValueError, match="annotation_contract_version"): + validator.validate_registration(drifted_parser) + + drifted_mapping = copy.deepcopy(registration) + _functional_metric(drifted_mapping)["label_mapping_contract_version"] = 2.0 + with pytest.raises(ValueError, match="label_mapping_contract_version"): + validator.validate_registration(drifted_mapping) From 03b26bf203c4784029f75296d308eed644d7a5eb Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 14:03:12 +0900 Subject: [PATCH 103/216] fix(mir): preregister functional-label evaluation contract --- scripts/research/validate_structure_noninferiority.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index 7b91e4dda..fc82dd7b5 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -40,6 +40,9 @@ }, "functional_label_accuracy": { "implementation": "mirex2025.frame_level_accuracy", + "frame_size_seconds": 0.1, + "annotation_contract_version": 1.0, + "label_mapping_contract_version": 1.0, }, "repetition_pairwise_f": { "implementation": "mir_eval.segment.pairwise", From 4936cefd1a2b13fcd75a25e0d622ca4ce0ba9373 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 14:03:50 +0900 Subject: [PATCH 104/216] test(mir): align registrations with functional-label contract --- .../tests/test_structure_noninferiority_policy.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_policy.py index c9523cedc..f00bf0214 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_policy.py @@ -41,6 +41,9 @@ def _registration() -> dict[str, object]: }, "functional_label_accuracy": { "implementation": "mirex2025.frame_level_accuracy", + "frame_size_seconds": 0.1, + "annotation_contract_version": 1.0, + "label_mapping_contract_version": 1.0, "noninferiority_margin": 0.02, }, "repetition_pairwise_f": { From 1839586fb1446f28654494c154182b1c415ab757 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 14:04:37 +0900 Subject: [PATCH 105/216] test(mir): align corpus fixtures with label contract --- .../analysis-engine/tests/test_structure_corpus_admission.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_corpus_admission.py b/services/analysis-engine/tests/test_structure_corpus_admission.py index eb0a457c2..c56019423 100644 --- a/services/analysis-engine/tests/test_structure_corpus_admission.py +++ b/services/analysis-engine/tests/test_structure_corpus_admission.py @@ -39,6 +39,9 @@ def _registration(audio_hashes: list[str], annotation_hashes: list[str]) -> dict }, "functional_label_accuracy": { "implementation": "mirex2025.frame_level_accuracy", + "frame_size_seconds": 0.1, + "annotation_contract_version": 1.0, + "label_mapping_contract_version": 1.0, "noninferiority_margin": 0.02, }, "repetition_pairwise_f": { From df3335929ce501772edd0e3876d75eff37610662 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 14:05:37 +0900 Subject: [PATCH 106/216] docs(mir): freeze functional-label ACC interpretation contract --- .../mir/functional-label-accuracy-contract.md | 61 +++++++++++++++++++ 1 file changed, 61 insertions(+) create mode 100644 docs/traceability/mir/functional-label-accuracy-contract.md diff --git a/docs/traceability/mir/functional-label-accuracy-contract.md b/docs/traceability/mir/functional-label-accuracy-contract.md new file mode 100644 index 000000000..2e82a2cbd --- /dev/null +++ b/docs/traceability/mir/functional-label-accuracy-contract.md @@ -0,0 +1,61 @@ +# Functional-label accuracy preregistration contract + +Status: Proposed +Owner: Signal-MIR Analysis +Tracking: #1225, #1228 +Parent: `docs/traceability/mir/structure-feature-noninferiority.md` + +## Problem + +The structure noninferiority registration previously identified functional-label accuracy only as `mirex2025.frame_level_accuracy`. That name is not a complete measurement contract. The MIREX 2025 task describes frame-level ACC by converting system and ground-truth segmentations to a time series at a fine temporal resolution and gives 10 ms or 100 ms as examples; it does not fix one unique frame size. The same task also requires source dataset labels to be mapped to a target functional vocabulary before evaluation. + +Without freezing the frame grid and annotation/mapping interpretation before candidate results exist, two otherwise identical result receipts could report different ACC values while both claiming the same preregistered metric identity. That is post-result measurement freedom, not reproducible evidence. + +## Decision + +Schema v1 now requires the `functional_label_accuracy` metric registration to contain all of the following in addition to its noninferiority margin: + +- `implementation = mirex2025.frame_level_accuracy`; +- `frame_size_seconds = 0.1`; +- `annotation_contract_version = 1.0`; +- `label_mapping_contract_version = 1.0`. + +The 100 ms frame grid is a BandScope preregistration choice, not a claim that MIREX mandates 100 ms. MIREX 2025 explicitly gives both 10 ms and 100 ms as examples. BandScope selects 100 ms before results because the same experiment already preregisters a 100 ms frame size for `mir_eval.segment.pairwise`; using one fixed temporal grid avoids an otherwise unnecessary second discretization convention in this experiment. + +`annotation_contract_version = 1.0` means the scientific runner must interpret the admitted annotation snapshot as an ordered functional segmentation in seconds. After parsing, the normalized representation must start at 0.0, preserve source segment order, contain finite non-negative boundaries, have no overlaps or gaps between adjacent segments, and cover the admitted decoded track duration. The parser implementation remains part of the exact registered source commit and lock/runtime identity. Corpus admission itself intentionally does not parse or rewrite annotations; it only binds the immutable admitted bytes. A future runner must reject annotation bytes that cannot satisfy this semantic contract rather than repairing them after candidate results are visible. + +`label_mapping_contract_version = 1.0` uses a fail-closed identity mapping at the measurement boundary. Normalized annotations presented to ACC must already use exactly one of these seven lowercase labels: + +`intro`, `verse`, `chorus`, `bridge`, `inst`, `outro`, `silence`. + +No case folding, stemming, prefix stripping (`verse1` → `verse`), synonym mapping (`solo` → `inst`), or catch-all remapping is allowed inside the acceptance run. If a rights-cleared source corpus uses another vocabulary, its mapping must be reviewed and applied before preregistration; the resulting normalized annotation bytes receive their own `annotation_sha256` and therefore become part of the frozen corpus identity. + +This deliberately avoids silently resolving an ambiguity in the current MIREX 2025 page: its narrative description mentions an `other` category, while its operational output-format section says submitted labels must be one of `intro`, `verse`, `chorus`, `bridge`, `inst`, `outro`, or `silence`. BandScope v1 follows the operational output-format list and rejects `other` rather than guessing how it should map. If the scientific review later adopts another mapping, that requires a new mapping-contract version and a new preregistration digest before results are inspected. + +## RED → GREEN lineage + +RED `6f19315d0092004499fea7e581e1f00cc6d021c1` added a focused policy test requiring functional-label ACC to freeze frame resolution plus annotation and label-mapping contract versions. The predecessor validator rejected those fields as unregistered and still accepted the older under-specified metric shape. + +GREEN `03b26bf203c4784029f75296d308eed644d7a5eb` makes those three fields mandatory closed-world configuration for `functional_label_accuracy`. Fixture alignment `4936cefd1a2b13fcd75a25e0d622ca4ce0ba9373` and `1839586fb1446f28654494c154182b1c415ab757` updates the shared evidence and corpus-admission registrations so successful tests exercise the new contract rather than bypassing it. + +The focused regression also requires missing fields and post-hoc changes to the 100 ms frame, annotation contract version, or label mapping contract version to fail validation. + +## Constraints and rejected alternatives + +Leaving frame resolution to the eventual runner was rejected because the same segment intervals can produce different frame-count weighting at 10 ms and 100 ms. A result receipt would then not identify one reproducible ACC statistic. + +Treating the MIREX task name as sufficient label-mapping authority was rejected because the current task page itself contains conflicting target-vocabulary prose. A scientific gate cannot depend on an implicit interpretation of that inconsistency. + +Automatically normalizing labels during the acceptance run was rejected. Mapping `verse1`, `solo`, `fade-out`, mixed case, or unknown labels after the corpus is frozen gives the runner post-hoc freedom over the dependent variable. Normalization belongs before preregistration and is content-addressed through the annotation digest. + +Changing corpus admission to parse annotations was rejected. Resource admission owns byte identity and immutable handoff; metric interpretation belongs to the Signal-MIR measurement runner. Combining them would blur the bounded-context boundary and make a byte-admission receipt look like scientific acceptance. + +## Claim boundary + +This contract freezes how a future BandScope runner must interpret functional-label ACC evidence. It does not establish that the chosen corpus is sufficiently broad, that the current synthetic unit-test margins are scientifically approved, that 100 ms is optimal for every MIR task, or that a production CQT→STFT switch is justified. Production acceptance still requires rights-cleared real decoded audio, independently reviewed normalized annotations, an implemented recognized metric runner consuming the exact admitted PCM/annotation snapshots, preregistered aggregation and paired uncertainty, and current-head release evidence. + +## References + +MIREX. (2025). *Music Structure Analysis*. International Music Information Retrieval Systems Evaluation Laboratory. https://music-ir.org/mirex/wiki/2025:Music_Structure_Analysis + +MIREX. (2025). *Music Structure Analysis Results*. International Music Information Retrieval Systems Evaluation Laboratory. https://music-ir.org/mirex/wiki/2025:Music_Structure_Analysis_Results From c083c37ca9a662113ad6d06c7775667fd7ac7e69 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 14:06:25 +0900 Subject: [PATCH 107/216] docs(mir): bind ACC frame and annotation semantics --- .../mir/structure-feature-noninferiority.md | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index 6653e410c..0b6bfdf86 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -20,7 +20,7 @@ A registration is valid only when it records all of the following before the res - baseline `chroma_cqt` and candidate `chroma_stft`; - a rights-cleared real-audio corpus with stable track IDs, content-unique audio SHA-256 identities, annotation SHA-256, rights basis, and provenance URI; - exact source commit and `uv.lock` identity plus Python, librosa, NumPy, sample rate, channel count, and host profile; -- the complete metric implementation contract and noninferiority/speed thresholds; +- the complete metric implementation contract and noninferiority/speed thresholds, including functional-label ACC frame resolution plus versioned annotation and label-mapping semantics; - the paired-uncertainty procedure identity, confidence level, resample count, and random seed; - a non-empty claim boundary stating the population/runtime scope to which a passing result may be applied. @@ -49,12 +49,12 @@ BandScope uses MIREX 2025 Music Structure Analysis as the functional-structure r | Boundary precision/recall/F at 0.5 s | `mir_eval.segment.detection`, 0.5 s window | F noninferiority | | Boundary precision/recall/F at 3.0 s | `mir_eval.segment.detection`, 3.0 s window | F noninferiority | | Boundary median deviation, both directions | `mir_eval.segment.deviation` convention | Report per track and aggregate | -| Functional-label accuracy | MIREX 2025 frame-level ACC convention | Noninferiority | +| Functional-label accuracy | MIREX 2025 frame-level ACC; 0.1 s grid; annotation contract v1; label-mapping contract v1 | Noninferiority | | Repetition/group consistency precision/recall/F | `mir_eval.segment.pairwise`, 0.1 s frame size | F noninferiority | | Latency | paired p50/p95 on the registered host | p95 ratio superiority | | Peak memory | peak RSS on the registered host | Report per track and aggregate | -MIREX 2025 evaluates functional structure with frame-level label accuracy plus boundary hit-rate F measures at 0.5 s and 3.0 s. `mir_eval.segment.detection` uses one-to-one boundary matching within the selected tolerance; `mir_eval.segment.deviation` reports the median nearest-boundary deviations in both directions; `mir_eval.segment.pairwise` measures structural grouping agreement. These are different questions and must not be collapsed into one score. +MIREX 2025 evaluates functional structure with frame-level label accuracy plus boundary hit-rate F measures at 0.5 s and 3.0 s. Its ACC description gives 10 ms and 100 ms as example frame resolutions rather than fixing a single grid. BandScope therefore freezes 100 ms before measurement and versions the annotation and label-mapping semantics instead of leaving those choices to the eventual runner. The detailed v1 contract and the MIREX vocabulary ambiguity are recorded in `docs/traceability/mir/functional-label-accuracy-contract.md`. `mir_eval.segment.detection` uses one-to-one boundary matching within the selected tolerance; `mir_eval.segment.deviation` reports the median nearest-boundary deviations in both directions; `mir_eval.segment.pairwise` measures structural grouping agreement. These are different questions and must not be collapsed into one score. Result receipts must carry the reported precision and recall alongside F for the boundary and repetition measures. The admission validator recomputes the harmonic mean and rejects an internally inconsistent P/R/F triplet. This does not replace the recognized metric implementation; it prevents a malformed receipt from claiming a metric value that its own reported components cannot support. @@ -96,7 +96,7 @@ The JSON `source_uri` value remains provenance metadata only; it is never derefe ## Reproducibility sequence -1. Review the rights basis, corpus composition, distinct audio-content identities, feature hypothesis, metric contract, host profile, numeric margins, aggregation procedure, paired-uncertainty procedure, and claim boundary **before** running the candidate. Freeze the procedure identifier, confidence level, resample count, random seed, and claim boundary in the registration. If repeated recordings, clustering, or any failure/exclusion tolerance is scientifically required, define and version that dependence/exclusion policy before measurement rather than aliasing or omitting observations after results are visible. +1. Review the rights basis, corpus composition, distinct audio-content identities, feature hypothesis, metric contract, functional-label frame/annotation/mapping contract, host profile, numeric margins, aggregation procedure, paired-uncertainty procedure, and claim boundary **before** running the candidate. Freeze the procedure identifier, confidence level, resample count, random seed, and claim boundary in the registration. If repeated recordings, clustering, source-label normalization, or any failure/exclusion tolerance is scientifically required, define and version that dependence/mapping/exclusion policy before measurement rather than remapping or omitting observations after results are visible. 2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain its canonical SHA-256. 3. Run baseline and candidate on the same decoded track identities and host profile. The experiment runner must compute the approved aggregation and paired uncertainty procedure from the complete measured-track evidence rather than accepting caller-authored aggregate numbers as scientific authority. Record recognized MIR metrics, p50/p95 latency, and peak RSS for successful tracks. If a registered track cannot produce a baseline/candidate measurement, preserve its ordered `track_id` receipt, add that exact ID to `failed_tracks`, and set the aggregate plus both CI summary fields to JSON `null`; do not invent metric values or compute a complete-case aggregate from the surviving tracks. Only a complete run records aggregate outputs and paired uncertainty using the preregistered procedure. 4. Put the registration digest, identical uncertainty-plan fields, and identical claim boundary in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. Under schema v1, any recorded failed track keeps the diagnostic receipt valid but makes acceptance fail closed before aggregate decision rules; undeclared missing measurements, non-null post-failure summaries, and claim-boundary drift are rejected. @@ -110,7 +110,7 @@ No step authorizes committing licensed audio to Git. Rights-cleared means BandSc - Evidence JSON is limited to 2 MiB, must be a regular file read through one open descriptor, must decode as UTF-8, and rejects duplicate keys plus non-standard `NaN`/`Infinity` constants. - `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute, relative, drive-relative, and `file:` forms are rejected; an explicit non-file URI scheme is required so transient workstation paths cannot become provenance authority or leak into review artifacts. - Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt; duplicate audio SHA-256 values under different track IDs are rejected so one recording cannot be silently counted multiple times. -- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, uncertainty-plan drift, claim-boundary drift, registration drift, undeclared missing measurements, measurement-bearing failed receipts, non-null aggregate/CI summaries after any failed track, unregistered registration/hypothesis/metric/corpus/runtime/result/track/aggregate/measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. +- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, functional-label frame/annotation/mapping drift, uncertainty-plan drift, claim-boundary drift, registration drift, undeclared missing measurements, measurement-bearing failed receipts, non-null aggregate/CI summaries after any failed track, unregistered registration/hypothesis/metric/corpus/runtime/result/track/aggregate/measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. - The validator does not claim that SHA-256 proves licensing, annotation validity, scientific adequacy, independence beyond exact-byte uniqueness, aggregate derivation, or that the registered statistical procedure is appropriate. Rights and scientific review remain separate gates. ## References From ff8aa8d478c9c8aa05d09d331a5b3ce71b31fdbd Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 14:09:11 +0900 Subject: [PATCH 108/216] docs(mir): explain best-effort cleanup boundaries --- scripts/research/verify_structure_corpus.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/scripts/research/verify_structure_corpus.py b/scripts/research/verify_structure_corpus.py index 414674fdc..35bbbea54 100644 --- a/scripts/research/verify_structure_corpus.py +++ b/scripts/research/verify_structure_corpus.py @@ -196,6 +196,7 @@ def _decode_pcm_identity(fd: int, target_sample_rate_hz: int) -> tuple[str, int, try: os.close(duplicate_fd) except OSError: + # The file object may already have closed the duplicate; preserve the decode error. pass raise if int(actual_sample_rate_hz) != target_sample_rate_hz: @@ -452,6 +453,7 @@ def _write_receipt_atomic(path: Path, receipt: Mapping[str, object]) -> None: try: temporary_path.unlink() except FileNotFoundError: + # Replacement or prior cleanup may have consumed the temp path already. pass raise From cc4a166cc9e1496cd562f7d79b9ffa30f6ca19c2 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 14:13:10 +0900 Subject: [PATCH 109/216] test(mir): define normalized functional annotation parser contract --- ..._structure_functional_annotation_parser.py | 85 +++++++++++++++++++ 1 file changed, 85 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_functional_annotation_parser.py diff --git a/services/analysis-engine/tests/test_structure_functional_annotation_parser.py b/services/analysis-engine/tests/test_structure_functional_annotation_parser.py new file mode 100644 index 000000000..6ccab0b1a --- /dev/null +++ b/services/analysis-engine/tests/test_structure_functional_annotation_parser.py @@ -0,0 +1,85 @@ +"""Tests for the preregistered functional-annotation interpretation contract.""" + +from __future__ import annotations + +from fractions import Fraction +from types import ModuleType + +import pytest +from conftest import load_module + + +def _parser() -> ModuleType: + """Load the repository-owned normalized functional annotation parser.""" + return load_module( + "scripts/research/parse_structure_functional_annotations.py", + "parse_structure_functional_annotations", + ) + + +def _readonly(payload: bytes) -> memoryview: + return memoryview(payload) + + +def test_parser_accepts_exact_mirex_tsv_and_preserves_rational_boundaries() -> None: + parser = _parser() + segments = parser.parse_functional_annotations( + _readonly(b"0.0\t1.5\tintro\n1.5\t4.0\tverse\n"), + decoded_frames=4000, + sample_rate_hz=1000, + ) + + assert [(segment.start, segment.end, segment.label) for segment in segments] == [ + (Fraction(0, 1), Fraction(3, 2), "intro"), + (Fraction(3, 2), Fraction(4, 1), "verse"), + ] + + +def test_parser_rejects_mapping_freedom_and_noncanonical_labels() -> None: + parser = _parser() + for label in ("Verse", "verse1", "solo", "other", " verse"): + payload = f"0.0\t1.0\t{label}\n".encode() + with pytest.raises(ValueError, match="functional label"): + parser.parse_functional_annotations( + _readonly(payload), + decoded_frames=1000, + sample_rate_hz=1000, + ) + + +def test_parser_rejects_gap_overlap_and_incomplete_duration() -> None: + parser = _parser() + invalid_payloads = ( + b"0.0\t1.0\tintro\n1.1\t2.0\tverse\n", + b"0.0\t1.1\tintro\n1.0\t2.0\tverse\n", + b"0.0\t1.5\tintro\n", + ) + for payload in invalid_payloads: + with pytest.raises(ValueError, match="continuous|duration"): + parser.parse_functional_annotations( + _readonly(payload), + decoded_frames=2000, + sample_rate_hz=1000, + ) + + +def test_parser_rejects_malformed_nonfinite_or_mutable_annotation_input() -> None: + parser = _parser() + with pytest.raises(ValueError, match="three tab-separated"): + parser.parse_functional_annotations( + _readonly(b"0.0\tintro\n"), + decoded_frames=1000, + sample_rate_hz=1000, + ) + with pytest.raises(ValueError, match="finite decimal"): + parser.parse_functional_annotations( + _readonly(b"0.0\tNaN\tintro\n"), + decoded_frames=1000, + sample_rate_hz=1000, + ) + with pytest.raises(ValueError, match="read-only"): + parser.parse_functional_annotations( + memoryview(bytearray(b"0.0\t1.0\tintro\n")), + decoded_frames=1000, + sample_rate_hz=1000, + ) From 0ee4a6aed49c8f001da451ca63c5f20e52832f5c Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 14:13:36 +0900 Subject: [PATCH 110/216] feat(mir): parse preregistered functional annotation contract --- .../parse_structure_functional_annotations.py | 116 ++++++++++++++++++ 1 file changed, 116 insertions(+) create mode 100644 scripts/research/parse_structure_functional_annotations.py diff --git a/scripts/research/parse_structure_functional_annotations.py b/scripts/research/parse_structure_functional_annotations.py new file mode 100644 index 000000000..e629688cb --- /dev/null +++ b/scripts/research/parse_structure_functional_annotations.py @@ -0,0 +1,116 @@ +#!/usr/bin/env python3 +"""Parse normalized functional annotations for the structure experiment. + +This module owns only the preregistered annotation-interpretation boundary. It +accepts the immutable annotation snapshot already admitted by the corpus tool +and returns exact rational segment boundaries. It does not normalize source +labels, reopen corpus paths, calculate MIR metrics, or choose scientific +thresholds. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from decimal import Decimal, InvalidOperation +from fractions import Fraction + +_ALLOWED_FUNCTIONAL_LABELS = frozenset( + {"intro", "verse", "chorus", "bridge", "inst", "outro", "silence"} +) + + +@dataclass(frozen=True, slots=True) +class FunctionalSegment: + """One normalized functional segment with exact rational boundaries.""" + + start: Fraction + end: Fraction + label: str + + +def _positive_integer(value: object, field: str) -> int: + """Return a positive integer while rejecting booleans and coercion.""" + if isinstance(value, bool) or not isinstance(value, int) or value <= 0: + raise ValueError(f"{field} must be a positive integer") + return value + + +def _finite_decimal(token: str, field: str) -> Fraction: + """Parse one finite decimal token to an exact rational value.""" + if token != token.strip() or not token: + raise ValueError(f"{field} must be a finite decimal without whitespace") + try: + value = Decimal(token) + except InvalidOperation as exc: + raise ValueError(f"{field} must be a finite decimal") from exc + if not value.is_finite(): + raise ValueError(f"{field} must be a finite decimal") + return Fraction(value) + + +def parse_functional_annotations( + annotation_bytes: memoryview, + *, + decoded_frames: int, + sample_rate_hz: int, +) -> tuple[FunctionalSegment, ...]: + """Parse the v1 MIREX-style TSV annotation contract. + + Each non-empty line must be ``startendlabel``. The admitted + annotation must be read-only, start at zero, remain gap/overlap-free, and + cover the decoded signal exactly. Functional labels are already normalized + before preregistration; this parser intentionally performs no case folding, + synonym mapping, or suffix stripping. + """ + if not isinstance(annotation_bytes, memoryview) or not annotation_bytes.readonly: + raise ValueError("annotation input must be a read-only memoryview") + + frames = _positive_integer(decoded_frames, "decoded_frames") + sample_rate = _positive_integer(sample_rate_hz, "sample_rate_hz") + track_duration = Fraction(frames, sample_rate) + + try: + text = bytes(annotation_bytes).decode("utf-8") + except UnicodeDecodeError as exc: + raise ValueError("annotation input must be valid UTF-8") from exc + lines = text.splitlines() + if not lines: + raise ValueError("annotation input must contain at least one segment") + + segments: list[FunctionalSegment] = [] + previous_end = Fraction(0, 1) + for index, line in enumerate(lines, start=1): + if not line: + raise ValueError(f"annotation line {index} must not be blank") + fields = line.split("\t") + if len(fields) != 3: + raise ValueError( + f"annotation line {index} must contain exactly three tab-separated fields" + ) + start_token, end_token, label = fields + start = _finite_decimal(start_token, f"annotation line {index} start") + end = _finite_decimal(end_token, f"annotation line {index} end") + if label not in _ALLOWED_FUNCTIONAL_LABELS: + raise ValueError( + f"annotation line {index} functional label is not registered: {label!r}" + ) + if start < 0 or end <= start: + raise ValueError( + f"annotation line {index} must have non-negative start and end > start" + ) + if index == 1 and start != 0: + raise ValueError("functional annotation must start at 0.0 seconds") + if start != previous_end: + raise ValueError( + f"functional annotation must be continuous at line {index}" + ) + if end > track_duration: + raise ValueError( + f"annotation line {index} exceeds decoded track duration" + ) + segments.append(FunctionalSegment(start=start, end=end, label=label)) + previous_end = end + + if previous_end != track_duration: + raise ValueError("functional annotation must cover the decoded track duration") + return tuple(segments) From 87e0cdebd43960420bd0b8d779d0e2960614111a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 14:14:02 +0900 Subject: [PATCH 111/216] docs(mir): bind normalized annotation parser implementation --- .../mir/functional-label-accuracy-contract.md | 14 +++++++++++--- 1 file changed, 11 insertions(+), 3 deletions(-) diff --git a/docs/traceability/mir/functional-label-accuracy-contract.md b/docs/traceability/mir/functional-label-accuracy-contract.md index 2e82a2cbd..89e3729f8 100644 --- a/docs/traceability/mir/functional-label-accuracy-contract.md +++ b/docs/traceability/mir/functional-label-accuracy-contract.md @@ -22,7 +22,7 @@ Schema v1 now requires the `functional_label_accuracy` metric registration to co The 100 ms frame grid is a BandScope preregistration choice, not a claim that MIREX mandates 100 ms. MIREX 2025 explicitly gives both 10 ms and 100 ms as examples. BandScope selects 100 ms before results because the same experiment already preregisters a 100 ms frame size for `mir_eval.segment.pairwise`; using one fixed temporal grid avoids an otherwise unnecessary second discretization convention in this experiment. -`annotation_contract_version = 1.0` means the scientific runner must interpret the admitted annotation snapshot as an ordered functional segmentation in seconds. After parsing, the normalized representation must start at 0.0, preserve source segment order, contain finite non-negative boundaries, have no overlaps or gaps between adjacent segments, and cover the admitted decoded track duration. The parser implementation remains part of the exact registered source commit and lock/runtime identity. Corpus admission itself intentionally does not parse or rewrite annotations; it only binds the immutable admitted bytes. A future runner must reject annotation bytes that cannot satisfy this semantic contract rather than repairing them after candidate results are visible. +`annotation_contract_version = 1.0` is implemented by `scripts/research/parse_structure_functional_annotations.py`. The parser consumes only the read-only annotation snapshot already admitted by the corpus boundary. Version 1 uses UTF-8 MIREX-style TSV with exactly three fields per segment: `startendlabel`. Times are parsed as finite decimal values into exact rational boundaries rather than binary floating-point approximations. The normalized segmentation must start at 0.0, preserve source order, have strictly positive segment duration, contain no gaps or overlaps, and end exactly at `decoded_frames / sample_rate_hz`. The parser never reopens a corpus path or repairs malformed evidence. `label_mapping_contract_version = 1.0` uses a fail-closed identity mapping at the measurement boundary. Normalized annotations presented to ACC must already use exactly one of these seven lowercase labels: @@ -32,13 +32,17 @@ No case folding, stemming, prefix stripping (`verse1` → `verse`), synonym mapp This deliberately avoids silently resolving an ambiguity in the current MIREX 2025 page: its narrative description mentions an `other` category, while its operational output-format section says submitted labels must be one of `intro`, `verse`, `chorus`, `bridge`, `inst`, `outro`, or `silence`. BandScope v1 follows the operational output-format list and rejects `other` rather than guessing how it should map. If the scientific review later adopts another mapping, that requires a new mapping-contract version and a new preregistration digest before results are inspected. +Corpus admission intentionally does not parse or rewrite annotations. Resource admission owns byte identity and immutable handoff; Signal-MIR owns interpretation and measurement. The future experiment runner must call the versioned parser on the admitted read-only annotation view and then evaluate both CQT and STFT against that same parsed evidence. + ## RED → GREEN lineage RED `6f19315d0092004499fea7e581e1f00cc6d021c1` added a focused policy test requiring functional-label ACC to freeze frame resolution plus annotation and label-mapping contract versions. The predecessor validator rejected those fields as unregistered and still accepted the older under-specified metric shape. GREEN `03b26bf203c4784029f75296d308eed644d7a5eb` makes those three fields mandatory closed-world configuration for `functional_label_accuracy`. Fixture alignment `4936cefd1a2b13fcd75a25e0d622ca4ce0ba9373` and `1839586fb1446f28654494c154182b1c415ab757` updates the shared evidence and corpus-admission registrations so successful tests exercise the new contract rather than bypassing it. -The focused regression also requires missing fields and post-hoc changes to the 100 ms frame, annotation contract version, or label mapping contract version to fail validation. +Parser RED `cc4a166cc9e1496cd562f7d79b9ffa30f6ca19c2` defines the executable v1 interpretation contract: exact three-column TSV, exact rational boundaries, continuous full-duration coverage, registered labels only, finite times, and read-only input. GREEN `0ee4a6aed49c8f001da451ca63c5f20e52832f5c` implements that contract without adding label normalization, filesystem access, metric computation, or scientific threshold logic. + +The focused regressions require missing preregistration fields and post-hoc changes to the 100 ms frame, annotation contract version, or label mapping contract version to fail validation, and they reject mapping freedom such as `Verse`, `verse1`, `solo`, or `other` at the parser boundary. ## Constraints and rejected alternatives @@ -48,12 +52,16 @@ Treating the MIREX task name as sufficient label-mapping authority was rejected Automatically normalizing labels during the acceptance run was rejected. Mapping `verse1`, `solo`, `fade-out`, mixed case, or unknown labels after the corpus is frozen gives the runner post-hoc freedom over the dependent variable. Normalization belongs before preregistration and is content-addressed through the annotation digest. -Changing corpus admission to parse annotations was rejected. Resource admission owns byte identity and immutable handoff; metric interpretation belongs to the Signal-MIR measurement runner. Combining them would blur the bounded-context boundary and make a byte-admission receipt look like scientific acceptance. +Parsing annotation bytes inside corpus admission was rejected. Resource admission owns byte identity and immutable handoff; metric interpretation belongs to the Signal-MIR measurement runner. Combining them would blur the bounded-context boundary and make a byte-admission receipt look like scientific acceptance. + +Using binary floating-point as the annotation-boundary authority was rejected. Decimal source timestamps are converted to exact rational values, while decoded duration is represented exactly as `decoded_frames / sample_rate_hz`; this makes gap/overlap and full-duration checks deterministic rather than tolerance-dependent. ## Claim boundary This contract freezes how a future BandScope runner must interpret functional-label ACC evidence. It does not establish that the chosen corpus is sufficiently broad, that the current synthetic unit-test margins are scientifically approved, that 100 ms is optimal for every MIR task, or that a production CQT→STFT switch is justified. Production acceptance still requires rights-cleared real decoded audio, independently reviewed normalized annotations, an implemented recognized metric runner consuming the exact admitted PCM/annotation snapshots, preregistered aggregation and paired uncertainty, and current-head release evidence. +The parser tests use synthetic annotation strings only to verify the interpretation boundary. They are unit evidence and are not counted as production scientific acceptance. + ## References MIREX. (2025). *Music Structure Analysis*. International Music Information Retrieval Systems Evaluation Laboratory. https://music-ir.org/mirex/wiki/2025:Music_Structure_Analysis From 52b6c7362baa00151f908a24cadbb1efeddd2101 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 14:15:51 +0900 Subject: [PATCH 112/216] docs(mir): distinguish local TSV from MIREX output format --- .../parse_structure_functional_annotations.py | 14 ++++++++------ 1 file changed, 8 insertions(+), 6 deletions(-) diff --git a/scripts/research/parse_structure_functional_annotations.py b/scripts/research/parse_structure_functional_annotations.py index e629688cb..48a68efc6 100644 --- a/scripts/research/parse_structure_functional_annotations.py +++ b/scripts/research/parse_structure_functional_annotations.py @@ -54,12 +54,14 @@ def parse_functional_annotations( decoded_frames: int, sample_rate_hz: int, ) -> tuple[FunctionalSegment, ...]: - """Parse the v1 MIREX-style TSV annotation contract. - - Each non-empty line must be ``startendlabel``. The admitted - annotation must be read-only, start at zero, remain gap/overlap-free, and - cover the decoded signal exactly. Functional labels are already normalized - before preregistration; this parser intentionally performs no case folding, + """Parse the BandScope v1 normalized three-column TSV contract. + + Each non-empty line must be ``startendlabel``. This local storage + contract is not the MIREX 2025 submission-file syntax; it freezes the same + start/end/label semantics before evaluation. The admitted annotation must be + read-only, start at zero, remain gap/overlap-free, and cover the decoded + signal exactly. Functional labels are already normalized before + preregistration; this parser intentionally performs no case folding, synonym mapping, or suffix stripping. """ if not isinstance(annotation_bytes, memoryview) or not annotation_bytes.readonly: From c84d3ffcba6a2e00da1209076e1a902f9d21a134 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 14:16:16 +0900 Subject: [PATCH 113/216] docs(mir): align local annotation format with MIREX authority --- .../mir/functional-label-accuracy-contract.md | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/docs/traceability/mir/functional-label-accuracy-contract.md b/docs/traceability/mir/functional-label-accuracy-contract.md index 89e3729f8..e38bb4468 100644 --- a/docs/traceability/mir/functional-label-accuracy-contract.md +++ b/docs/traceability/mir/functional-label-accuracy-contract.md @@ -22,7 +22,9 @@ Schema v1 now requires the `functional_label_accuracy` metric registration to co The 100 ms frame grid is a BandScope preregistration choice, not a claim that MIREX mandates 100 ms. MIREX 2025 explicitly gives both 10 ms and 100 ms as examples. BandScope selects 100 ms before results because the same experiment already preregisters a 100 ms frame size for `mir_eval.segment.pairwise`; using one fixed temporal grid avoids an otherwise unnecessary second discretization convention in this experiment. -`annotation_contract_version = 1.0` is implemented by `scripts/research/parse_structure_functional_annotations.py`. The parser consumes only the read-only annotation snapshot already admitted by the corpus boundary. Version 1 uses UTF-8 MIREX-style TSV with exactly three fields per segment: `startendlabel`. Times are parsed as finite decimal values into exact rational boundaries rather than binary floating-point approximations. The normalized segmentation must start at 0.0, preserve source order, have strictly positive segment duration, contain no gaps or overlaps, and end exactly at `decoded_frames / sample_rate_hz`. The parser never reopens a corpus path or repairs malformed evidence. +`annotation_contract_version = 1.0` is implemented by `scripts/research/parse_structure_functional_annotations.py`. The parser consumes only the read-only annotation snapshot already admitted by the corpus boundary. Version 1 uses a BandScope-local UTF-8 normalized TSV with exactly three fields per segment: `startendlabel`. This is **not** the MIREX 2025 submission-file syntax: the MIREX task specifies a text/JSON-parsable list of per-track segment predictions. BandScope's local TSV is a deliberately simpler frozen corpus-annotation representation that preserves the same start/end/label semantics before evaluation. + +Times are parsed as finite decimal values into exact rational boundaries rather than binary floating-point approximations. The normalized segmentation must start at 0.0, preserve source order, have strictly positive segment duration, contain no gaps or overlaps, and end exactly at `decoded_frames / sample_rate_hz`. The parser never reopens a corpus path or repairs malformed evidence. `label_mapping_contract_version = 1.0` uses a fail-closed identity mapping at the measurement boundary. Normalized annotations presented to ACC must already use exactly one of these seven lowercase labels: @@ -40,7 +42,7 @@ RED `6f19315d0092004499fea7e581e1f00cc6d021c1` added a focused policy test requi GREEN `03b26bf203c4784029f75296d308eed644d7a5eb` makes those three fields mandatory closed-world configuration for `functional_label_accuracy`. Fixture alignment `4936cefd1a2b13fcd75a25e0d622ca4ce0ba9373` and `1839586fb1446f28654494c154182b1c415ab757` updates the shared evidence and corpus-admission registrations so successful tests exercise the new contract rather than bypassing it. -Parser RED `cc4a166cc9e1496cd562f7d79b9ffa30f6ca19c2` defines the executable v1 interpretation contract: exact three-column TSV, exact rational boundaries, continuous full-duration coverage, registered labels only, finite times, and read-only input. GREEN `0ee4a6aed49c8f001da451ca63c5f20e52832f5c` implements that contract without adding label normalization, filesystem access, metric computation, or scientific threshold logic. +Parser RED `cc4a166cc9e1496cd562f7d79b9ffa30f6ca19c2` defines the executable v1 interpretation contract: exact three-column local TSV, exact rational boundaries, continuous full-duration coverage, registered labels only, finite times, and read-only input. GREEN `0ee4a6aed49c8f001da451ca63c5f20e52832f5c` implements that contract without adding label normalization, filesystem access, metric computation, or scientific threshold logic. Authority correction `52b6c7362baa00151f908a24cadbb1efeddd2101` removes the inaccurate implication that this local TSV is the MIREX 2025 submission format while retaining the same executable semantics. The focused regressions require missing preregistration fields and post-hoc changes to the 100 ms frame, annotation contract version, or label mapping contract version to fail validation, and they reject mapping freedom such as `Verse`, `verse1`, `solo`, or `other` at the parser boundary. @@ -56,6 +58,8 @@ Parsing annotation bytes inside corpus admission was rejected. Resource admissio Using binary floating-point as the annotation-boundary authority was rejected. Decimal source timestamps are converted to exact rational values, while decoded duration is represented exactly as `decoded_frames / sample_rate_hz`; this makes gap/overlap and full-duration checks deterministic rather than tolerance-dependent. +Copying the MIREX submission envelope into local corpus annotations was rejected. Submission transport syntax and local preregistered annotation storage solve different problems; the scientific contract is the segment boundaries, labels, normalization, and evaluation semantics, all of which remain explicitly bound here. + ## Claim boundary This contract freezes how a future BandScope runner must interpret functional-label ACC evidence. It does not establish that the chosen corpus is sufficiently broad, that the current synthetic unit-test margins are scientifically approved, that 100 ms is optimal for every MIR task, or that a production CQT→STFT switch is justified. Production acceptance still requires rights-cleared real decoded audio, independently reviewed normalized annotations, an implemented recognized metric runner consuming the exact admitted PCM/annotation snapshots, preregistered aggregation and paired uncertainty, and current-head release evidence. From 73ddb084c6e7279f4fc3d92ee9c8d0c8edf50071 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 14:16:39 +0900 Subject: [PATCH 114/216] test(mir): freeze functional ACC frame-grid semantics --- .../test_structure_functional_label_contract_policy.py | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py b/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py index 35d614c43..2f74d9827 100644 --- a/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py +++ b/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py @@ -11,6 +11,7 @@ _FUNCTIONAL_ACC_CONTRACT = { "frame_size_seconds": 0.1, + "frame_grid_contract_version": 1.0, "annotation_contract_version": 1.0, "label_mapping_contract_version": 1.0, } @@ -49,6 +50,11 @@ def test_functional_label_accuracy_rejects_post_hoc_contract_drift() -> None: with pytest.raises(ValueError, match="frame_size_seconds"): validator.validate_registration(drifted_frame) + drifted_grid = copy.deepcopy(registration) + _functional_metric(drifted_grid)["frame_grid_contract_version"] = 2.0 + with pytest.raises(ValueError, match="frame_grid_contract_version"): + validator.validate_registration(drifted_grid) + drifted_parser = copy.deepcopy(registration) _functional_metric(drifted_parser)["annotation_contract_version"] = 2.0 with pytest.raises(ValueError, match="annotation_contract_version"): From cd069263a947a762a1cc66e8f85bfee7368180a8 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 14:17:12 +0900 Subject: [PATCH 115/216] test(mir): defer ACC frame-grid rule pending evaluator contract --- .../test_structure_functional_label_contract_policy.py | 6 ------ 1 file changed, 6 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py b/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py index 2f74d9827..35d614c43 100644 --- a/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py +++ b/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py @@ -11,7 +11,6 @@ _FUNCTIONAL_ACC_CONTRACT = { "frame_size_seconds": 0.1, - "frame_grid_contract_version": 1.0, "annotation_contract_version": 1.0, "label_mapping_contract_version": 1.0, } @@ -50,11 +49,6 @@ def test_functional_label_accuracy_rejects_post_hoc_contract_drift() -> None: with pytest.raises(ValueError, match="frame_size_seconds"): validator.validate_registration(drifted_frame) - drifted_grid = copy.deepcopy(registration) - _functional_metric(drifted_grid)["frame_grid_contract_version"] = 2.0 - with pytest.raises(ValueError, match="frame_grid_contract_version"): - validator.validate_registration(drifted_grid) - drifted_parser = copy.deepcopy(registration) _functional_metric(drifted_parser)["annotation_contract_version"] = 2.0 with pytest.raises(ValueError, match="annotation_contract_version"): From 27304ac52ad529f7e96d407af2440c808e1d7aae Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 14:17:44 +0900 Subject: [PATCH 116/216] docs(mir): record unresolved ACC frame-grid evaluator choice --- .../mir/functional-label-accuracy-contract.md | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/docs/traceability/mir/functional-label-accuracy-contract.md b/docs/traceability/mir/functional-label-accuracy-contract.md index e38bb4468..278f90ae7 100644 --- a/docs/traceability/mir/functional-label-accuracy-contract.md +++ b/docs/traceability/mir/functional-label-accuracy-contract.md @@ -46,6 +46,14 @@ Parser RED `cc4a166cc9e1496cd562f7d79b9ffa30f6ca19c2` defines the executable v1 The focused regressions require missing preregistration fields and post-hoc changes to the 100 ms frame, annotation contract version, or label mapping contract version to fail validation, and they reject mapping freedom such as `Verse`, `verse1`, `solo`, or `other` at the parser boundary. +## Open evaluator decision: frame-grid sampling semantics + +Freezing `frame_size_seconds = 0.1` is necessary but not yet sufficient to implement ACC. MIREX 2025 does not specify the exact sample-point convention at segment boundaries or how a final partial 100 ms interval is represented. Current `mir_eval.util.intervals_to_samples` is a relevant implementation reference: it exposes `offset=0`, `sample_size=0.1`, creates sample times from zero, and uses `floor(max_interval / sample_size)` samples. That behavior is authoritative for that utility, but the MIREX 2025 task page does not state that ACC is computed by this exact helper. + +An exploratory RED `73ddb084c6e7279f4fc3d92ee9c8d0c8edf50071` tried to add `frame_grid_contract_version` before the evaluator convention itself had been selected. Ordinary descendant `cd069263a947a762a1cc66e8f85bfee7368180a8` restored the current schema rather than invent a version number with no defined semantics. No force-push or history rewrite was used. + +Before the recognized ACC runner is implemented, the scientific contract must explicitly choose and test at least: grid origin/offset, sample-point placement at exact segment boundaries, treatment of a final partial frame, and whether BandScope follows `mir_eval.util.intervals_to_samples` exactly or another reviewed evaluator convention. Only then should a frame-grid contract version enter the registration digest. This is a deliberate unresolved scientific prerequisite, not permission for the runner to choose defaults after seeing results. + ## Constraints and rejected alternatives Leaving frame resolution to the eventual runner was rejected because the same segment intervals can produce different frame-count weighting at 10 ms and 100 ms. A result receipt would then not identify one reproducible ACC statistic. @@ -62,7 +70,7 @@ Copying the MIREX submission envelope into local corpus annotations was rejected ## Claim boundary -This contract freezes how a future BandScope runner must interpret functional-label ACC evidence. It does not establish that the chosen corpus is sufficiently broad, that the current synthetic unit-test margins are scientifically approved, that 100 ms is optimal for every MIR task, or that a production CQT→STFT switch is justified. Production acceptance still requires rights-cleared real decoded audio, independently reviewed normalized annotations, an implemented recognized metric runner consuming the exact admitted PCM/annotation snapshots, preregistered aggregation and paired uncertainty, and current-head release evidence. +This contract freezes how a future BandScope runner must interpret functional-label annotation evidence and the selected 100 ms resolution. It does not yet freeze the ACC sample-grid convention described above, establish that the chosen corpus is sufficiently broad, approve the current synthetic unit-test margins, prove that 100 ms is optimal for every MIR task, or justify a production CQT→STFT switch. Production acceptance still requires the frame-grid decision, rights-cleared real decoded audio, independently reviewed normalized annotations, an implemented recognized metric runner consuming the exact admitted PCM/annotation snapshots, preregistered aggregation and paired uncertainty, and current-head release evidence. The parser tests use synthetic annotation strings only to verify the interpretation boundary. They are unit evidence and are not counted as production scientific acceptance. @@ -70,4 +78,6 @@ The parser tests use synthetic annotation strings only to verify the interpretat MIREX. (2025). *Music Structure Analysis*. International Music Information Retrieval Systems Evaluation Laboratory. https://music-ir.org/mirex/wiki/2025:Music_Structure_Analysis +mir_eval contributors. (n.d.). *mir_eval.util.intervals_to_samples*. https://github.com/mir-evaluation/mir_eval/blob/main/mir_eval/util.py + MIREX. (2025). *Music Structure Analysis Results*. International Music Information Retrieval Systems Evaluation Laboratory. https://music-ir.org/mirex/wiki/2025:Music_Structure_Analysis_Results From 6c20500bf31be88cd145f30fe7355a915066bdfc Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 15:01:33 +0900 Subject: [PATCH 117/216] test(mir): pin functional ACC to official evaluator grid --- ...ucture_functional_label_contract_policy.py | 45 +++++++++++++++---- 1 file changed, 36 insertions(+), 9 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py b/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py index 35d614c43..67c6a6d35 100644 --- a/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py +++ b/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py @@ -9,8 +9,14 @@ from test_structure_noninferiority_policy import _metrics, _registration, _validator +_MIREX_2025_ACC_IMPLEMENTATION = ( + "ismir-mirex/mirex-evaluation@" + "b9fa0b0b32e2145af31f35830f78fc9d09a4301b:" + "music_structure_analysis.eval_script.calculate_accuracy" +) _FUNCTIONAL_ACC_CONTRACT = { - "frame_size_seconds": 0.1, + "frame_size_seconds": 0.2, + "frame_grid_contract_version": 1.0, "annotation_contract_version": 1.0, "label_mapping_contract_version": 1.0, } @@ -23,14 +29,24 @@ def _functional_metric(registration: dict[str, object]) -> dict[str, object]: return metric -def test_functional_label_accuracy_freezes_frame_and_annotation_semantics() -> None: - """ACC cannot be preregistered without one exact frame/parser/mapping contract.""" - validator = _validator() +def _registered_functional_metric() -> dict[str, object]: + """Return a registration bound to the official MIREX 2025 ACC implementation.""" registration = _registration() - _functional_metric(registration).update(_FUNCTIONAL_ACC_CONTRACT) + metric = _functional_metric(registration) + metric["implementation"] = _MIREX_2025_ACC_IMPLEMENTATION + metric.update(_FUNCTIONAL_ACC_CONTRACT) + return registration + + +def test_functional_label_accuracy_freezes_official_grid_and_annotation_semantics() -> None: + """ACC must pin the official 200 ms grid plus parser/mapping interpretation.""" + validator = _validator() + registration = _registered_functional_metric() validator.validate_registration(registration) + metric = _functional_metric(registration) + assert metric["implementation"] == _MIREX_2025_ACC_IMPLEMENTATION for field in _FUNCTIONAL_ACC_CONTRACT: missing = copy.deepcopy(registration) del _functional_metric(missing)[field] @@ -39,16 +55,27 @@ def test_functional_label_accuracy_freezes_frame_and_annotation_semantics() -> N def test_functional_label_accuracy_rejects_post_hoc_contract_drift() -> None: - """Frame resolution and interpretation versions are fixed before results.""" + """Evaluator identity, frame grid, and interpretation cannot drift after registration.""" validator = _validator() - registration = _registration() - _functional_metric(registration).update(_FUNCTIONAL_ACC_CONTRACT) + registration = _registered_functional_metric() + + drifted_implementation = copy.deepcopy(registration) + _functional_metric(drifted_implementation)["implementation"] = ( + "mirex2025.frame_level_accuracy" + ) + with pytest.raises(ValueError, match="implementation"): + validator.validate_registration(drifted_implementation) drifted_frame = copy.deepcopy(registration) - _functional_metric(drifted_frame)["frame_size_seconds"] = 0.01 + _functional_metric(drifted_frame)["frame_size_seconds"] = 0.1 with pytest.raises(ValueError, match="frame_size_seconds"): validator.validate_registration(drifted_frame) + drifted_grid = copy.deepcopy(registration) + _functional_metric(drifted_grid)["frame_grid_contract_version"] = 2.0 + with pytest.raises(ValueError, match="frame_grid_contract_version"): + validator.validate_registration(drifted_grid) + drifted_parser = copy.deepcopy(registration) _functional_metric(drifted_parser)["annotation_contract_version"] = 2.0 with pytest.raises(ValueError, match="annotation_contract_version"): From 0f1e3e732154ff0ea94beb5a60e5702ed46c2453 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 15:02:50 +0900 Subject: [PATCH 118/216] fix(mir): align functional ACC registration with official grid --- scripts/research/validate_structure_noninferiority.py | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index fc82dd7b5..f3f344164 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -39,8 +39,13 @@ "window_seconds": 3.0, }, "functional_label_accuracy": { - "implementation": "mirex2025.frame_level_accuracy", - "frame_size_seconds": 0.1, + "implementation": ( + "ismir-mirex/mirex-evaluation@" + "b9fa0b0b32e2145af31f35830f78fc9d09a4301b:" + "music_structure_analysis.eval_script.calculate_accuracy" + ), + "frame_size_seconds": 0.2, + "frame_grid_contract_version": 1.0, "annotation_contract_version": 1.0, "label_mapping_contract_version": 1.0, }, From 86e6be0b41547ecd14b4f837ff6c88cf8f770b0b Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 15:04:58 +0900 Subject: [PATCH 119/216] test(mir): align shared registration with official ACC --- .../tests/test_structure_noninferiority_policy.py | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_policy.py index f00bf0214..80f0bbfdb 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_policy.py @@ -40,8 +40,13 @@ def _registration() -> dict[str, object]: "noninferiority_margin": 0.02, }, "functional_label_accuracy": { - "implementation": "mirex2025.frame_level_accuracy", - "frame_size_seconds": 0.1, + "implementation": ( + "ismir-mirex/mirex-evaluation@" + "b9fa0b0b32e2145af31f35830f78fc9d09a4301b:" + "music_structure_analysis.eval_script.calculate_accuracy" + ), + "frame_size_seconds": 0.2, + "frame_grid_contract_version": 1.0, "annotation_contract_version": 1.0, "label_mapping_contract_version": 1.0, "noninferiority_margin": 0.02, From 02c73b1fff55def189bed8fe2d34d4b5ee83cb0a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 15:05:33 +0900 Subject: [PATCH 120/216] test(mir): align corpus registration with official ACC --- .../tests/test_structure_corpus_admission.py | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_corpus_admission.py b/services/analysis-engine/tests/test_structure_corpus_admission.py index c56019423..4fd34e7ec 100644 --- a/services/analysis-engine/tests/test_structure_corpus_admission.py +++ b/services/analysis-engine/tests/test_structure_corpus_admission.py @@ -38,8 +38,13 @@ def _registration(audio_hashes: list[str], annotation_hashes: list[str]) -> dict "noninferiority_margin": 0.02, }, "functional_label_accuracy": { - "implementation": "mirex2025.frame_level_accuracy", - "frame_size_seconds": 0.1, + "implementation": ( + "ismir-mirex/mirex-evaluation@" + "b9fa0b0b32e2145af31f35830f78fc9d09a4301b:" + "music_structure_analysis.eval_script.calculate_accuracy" + ), + "frame_size_seconds": 0.2, + "frame_grid_contract_version": 1.0, "annotation_contract_version": 1.0, "label_mapping_contract_version": 1.0, "noninferiority_margin": 0.02, From 7f221ea57893df22fef3bb7becf69a9329cf5107 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 15:06:13 +0900 Subject: [PATCH 121/216] docs(mir): bind ACC to official evaluator semantics --- .../mir/functional-label-accuracy-contract.md | 74 ++++++++++--------- 1 file changed, 38 insertions(+), 36 deletions(-) diff --git a/docs/traceability/mir/functional-label-accuracy-contract.md b/docs/traceability/mir/functional-label-accuracy-contract.md index 278f90ae7..5107638c3 100644 --- a/docs/traceability/mir/functional-label-accuracy-contract.md +++ b/docs/traceability/mir/functional-label-accuracy-contract.md @@ -7,77 +7,79 @@ Parent: `docs/traceability/mir/structure-feature-noninferiority.md` ## Problem -The structure noninferiority registration previously identified functional-label accuracy only as `mirex2025.frame_level_accuracy`. That name is not a complete measurement contract. The MIREX 2025 task describes frame-level ACC by converting system and ground-truth segmentations to a time series at a fine temporal resolution and gives 10 ms or 100 ms as examples; it does not fix one unique frame size. The same task also requires source dataset labels to be mapped to a target functional vocabulary before evaluation. +The structure noninferiority registration originally named functional-label accuracy as `mirex2025.frame_level_accuracy` and later froze a BandScope-selected 100 ms frame size. That was still not a reproducible implementation identity. The MIREX task page describes frame-level ACC conceptually and gives 10 ms or 100 ms as examples, but the current official `ismir-mirex/mirex-evaluation` repository contains the executable MIREX 2025 reproduction used as the stronger authority for evaluator behavior. -Without freezing the frame grid and annotation/mapping interpretation before candidate results exist, two otherwise identical result receipts could report different ACC values while both claiming the same preregistered metric identity. That is post-result measurement freedom, not reproducible evidence. +At `ismir-mirex/mirex-evaluation@b9fa0b0b32e2145af31f35830f78fc9d09a4301b`, `music_structure_analysis/eval_script.py::calculate_accuracy` uses a default `frame_hop` of 0.2 seconds. It creates frame times with `np.arange(0, gt_duration, frame_hop)`, advances a segment while `t >= segment_end`, and counts all reference-grid frames in the denominator. Therefore the former 100 ms registration did not reproduce the current official evaluator and left the exact time-grid authority implicit. ## Decision -Schema v1 now requires the `functional_label_accuracy` metric registration to contain all of the following in addition to its noninferiority margin: +Schema v1 now requires the `functional_label_accuracy` registration to contain, in addition to its noninferiority margin: -- `implementation = mirex2025.frame_level_accuracy`; -- `frame_size_seconds = 0.1`; +- `implementation = ismir-mirex/mirex-evaluation@b9fa0b0b32e2145af31f35830f78fc9d09a4301b:music_structure_analysis.eval_script.calculate_accuracy`; +- `frame_size_seconds = 0.2`; +- `frame_grid_contract_version = 1.0`; - `annotation_contract_version = 1.0`; - `label_mapping_contract_version = 1.0`. -The 100 ms frame grid is a BandScope preregistration choice, not a claim that MIREX mandates 100 ms. MIREX 2025 explicitly gives both 10 ms and 100 ms as examples. BandScope selects 100 ms before results because the same experiment already preregisters a 100 ms frame size for `mir_eval.segment.pairwise`; using one fixed temporal grid avoids an otherwise unnecessary second discretization convention in this experiment. +`frame_grid_contract_version = 1.0` means the MIREX 2025 reproduction semantics at that pinned upstream commit: -`annotation_contract_version = 1.0` is implemented by `scripts/research/parse_structure_functional_annotations.py`. The parser consumes only the read-only annotation snapshot already admitted by the corpus boundary. Version 1 uses a BandScope-local UTF-8 normalized TSV with exactly three fields per segment: `startendlabel`. This is **not** the MIREX 2025 submission-file syntax: the MIREX task specifies a text/JSON-parsable list of per-track segment predictions. BandScope's local TSV is a deliberately simpler frozen corpus-annotation representation that preserves the same start/end/label semantics before evaluation. +- the grid origin is 0 seconds; +- frame points are `0.0, 0.2, 0.4, ...` while the point is strictly less than the ground-truth duration; +- an exact segment-end boundary belongs to the following segment because the evaluator advances while `t >= segment_end`; +- a final partial 200 ms span contributes a frame when its starting grid point is still below the ground-truth duration; +- every reference-grid frame contributes to the denominator; +- a frame contributes to the numerator only when the reference and prediction labels match and the reference label is not `other`. -Times are parsed as finite decimal values into exact rational boundaries rather than binary floating-point approximations. The normalized segmentation must start at 0.0, preserve source order, have strictly positive segment duration, contain no gaps or overlaps, and end exactly at `decoded_frames / sample_rate_hz`. The parser never reopens a corpus path or repairs malformed evidence. +This closes the previously open frame-grid decision. The 200 ms value is not inferred from the prose examples on the MIREX wiki; it is bound to the executable standardized-evaluation source and its exact commit. -`label_mapping_contract_version = 1.0` uses a fail-closed identity mapping at the measurement boundary. Normalized annotations presented to ACC must already use exactly one of these seven lowercase labels: +## Annotation and label-mapping boundary -`intro`, `verse`, `chorus`, `bridge`, `inst`, `outro`, `silence`. - -No case folding, stemming, prefix stripping (`verse1` → `verse`), synonym mapping (`solo` → `inst`), or catch-all remapping is allowed inside the acceptance run. If a rights-cleared source corpus uses another vocabulary, its mapping must be reviewed and applied before preregistration; the resulting normalized annotation bytes receive their own `annotation_sha256` and therefore become part of the frozen corpus identity. +`annotation_contract_version = 1.0` remains implemented by `scripts/research/parse_structure_functional_annotations.py`. It consumes the already-admitted read-only annotation snapshot and parses BandScope-local UTF-8 `startendlabel` rows into exact rational boundaries. This local TSV is not the MIREX submission transport syntax. -This deliberately avoids silently resolving an ambiguity in the current MIREX 2025 page: its narrative description mentions an `other` category, while its operational output-format section says submitted labels must be one of `intro`, `verse`, `chorus`, `bridge`, `inst`, `outro`, or `silence`. BandScope v1 follows the operational output-format list and rejects `other` rather than guessing how it should map. If the scientific review later adopts another mapping, that requires a new mapping-contract version and a new preregistration digest before results are inspected. +The normalized segmentation must start at 0.0, preserve source order, contain strictly positive segments with no gaps or overlaps, and end exactly at `decoded_frames / sample_rate_hz`. The parser never reopens a corpus path or repairs malformed evidence. -Corpus admission intentionally does not parse or rewrite annotations. Resource admission owns byte identity and immutable handoff; Signal-MIR owns interpretation and measurement. The future experiment runner must call the versioned parser on the admitted read-only annotation view and then evaluate both CQT and STFT against that same parsed evidence. +`label_mapping_contract_version = 1.0` remains a fail-closed identity mapping at the acceptance boundary. The admitted normalized annotation bytes must already contain one of: -## RED → GREEN lineage +`intro`, `verse`, `chorus`, `bridge`, `inst`, `outro`, `silence`. -RED `6f19315d0092004499fea7e581e1f00cc6d021c1` added a focused policy test requiring functional-label ACC to freeze frame resolution plus annotation and label-mapping contract versions. The predecessor validator rejected those fields as unregistered and still accepted the older under-specified metric shape. +The official evaluator's raw-label preprocessing is broader: it case-folds labels, applies substring mappings such as `solo` → `inst`, and maps otherwise unknown labels to `other`; `other` reference frames remain in the ACC denominator but cannot contribute to the numerator. BandScope does not run that mutable normalization after preregistration. If a source corpus needs the official mapping, that transformation must be reviewed and applied before preregistration, and the resulting normalized annotation bytes receive their own `annotation_sha256`. -GREEN `03b26bf203c4784029f75296d308eed644d7a5eb` makes those three fields mandatory closed-world configuration for `functional_label_accuracy`. Fixture alignment `4936cefd1a2b13fcd75a25e0d622ca4ce0ba9373` and `1839586fb1446f28654494c154182b1c415ab757` updates the shared evidence and corpus-admission registrations so successful tests exercise the new contract rather than bypassing it. +This separation is deliberate. The pinned upstream evaluator is authoritative for ACC time-grid and scoring semantics, while BandScope's content-addressed corpus boundary prevents label-mapping decisions from changing after candidate results are visible. -Parser RED `cc4a166cc9e1496cd562f7d79b9ffa30f6ca19c2` defines the executable v1 interpretation contract: exact three-column local TSV, exact rational boundaries, continuous full-duration coverage, registered labels only, finite times, and read-only input. GREEN `0ee4a6aed49c8f001da451ca63c5f20e52832f5c` implements that contract without adding label normalization, filesystem access, metric computation, or scientific threshold logic. Authority correction `52b6c7362baa00151f908a24cadbb1efeddd2101` removes the inaccurate implication that this local TSV is the MIREX 2025 submission format while retaining the same executable semantics. +The MIREX task page also continues to expose a vocabulary inconsistency: descriptive prose includes `other`, while the operational seven-label output list includes `silence`. The local v1 annotation contract therefore does not guess between them at run time. A different mapping policy requires a new mapping-contract version and preregistration digest. -The focused regressions require missing preregistration fields and post-hoc changes to the 100 ms frame, annotation contract version, or label mapping contract version to fail validation, and they reject mapping freedom such as `Verse`, `verse1`, `solo`, or `other` at the parser boundary. +## RED → GREEN lineage -## Open evaluator decision: frame-grid sampling semantics +The predecessor branch froze 100 ms but left the official implementation identity and frame-grid behavior unresolved. Fresh inspection of the current standardized evaluator at `b9fa0b0b32e2145af31f35830f78fc9d09a4301b` showed that its MIREX 2025 reproduction uses a 200 ms hop with explicit `np.arange` and segment-boundary behavior. -Freezing `frame_size_seconds = 0.1` is necessary but not yet sufficient to implement ACC. MIREX 2025 does not specify the exact sample-point convention at segment boundaries or how a final partial 100 ms interval is represented. Current `mir_eval.util.intervals_to_samples` is a relevant implementation reference: it exposes `offset=0`, `sample_size=0.1`, creates sample times from zero, and uses `floor(max_interval / sample_size)` samples. That behavior is authoritative for that utility, but the MIREX 2025 task page does not state that ACC is computed by this exact helper. +RED `6c20500bf31be88cd145f30fe7355a915066bdfc` changed the focused policy test to require the exact upstream commit/function identity, `frame_size_seconds = 0.2`, and `frame_grid_contract_version = 1.0`; the predecessor validator rejected that registration and still accepted the stale generic 100 ms contract. -An exploratory RED `73ddb084c6e7279f4fc3d92ee9c8d0c8edf50071` tried to add `frame_grid_contract_version` before the evaluator convention itself had been selected. Ordinary descendant `cd069263a947a762a1cc66e8f85bfee7368180a8` restored the current schema rather than invent a version number with no defined semantics. No force-push or history rewrite was used. +GREEN `0f1e3e732154ff0ea94beb5a60e5702ed46c2453` changed the closed-world validator contract to the pinned official evaluator and 200 ms grid. Fixture alignment `86e6be0b41547ecd14b4f837ff6c88cf8f770b0b` and `02c73b1fff55def189bed8fe2d34d4b5ee83cb0a` moved the shared evidence and corpus-admission registrations onto the same contract instead of leaving successful tests on the stale 100 ms shape. -Before the recognized ACC runner is implemented, the scientific contract must explicitly choose and test at least: grid origin/offset, sample-point placement at exact segment boundaries, treatment of a final partial frame, and whether BandScope follows `mir_eval.util.intervals_to_samples` exactly or another reviewed evaluator convention. Only then should a frame-grid contract version enter the registration digest. This is a deliberate unresolved scientific prerequisite, not permission for the runner to choose defaults after seeing results. +The earlier immutable parser lineage remains valid: parser RED `cc4a166cc9e1496cd562f7d79b9ffa30f6ca19c2` → GREEN `0ee4a6aed49c8f001da451ca63c5f20e52832f5c`, with authority correction `52b6c7362baa00151f908a24cadbb1efeddd2101` separating the BandScope TSV representation from the MIREX submission format. ## Constraints and rejected alternatives -Leaving frame resolution to the eventual runner was rejected because the same segment intervals can produce different frame-count weighting at 10 ms and 100 ms. A result receipt would then not identify one reproducible ACC statistic. +Keeping 100 ms merely because the MIREX wiki lists it as an example was rejected. The executable standardized evaluator is more specific and currently uses 200 ms for the MIREX 2025 reproduction. -Treating the MIREX task name as sufficient label-mapping authority was rejected because the current task page itself contains conflicting target-vocabulary prose. A scientific gate cannot depend on an implicit interpretation of that inconsistency. +Keeping only a symbolic `mirex2025.frame_level_accuracy` implementation name was rejected because it cannot identify the code revision or distinguish future evaluator changes. -Automatically normalizing labels during the acceptance run was rejected. Mapping `verse1`, `solo`, `fade-out`, mixed case, or unknown labels after the corpus is frozen gives the runner post-hoc freedom over the dependent variable. Normalization belongs before preregistration and is content-addressed through the annotation digest. +Using `mir_eval.util.intervals_to_samples` for ACC was rejected for this contract. It is a useful MIR utility, but the current official MIREX evaluator implements ACC with its own `np.arange` grid and pointer traversal; substituting another grid would no longer be exact evaluator parity. -Parsing annotation bytes inside corpus admission was rejected. Resource admission owns byte identity and immutable handoff; metric interpretation belongs to the Signal-MIR measurement runner. Combining them would blur the bounded-context boundary and make a byte-admission receipt look like scientific acceptance. +Automatically normalizing labels during the acceptance run was rejected. Mapping `verse1`, `solo`, mixed case, or unknown labels after corpus bytes are frozen would reintroduce post-result freedom over the dependent variable. Source-vocabulary normalization belongs before preregistration and is content-addressed through `annotation_sha256`. -Using binary floating-point as the annotation-boundary authority was rejected. Decimal source timestamps are converted to exact rational values, while decoded duration is represented exactly as `decoded_frames / sample_rate_hz`; this makes gap/overlap and full-duration checks deterministic rather than tolerance-dependent. +Parsing annotations inside resource admission was rejected. Resource admission owns byte identity and immutable handoff; Signal-MIR owns scientific interpretation and measurement. -Copying the MIREX submission envelope into local corpus annotations was rejected. Submission transport syntax and local preregistered annotation storage solve different problems; the scientific contract is the segment boundaries, labels, normalization, and evaluation semantics, all of which remain explicitly bound here. +## Claim boundary and next work -## Claim boundary +This contract now freezes the ACC evaluator identity, 200 ms grid, boundary-point behavior, final-partial-frame behavior, annotation interpretation, and label-mapping boundary. It does not yet implement the BandScope runner that must demonstrate parity with the pinned evaluator on admitted normalized annotations, approve the corpus or noninferiority margins, derive aggregate/paired uncertainty, or justify a production CQT → STFT switch. -This contract freezes how a future BandScope runner must interpret functional-label annotation evidence and the selected 100 ms resolution. It does not yet freeze the ACC sample-grid convention described above, establish that the chosen corpus is sufficiently broad, approve the current synthetic unit-test margins, prove that 100 ms is optimal for every MIR task, or justify a production CQT→STFT switch. Production acceptance still requires the frame-grid decision, rights-cleared real decoded audio, independently reviewed normalized annotations, an implemented recognized metric runner consuming the exact admitted PCM/annotation snapshots, preregistered aggregation and paired uncertainty, and current-head release evidence. +The next scientific slice is an executable parity-tested ACC runner that consumes the exact admitted PCM/annotation snapshots and reproduces the pinned evaluator semantics without reopening workstation paths or performing post-preregistration label mapping. Production acceptance still requires rights-cleared real decoded audio, independently reviewed normalized annotations, preregistered aggregation and paired uncertainty, paired CQT/STFT execution on the same admitted signal/runtime identity, and current-head release evidence. -The parser tests use synthetic annotation strings only to verify the interpretation boundary. They are unit evidence and are not counted as production scientific acceptance. +Synthetic annotation fixtures remain unit evidence only and are not counted as production scientific acceptance. ## References -MIREX. (2025). *Music Structure Analysis*. International Music Information Retrieval Systems Evaluation Laboratory. https://music-ir.org/mirex/wiki/2025:Music_Structure_Analysis - -mir_eval contributors. (n.d.). *mir_eval.util.intervals_to_samples*. https://github.com/mir-evaluation/mir_eval/blob/main/mir_eval/util.py +MIREX. (2026). *Music Structure Analysis*. International Music Information Retrieval Systems Evaluation Laboratory. https://music-ir.org/mirex/wiki/2026:Music_Structure_Analysis -MIREX. (2025). *Music Structure Analysis Results*. International Music Information Retrieval Systems Evaluation Laboratory. https://music-ir.org/mirex/wiki/2025:Music_Structure_Analysis_Results +MIREX Evaluation contributors. (2026). *music_structure_analysis/eval_script.py* (commit `b9fa0b0b32e2145af31f35830f78fc9d09a4301b`). GitHub. https://github.com/ismir-mirex/mirex-evaluation/blob/b9fa0b0b32e2145af31f35830f78fc9d09a4301b/music_structure_analysis/eval_script.py From 5cffef1a5db37b700ee7a27012c0f8324855665f Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 15:07:29 +0900 Subject: [PATCH 122/216] test(mir): define official ACC runner parity --- ...st_structure_functional_accuracy_runner.py | 100 ++++++++++++++++++ 1 file changed, 100 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_functional_accuracy_runner.py diff --git a/services/analysis-engine/tests/test_structure_functional_accuracy_runner.py b/services/analysis-engine/tests/test_structure_functional_accuracy_runner.py new file mode 100644 index 000000000..020d03180 --- /dev/null +++ b/services/analysis-engine/tests/test_structure_functional_accuracy_runner.py @@ -0,0 +1,100 @@ +"""Parity contract for the pinned MIREX 2025 functional ACC evaluator.""" + +from __future__ import annotations + +from fractions import Fraction +from types import ModuleType + +import pytest +from conftest import load_module + + +def _runner() -> ModuleType: + """Load the repository-owned functional ACC evaluator adapter.""" + return load_module( + "scripts/research/evaluate_structure_functional_accuracy.py", + "evaluate_structure_functional_accuracy", + ) + + +def _segment(start: str, end: str, label: str) -> object: + """Create one runner segment without importing the production parser.""" + runner = _runner() + return runner.FunctionalAccuracySegment( + start=Fraction(start), + end=Fraction(end), + label=label, + ) + + +def test_runner_pins_official_mirex_2025_evaluator_identity_and_200ms_grid() -> None: + """The acceptance runner must not silently drift from the pinned evaluator.""" + runner = _runner() + + assert runner.OFFICIAL_MIREX_2025_EVALUATOR_COMMIT == ( + "b9fa0b0b32e2145af31f35830f78fc9d09a4301b" + ) + assert runner.FRAME_HOP_SECONDS == 0.2 + + +def test_runner_matches_official_boundary_and_final_partial_frame_semantics() -> None: + """Exact ends advance to the next segment and a trailing partial grid span counts.""" + runner = _runner() + reference = ( + _segment("0.0", "0.4", "verse"), + _segment("0.4", "0.7", "chorus"), + ) + estimate = ( + _segment("0.0", "0.2", "verse"), + _segment("0.2", "0.7", "chorus"), + ) + + result = runner.calculate_functional_accuracy( + reference, + estimate, + duration_seconds=Fraction("0.7"), + ) + + assert result.frame_times_seconds == pytest.approx((0.0, 0.2, 0.4, 0.6)) + assert result.total_frames == 4 + assert result.correct_frames == 3 + assert result.accuracy == pytest.approx(0.75) + + +def test_runner_does_not_add_a_frame_at_exact_track_duration() -> None: + """The official np.arange grid excludes a frame exactly at the track duration.""" + runner = _runner() + segments = ( + _segment("0.0", "0.8", "verse"), + ) + + result = runner.calculate_functional_accuracy( + segments, + segments, + duration_seconds=Fraction("0.8"), + ) + + assert result.frame_times_seconds == pytest.approx((0.0, 0.2, 0.4, 0.6)) + assert result.total_frames == 4 + assert result.accuracy == pytest.approx(1.0) + + +def test_runner_rejects_post_preregistration_mapping_or_incomplete_segments() -> None: + """Raw `other` labels and discontinuous segments cannot enter the acceptance metric.""" + runner = _runner() + with pytest.raises(ValueError, match="functional label"): + runner.calculate_functional_accuracy( + (_segment("0.0", "0.4", "other"),), + (_segment("0.0", "0.4", "verse"),), + duration_seconds=Fraction("0.4"), + ) + + with pytest.raises(ValueError, match="continuous"): + runner.calculate_functional_accuracy( + ( + _segment("0.0", "0.2", "verse"), + _segment("0.3", "0.4", "chorus"), + ), + (_segment("0.0", "0.4", "verse"),), + duration_seconds=Fraction("0.4"), + ) From e72b29565a6d906a9955907656cbae97cffa471a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 15:07:56 +0900 Subject: [PATCH 123/216] feat(mir): add pinned functional ACC evaluator adapter --- .../evaluate_structure_functional_accuracy.py | 143 ++++++++++++++++++ 1 file changed, 143 insertions(+) create mode 100644 scripts/research/evaluate_structure_functional_accuracy.py diff --git a/scripts/research/evaluate_structure_functional_accuracy.py b/scripts/research/evaluate_structure_functional_accuracy.py new file mode 100644 index 000000000..4b19fc11f --- /dev/null +++ b/scripts/research/evaluate_structure_functional_accuracy.py @@ -0,0 +1,143 @@ +#!/usr/bin/env python3 +"""Evaluate preregistered functional-label ACC on normalized segments. + +This adapter reproduces the time-grid and pointer semantics of the pinned MIREX +2025 standardized evaluator for BandScope's already-normalized seven-label +annotation boundary. It performs no source-vocabulary mapping and no file I/O. +""" + +from __future__ import annotations + +from collections.abc import Sequence +from dataclasses import dataclass +from fractions import Fraction + +import numpy as np + +OFFICIAL_MIREX_2025_EVALUATOR_COMMIT = ( + "b9fa0b0b32e2145af31f35830f78fc9d09a4301b" +) +FRAME_HOP_SECONDS = 0.2 +_ALLOWED_FUNCTIONAL_LABELS = frozenset( + {"intro", "verse", "chorus", "bridge", "inst", "outro", "silence"} +) + + +@dataclass(frozen=True, slots=True) +class FunctionalAccuracySegment: + """One preregistered normalized segment for ACC evaluation.""" + + start: Fraction + end: Fraction + label: str + + +@dataclass(frozen=True, slots=True) +class FunctionalAccuracyResult: + """Deterministic track-level ACC evidence for one normalized segmentation pair.""" + + accuracy: float + correct_frames: int + total_frames: int + frame_times_seconds: tuple[float, ...] + + +def _validate_segments( + segments: Sequence[FunctionalAccuracySegment], + *, + duration_seconds: Fraction, + field: str, +) -> tuple[FunctionalAccuracySegment, ...]: + """Require one continuous full-duration preregistered segmentation.""" + if not segments: + raise ValueError(f"{field} must contain at least one segment") + + normalized = tuple(segments) + previous_end = Fraction(0, 1) + for index, segment in enumerate(normalized): + if not isinstance(segment.start, Fraction) or not isinstance(segment.end, Fraction): + raise ValueError(f"{field}[{index}] boundaries must be exact Fractions") + if segment.label not in _ALLOWED_FUNCTIONAL_LABELS: + raise ValueError( + f"{field}[{index}] functional label is not preregistered: " + f"{segment.label!r}" + ) + if segment.start != previous_end: + raise ValueError(f"{field} must be continuous at segment {index}") + if segment.end <= segment.start: + raise ValueError(f"{field}[{index}] must have end > start") + if segment.end > duration_seconds: + raise ValueError(f"{field}[{index}] exceeds track duration") + previous_end = segment.end + + if normalized[0].start != 0: + raise ValueError(f"{field} must start at 0 seconds") + if previous_end != duration_seconds: + raise ValueError(f"{field} must cover the full track duration") + return normalized + + +def calculate_functional_accuracy( + reference_segments: Sequence[FunctionalAccuracySegment], + estimated_segments: Sequence[FunctionalAccuracySegment], + *, + duration_seconds: Fraction, +) -> FunctionalAccuracyResult: + """Return ACC using the pinned MIREX 2025 200 ms frame-grid semantics. + + The official evaluator constructs frame points with + ``np.arange(0, gt_duration, 0.2)`` and advances a segment while the frame + point is greater than or equal to that segment's end. BandScope reproduces + that behavior only after source labels have been normalized and content- + addressed by the preregistration boundary. + """ + if not isinstance(duration_seconds, Fraction) or duration_seconds <= 0: + raise ValueError("duration_seconds must be a positive Fraction") + + reference = _validate_segments( + reference_segments, + duration_seconds=duration_seconds, + field="reference_segments", + ) + estimate = _validate_segments( + estimated_segments, + duration_seconds=duration_seconds, + field="estimated_segments", + ) + + frame_times = np.arange( + 0.0, + float(duration_seconds), + FRAME_HOP_SECONDS, + dtype=np.float64, + ) + reference_index = 0 + estimate_index = 0 + correct_frames = 0 + + for frame_time_value in frame_times: + frame_time = float(frame_time_value) + while ( + reference_index < len(reference) + and frame_time >= float(reference[reference_index].end) + ): + reference_index += 1 + while ( + estimate_index < len(estimate) + and frame_time >= float(estimate[estimate_index].end) + ): + estimate_index += 1 + + if reference_index >= len(reference) or estimate_index >= len(estimate): + raise ValueError("frame grid exceeded a validated segmentation") + if reference[reference_index].label == estimate[estimate_index].label: + correct_frames += 1 + + total_frames = int(frame_times.size) + accuracy = correct_frames / total_frames if total_frames else 0.0 + return FunctionalAccuracyResult( + accuracy=accuracy, + correct_frames=correct_frames, + total_frames=total_frames, + frame_times_seconds=tuple(float(value) for value in frame_times), + ) From c84c0acdf25159f9aa066e21f9420d9fdf72c50a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 15:08:33 +0900 Subject: [PATCH 124/216] test(mir): consume canonical functional segment value object --- .../test_structure_functional_accuracy_runner.py | 14 +++++++++++--- 1 file changed, 11 insertions(+), 3 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_functional_accuracy_runner.py b/services/analysis-engine/tests/test_structure_functional_accuracy_runner.py index 020d03180..82b7f3087 100644 --- a/services/analysis-engine/tests/test_structure_functional_accuracy_runner.py +++ b/services/analysis-engine/tests/test_structure_functional_accuracy_runner.py @@ -17,10 +17,18 @@ def _runner() -> ModuleType: ) +def _parser() -> ModuleType: + """Load the canonical preregistered functional-annotation parser.""" + return load_module( + "scripts/research/parse_structure_functional_annotations.py", + "parse_structure_functional_annotations_for_acc", + ) + + def _segment(start: str, end: str, label: str) -> object: - """Create one runner segment without importing the production parser.""" - runner = _runner() - return runner.FunctionalAccuracySegment( + """Create one canonical parser-owned functional segment value object.""" + parser = _parser() + return parser.FunctionalSegment( start=Fraction(start), end=Fraction(end), label=label, From fe3ed61a3b1926af6b39b6302d6f857979f9bccc Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 15:08:58 +0900 Subject: [PATCH 125/216] refactor(mir): reuse parser-owned functional segment contract --- .../evaluate_structure_functional_accuracy.py | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/scripts/research/evaluate_structure_functional_accuracy.py b/scripts/research/evaluate_structure_functional_accuracy.py index 4b19fc11f..631a3cc64 100644 --- a/scripts/research/evaluate_structure_functional_accuracy.py +++ b/scripts/research/evaluate_structure_functional_accuracy.py @@ -11,6 +11,7 @@ from collections.abc import Sequence from dataclasses import dataclass from fractions import Fraction +from typing import Protocol import numpy as np @@ -23,9 +24,8 @@ ) -@dataclass(frozen=True, slots=True) -class FunctionalAccuracySegment: - """One preregistered normalized segment for ACC evaluation.""" +class FunctionalSegmentLike(Protocol): + """Structural view of the parser-owned functional segment value object.""" start: Fraction end: Fraction @@ -43,11 +43,11 @@ class FunctionalAccuracyResult: def _validate_segments( - segments: Sequence[FunctionalAccuracySegment], + segments: Sequence[FunctionalSegmentLike], *, duration_seconds: Fraction, field: str, -) -> tuple[FunctionalAccuracySegment, ...]: +) -> tuple[FunctionalSegmentLike, ...]: """Require one continuous full-duration preregistered segmentation.""" if not segments: raise ValueError(f"{field} must contain at least one segment") @@ -78,8 +78,8 @@ def _validate_segments( def calculate_functional_accuracy( - reference_segments: Sequence[FunctionalAccuracySegment], - estimated_segments: Sequence[FunctionalAccuracySegment], + reference_segments: Sequence[FunctionalSegmentLike], + estimated_segments: Sequence[FunctionalSegmentLike], *, duration_seconds: Fraction, ) -> FunctionalAccuracyResult: From f03a6cedd9f8ba00dd646485323be71d417361cf Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 15:10:39 +0900 Subject: [PATCH 126/216] docs(mir): make structure gate evaluator-current --- .../mir/structure-feature-noninferiority.md | 20 ++++++++++++------- 1 file changed, 13 insertions(+), 7 deletions(-) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index 0b6bfdf86..26e78bbb1 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -20,7 +20,7 @@ A registration is valid only when it records all of the following before the res - baseline `chroma_cqt` and candidate `chroma_stft`; - a rights-cleared real-audio corpus with stable track IDs, content-unique audio SHA-256 identities, annotation SHA-256, rights basis, and provenance URI; - exact source commit and `uv.lock` identity plus Python, librosa, NumPy, sample rate, channel count, and host profile; -- the complete metric implementation contract and noninferiority/speed thresholds, including functional-label ACC frame resolution plus versioned annotation and label-mapping semantics; +- the complete metric implementation contract and noninferiority/speed thresholds, including the pinned MIREX functional-ACC evaluator commit/function, 200 ms frame-grid contract v1, and versioned annotation/label-mapping semantics; - the paired-uncertainty procedure identity, confidence level, resample count, and random seed; - a non-empty claim boundary stating the population/runtime scope to which a passing result may be applied. @@ -42,19 +42,21 @@ Schema v1 does not contain a preregistered dropout, exclusion, or missing-track ## Metrics -BandScope uses MIREX 2025 Music Structure Analysis as the functional-structure reference point. The registered quality gate requires: +BandScope uses the MIREX Music Structure Analysis task and its standardized evaluator repository as the functional-structure reference point. The registered quality gate requires: | Evidence | Registered implementation / convention | Decision use | | --- | --- | --- | | Boundary precision/recall/F at 0.5 s | `mir_eval.segment.detection`, 0.5 s window | F noninferiority | | Boundary precision/recall/F at 3.0 s | `mir_eval.segment.detection`, 3.0 s window | F noninferiority | | Boundary median deviation, both directions | `mir_eval.segment.deviation` convention | Report per track and aggregate | -| Functional-label accuracy | MIREX 2025 frame-level ACC; 0.1 s grid; annotation contract v1; label-mapping contract v1 | Noninferiority | +| Functional-label accuracy | `ismir-mirex/mirex-evaluation@b9fa0b0...:music_structure_analysis.eval_script.calculate_accuracy`; 0.2 s grid; frame-grid v1; annotation v1; label-mapping v1 | Noninferiority | | Repetition/group consistency precision/recall/F | `mir_eval.segment.pairwise`, 0.1 s frame size | F noninferiority | | Latency | paired p50/p95 on the registered host | p95 ratio superiority | | Peak memory | peak RSS on the registered host | Report per track and aggregate | -MIREX 2025 evaluates functional structure with frame-level label accuracy plus boundary hit-rate F measures at 0.5 s and 3.0 s. Its ACC description gives 10 ms and 100 ms as example frame resolutions rather than fixing a single grid. BandScope therefore freezes 100 ms before measurement and versions the annotation and label-mapping semantics instead of leaving those choices to the eventual runner. The detailed v1 contract and the MIREX vocabulary ambiguity are recorded in `docs/traceability/mir/functional-label-accuracy-contract.md`. `mir_eval.segment.detection` uses one-to-one boundary matching within the selected tolerance; `mir_eval.segment.deviation` reports the median nearest-boundary deviations in both directions; `mir_eval.segment.pairwise` measures structural grouping agreement. These are different questions and must not be collapsed into one score. +The MIREX task page describes frame-level ACC conceptually and gives finer resolutions as examples, but the current official `ismir-mirex/mirex-evaluation` MIREX-2025 reproduction is more specific: at commit `b9fa0b0b32e2145af31f35830f78fc9d09a4301b`, `calculate_accuracy` uses a 0.2 s hop, `np.arange(0, gt_duration, frame_hop)`, and advances a segment when a frame point is greater than or equal to the segment end. BandScope therefore pins that exact evaluator identity and frame-grid contract rather than selecting 100 ms from prose examples. The local annotation mapping remains preregistered and content-addressed before evaluation; raw-label normalization is not allowed to drift after candidate results are visible. The detailed contract is recorded in `docs/traceability/mir/functional-label-accuracy-contract.md`, and `scripts/research/evaluate_structure_functional_accuracy.py` is the repository-owned adapter for the normalized seven-label subset. + +`mir_eval.segment.detection` uses one-to-one boundary matching within the selected tolerance; `mir_eval.segment.deviation` reports the median nearest-boundary deviations in both directions; `mir_eval.segment.pairwise` measures structural grouping agreement. These are different questions and must not be collapsed into one score. Result receipts must carry the reported precision and recall alongside F for the boundary and repetition measures. The admission validator recomputes the harmonic mean and rejects an internally inconsistent P/R/F triplet. This does not replace the recognized metric implementation; it prevents a malformed receipt from claiming a metric value that its own reported components cannot support. @@ -96,9 +98,9 @@ The JSON `source_uri` value remains provenance metadata only; it is never derefe ## Reproducibility sequence -1. Review the rights basis, corpus composition, distinct audio-content identities, feature hypothesis, metric contract, functional-label frame/annotation/mapping contract, host profile, numeric margins, aggregation procedure, paired-uncertainty procedure, and claim boundary **before** running the candidate. Freeze the procedure identifier, confidence level, resample count, random seed, and claim boundary in the registration. If repeated recordings, clustering, source-label normalization, or any failure/exclusion tolerance is scientifically required, define and version that dependence/mapping/exclusion policy before measurement rather than remapping or omitting observations after results are visible. +1. Review the rights basis, corpus composition, distinct audio-content identities, feature hypothesis, metric contract, pinned functional-ACC evaluator/grid/annotation/mapping contract, host profile, numeric margins, aggregation procedure, paired-uncertainty procedure, and claim boundary **before** running the candidate. Freeze the procedure identifier, confidence level, resample count, random seed, and claim boundary in the registration. If repeated recordings, clustering, source-label normalization, or any failure/exclusion tolerance is scientifically required, define and version that dependence/mapping/exclusion policy before measurement rather than remapping or omitting observations after results are visible. 2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain its canonical SHA-256. -3. Run baseline and candidate on the same decoded track identities and host profile. The experiment runner must compute the approved aggregation and paired uncertainty procedure from the complete measured-track evidence rather than accepting caller-authored aggregate numbers as scientific authority. Record recognized MIR metrics, p50/p95 latency, and peak RSS for successful tracks. If a registered track cannot produce a baseline/candidate measurement, preserve its ordered `track_id` receipt, add that exact ID to `failed_tracks`, and set the aggregate plus both CI summary fields to JSON `null`; do not invent metric values or compute a complete-case aggregate from the surviving tracks. Only a complete run records aggregate outputs and paired uncertainty using the preregistered procedure. +3. Run baseline and candidate on the same decoded track identities and host profile. Functional ACC must consume the admitted normalized annotation through `scripts/research/evaluate_structure_functional_accuracy.py` and preserve the pinned 200 ms MIREX grid semantics; it must not reopen corpus paths or remap labels after preregistration. The experiment runner must compute the approved aggregation and paired uncertainty procedure from the complete measured-track evidence rather than accepting caller-authored aggregate numbers as scientific authority. Record recognized MIR metrics, p50/p95 latency, and peak RSS for successful tracks. If a registered track cannot produce a baseline/candidate measurement, preserve its ordered `track_id` receipt, add that exact ID to `failed_tracks`, and set the aggregate plus both CI summary fields to JSON `null`; do not invent metric values or compute a complete-case aggregate from the surviving tracks. Only a complete run records aggregate outputs and paired uncertainty using the preregistered procedure. 4. Put the registration digest, identical uncertainty-plan fields, and identical claim boundary in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. Under schema v1, any recorded failed track keeps the diagnostic receipt valid but makes acceptance fail closed before aggregate decision rules; undeclared missing measurements, non-null post-failure summaries, and claim-boundary drift are rejected. 5. Preserve the registration, result receipt, corpus/annotation hashes, exact source commit, lock hash, aggregation implementation identity, uncertainty procedure, and claim boundary together. A production feature switch requires this evidence plus normal code review and protected-head checks. @@ -110,7 +112,7 @@ No step authorizes committing licensed audio to Git. Rights-cleared means BandSc - Evidence JSON is limited to 2 MiB, must be a regular file read through one open descriptor, must decode as UTF-8, and rejects duplicate keys plus non-standard `NaN`/`Infinity` constants. - `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute, relative, drive-relative, and `file:` forms are rejected; an explicit non-file URI scheme is required so transient workstation paths cannot become provenance authority or leak into review artifacts. - Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt; duplicate audio SHA-256 values under different track IDs are rejected so one recording cannot be silently counted multiple times. -- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, functional-label frame/annotation/mapping drift, uncertainty-plan drift, claim-boundary drift, registration drift, undeclared missing measurements, measurement-bearing failed receipts, non-null aggregate/CI summaries after any failed track, unregistered registration/hypothesis/metric/corpus/runtime/result/track/aggregate/measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. +- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, functional-label evaluator/grid/annotation/mapping drift, uncertainty-plan drift, claim-boundary drift, registration drift, undeclared missing measurements, measurement-bearing failed receipts, non-null aggregate/CI summaries after any failed track, unregistered registration/hypothesis/metric/corpus/runtime/result/track/aggregate/measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. - The validator does not claim that SHA-256 proves licensing, annotation validity, scientific adequacy, independence beyond exact-byte uniqueness, aggregate derivation, or that the registered statistical procedure is appropriate. Rights and scientific review remain separate gates. ## References @@ -125,6 +127,10 @@ MIREX. (2025). *Music Structure Analysis*. https://music-ir.org/mirex/wiki/2025: MIREX. (2025). *Music Structure Analysis Results*. https://music-ir.org/mirex/wiki/2025:Music_Structure_Analysis_Results +MIREX. (2026). *Music Structure Analysis*. https://music-ir.org/mirex/wiki/2026:Music_Structure_Analysis + +MIREX Evaluation contributors. (2026). *music_structure_analysis/eval_script.py* (commit `b9fa0b0b32e2145af31f35830f78fc9d09a4301b`). GitHub. https://github.com/ismir-mirex/mirex-evaluation/blob/b9fa0b0b32e2145af31f35830f78fc9d09a4301b/music_structure_analysis/eval_script.py + mir_eval contributors. (n.d.). *mir_eval.segment: Structural segmentation evaluation*. https://github.com/mir-evaluation/mir_eval/blob/main/mir_eval/segment.py Wang, J.-C., Hung, Y.-N., & Smith, J. B. L. (2022). To catch a chorus, verse, intro, or anything else: Analyzing a song with structural functions. In *ICASSP 2022—2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)* (pp. 416–420). IEEE. https://arxiv.org/abs/2205.14700 From 2b1aa6c7e230818951804f53b9425330a1bd7c54 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 15:11:13 +0900 Subject: [PATCH 127/216] docs(mir): record ACC evaluator adapter parity --- .../mir/functional-label-accuracy-contract.md | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/docs/traceability/mir/functional-label-accuracy-contract.md b/docs/traceability/mir/functional-label-accuracy-contract.md index 5107638c3..b33d92f0c 100644 --- a/docs/traceability/mir/functional-label-accuracy-contract.md +++ b/docs/traceability/mir/functional-label-accuracy-contract.md @@ -48,6 +48,12 @@ This separation is deliberate. The pinned upstream evaluator is authoritative fo The MIREX task page also continues to expose a vocabulary inconsistency: descriptive prose includes `other`, while the operational seven-label output list includes `silence`. The local v1 annotation contract therefore does not guess between them at run time. A different mapping policy requires a new mapping-contract version and preregistration digest. +## Executable evaluator adapter + +`scripts/research/evaluate_structure_functional_accuracy.py` now owns the repository-side adapter for the preregistered normalized seven-label subset. It does not read files, map source labels, or duplicate the parser's `FunctionalSegment` value object. It accepts the parser-owned structural contract, validates continuous full-duration normalized segmentations, uses NumPy's 200 ms `arange` grid and the pinned evaluator's `t >= segment_end` pointer semantics, and returns track-level ACC plus frame counts and frame times. + +The adapter deliberately rejects `other` at its input boundary because `other` is not part of the preregistered normalized corpus vocabulary. This is not a claim that the official evaluator lacks `other`; it means upstream raw-label mapping must be completed before preregistration so the acceptance run cannot change labels after results are visible. + ## RED → GREEN lineage The predecessor branch froze 100 ms but left the official implementation identity and frame-grid behavior unresolved. Fresh inspection of the current standardized evaluator at `b9fa0b0b32e2145af31f35830f78fc9d09a4301b` showed that its MIREX 2025 reproduction uses a 200 ms hop with explicit `np.arange` and segment-boundary behavior. @@ -56,6 +62,8 @@ RED `6c20500bf31be88cd145f30fe7355a915066bdfc` changed the focused policy test t GREEN `0f1e3e732154ff0ea94beb5a60e5702ed46c2453` changed the closed-world validator contract to the pinned official evaluator and 200 ms grid. Fixture alignment `86e6be0b41547ecd14b4f837ff6c88cf8f770b0b` and `02c73b1fff55def189bed8fe2d34d4b5ee83cb0a` moved the shared evidence and corpus-admission registrations onto the same contract instead of leaving successful tests on the stale 100 ms shape. +Evaluator RED `5cffef1a5db37b700ee7a27012c0f8324855665f` added executable parity cases for the pinned upstream identity, exact-boundary advancement, exclusion of a frame at exact track duration, inclusion of a trailing partial span when its grid point is below duration, and fail-closed normalized labels. GREEN `e72b29565a6d906a9955907656cbae97cffa471a` implemented the adapter. Consolidation `c84c0acdf25159f9aa066e21f9420d9fdf72c50a` / `fe3ed61a3b1926af6b39b6302d6f857979f9bccc` removed a duplicate segment value object so the evaluator consumes the parser-owned segment contract instead. + The earlier immutable parser lineage remains valid: parser RED `cc4a166cc9e1496cd562f7d79b9ffa30f6ca19c2` → GREEN `0ee4a6aed49c8f001da451ca63c5f20e52832f5c`, with authority correction `52b6c7362baa00151f908a24cadbb1efeddd2101` separating the BandScope TSV representation from the MIREX submission format. ## Constraints and rejected alternatives @@ -72,9 +80,9 @@ Parsing annotations inside resource admission was rejected. Resource admission o ## Claim boundary and next work -This contract now freezes the ACC evaluator identity, 200 ms grid, boundary-point behavior, final-partial-frame behavior, annotation interpretation, and label-mapping boundary. It does not yet implement the BandScope runner that must demonstrate parity with the pinned evaluator on admitted normalized annotations, approve the corpus or noninferiority margins, derive aggregate/paired uncertainty, or justify a production CQT → STFT switch. +This contract now freezes the ACC evaluator identity, 200 ms grid, boundary-point behavior, final-partial-frame behavior, annotation interpretation, label-mapping boundary, and a repository-owned adapter for the preregistered normalized subset. It does not yet wire that adapter into the admitted-track consumer and paired CQT/STFT experiment, approve the corpus or noninferiority margins, derive aggregate/paired uncertainty, or justify a production CQT → STFT switch. -The next scientific slice is an executable parity-tested ACC runner that consumes the exact admitted PCM/annotation snapshots and reproduces the pinned evaluator semantics without reopening workstation paths or performing post-preregistration label mapping. Production acceptance still requires rights-cleared real decoded audio, independently reviewed normalized annotations, preregistered aggregation and paired uncertainty, paired CQT/STFT execution on the same admitted signal/runtime identity, and current-head release evidence. +The next scientific slice is to integrate the parser and ACC adapter into the exact admitted PCM/annotation consumer alongside the remaining recognized structure metrics, then implement the reviewed aggregation/paired-uncertainty procedure. Production acceptance still requires rights-cleared real decoded audio, independently reviewed normalized annotations, paired CQT/STFT execution on the same admitted signal/runtime identity, and current-head release evidence. Synthetic annotation fixtures remain unit evidence only and are not counted as production scientific acceptance. From b42297e2300d3f9b085206ccfc796fd8bb7713b6 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 16:05:53 +0900 Subject: [PATCH 128/216] test(mir): require admitted-track paired functional accuracy --- ...ture_admitted_track_functional_accuracy.py | 129 ++++++++++++++++++ 1 file changed, 129 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_admitted_track_functional_accuracy.py diff --git a/services/analysis-engine/tests/test_structure_admitted_track_functional_accuracy.py b/services/analysis-engine/tests/test_structure_admitted_track_functional_accuracy.py new file mode 100644 index 000000000..f8c9833bf --- /dev/null +++ b/services/analysis-engine/tests/test_structure_admitted_track_functional_accuracy.py @@ -0,0 +1,129 @@ +"""Contract for paired functional ACC on the exact admitted PCM handoff.""" + +from __future__ import annotations + +import hashlib +import struct +from fractions import Fraction +from types import ModuleType, SimpleNamespace +from typing import Any + +import pytest +from conftest import load_module + + +def _consumer_module() -> ModuleType: + """Load the repository-owned admitted-track functional ACC consumer.""" + return load_module( + "scripts/research/evaluate_admitted_structure_track.py", + "evaluate_admitted_structure_track", + ) + + +def _segments(*rows: tuple[str, str, str]) -> tuple[SimpleNamespace, ...]: + """Build protocol-compatible functional segments without copying production types.""" + return tuple( + SimpleNamespace(start=Fraction(start), end=Fraction(end), label=label) + for start, end, label in rows + ) + + +def test_consumer_scores_baseline_and_candidate_on_the_same_admitted_pcm() -> None: + """Both feature lanes must consume the exact read-only PCM handed off by admission.""" + module = _consumer_module() + pcm = memoryview(struct.pack("<8f", *([0.0] * 8))) + annotation = memoryview(b"0.0\t0.4\tverse\n0.4\t0.8\tchorus\n") + observed: list[tuple[str, memoryview, int, Fraction]] = [] + + def baseline_segmenter( + decoded_pcm: memoryview, + sample_rate_hz: int, + duration_seconds: Fraction, + ) -> tuple[SimpleNamespace, ...]: + """Return the baseline segmentation while recording the exact input identity.""" + observed.append(("baseline", decoded_pcm, sample_rate_hz, duration_seconds)) + return _segments(("0.0", "0.4", "verse"), ("0.4", "0.8", "chorus")) + + def candidate_segmenter( + decoded_pcm: memoryview, + sample_rate_hz: int, + duration_seconds: Fraction, + ) -> tuple[SimpleNamespace, ...]: + """Return a known candidate segmentation on the same immutable signal.""" + observed.append(("candidate", decoded_pcm, sample_rate_hz, duration_seconds)) + return _segments(("0.0", "0.2", "verse"), ("0.2", "0.8", "chorus")) + + consumer = module.PairedFunctionalAccuracyTrackConsumer( + baseline_segmenter=baseline_segmenter, + candidate_segmenter=candidate_segmenter, + ) + consumer("track-01", pcm, annotation, 10) + + assert [name for name, *_ in observed] == ["baseline", "candidate"] + assert all(view is pcm for _, view, _, _ in observed) + assert all(view.readonly for _, view, _, _ in observed) + assert all(sample_rate == 10 for _, _, sample_rate, _ in observed) + assert all(duration == Fraction(4, 5) for _, _, _, duration in observed) + + evidence = consumer.evidence + assert len(evidence) == 1 + assert evidence[0].track_id == "track-01" + assert evidence[0].decoded_pcm_sha256 == hashlib.sha256(pcm).hexdigest() + assert evidence[0].annotation_sha256 == hashlib.sha256(annotation).hexdigest() + assert evidence[0].decoded_frames == 8 + assert evidence[0].sample_rate_hz == 10 + assert evidence[0].baseline_accuracy == pytest.approx(1.0) + assert evidence[0].candidate_accuracy == pytest.approx(0.75) + assert evidence[0].baseline_total_frames == 4 + assert evidence[0].candidate_total_frames == 4 + + +def test_consumer_fails_closed_before_measurement_on_mutable_or_malformed_pcm() -> None: + """The measurement boundary must not accept mutable or non-float32-shaped handoffs.""" + module = _consumer_module() + calls: list[str] = [] + + def should_not_run(*_args: Any) -> tuple[SimpleNamespace, ...]: + """Record an invalid call if admission-shape checks fail to run first.""" + calls.append("called") + return _segments(("0.0", "0.4", "verse")) + + consumer = module.PairedFunctionalAccuracyTrackConsumer( + baseline_segmenter=should_not_run, + candidate_segmenter=should_not_run, + ) + annotation = memoryview(b"0.0\t0.4\tverse\n") + + with pytest.raises(ValueError, match="read-only"): + consumer("track-01", memoryview(bytearray(16)), annotation, 10) + with pytest.raises(ValueError, match="float32"): + consumer("track-01", memoryview(b"abc"), annotation, 10) + + assert calls == [] + assert consumer.evidence == () + + +def test_consumer_rejects_duplicate_track_measurement_identity() -> None: + """A registered track must not be counted twice in paired evidence.""" + module = _consumer_module() + pcm = memoryview(struct.pack("<4f", *([0.0] * 4))) + annotation = memoryview(b"0.0\t0.4\tverse\n") + + def same_segmenter( + _decoded_pcm: memoryview, + _sample_rate_hz: int, + _duration_seconds: Fraction, + ) -> tuple[SimpleNamespace, ...]: + """Return one complete normalized segment.""" + return _segments(("0.0", "0.4", "verse")) + + consumer = module.PairedFunctionalAccuracyTrackConsumer( + baseline_segmenter=same_segmenter, + candidate_segmenter=same_segmenter, + ) + consumer("track-01", pcm, annotation, 10) + + with pytest.raises(ValueError, match="already measured"): + consumer("track-01", pcm, annotation, 10) + + assert len(consumer.evidence) == 1 From a3bfbe2b4e76b62ba3446241bdfb467841bdaa93 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 16:06:44 +0900 Subject: [PATCH 129/216] feat(mir): score paired ACC on admitted PCM handoff --- .../evaluate_admitted_structure_track.py | 183 ++++++++++++++++++ 1 file changed, 183 insertions(+) create mode 100644 scripts/research/evaluate_admitted_structure_track.py diff --git a/scripts/research/evaluate_admitted_structure_track.py b/scripts/research/evaluate_admitted_structure_track.py new file mode 100644 index 000000000..5df46b6e2 --- /dev/null +++ b/scripts/research/evaluate_admitted_structure_track.py @@ -0,0 +1,183 @@ +#!/usr/bin/env python3 +"""Score paired functional ACC on the exact corpus-admission handoff. + +This module bridges resource admission to Signal-MIR measurement without +reopening workstation paths. The baseline and candidate segmenters receive the +same immutable canonical PCM memoryview, while the reference segmentation is +parsed from the same immutable admitted annotation snapshot. Scientific corpus +choice, margins, aggregation, uncertainty, and the production feature switch +remain outside this boundary. +""" + +from __future__ import annotations + +import hashlib +import importlib.util +import sys +from collections.abc import Callable, Sequence +from dataclasses import dataclass +from fractions import Fraction +from pathlib import Path +from types import ModuleType +from typing import Any + +Segmenter = Callable[[memoryview, int, Fraction], Sequence[Any]] + + +def _load_sibling(filename: str, module_name: str) -> ModuleType: + """Load one repository-owned sibling script under a stable private module name.""" + existing = sys.modules.get(module_name) + if existing is not None: + return existing + + path = Path(__file__).with_name(filename) + spec = importlib.util.spec_from_file_location(module_name, path) + if spec is None or spec.loader is None: + raise RuntimeError(f"could not load research module: {filename}") + module = importlib.util.module_from_spec(spec) + sys.modules[module_name] = module + try: + spec.loader.exec_module(module) + except Exception: + sys.modules.pop(module_name, None) + raise + return module + + +_PARSER = _load_sibling( + "parse_structure_functional_annotations.py", + "_bandscope_structure_functional_annotations", +) +_EVALUATOR = _load_sibling( + "evaluate_structure_functional_accuracy.py", + "_bandscope_structure_functional_accuracy", +) + + +@dataclass(frozen=True, slots=True) +class PairedFunctionalAccuracyEvidence: + """Path-free track evidence bound to exact admitted PCM and annotation bytes.""" + + track_id: str + decoded_pcm_sha256: str + annotation_sha256: str + decoded_frames: int + sample_rate_hz: int + baseline_accuracy: float + baseline_correct_frames: int + baseline_total_frames: int + candidate_accuracy: float + candidate_correct_frames: int + candidate_total_frames: int + + +def _track_id(value: object) -> str: + """Return one canonical non-empty track identifier without silent trimming.""" + if not isinstance(value, str) or not value or value != value.strip(): + raise ValueError("track_id must be non-empty text without surrounding whitespace") + return value + + +def _sample_rate(value: object) -> int: + """Return one positive sample rate while rejecting booleans and coercion.""" + if isinstance(value, bool) or not isinstance(value, int) or value <= 0: + raise ValueError("sample_rate_hz must be a positive integer") + return value + + +def _pcm_shape(decoded_pcm: object) -> tuple[memoryview, int]: + """Validate the immutable mono-float32 byte shape guaranteed by corpus admission.""" + if not isinstance(decoded_pcm, memoryview) or not decoded_pcm.readonly: + raise ValueError("decoded PCM must be a read-only memoryview") + if not decoded_pcm.c_contiguous: + raise ValueError("decoded PCM must be a contiguous read-only memoryview") + if decoded_pcm.nbytes == 0 or decoded_pcm.nbytes % 4 != 0: + raise ValueError("decoded PCM must contain non-empty mono float32 bytes") + return decoded_pcm, decoded_pcm.nbytes // 4 + + +def _annotation_view(annotation_bytes: object) -> memoryview: + """Require the immutable annotation snapshot emitted by corpus admission.""" + if not isinstance(annotation_bytes, memoryview) or not annotation_bytes.readonly: + raise ValueError("annotation input must be a read-only memoryview") + if not annotation_bytes.c_contiguous: + raise ValueError("annotation input must be a contiguous read-only memoryview") + return annotation_bytes + + +class PairedFunctionalAccuracyTrackConsumer: + """Collect paired functional ACC evidence from corpus-admission callbacks.""" + + def __init__( + self, + *, + baseline_segmenter: Segmenter, + candidate_segmenter: Segmenter, + ) -> None: + """Bind the two preregistered segmentation lanes without track-specific dispatch.""" + if not callable(baseline_segmenter) or not callable(candidate_segmenter): + raise TypeError("baseline_segmenter and candidate_segmenter must be callable") + self._baseline_segmenter = baseline_segmenter + self._candidate_segmenter = candidate_segmenter + self._evidence: list[PairedFunctionalAccuracyEvidence] = [] + self._measured_track_ids: set[str] = set() + + @property + def evidence(self) -> tuple[PairedFunctionalAccuracyEvidence, ...]: + """Return immutable ordered evidence accumulated from admitted tracks.""" + return tuple(self._evidence) + + def __call__( + self, + track_id: str, + decoded_pcm: memoryview, + annotation_bytes: memoryview, + sample_rate_hz: int, + ) -> None: + """Measure both lanes on one exact admitted signal and normalized annotation.""" + normalized_track_id = _track_id(track_id) + if normalized_track_id in self._measured_track_ids: + raise ValueError(f"track_id already measured: {normalized_track_id}") + + pcm, decoded_frames = _pcm_shape(decoded_pcm) + annotation = _annotation_view(annotation_bytes) + sample_rate = _sample_rate(sample_rate_hz) + duration_seconds = Fraction(decoded_frames, sample_rate) + + reference_segments = _PARSER.parse_functional_annotations( + annotation, + decoded_frames=decoded_frames, + sample_rate_hz=sample_rate, + ) + baseline_segments = tuple( + self._baseline_segmenter(pcm, sample_rate, duration_seconds) + ) + candidate_segments = tuple( + self._candidate_segmenter(pcm, sample_rate, duration_seconds) + ) + baseline_result = _EVALUATOR.calculate_functional_accuracy( + reference_segments, + baseline_segments, + duration_seconds=duration_seconds, + ) + candidate_result = _EVALUATOR.calculate_functional_accuracy( + reference_segments, + candidate_segments, + duration_seconds=duration_seconds, + ) + + evidence = PairedFunctionalAccuracyEvidence( + track_id=normalized_track_id, + decoded_pcm_sha256=hashlib.sha256(pcm).hexdigest(), + annotation_sha256=hashlib.sha256(annotation).hexdigest(), + decoded_frames=decoded_frames, + sample_rate_hz=sample_rate, + baseline_accuracy=float(baseline_result.accuracy), + baseline_correct_frames=int(baseline_result.correct_frames), + baseline_total_frames=int(baseline_result.total_frames), + candidate_accuracy=float(candidate_result.accuracy), + candidate_correct_frames=int(candidate_result.correct_frames), + candidate_total_frames=int(candidate_result.total_frames), + ) + self._evidence.append(evidence) + self._measured_track_ids.add(normalized_track_id) From 532b82711654add9961a62bf11fe0877a141aa0b Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 16:09:04 +0900 Subject: [PATCH 130/216] docs(mir): bind ACC to admitted-track paired consumer --- .../mir/functional-label-accuracy-contract.md | 55 ++++++++++++++++--- 1 file changed, 46 insertions(+), 9 deletions(-) diff --git a/docs/traceability/mir/functional-label-accuracy-contract.md b/docs/traceability/mir/functional-label-accuracy-contract.md index b33d92f0c..b9c490176 100644 --- a/docs/traceability/mir/functional-label-accuracy-contract.md +++ b/docs/traceability/mir/functional-label-accuracy-contract.md @@ -11,9 +11,11 @@ The structure noninferiority registration originally named functional-label accu At `ismir-mirex/mirex-evaluation@b9fa0b0b32e2145af31f35830f78fc9d09a4301b`, `music_structure_analysis/eval_script.py::calculate_accuracy` uses a default `frame_hop` of 0.2 seconds. It creates frame times with `np.arange(0, gt_duration, frame_hop)`, advances a segment while `t >= segment_end`, and counts all reference-grid frames in the denominator. Therefore the former 100 ms registration did not reproduce the current official evaluator and left the exact time-grid authority implicit. +Even after the evaluator and annotation parser were pinned, a second reproducibility gap remained: the ACC adapter was not connected to the corpus-admission callback. A later experiment runner could therefore have reopened local audio or annotation paths, decoded again, or supplied different PCM to the baseline and candidate lanes while still producing syntactically valid metric receipts. + ## Decision -Schema v1 now requires the `functional_label_accuracy` registration to contain, in addition to its noninferiority margin: +Schema v1 requires the `functional_label_accuracy` registration to contain, in addition to its noninferiority margin: - `implementation = ismir-mirex/mirex-evaluation@b9fa0b0b32e2145af31f35830f78fc9d09a4301b:music_structure_analysis.eval_script.calculate_accuracy`; - `frame_size_seconds = 0.2`; @@ -30,15 +32,15 @@ Schema v1 now requires the `functional_label_accuracy` registration to contain, - every reference-grid frame contributes to the denominator; - a frame contributes to the numerator only when the reference and prediction labels match and the reference label is not `other`. -This closes the previously open frame-grid decision. The 200 ms value is not inferred from the prose examples on the MIREX wiki; it is bound to the executable standardized-evaluation source and its exact commit. +This closes the frame-grid decision. The 200 ms value is not inferred from prose examples on the MIREX wiki; it is bound to the executable standardized-evaluation source and its exact commit. ## Annotation and label-mapping boundary -`annotation_contract_version = 1.0` remains implemented by `scripts/research/parse_structure_functional_annotations.py`. It consumes the already-admitted read-only annotation snapshot and parses BandScope-local UTF-8 `startendlabel` rows into exact rational boundaries. This local TSV is not the MIREX submission transport syntax. +`annotation_contract_version = 1.0` is implemented by `scripts/research/parse_structure_functional_annotations.py`. It consumes the already-admitted read-only annotation snapshot and parses BandScope-local UTF-8 `startendlabel` rows into exact rational boundaries. This local TSV is not the MIREX submission transport syntax. The normalized segmentation must start at 0.0, preserve source order, contain strictly positive segments with no gaps or overlaps, and end exactly at `decoded_frames / sample_rate_hz`. The parser never reopens a corpus path or repairs malformed evidence. -`label_mapping_contract_version = 1.0` remains a fail-closed identity mapping at the acceptance boundary. The admitted normalized annotation bytes must already contain one of: +`label_mapping_contract_version = 1.0` is a fail-closed identity mapping at the acceptance boundary. The admitted normalized annotation bytes must already contain one of: `intro`, `verse`, `chorus`, `bridge`, `inst`, `outro`, `silence`. @@ -50,10 +52,28 @@ The MIREX task page also continues to expose a vocabulary inconsistency: descrip ## Executable evaluator adapter -`scripts/research/evaluate_structure_functional_accuracy.py` now owns the repository-side adapter for the preregistered normalized seven-label subset. It does not read files, map source labels, or duplicate the parser's `FunctionalSegment` value object. It accepts the parser-owned structural contract, validates continuous full-duration normalized segmentations, uses NumPy's 200 ms `arange` grid and the pinned evaluator's `t >= segment_end` pointer semantics, and returns track-level ACC plus frame counts and frame times. +`scripts/research/evaluate_structure_functional_accuracy.py` owns the repository-side adapter for the preregistered normalized seven-label subset. It does not read files, map source labels, or duplicate the parser's `FunctionalSegment` value object. It accepts the parser-owned structural contract, validates continuous full-duration normalized segmentations, uses NumPy's 200 ms `arange` grid and the pinned evaluator's `t >= segment_end` pointer semantics, and returns track-level ACC plus frame counts and frame times. The adapter deliberately rejects `other` at its input boundary because `other` is not part of the preregistered normalized corpus vocabulary. This is not a claim that the official evaluator lacks `other`; it means upstream raw-label mapping must be completed before preregistration so the acceptance run cannot change labels after results are visible. +## Exact admitted-track paired consumer + +`scripts/research/evaluate_admitted_structure_track.py` now implements the functional-ACC consumer for `verify_structure_corpus.py`'s in-process `track_consumer` boundary. It receives only the already-verified track ID, immutable canonical PCM memoryview, immutable annotation snapshot, and registered sample rate. It does not receive or reopen workstation paths. + +For each admitted track it: + +- requires the PCM and annotation handoffs to be read-only contiguous memoryviews; +- derives `decoded_frames` from the canonical mono-float32 byte length and the exact duration as `decoded_frames / sample_rate_hz`; +- parses the reference segmentation from the admitted annotation bytes with the v1 parser; +- invokes the preregistered baseline and candidate segmentation lanes on the **same PCM memoryview object**, sample rate, and exact duration; +- evaluates both outputs with the pinned 200 ms functional-ACC adapter; +- records path-free evidence containing the track ID, PCM SHA-256, annotation SHA-256, frame/sample-rate identity, and baseline/candidate ACC frame counts; +- rejects duplicate measurement of the same track ID rather than letting one registered item inflate paired evidence. + +The segmentation callbacks intentionally do not receive `track_id`. The measurement boundary therefore does not offer a built-in track-specific dispatch key that could select a different algorithm after corpus identity is known. This does not prove that arbitrary caller code is scientifically valid; the eventual baseline and candidate segmenter implementations still have to be pinned by the preregistration and reviewed before real-audio execution. + +This consumer is research measurement infrastructure, not a production feature switch. It does not select the corpus, choose noninferiority margins, implement aggregation or uncertainty, alter `sections/segmenter.py`, or count synthetic fixtures as scientific acceptance. + ## RED → GREEN lineage The predecessor branch froze 100 ms but left the official implementation identity and frame-grid behavior unresolved. Fresh inspection of the current standardized evaluator at `b9fa0b0b32e2145af31f35830f78fc9d09a4301b` showed that its MIREX 2025 reproduction uses a 200 ms hop with explicit `np.arange` and segment-boundary behavior. @@ -64,7 +84,9 @@ GREEN `0f1e3e732154ff0ea94beb5a60e5702ed46c2453` changed the closed-world valida Evaluator RED `5cffef1a5db37b700ee7a27012c0f8324855665f` added executable parity cases for the pinned upstream identity, exact-boundary advancement, exclusion of a frame at exact track duration, inclusion of a trailing partial span when its grid point is below duration, and fail-closed normalized labels. GREEN `e72b29565a6d906a9955907656cbae97cffa471a` implemented the adapter. Consolidation `c84c0acdf25159f9aa066e21f9420d9fdf72c50a` / `fe3ed61a3b1926af6b39b6302d6f857979f9bccc` removed a duplicate segment value object so the evaluator consumes the parser-owned segment contract instead. -The earlier immutable parser lineage remains valid: parser RED `cc4a166cc9e1496cd562f7d79b9ffa30f6ca19c2` → GREEN `0ee4a6aed49c8f001da451ca63c5f20e52832f5c`, with authority correction `52b6c7362baa00151f908a24cadbb1efeddd2101` separating the BandScope TSV representation from the MIREX submission format. +The immutable parser lineage remains parser RED `cc4a166cc9e1496cd562f7d79b9ffa30f6ca19c2` → GREEN `0ee4a6aed49c8f001da451ca63c5f20e52832f5c`, with authority correction `52b6c7362baa00151f908a24cadbb1efeddd2101` separating the BandScope TSV representation from the MIREX submission format. + +Admitted-track integration RED `b42297e2300d3f9b085206ccfc796fd8bb7713b6` requires both measurement lanes to receive the exact same read-only PCM handoff, binds evidence to PCM/annotation hashes, rejects malformed or mutable PCM before either segmenter executes, and rejects duplicate track measurement. GREEN `a3bfbe2b4e76b62ba3446241bdfb467841bdaa93` implements that paired consumer without changing production segmentation. ## Constraints and rejected alternatives @@ -78,13 +100,28 @@ Automatically normalizing labels during the acceptance run was rejected. Mapping Parsing annotations inside resource admission was rejected. Resource admission owns byte identity and immutable handoff; Signal-MIR owns scientific interpretation and measurement. +Reopening or re-decoding the corpus separately for CQT and STFT was rejected. Paired measurement must consume the same immutable admitted PCM identity; otherwise a feature comparison can be confounded by decode or local-file drift. + +Copying the PCM into independent baseline/candidate input buffers at this boundary was rejected. Both lanes now receive the same read-only memoryview object, making input identity explicit while preventing mutation through the consumer API. + +Computing an aggregate or confidence interval in the admitted-track consumer was rejected. Aggregation and paired uncertainty remain unapproved scientific decisions and must be preregistered before candidate results are visible. + +## Security Notes + +- The consumer adds no filesystem, subprocess, network, model-download, or generic execution capability. +- PCM and annotation material are accepted only as immutable in-process views from resource admission; malformed shape or mutability fails before either measurement lane executes. +- Durable functional evidence contains content digests and measurement values, not workstation paths or raw licensed audio. +- The injected segmentation callables remain a code-review boundary. Real-audio execution must use preregistered repository-owned implementations; arbitrary runtime plugin loading is not introduced here. + ## Claim boundary and next work -This contract now freezes the ACC evaluator identity, 200 ms grid, boundary-point behavior, final-partial-frame behavior, annotation interpretation, label-mapping boundary, and a repository-owned adapter for the preregistered normalized subset. It does not yet wire that adapter into the admitted-track consumer and paired CQT/STFT experiment, approve the corpus or noninferiority margins, derive aggregate/paired uncertainty, or justify a production CQT → STFT switch. +This contract now freezes the ACC evaluator identity, 200 ms grid, boundary-point behavior, final-partial-frame behavior, annotation interpretation, label-mapping boundary, repository-owned adapter, and the exact admitted-track paired functional-ACC handoff. It still does not implement the actual paired CQT/STFT segmentation lanes, the remaining recognized boundary/repetition metrics, approved aggregation/paired uncertainty, corpus/margin approval, or a production CQT → STFT switch. + +The next causal scientific slice is to implement repository-owned baseline/candidate structure-measurement lanes over this same admitted PCM boundary and add the recognized boundary/deviation/repetition metrics without changing production behavior. Aggregation and paired uncertainty must then be reviewed and preregistered before any rights-cleared real-corpus candidate result is inspected. -The next scientific slice is to integrate the parser and ACC adapter into the exact admitted PCM/annotation consumer alongside the remaining recognized structure metrics, then implement the reviewed aggregation/paired-uncertainty procedure. Production acceptance still requires rights-cleared real decoded audio, independently reviewed normalized annotations, paired CQT/STFT execution on the same admitted signal/runtime identity, and current-head release evidence. +Production acceptance still requires rights-cleared real decoded audio, independently reviewed normalized annotations, paired CQT/STFT execution on the same admitted signal/runtime identity, recognized track-level metrics, approved aggregate/CI evidence, current-head protected checks, and independent review. -Synthetic annotation fixtures remain unit evidence only and are not counted as production scientific acceptance. +Synthetic fixtures exercise only the measurement contract and are not production scientific acceptance. ## References From 4a0a09d627acc824a96d7b6d7d69417521e183e5 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 16:31:39 +0900 Subject: [PATCH 131/216] test(mir): require repository-owned paired feature lanes --- ...est_structure_feature_measurement_lanes.py | 117 ++++++++++++++++++ 1 file changed, 117 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_feature_measurement_lanes.py diff --git a/services/analysis-engine/tests/test_structure_feature_measurement_lanes.py b/services/analysis-engine/tests/test_structure_feature_measurement_lanes.py new file mode 100644 index 000000000..be13b7728 --- /dev/null +++ b/services/analysis-engine/tests/test_structure_feature_measurement_lanes.py @@ -0,0 +1,117 @@ +"""Contracts for repository-owned CQT/STFT structure-measurement lanes.""" + +from __future__ import annotations + +import struct +from fractions import Fraction +from types import ModuleType +from typing import Any + +import numpy as np +import pytest +from conftest import load_module + +from bandscope_analysis.sections import segmenter + + +def _lane_module() -> ModuleType: + """Load the repository-owned admitted-PCM structure feature lanes.""" + return load_module( + "scripts/research/measure_structure_feature_lanes.py", + "measure_structure_feature_lanes", + ) + + +def test_research_lanes_bind_cqt_and_stft_without_reopening_audio( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """The paired hypothesis must differ only by the registered chroma representation.""" + module = _lane_module() + pcm = memoryview(struct.pack("<8f", *([0.0] * 8))) + observed: list[tuple[str, int, float, bool, int]] = [] + + def fake_segment_with_boundaries( + audio: np.ndarray[Any, np.dtype[np.float32]], + sr: int, + duration: float | None = None, + *, + chroma_feature: str = "cqt", + ) -> tuple[list[dict[str, Any]], list[tuple[float, float]]]: + assert duration is not None + observed.append( + (chroma_feature, sr, duration, bool(audio.flags.writeable), audio.size) + ) + return ( + [ + {"form_label": "verse"}, + {"form_label": "chorus"}, + ], + [(0.0, 0.4), (0.4, 0.8)], + ) + + monkeypatch.setattr(segmenter, "segment_with_boundaries", fake_segment_with_boundaries) + + baseline = module.repository_structure_segmenter("cqt") + candidate = module.repository_structure_segmenter("stft") + baseline_segments = baseline(pcm, 10, Fraction(4, 5)) + candidate_segments = candidate(pcm, 10, Fraction(4, 5)) + + assert [segment.label for segment in baseline_segments] == ["verse", "chorus"] + assert [segment.label for segment in candidate_segments] == ["verse", "chorus"] + assert [(segment.start, segment.end) for segment in baseline_segments] == [ + (Fraction(0), Fraction(2, 5)), + (Fraction(2, 5), Fraction(4, 5)), + ] + assert observed == [ + ("cqt", 10, 0.8, False, 8), + ("stft", 10, 0.8, False, 8), + ] + + +def test_research_lane_rejects_unregistered_feature() -> None: + """Scientific feature identity must be closed-world before candidate results exist.""" + module = _lane_module() + + with pytest.raises(ValueError, match="registered chroma feature"): + module.repository_structure_segmenter("cens") + + +def test_segmenter_feature_selector_controls_boundary_and_repetition_chroma( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """One selected representation must drive both boundary and repetition semantics.""" + audio = np.zeros(8192, dtype=np.float32) + calls: list[str] = [] + chroma = np.ones((12, 128), dtype=np.float64) + + def fake_cqt(**_kwargs: Any) -> np.ndarray[Any, np.dtype[np.float64]]: + calls.append("cqt") + return chroma + + def fake_stft(**_kwargs: Any) -> np.ndarray[Any, np.dtype[np.float64]]: + calls.append("stft") + return chroma + + monkeypatch.setattr(segmenter.librosa.feature, "chroma_cqt", fake_cqt) + monkeypatch.setattr(segmenter.librosa.feature, "chroma_stft", fake_stft) + monkeypatch.setattr( + segmenter.librosa.segment, + "recurrence_matrix", + lambda *_args, **_kwargs: np.eye(128, dtype=np.float64), + ) + monkeypatch.setattr( + segmenter, + "_checkerboard_novelty", + lambda _ssm: np.zeros(128, dtype=np.float64), + ) + + segmenter.compute_novelty_curve(audio, 8_000, chroma_feature="stft") + segmenter._segment_repetition_groups( + audio, + 8_000, + [0.0], + 1.024, + chroma_feature="stft", + ) + + assert calls == ["stft", "stft"] From d9f7604ae89f9851ae93a8ba8df7dc54e832262a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 16:34:42 +0900 Subject: [PATCH 132/216] feat(mir): parameterize structure chroma representation --- .../bandscope_analysis/sections/segmenter.py | 112 ++++++++++++++++-- 1 file changed, 100 insertions(+), 12 deletions(-) diff --git a/services/analysis-engine/src/bandscope_analysis/sections/segmenter.py b/services/analysis-engine/src/bandscope_analysis/sections/segmenter.py index 7841e20d2..04228f5eb 100644 --- a/services/analysis-engine/src/bandscope_analysis/sections/segmenter.py +++ b/services/analysis-engine/src/bandscope_analysis/sections/segmenter.py @@ -34,6 +34,8 @@ # Maximum frame count for dense SSM construction. MAX_SSM_FRAMES = 4096 +ChromaFeature = Literal["cqt", "stft"] + # Canonical section label assignment order for repeating patterns. _LABEL_ORDER: tuple[str, ...] = ( "intro", @@ -47,10 +49,33 @@ ) +def _validated_chroma_feature(value: object) -> ChromaFeature: + """Return one registered chroma representation or fail closed.""" + if value == "cqt" or value == "stft": + return value + raise ValueError("chroma_feature must be one of: cqt, stft") + + +def _extract_chroma( + audio: NDArray[np.floating[Any]], + sr: int, + hop_length: int, + *, + chroma_feature: ChromaFeature, +) -> NDArray[np.floating[Any]]: + """Extract the registered chroma representation for one structure lane.""" + feature = _validated_chroma_feature(chroma_feature) + if feature == "cqt": + return librosa.feature.chroma_cqt(y=audio, sr=sr, hop_length=hop_length) + return librosa.feature.chroma_stft(y=audio, sr=sr, hop_length=hop_length) + + def compute_novelty_curve( audio: NDArray[np.floating[Any]], sr: int, hop_length: int = 512, + *, + chroma_feature: ChromaFeature = "cqt", ) -> tuple[NDArray[np.floating[Any]], NDArray[np.floating[Any]]]: """Compute a novelty curve from the self-similarity matrix of chroma features. @@ -58,14 +83,20 @@ def compute_novelty_curve( audio: Mono audio signal as a 1D float array. sr: Sample rate. hop_length: Hop length for feature extraction. + chroma_feature: Registered chroma representation. Production defaults to CQT. Returns: Tuple of (novelty_curve, frame_times). """ + feature = _validated_chroma_feature(chroma_feature) effective_hop_length = max(hop_length, math.ceil(audio.size / MAX_SSM_FRAMES)) - # Extract chroma features for structural comparison - chroma = librosa.feature.chroma_cqt(y=audio, sr=sr, hop_length=effective_hop_length) + chroma = _extract_chroma( + audio, + sr, + effective_hop_length, + chroma_feature=feature, + ) # Build self-similarity matrix from chroma ssm = librosa.segment.recurrence_matrix( @@ -211,27 +242,32 @@ def _segment_repetition_groups( sr: int, boundaries: list[float], duration: float, + *, + chroma_feature: ChromaFeature = "cqt", ) -> list[int]: """Group segments that repeat, by mean-chroma similarity. Returns a group id per segment; segments sharing an id are acoustically - similar (a repeated section). Reuses the chroma the boundary detector relies - on, so labels reflect the audio rather than a segment's position. + similar (a repeated section). Reuses the same registered chroma representation + as boundary detection so labels reflect the measured lane rather than a mixed + CQT/STFT pipeline. Args: audio: Mono audio signal. sr: Sample rate. boundaries: Sorted boundary start times. duration: Total audio duration. + chroma_feature: Registered chroma representation. Returns: A group id per segment, in segment order. """ + feature = _validated_chroma_feature(chroma_feature) n = len(boundaries) if n == 0: return [] hop = max(512, math.ceil(audio.size / MAX_SSM_FRAMES)) - chroma = librosa.feature.chroma_cqt(y=audio, sr=sr, hop_length=hop) + chroma = _extract_chroma(audio, sr, hop, chroma_feature=feature) n_frames = chroma.shape[1] reps: list[NDArray[np.floating[Any]]] = [] groups: list[int] = [] @@ -358,6 +394,8 @@ def segment_audio( audio: NDArray[np.floating[Any]], sr: int, duration: float | None = None, + *, + chroma_feature: ChromaFeature = "cqt", ) -> list[SectionCandidate]: """Run full structural segmentation pipeline on audio. @@ -365,10 +403,12 @@ def segment_audio( audio: Mono audio signal. sr: Sample rate. duration: Optional pre-computed duration. Calculated if not provided. + chroma_feature: Registered chroma representation. Production defaults to CQT. Returns: List of SectionCandidate dicts with detected boundaries and labels. """ + feature = _validated_chroma_feature(chroma_feature) if audio.size == 0: return [] @@ -379,18 +419,31 @@ def segment_audio( return _single_section_fallback("Audio too short for structural analysis") try: - boundaries = _compute_boundaries(audio, sr, duration) + boundaries = _compute_boundaries( + audio, + sr, + duration, + chroma_feature=feature, + ) except Exception as e: logger.warning("Structural segmentation failed, falling back to single section: %s", e) return _single_section_fallback(f"Segmentation fallback: {e}") - return _sections_from_boundaries(boundaries, duration, audio, sr) + return _sections_from_boundaries( + boundaries, + duration, + audio, + sr, + chroma_feature=feature, + ) def segment_boundaries_from_audio( audio: NDArray[np.floating[Any]], sr: int, duration: float | None = None, + *, + chroma_feature: ChromaFeature = "cqt", ) -> list[tuple[float, float]]: """Return raw (start, end) boundary pairs from audio segmentation. @@ -400,10 +453,12 @@ def segment_boundaries_from_audio( audio: Mono audio signal. sr: Sample rate. duration: Optional pre-computed duration. + chroma_feature: Registered chroma representation. Production defaults to CQT. Returns: List of (start_seconds, end_seconds) tuples for each segment. """ + feature = _validated_chroma_feature(chroma_feature) if audio.size == 0: return [] @@ -414,7 +469,12 @@ def segment_boundaries_from_audio( return [(0.0, duration)] try: - boundaries = _compute_boundaries(audio, sr, duration) + boundaries = _compute_boundaries( + audio, + sr, + duration, + chroma_feature=feature, + ) except Exception as e: logger.warning("Boundary detection failed: %s", e) return [(0.0, duration)] @@ -426,6 +486,8 @@ def segment_with_boundaries( audio: NDArray[np.floating[Any]], sr: int, duration: float | None = None, + *, + chroma_feature: ChromaFeature = "cqt", ) -> tuple[list[SectionCandidate], list[tuple[float, float]]]: """Run segmentation and return both section candidates and boundary pairs. @@ -436,10 +498,12 @@ def segment_with_boundaries( audio: Mono audio signal. sr: Sample rate. duration: Optional pre-computed duration. + chroma_feature: Registered chroma representation. Production defaults to CQT. Returns: Tuple of (section_candidates, boundary_pairs). """ + feature = _validated_chroma_feature(chroma_feature) if audio.size == 0: return [], [] @@ -452,13 +516,22 @@ def segment_with_boundaries( ] try: - boundaries = _compute_boundaries(audio, sr, duration) + boundaries = _compute_boundaries( + audio, + sr, + duration, + chroma_feature=feature, + ) except Exception as e: logger.warning("Structural segmentation failed, falling back to single section: %s", e) return _single_section_fallback(f"Segmentation fallback: {e}"), [(0.0, duration)] return _sections_from_boundaries( - boundaries, duration, audio, sr + boundaries, + duration, + audio, + sr, + chroma_feature=feature, ), _boundary_pairs_from_boundaries(boundaries, duration) @@ -486,9 +559,17 @@ def _sections_from_boundaries( duration: float, audio: NDArray[np.floating[Any]], sr: int, + *, + chroma_feature: ChromaFeature = "cqt", ) -> list[SectionCandidate]: """Build section candidates from precomputed boundary start times.""" - groups = _segment_repetition_groups(audio, sr, boundaries, duration) + groups = _segment_repetition_groups( + audio, + sr, + boundaries, + duration, + chroma_feature=chroma_feature, + ) labels = assign_section_labels(boundaries, duration, groups) sections: list[SectionCandidate] = [] n_boundaries = len(boundaries) @@ -538,6 +619,8 @@ def _compute_boundaries( audio: NDArray[np.floating[Any]], sr: int, duration: float, + *, + chroma_feature: ChromaFeature = "cqt", ) -> list[float]: """Compute raw boundary times from audio (shared implementation). @@ -545,9 +628,14 @@ def _compute_boundaries( audio: Mono audio signal. sr: Sample rate. duration: Total audio duration. + chroma_feature: Registered chroma representation. Returns: Sorted list of boundary start times. """ - novelty, frame_times = compute_novelty_curve(audio, sr) + novelty, frame_times = compute_novelty_curve( + audio, + sr, + chroma_feature=chroma_feature, + ) return detect_boundaries(novelty, frame_times, duration) From 3b5f82dab83653176851abb087dd60b7ec7e78d1 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 16:35:29 +0900 Subject: [PATCH 133/216] feat(mir): bind admitted PCM to CQT and STFT lanes --- .../measure_structure_feature_lanes.py | 174 ++++++++++++++++++ 1 file changed, 174 insertions(+) create mode 100644 scripts/research/measure_structure_feature_lanes.py diff --git a/scripts/research/measure_structure_feature_lanes.py b/scripts/research/measure_structure_feature_lanes.py new file mode 100644 index 000000000..f448b1836 --- /dev/null +++ b/scripts/research/measure_structure_feature_lanes.py @@ -0,0 +1,174 @@ +#!/usr/bin/env python3 +"""Bind preregistered CQT/STFT structure lanes to admitted PCM. + +The experiment hypothesis from #1225 changes only the chroma representation +used by the repository-owned production segmentation pipeline. This adapter +turns the immutable canonical little-endian float32 PCM snapshot into a +zero-copy NumPy view, selects the registered CQT or STFT lane, and converts the +result back to the normalized functional-segment value object used by the +scientific evaluator. + +It does not open paths, decode audio, download models, perform label mapping, or +choose margins, aggregation, uncertainty, or production defaults. + +Security Notes: +- Input is the read-only PCM memoryview already admitted by the corpus boundary. +- No filesystem, network, subprocess, plugin-loading, or generic execution path + is introduced here. +- Feature identity is closed-world (``cqt`` or ``stft``) and invalid values fail + before analysis. +""" + +from __future__ import annotations + +import importlib.util +import math +import sys +from collections.abc import Callable +from fractions import Fraction +from pathlib import Path +from types import ModuleType +from typing import Any, Literal + +import numpy as np + +from bandscope_analysis.sections import segmenter + +ChromaFeature = Literal["cqt", "stft"] +StructureSegmenter = Callable[[memoryview, int, Fraction], tuple[Any, ...]] + + +def _load_annotation_module() -> ModuleType: + """Load the canonical research functional-segment value object.""" + module_name = "_bandscope_structure_feature_lane_annotations" + existing = sys.modules.get(module_name) + if existing is not None: + return existing + + path = Path(__file__).with_name("parse_structure_functional_annotations.py") + spec = importlib.util.spec_from_file_location(module_name, path) + if spec is None or spec.loader is None: + raise RuntimeError("could not load functional annotation contract") + module = importlib.util.module_from_spec(spec) + sys.modules[module_name] = module + try: + spec.loader.exec_module(module) + except Exception: + sys.modules.pop(module_name, None) + raise + return module + + +_ANNOTATIONS = _load_annotation_module() + + +def _registered_feature(value: object) -> ChromaFeature: + """Return one preregistered feature identity or fail closed.""" + if value == "cqt" or value == "stft": + return value + raise ValueError("registered chroma feature must be one of: cqt, stft") + + +def _pcm_view(decoded_pcm: object) -> memoryview: + """Require the immutable canonical mono-float32 handoff shape.""" + if not isinstance(decoded_pcm, memoryview) or not decoded_pcm.readonly: + raise ValueError("decoded PCM must be a read-only memoryview") + if not decoded_pcm.c_contiguous: + raise ValueError("decoded PCM must be contiguous") + if decoded_pcm.nbytes == 0 or decoded_pcm.nbytes % 4 != 0: + raise ValueError("decoded PCM must contain non-empty mono float32 bytes") + return decoded_pcm + + +def _positive_sample_rate(value: object) -> int: + """Return a positive integer sample rate without coercion.""" + if isinstance(value, bool) or not isinstance(value, int) or value <= 0: + raise ValueError("sample_rate_hz must be a positive integer") + return value + + +def _boundary_fraction(value: object, field: str) -> Fraction: + """Convert one finite production boundary float to a stable exact rational.""" + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise ValueError(f"{field} must be a finite numeric boundary") + numeric = float(value) + if not math.isfinite(numeric): + raise ValueError(f"{field} must be a finite numeric boundary") + return Fraction(str(numeric)) + + +def repository_structure_segmenter(chroma_feature: object) -> StructureSegmenter: + """Return the repository-owned lane for one preregistered chroma feature.""" + feature = _registered_feature(chroma_feature) + + def measure( + decoded_pcm: memoryview, + sample_rate_hz: int, + duration_seconds: Fraction, + ) -> tuple[Any, ...]: + pcm = _pcm_view(decoded_pcm) + sample_rate = _positive_sample_rate(sample_rate_hz) + decoded_frames = pcm.nbytes // 4 + expected_duration = Fraction(decoded_frames, sample_rate) + if duration_seconds != expected_duration: + raise ValueError("duration_seconds must match the admitted PCM identity") + + audio = np.frombuffer(pcm, dtype=" expected_duration: + raise RuntimeError("structure lane returned an invalid boundary interval") + + normalized.append( + _ANNOTATIONS.FunctionalSegment( + start=start, + end=end, + label=label, + ) + ) + previous_end = end + + if previous_end != expected_duration: + raise RuntimeError("structure lane did not cover the admitted duration") + return tuple(normalized) + + return measure From 21eeb46ffbd5a17f1cdf7f47fd43a88f965f1d43 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 16:38:09 +0900 Subject: [PATCH 134/216] feat(mir): bind paired ACC to registered CQT-STFT lanes --- .../research/evaluate_admitted_structure_track.py | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/scripts/research/evaluate_admitted_structure_track.py b/scripts/research/evaluate_admitted_structure_track.py index 5df46b6e2..1ef140170 100644 --- a/scripts/research/evaluate_admitted_structure_track.py +++ b/scripts/research/evaluate_admitted_structure_track.py @@ -52,6 +52,10 @@ def _load_sibling(filename: str, module_name: str) -> ModuleType: "evaluate_structure_functional_accuracy.py", "_bandscope_structure_functional_accuracy", ) +_LANES = _load_sibling( + "measure_structure_feature_lanes.py", + "_bandscope_structure_feature_lanes", +) @dataclass(frozen=True, slots=True) @@ -122,6 +126,14 @@ def __init__( self._evidence: list[PairedFunctionalAccuracyEvidence] = [] self._measured_track_ids: set[str] = set() + @classmethod + def for_registered_cqt_stft_hypothesis(cls) -> PairedFunctionalAccuracyTrackConsumer: + """Bind the canonical #1225 baseline/candidate feature identities.""" + return cls( + baseline_segmenter=_LANES.repository_structure_segmenter("cqt"), + candidate_segmenter=_LANES.repository_structure_segmenter("stft"), + ) + @property def evidence(self) -> tuple[PairedFunctionalAccuracyEvidence, ...]: """Return immutable ordered evidence accumulated from admitted tracks.""" From 0b0f05f1f19202ce291d36b3891e4aa158183c8e Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 16:38:34 +0900 Subject: [PATCH 135/216] test(mir): freeze CQT-STFT hypothesis factory --- ...ture_admitted_track_functional_accuracy.py | 27 +++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_admitted_track_functional_accuracy.py b/services/analysis-engine/tests/test_structure_admitted_track_functional_accuracy.py index f8c9833bf..804d77bcb 100644 --- a/services/analysis-engine/tests/test_structure_admitted_track_functional_accuracy.py +++ b/services/analysis-engine/tests/test_structure_admitted_track_functional_accuracy.py @@ -78,6 +78,33 @@ def candidate_segmenter( assert evidence[0].candidate_total_frames == 4 +def test_registered_hypothesis_factory_binds_exact_cqt_then_stft_lanes( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """The canonical experiment factory must not leave feature identity to callers.""" + module = _consumer_module() + requested: list[str] = [] + + def lane(feature: str) -> Any: + requested.append(feature) + + def segmenter( + _pcm: memoryview, + _sample_rate: int, + _duration: Fraction, + ) -> tuple[SimpleNamespace, ...]: + return _segments(("0.0", "0.4", "verse")) + + return segmenter + + monkeypatch.setattr(module._LANES, "repository_structure_segmenter", lane) + + consumer = module.PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis() + + assert isinstance(consumer, module.PairedFunctionalAccuracyTrackConsumer) + assert requested == ["cqt", "stft"] + + def test_consumer_fails_closed_before_measurement_on_mutable_or_malformed_pcm() -> None: """The measurement boundary must not accept mutable or non-float32-shaped handoffs.""" module = _consumer_module() From bd08323ff41b0aa4a1430756c6bcd7b5b9945a33 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 16:39:57 +0900 Subject: [PATCH 136/216] docs(mir): trace registered CQT-STFT measurement lanes --- .../structure-feature-measurement-lanes.md | 64 +++++++++++++++++++ 1 file changed, 64 insertions(+) create mode 100644 docs/traceability/mir/structure-feature-measurement-lanes.md diff --git a/docs/traceability/mir/structure-feature-measurement-lanes.md b/docs/traceability/mir/structure-feature-measurement-lanes.md new file mode 100644 index 000000000..3240bffa2 --- /dev/null +++ b/docs/traceability/mir/structure-feature-measurement-lanes.md @@ -0,0 +1,64 @@ +# Structure feature measurement lanes + +Status: Proposed +Owner: Signal-MIR Analysis +Tracking: #1225, #1228 +Production default: `chroma_cqt` + +## Problem + +The admitted-track functional-ACC consumer already guaranteed that baseline and candidate received the same immutable decoded PCM object, sample rate, exact duration, and normalized annotation snapshot. It still accepted arbitrary injected segmentation callbacks, however. That left the scientific identity of the actual CQT baseline and STFT candidate outside the repository-owned measurement boundary. + +The earlier #1223 optimization shows the intended hypothesis precisely: replace both structure-pipeline uses of `librosa.feature.chroma_cqt` with `librosa.feature.chroma_stft`. Production reverted that change because synthetic timing did not establish real-audio noninferiority. The experiment therefore needs two repository-owned lanes that differ in that representation and nothing else, without changing the production default. + +## Decision + +BandScope now exposes a closed `ChromaFeature = Literal["cqt", "stft"]` selector inside the existing structure segmenter. The selector is keyword-only on the public segmentation entry points and defaults to `cqt`, so existing production callers retain the protected CQT behavior. + +The selected representation propagates through both places where structure semantics depend on chroma: + +1. self-similarity/novelty boundary extraction; and +2. mean-chroma repetition grouping used to derive functional section labels. + +Using STFT for boundary detection while silently returning to CQT for repetition grouping would not reproduce the #1223 candidate and would make functional-label evidence scientifically ambiguous, so mixed-representation execution is not an allowed lane. + +`scripts/research/measure_structure_feature_lanes.py` binds the experiment to the repository-owned segmenter. It accepts only the read-only admitted canonical little-endian float32 PCM memoryview, checks that `duration_seconds == decoded_frames / sample_rate_hz`, creates a zero-copy read-only NumPy view, invokes the selected repository lane, and converts its section labels and boundaries to the canonical research `FunctionalSegment` value object. It opens no path and performs no decoding, networking, subprocess execution, model download, dynamic plugin discovery, label normalization, aggregation, or inferential decision. + +`PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis()` is the canonical factory for #1225. It fixes the baseline lane to `cqt` and candidate lane to `stft` rather than asking an experiment caller to select those identities at runtime. The generic constructor remains injectable for focused tests, but production scientific execution of this hypothesis should use the registered factory. + +## Alternatives rejected + +- **Copy the #1223 STFT implementation into a research-only segmenter.** Rejected because duplicated structure logic would drift from the production baseline and would not test the actual code path a later feature-switch PR would use. +- **Switch production to STFT behind the scientific PR.** Rejected because the experiment exists precisely because production acceptance has not yet been established. +- **Change only boundary chroma and leave repetition grouping on CQT.** Rejected because functional section labels would then describe a mixed feature pipeline rather than the registered candidate. +- **Let the experiment runner inject arbitrary callbacks.** Retained only as a test seam; rejected as the canonical scientific identity because callback provenance is not equivalent to a reviewed repository-owned CQT/STFT lane. + +## RED → GREEN evidence + +- RED `4a0a09d627acc824a96d7b6d7d69417521e183e5` adds executable contracts requiring a repository-owned CQT/STFT lane factory and requiring the selected STFT representation to drive both novelty/boundary and repetition-group extraction. The source at that commit had neither the lane module nor the feature selector. +- GREEN `d9f7604ae89f9851ae93a8ba8df7dc54e832262a` parameterizes the existing structure pipeline while preserving `cqt` as the production default. +- GREEN `3b5f82dab83653176851abb087dd60b7ec7e78d1` adds the admitted-PCM lane adapter with no file/network/subprocess capability. +- `21eeb46ffbd5a17f1cdf7f47fd43a88f965f1d43` binds the paired functional-ACC consumer to those repository-owned lanes through a canonical hypothesis factory. +- `0b0f05f1f19202ce291d36b3891e4aa158183c8e` freezes the factory order as baseline `cqt`, candidate `stft` in regression coverage. + +Predecessor workflow results do not transfer to later heads. Hosted evidence must be evaluated on the exact final head of #1228. + +## Claim boundary + +This change establishes **measurement identity**, not noninferiority. It does not show that STFT preserves rehearsal-relevant structure quality, does not establish a latency advantage, and does not authorize a production default change. + +Production acceptance still requires the preregistered rights-cleared real-audio corpus, recognized boundary/deviation/repetition metrics, functional ACC, registered latency/RSS measurements, reviewed aggregation and paired uncertainty, exact runtime/source identities, complete-track fail-closed handling, and a qualifying current-head independent review. + +Synthetic fixtures in the lane tests demonstrate wiring and invariants only. They are not scientific acceptance evidence. + +## Security Notes + +- The lane adapter receives only the immutable decoded PCM handoff already admitted by `verify_structure_corpus.py`; it receives no workstation audio path. +- Feature identity is a two-value closed world and invalid values fail before feature extraction. +- The adapter adds no filesystem, URL, subprocess, network, model-download, plugin-loading, IPC, or export capability. +- NumPy reads directly from the immutable PCM buffer; the adapter rejects a mutable view rather than copying into an untracked scientific input. +- Boundary/label output is validated for cardinality, continuity, finite values, and full admitted-duration coverage before it becomes functional-ACC input. + +## Follow-up + +The next scientific implementation slice is to add the recognized boundary retrieval/deviation and repetition-group metric adapters to the same admitted-track CQT/STFT lane output, then implement the reviewed preregistered aggregation and paired-uncertainty procedure. Only after those contracts are frozen should the rights-cleared corpus be measured and interpreted within its registered claim boundary. From 8147c2bbbb50c6109928a5df2edd5a01bc181842 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 16:41:35 +0900 Subject: [PATCH 137/216] fix(mir): make feature literal validation type-safe --- .../src/bandscope_analysis/sections/segmenter.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/services/analysis-engine/src/bandscope_analysis/sections/segmenter.py b/services/analysis-engine/src/bandscope_analysis/sections/segmenter.py index 04228f5eb..f9b452451 100644 --- a/services/analysis-engine/src/bandscope_analysis/sections/segmenter.py +++ b/services/analysis-engine/src/bandscope_analysis/sections/segmenter.py @@ -51,8 +51,10 @@ def _validated_chroma_feature(value: object) -> ChromaFeature: """Return one registered chroma representation or fail closed.""" - if value == "cqt" or value == "stft": - return value + if value == "cqt": + return "cqt" + if value == "stft": + return "stft" raise ValueError("chroma_feature must be one of: cqt, stft") From eaadd47315fa716c61721cfdb5feb446c103f004 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 16:42:28 +0900 Subject: [PATCH 138/216] fix(mir): make registered feature literal type-safe --- scripts/research/measure_structure_feature_lanes.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/scripts/research/measure_structure_feature_lanes.py b/scripts/research/measure_structure_feature_lanes.py index f448b1836..216b66691 100644 --- a/scripts/research/measure_structure_feature_lanes.py +++ b/scripts/research/measure_structure_feature_lanes.py @@ -64,8 +64,10 @@ def _load_annotation_module() -> ModuleType: def _registered_feature(value: object) -> ChromaFeature: """Return one preregistered feature identity or fail closed.""" - if value == "cqt" or value == "stft": - return value + if value == "cqt": + return "cqt" + if value == "stft": + return "stft" raise ValueError("registered chroma feature must be one of: cqt, stft") From 5d509998e094348216ea81fd448fc294deacc6b4 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 16:48:21 +0900 Subject: [PATCH 139/216] test(mir): lock public lane feature propagation --- ...est_structure_feature_measurement_lanes.py | 43 +++++++++++++++++++ 1 file changed, 43 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_feature_measurement_lanes.py b/services/analysis-engine/tests/test_structure_feature_measurement_lanes.py index be13b7728..17c42acd2 100644 --- a/services/analysis-engine/tests/test_structure_feature_measurement_lanes.py +++ b/services/analysis-engine/tests/test_structure_feature_measurement_lanes.py @@ -115,3 +115,46 @@ def fake_stft(**_kwargs: Any) -> np.ndarray[Any, np.dtype[np.float64]]: ) assert calls == ["stft", "stft"] + + +def test_public_segmenter_propagates_one_feature_to_boundary_and_repetition( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """The public paired lane must not split STFT boundaries from CQT repetition labels.""" + audio = np.zeros(160_000, dtype=np.float32) + observed: list[tuple[str, str]] = [] + + def fake_boundaries( + _audio: np.ndarray[Any, np.dtype[np.float32]], + _sr: int, + _duration: float, + *, + chroma_feature: str = "cqt", + ) -> list[float]: + observed.append(("boundary", chroma_feature)) + return [0.0, 10.0] + + def fake_groups( + _audio: np.ndarray[Any, np.dtype[np.float32]], + _sr: int, + _boundaries: list[float], + _duration: float, + *, + chroma_feature: str = "cqt", + ) -> list[int]: + observed.append(("repetition", chroma_feature)) + return [0, 0] + + monkeypatch.setattr(segmenter, "_compute_boundaries", fake_boundaries) + monkeypatch.setattr(segmenter, "_segment_repetition_groups", fake_groups) + + sections, boundaries = segmenter.segment_with_boundaries( + audio, + 8_000, + duration=20.0, + chroma_feature="stft", + ) + + assert observed == [("boundary", "stft"), ("repetition", "stft")] + assert boundaries == [(0.0, 10.0), (10.0, 20.0)] + assert [section["form_label"] for section in sections] == ["chorus", "chorus"] From 3cdf045fc48b64cc8340dc7c7ba71ae44f07c2db Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 17:01:50 +0900 Subject: [PATCH 140/216] test(mir): preregister boundary and repetition metric adapter --- ...t_structure_segmentation_metric_adapter.py | 151 ++++++++++++++++++ 1 file changed, 151 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_segmentation_metric_adapter.py diff --git a/services/analysis-engine/tests/test_structure_segmentation_metric_adapter.py b/services/analysis-engine/tests/test_structure_segmentation_metric_adapter.py new file mode 100644 index 000000000..66b5c7dd4 --- /dev/null +++ b/services/analysis-engine/tests/test_structure_segmentation_metric_adapter.py @@ -0,0 +1,151 @@ +"""Contract for recognized boundary/repetition metrics on normalized structure segments.""" + +from __future__ import annotations + +from fractions import Fraction +from types import ModuleType, SimpleNamespace +from typing import Any + +import pytest +from conftest import load_module + + +def _adapter() -> ModuleType: + """Load the repository-owned structure metric adapter.""" + return load_module( + "scripts/research/evaluate_structure_segmentation_metrics.py", + "evaluate_structure_segmentation_metrics", + ) + + +def _segments(*rows: tuple[str, str, str]) -> tuple[SimpleNamespace, ...]: + """Build protocol-compatible normalized functional segments.""" + return tuple( + SimpleNamespace(start=Fraction(start), end=Fraction(end), label=label) + for start, end, label in rows + ) + + +class _FakeSegmentMetrics: + """Record the exact mir_eval calls made by the adapter.""" + + def __init__(self) -> None: + self.calls: list[tuple[str, dict[str, Any]]] = [] + + def detection( + self, + reference_intervals: Any, + estimated_intervals: Any, + **kwargs: Any, + ) -> tuple[float, float, float]: + self.calls.append(("detection", kwargs)) + window = kwargs["window"] + if window == 0.5: + return (0.8, 0.6, 0.6857142857142857) + return (0.9, 0.75, 0.8181818181818182) + + def deviation( + self, + reference_intervals: Any, + estimated_intervals: Any, + **kwargs: Any, + ) -> tuple[float, float]: + self.calls.append(("deviation", kwargs)) + return (0.12, 0.18) + + def pairwise( + self, + reference_intervals: Any, + reference_labels: Any, + estimated_intervals: Any, + estimated_labels: Any, + **kwargs: Any, + ) -> tuple[float, float, float]: + self.calls.append(("pairwise", kwargs)) + return (0.7, 0.8, 0.7466666666666666) + + +class _FakeMirEval: + """Minimal versioned mir_eval surface used as a deterministic test boundary.""" + + __version__ = "0.8.2" + + def __init__(self) -> None: + self.segment = _FakeSegmentMetrics() + + +def test_adapter_pins_mir_eval_082_and_nontrivial_boundary_semantics( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """Boundary and repetition metrics must use the preregistered exact arguments.""" + module = _adapter() + backend = _FakeMirEval() + monkeypatch.setattr(module, "_load_mir_eval", lambda: backend) + + reference = _segments( + ("0", "10", "intro"), + ("10", "20", "verse"), + ("20", "30", "chorus"), + ) + estimate = _segments( + ("0", "10.2", "intro"), + ("10.2", "19.8", "verse"), + ("19.8", "30", "chorus"), + ) + + result = module.calculate_structure_segmentation_metrics(reference, estimate) + + assert backend.segment.calls == [ + ("detection", {"window": 0.5, "beta": 1.0, "trim": True}), + ("detection", {"window": 3.0, "beta": 1.0, "trim": True}), + ("deviation", {"trim": True}), + ("pairwise", {"frame_size": 0.1, "beta": 1.0}), + ] + assert result.boundary_precision_0_5 == pytest.approx(0.8) + assert result.boundary_recall_0_5 == pytest.approx(0.6) + assert result.boundary_f_0_5 == pytest.approx(0.6857142857142857) + assert result.boundary_precision_3_0 == pytest.approx(0.9) + assert result.boundary_recall_3_0 == pytest.approx(0.75) + assert result.boundary_f_3_0 == pytest.approx(0.8181818181818182) + assert result.reference_to_estimate_median_deviation_seconds == pytest.approx(0.12) + assert result.estimate_to_reference_median_deviation_seconds == pytest.approx(0.18) + assert result.repetition_pairwise_precision == pytest.approx(0.7) + assert result.repetition_pairwise_recall == pytest.approx(0.8) + assert result.repetition_pairwise_f == pytest.approx(0.7466666666666666) + + +def test_adapter_fails_closed_on_unregistered_mir_eval_version( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """A dependency drift must fail before scientific scores are accepted.""" + module = _adapter() + backend = _FakeMirEval() + backend.__version__ = "0.8.3" + monkeypatch.setattr(module, "_load_mir_eval", lambda: backend) + + segments = _segments(("0", "10", "verse"), ("10", "20", "chorus")) + + with pytest.raises(RuntimeError, match="mir_eval 0.8.2"): + module.calculate_structure_segmentation_metrics(segments, segments) + + assert backend.segment.calls == [] + + +def test_adapter_rejects_discontinuous_or_duration_mismatched_segments( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """Metric code must not silently repair malformed scientific segmentations.""" + module = _adapter() + backend = _FakeMirEval() + monkeypatch.setattr(module, "_load_mir_eval", lambda: backend) + + reference = _segments(("0", "10", "verse"), ("10", "20", "chorus")) + gap = _segments(("0", "9", "verse"), ("10", "20", "chorus")) + shorter = _segments(("0", "10", "verse"), ("10", "19", "chorus")) + + with pytest.raises(ValueError, match="continuous"): + module.calculate_structure_segmentation_metrics(reference, gap) + with pytest.raises(ValueError, match="same duration"): + module.calculate_structure_segmentation_metrics(reference, shorter) + + assert backend.segment.calls == [] From e570c414eb8cbc76beec4be7c4424814a09d3694 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 17:02:41 +0900 Subject: [PATCH 141/216] feat(mir): add pinned boundary and repetition metric adapter --- ...evaluate_structure_segmentation_metrics.py | 224 ++++++++++++++++++ 1 file changed, 224 insertions(+) create mode 100644 scripts/research/evaluate_structure_segmentation_metrics.py diff --git a/scripts/research/evaluate_structure_segmentation_metrics.py b/scripts/research/evaluate_structure_segmentation_metrics.py new file mode 100644 index 000000000..986a78db9 --- /dev/null +++ b/scripts/research/evaluate_structure_segmentation_metrics.py @@ -0,0 +1,224 @@ +#!/usr/bin/env python3 +"""Evaluate preregistered structure boundary and repetition metrics. + +This adapter consumes normalized in-memory segment values produced from the +admitted corpus handoff. It deliberately does not read paths, remap labels, or +choose experiment margins, aggregation, or uncertainty. The scientific metric +backend is pinned to mir_eval 0.8.2 and every non-default argument that affects +the registered result is supplied explicitly. + +Security Notes: +- Inputs are normalized segment value objects; no filesystem, network, + subprocess, model-download, or plugin-loading path is exposed. +- Dependency drift fails closed before metric execution. +- Malformed, discontinuous, or duration-mismatched segmentations are rejected + rather than repaired inside the evaluator. +""" + +from __future__ import annotations + +import importlib +import importlib.metadata +import math +from collections.abc import Sequence +from dataclasses import dataclass +from fractions import Fraction +from types import ModuleType +from typing import Any + +import numpy as np + +MIR_EVAL_VERSION = "0.8.2" +BOUNDARY_WINDOWS_SECONDS = (0.5, 3.0) +BOUNDARY_BETA = 1.0 +BOUNDARY_TRIM = True +PAIRWISE_FRAME_SIZE_SECONDS = 0.1 +PAIRWISE_BETA = 1.0 + + +@dataclass(frozen=True, slots=True) +class StructureSegmentationMetrics: + """Track-level recognized boundary, deviation, and repetition measurements.""" + + boundary_precision_0_5: float + boundary_recall_0_5: float + boundary_f_0_5: float + boundary_precision_3_0: float + boundary_recall_3_0: float + boundary_f_3_0: float + reference_to_estimate_median_deviation_seconds: float + estimate_to_reference_median_deviation_seconds: float + repetition_pairwise_precision: float + repetition_pairwise_recall: float + repetition_pairwise_f: float + + +def _load_mir_eval() -> ModuleType: + """Load the reviewed metric package only when scientific evaluation runs.""" + try: + return importlib.import_module("mir_eval") + except ModuleNotFoundError as exc: + raise RuntimeError( + f"mir_eval {MIR_EVAL_VERSION} is required for structure metric evaluation" + ) from exc + + +def _mir_eval_version() -> str: + """Return installed distribution identity instead of trusting module globals.""" + try: + return importlib.metadata.version("mir_eval") + except importlib.metadata.PackageNotFoundError as exc: + raise RuntimeError( + f"mir_eval {MIR_EVAL_VERSION} is required for structure metric evaluation" + ) from exc + + +def _boundary(value: object, field: str) -> Fraction: + """Return one finite exact boundary without bool or string coercion.""" + if isinstance(value, bool) or not isinstance(value, (int, float, Fraction)): + raise ValueError(f"{field} must be a finite numeric boundary") + if isinstance(value, Fraction): + return value + numeric = float(value) + if not math.isfinite(numeric): + raise ValueError(f"{field} must be a finite numeric boundary") + return Fraction(str(numeric)) + + +def _normalize_segments( + raw_segments: object, + field: str, +) -> tuple[np.ndarray[Any, np.dtype[np.float64]], list[str], Fraction]: + """Validate one complete normalized segmentation for recognized metrics.""" + if isinstance(raw_segments, (str, bytes)) or not isinstance(raw_segments, Sequence): + raise ValueError(f"{field} must be a segment sequence") + if len(raw_segments) < 2: + raise ValueError(f"{field} must contain at least two structural segments") + + intervals: list[tuple[float, float]] = [] + labels: list[str] = [] + previous_end = Fraction(0, 1) + for index, segment in enumerate(raw_segments): + start = _boundary(getattr(segment, "start", None), f"{field}[{index}].start") + end = _boundary(getattr(segment, "end", None), f"{field}[{index}].end") + label = getattr(segment, "label", None) + if not isinstance(label, str) or not label: + raise ValueError(f"{field}[{index}].label must be non-empty text") + if start != previous_end: + raise ValueError(f"{field} must be continuous from zero without gaps or overlaps") + if end <= start: + raise ValueError(f"{field}[{index}] must have positive duration") + intervals.append((float(start), float(end))) + labels.append(label) + previous_end = end + + return np.asarray(intervals, dtype=np.float64), labels, previous_end + + +def _score(value: object, field: str) -> float: + """Return one finite metric score in the closed 0..1 interval.""" + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise RuntimeError(f"{field} returned a non-numeric score") + score = float(value) + if not math.isfinite(score) or not 0.0 <= score <= 1.0: + raise RuntimeError(f"{field} returned a score outside 0..1") + return score + + +def _deviation(value: object, field: str) -> float: + """Return one finite non-negative boundary-deviation measurement.""" + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise RuntimeError(f"{field} returned a non-numeric deviation") + deviation = float(value) + if not math.isfinite(deviation) or deviation < 0.0: + raise RuntimeError(f"{field} returned an invalid deviation") + return deviation + + +def calculate_structure_segmentation_metrics( + reference_segments: object, + estimated_segments: object, +) -> StructureSegmentationMetrics: + """Evaluate one normalized estimate with the frozen mir_eval 0.8.2 contract. + + ``trim=True`` intentionally removes the shared start/end markers from boundary + hit-rate and deviation scoring. Those markers are guaranteed by the normalized + full-duration segment contract and would otherwise add two trivial hits that do + not measure a model's ability to locate internal rehearsal structure. + """ + reference_intervals, reference_labels, reference_duration = _normalize_segments( + reference_segments, + "reference_segments", + ) + estimated_intervals, estimated_labels, estimated_duration = _normalize_segments( + estimated_segments, + "estimated_segments", + ) + if reference_duration != estimated_duration: + raise ValueError("reference and estimated segmentations must cover the same duration") + + backend = _load_mir_eval() + observed_version = _mir_eval_version() + if observed_version != MIR_EVAL_VERSION: + raise RuntimeError( + f"structure metrics require mir_eval {MIR_EVAL_VERSION}; " + f"observed {observed_version}" + ) + + segment = getattr(backend, "segment", None) + if segment is None: + raise RuntimeError("mir_eval segment metric module is unavailable") + + p_05, r_05, f_05 = segment.detection( + reference_intervals, + estimated_intervals, + window=BOUNDARY_WINDOWS_SECONDS[0], + beta=BOUNDARY_BETA, + trim=BOUNDARY_TRIM, + ) + p_3, r_3, f_3 = segment.detection( + reference_intervals, + estimated_intervals, + window=BOUNDARY_WINDOWS_SECONDS[1], + beta=BOUNDARY_BETA, + trim=BOUNDARY_TRIM, + ) + reference_to_estimate, estimate_to_reference = segment.deviation( + reference_intervals, + estimated_intervals, + trim=BOUNDARY_TRIM, + ) + pairwise_precision, pairwise_recall, pairwise_f = segment.pairwise( + reference_intervals, + reference_labels, + estimated_intervals, + estimated_labels, + frame_size=PAIRWISE_FRAME_SIZE_SECONDS, + beta=PAIRWISE_BETA, + ) + + return StructureSegmentationMetrics( + boundary_precision_0_5=_score(p_05, "boundary precision at 0.5 s"), + boundary_recall_0_5=_score(r_05, "boundary recall at 0.5 s"), + boundary_f_0_5=_score(f_05, "boundary F at 0.5 s"), + boundary_precision_3_0=_score(p_3, "boundary precision at 3.0 s"), + boundary_recall_3_0=_score(r_3, "boundary recall at 3.0 s"), + boundary_f_3_0=_score(f_3, "boundary F at 3.0 s"), + reference_to_estimate_median_deviation_seconds=_deviation( + reference_to_estimate, + "reference-to-estimate boundary deviation", + ), + estimate_to_reference_median_deviation_seconds=_deviation( + estimate_to_reference, + "estimate-to-reference boundary deviation", + ), + repetition_pairwise_precision=_score( + pairwise_precision, + "repetition pairwise precision", + ), + repetition_pairwise_recall=_score( + pairwise_recall, + "repetition pairwise recall", + ), + repetition_pairwise_f=_score(pairwise_f, "repetition pairwise F"), + ) From 93df1d4b1b617bdfa1d2f62d720656d10a67fa97 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 17:03:20 +0900 Subject: [PATCH 142/216] test(mir): isolate metric backend version identity --- .../tests/test_structure_segmentation_metric_adapter.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_segmentation_metric_adapter.py b/services/analysis-engine/tests/test_structure_segmentation_metric_adapter.py index 66b5c7dd4..2dfa027a9 100644 --- a/services/analysis-engine/tests/test_structure_segmentation_metric_adapter.py +++ b/services/analysis-engine/tests/test_structure_segmentation_metric_adapter.py @@ -66,9 +66,7 @@ def pairwise( class _FakeMirEval: - """Minimal versioned mir_eval surface used as a deterministic test boundary.""" - - __version__ = "0.8.2" + """Minimal mir_eval surface used as a deterministic test boundary.""" def __init__(self) -> None: self.segment = _FakeSegmentMetrics() @@ -81,6 +79,7 @@ def test_adapter_pins_mir_eval_082_and_nontrivial_boundary_semantics( module = _adapter() backend = _FakeMirEval() monkeypatch.setattr(module, "_load_mir_eval", lambda: backend) + monkeypatch.setattr(module, "_mir_eval_version", lambda: "0.8.2") reference = _segments( ("0", "10", "intro"), @@ -120,8 +119,8 @@ def test_adapter_fails_closed_on_unregistered_mir_eval_version( """A dependency drift must fail before scientific scores are accepted.""" module = _adapter() backend = _FakeMirEval() - backend.__version__ = "0.8.3" monkeypatch.setattr(module, "_load_mir_eval", lambda: backend) + monkeypatch.setattr(module, "_mir_eval_version", lambda: "0.8.3") segments = _segments(("0", "10", "verse"), ("10", "20", "chorus")) @@ -138,6 +137,7 @@ def test_adapter_rejects_discontinuous_or_duration_mismatched_segments( module = _adapter() backend = _FakeMirEval() monkeypatch.setattr(module, "_load_mir_eval", lambda: backend) + monkeypatch.setattr(module, "_mir_eval_version", lambda: "0.8.2") reference = _segments(("0", "10", "verse"), ("10", "20", "chorus")) gap = _segments(("0", "9", "verse"), ("10", "20", "chorus")) From e4729c3eebf14ee7ec23270bc09a0837108a6caa Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 17:04:23 +0900 Subject: [PATCH 143/216] docs(mir): trace boundary and repetition metric authority --- .../structure-segmentation-metric-adapter.md | 67 +++++++++++++++++++ 1 file changed, 67 insertions(+) create mode 100644 docs/traceability/mir/structure-segmentation-metric-adapter.md diff --git a/docs/traceability/mir/structure-segmentation-metric-adapter.md b/docs/traceability/mir/structure-segmentation-metric-adapter.md new file mode 100644 index 000000000..c960b7c44 --- /dev/null +++ b/docs/traceability/mir/structure-segmentation-metric-adapter.md @@ -0,0 +1,67 @@ +# Structure segmentation metric adapter + +Status: Proposed +Owner: Signal-MIR Analysis +Tracking: #1225, #1228 +Parent: `docs/traceability/mir/structure-feature-noninferiority.md` + +## Problem + +The noninferiority registration names `mir_eval.segment.detection`, `mir_eval.segment.deviation`, and `mir_eval.segment.pairwise`, but a function name alone is not a reproducible scientific implementation identity. Boundary detection changes materially with `window`, `beta`, and especially `trim`; pairwise grouping changes with `frame_size` and `beta`. Leaving those arguments at library defaults would let a future environment or caller alter the dependent variable without changing the registration digest. + +There is also an authority distinction that must not be blurred. The official `ismir-mirex/mirex-evaluation` MIREX-2025 reproduction at commit `b9fa0b0b32e2145af31f35830f78fc9d09a4301b` implements ACC and HR.5/HR3 itself. Its hit-rate function includes start/end boundaries and is not the same implementation as `mir_eval.segment.detection`, which uses one-to-one boundary matching and exposes an explicit `trim` option. BandScope already chose `mir_eval` for the recognized boundary/deviation/repetition measures; therefore its scores must not be described as numerically identical to the official MIREX HR.5/HR3 table merely because the tolerance names are the same. + +## Decision + +`scripts/research/evaluate_structure_segmentation_metrics.py` is the repository-owned adapter for the currently selected recognized metrics. It requires installed distribution identity `mir_eval==0.8.2` and calls the public API with every result-affecting argument explicit: + +- boundary precision/recall/F at 0.5 s: `mir_eval.segment.detection(window=0.5, beta=1.0, trim=True)`; +- boundary precision/recall/F at 3.0 s: `mir_eval.segment.detection(window=3.0, beta=1.0, trim=True)`; +- bidirectional median boundary deviation: `mir_eval.segment.deviation(trim=True)`; +- repetition/grouping precision/recall/F: `mir_eval.segment.pairwise(frame_size=0.1, beta=1.0)`. + +`trim=True` is deliberate. The normalized BandScope segment contract already guarantees a boundary at 0 and exact full-track coverage. Counting those shared endpoints as boundary hits would reward two invariants supplied by the adapter contract rather than the model's ability to locate internal rehearsal structure. `mir_eval` documents those endpoints as the typical reason to enable trimming. + +The adapter validates both segmentations as continuous from zero with positive segment durations and requires exactly the same total duration. It does not repair gaps, overlaps, duration drift, or malformed labels. It returns only finite bounded scores and non-negative deviations. + +## RED → GREEN + +RED `3cdf045fc48b64cc8340dc7c7ba71ae44f07c2db` added a focused contract before the adapter existed. The regression requires exact 0.5 s/3.0 s detection calls with `beta=1.0` and `trim=True`, trimmed deviation, 100 ms pairwise grouping, dependency-version fail-closed behavior, and rejection of malformed segmentation geometry. + +GREEN `e570c414eb8cbc76beec4be7c4424814a09d3694` added the metric adapter. Follow-up `93df1d4b1b617bdfa1d2f62d720656d10a67fa97` isolated installed-distribution version evidence from the fake test backend so unit tests do not pretend that a locally injected test double is package provenance. + +These are source-level TDD commits. They are not hosted scientific acceptance and do not show a rights-cleared corpus result. + +## Constraints and rejected alternatives + +Using the official MIREX 2025 `calculate_hit_rate` while continuing to label the registration as `mir_eval.segment.detection` was rejected. They have different matching semantics and would make the registration false. + +Leaving `trim=False` at the mir_eval default was rejected for the BandScope noninferiority decision because the guaranteed start/end markers would dilute the internal-boundary signal, especially on tracks with few sections. + +Reimplementing mir_eval formulas in BandScope was rejected. The adapter owns argument selection, validation, and evidence normalization, not a fork of the recognized metric library. + +Adding an unpinned `mir_eval>=...` dependency was rejected. Scientific execution must bind the exact reviewed version rather than accepting a later package with the same API surface. + +## Current limitation and next causal step + +The main registration validator still records generic `mir_eval.segment.*` implementation strings and the analysis-engine lock does not yet carry an exact `mir_eval==0.8.2` research-runtime dependency. Therefore the new adapter is **not yet admissible for the rights-cleared experiment**. Before any candidate real-audio result is inspected, the canonical registration schema and runtime identity must bind: + +- mir_eval version 0.8.2; +- detection `beta=1.0` and `trim=True` for both registered windows; +- deviation implementation plus `trim=True`; +- pairwise `frame_size=0.1` and `beta=1.0`; +- the exact locked dependency identity used by the experiment runtime. + +After that schema/runtime repair, this adapter can be wired into the same admitted-track CQT/STFT consumer that already produces functional ACC. Aggregation and paired uncertainty remain a later preregistered decision and must still be frozen before candidate outcomes are inspected. + +## Security Notes + +The adapter receives normalized in-memory segments only. It adds no filesystem, network, subprocess, model-download, credential, or generic plugin path. The private test seam replaces the module loader only inside focused tests; production execution resolves the installed `mir_eval` distribution and rejects any version other than 0.8.2 before metric calls. + +## References + +Raffel, C., McFee, B., Humphrey, E. J., Salamon, J., Nieto, O., Liang, D., & Ellis, D. P. W. (2014). *mir_eval: A transparent implementation of common MIR metrics*. Proceedings of the 15th International Society for Music Information Retrieval Conference. + +mir-evaluation contributors. (2026). *mir_eval 0.8.2: segment evaluation API*. https://mir-eval.readthedocs.io/latest/api/segment.html + +MIREX Evaluation contributors. (2026). *Music Structure Analysis evaluation script* (commit `b9fa0b0b32e2145af31f35830f78fc9d09a4301b`). https://github.com/ismir-mirex/mirex-evaluation/blob/b9fa0b0b32e2145af31f35830f78fc9d09a4301b/music_structure_analysis/eval_script.py From 37f3549d15acb33f397245a88da16dcd8e274cf7 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 17:24:31 +0900 Subject: [PATCH 144/216] test(mir): require exact structure metric runtime lock --- .../test_structure_metric_runtime_lock.py | 66 +++++++++++++++++++ 1 file changed, 66 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_metric_runtime_lock.py diff --git a/services/analysis-engine/tests/test_structure_metric_runtime_lock.py b/services/analysis-engine/tests/test_structure_metric_runtime_lock.py new file mode 100644 index 000000000..38846df46 --- /dev/null +++ b/services/analysis-engine/tests/test_structure_metric_runtime_lock.py @@ -0,0 +1,66 @@ +"""Contract tests for the structure-metric research runtime lock.""" + +from __future__ import annotations + +from pathlib import Path +from types import ModuleType + +import pytest +from conftest import load_module + +_REPOSITORY_ROOT = Path(__file__).resolve().parents[3] +_LOCK_PATH = _REPOSITORY_ROOT / "services/analysis-engine/requirements-structure-metrics.lock" +_EXPECTED_VERSION = "0.8.2" +_EXPECTED_WHEEL_SHA256 = "114cda33d8e17408c170598e0b36ed0d71ff4a2fee8eaf9e165b58ecf1c87170" +_EXPECTED_SOURCE_COMMIT = "8db0b3812e2032544c1fc00d02d4256cab043f3d" +_EXPECTED_TRANSPARENCY_ENTRY = 174236906 + + +def _verifier() -> ModuleType: + """Load the repository-owned structure metric runtime verifier.""" + return load_module( + "scripts/research/verify_structure_metric_runtime_lock.py", + "verify_structure_metric_runtime_lock", + ) + + +def test_runtime_lock_binds_exact_trusted_pypi_wheel(monkeypatch: pytest.MonkeyPatch) -> None: + """Research execution must resolve one reviewed mir_eval artifact identity.""" + verifier = _verifier() + monkeypatch.setattr(verifier, "_installed_mir_eval_version", lambda: _EXPECTED_VERSION) + + identity = verifier.verify_structure_metric_runtime_lock(_LOCK_PATH) + + assert identity.version == _EXPECTED_VERSION + assert identity.wheel_sha256 == _EXPECTED_WHEEL_SHA256 + assert identity.source_commit == _EXPECTED_SOURCE_COMMIT + assert identity.pypi_transparency_entry == _EXPECTED_TRANSPARENCY_ENTRY + assert len(identity.lock_sha256) == 64 + + +def test_runtime_lock_rejects_installed_distribution_drift( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """A same-API but different mir_eval version is not an admissible runtime.""" + verifier = _verifier() + monkeypatch.setattr(verifier, "_installed_mir_eval_version", lambda: "0.8.1") + + with pytest.raises(RuntimeError, match="requires mir_eval 0.8.2"): + verifier.verify_structure_metric_runtime_lock(_LOCK_PATH) + + +def test_runtime_lock_rejects_artifact_drift( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + """Changing the wheel hash must invalidate the scientific runtime identity.""" + verifier = _verifier() + monkeypatch.setattr(verifier, "_installed_mir_eval_version", lambda: _EXPECTED_VERSION) + drifted = tmp_path / "requirements-structure-metrics.lock" + drifted.write_text( + _LOCK_PATH.read_text(encoding="utf-8").replace(_EXPECTED_WHEEL_SHA256, "0" * 64), + encoding="utf-8", + ) + + with pytest.raises(ValueError, match="wheel identity"): + verifier.verify_structure_metric_runtime_lock(drifted) From 8b495913bd6b7fe66f2855afb6fc5d095ae8433d Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 17:24:59 +0900 Subject: [PATCH 145/216] feat(mir): verify exact research metric runtime artifact --- .../verify_structure_metric_runtime_lock.py | 96 +++++++++++++++++++ 1 file changed, 96 insertions(+) create mode 100644 scripts/research/verify_structure_metric_runtime_lock.py diff --git a/scripts/research/verify_structure_metric_runtime_lock.py b/scripts/research/verify_structure_metric_runtime_lock.py new file mode 100644 index 000000000..86f10256a --- /dev/null +++ b/scripts/research/verify_structure_metric_runtime_lock.py @@ -0,0 +1,96 @@ +#!/usr/bin/env python3 +"""Verify the exact research-only mir_eval runtime used by structure evidence. + +The production analysis environment remains governed by the analysis-engine +``uv.lock``. Structure noninferiority evaluation adds one reviewed, pure-Python +metric wheel as a no-dependency research overlay so scientific evaluation does +not widen production dependencies. This verifier binds that overlay to the +exact PyPI artifact and the installed distribution version before any metric is +computed. + +Security Notes: +- The verifier performs no network access and never installs packages. +- The lock file is size-bounded and must match the reviewed direct-wheel + requirement exactly; mutable indexes, ranges, and alternate artifacts fail + closed. +""" + +from __future__ import annotations + +import hashlib +import importlib.metadata +from dataclasses import dataclass +from pathlib import Path + +MIR_EVAL_VERSION = "0.8.2" +MIR_EVAL_WHEEL_SHA256 = "114cda33d8e17408c170598e0b36ed0d71ff4a2fee8eaf9e165b58ecf1c87170" +MIR_EVAL_SOURCE_COMMIT = "8db0b3812e2032544c1fc00d02d4256cab043f3d" +MIR_EVAL_PYPI_TRANSPARENCY_ENTRY = 174236906 +MIR_EVAL_WHEEL_URL = ( + "https://files.pythonhosted.org/packages/1b/5a/" + "69ce896a32ebc8c75deae00b1fba9837567405fa6ef37b377f2e85b856ae/" + "mir_eval-0.8.2-py3-none-any.whl" +) +MAX_LOCK_BYTES = 16 * 1024 +EXPECTED_REQUIREMENT = ( + f"mir-eval @ {MIR_EVAL_WHEEL_URL}#sha256={MIR_EVAL_WHEEL_SHA256}" +) +EXPECTED_LOCK_TEXT = ( + "# BandScope structure noninferiority research-metric overlay.\n" + "# Install only after the frozen analysis-engine environment is synced, using --no-deps.\n" + f"# Upstream source: mir-evaluation/mir_eval@{MIR_EVAL_SOURCE_COMMIT}\n" + f"# PyPI Sigstore transparency entry: {MIR_EVAL_PYPI_TRANSPARENCY_ENTRY}\n" + f"{EXPECTED_REQUIREMENT}\n" +) + + +@dataclass(frozen=True, slots=True) +class StructureMetricRuntimeIdentity: + """Content-addressed identity for the reviewed structure metric overlay.""" + + version: str + wheel_sha256: str + source_commit: str + pypi_transparency_entry: int + lock_sha256: str + + +def _installed_mir_eval_version() -> str: + """Return the installed distribution version from package metadata.""" + try: + return importlib.metadata.version("mir_eval") + except importlib.metadata.PackageNotFoundError as exc: + raise RuntimeError(f"structure metrics require mir_eval {MIR_EVAL_VERSION}") from exc + + +def verify_structure_metric_runtime_lock( + lock_path: Path, +) -> StructureMetricRuntimeIdentity: + """Validate the reviewed wheel lock and installed mir_eval distribution.""" + raw = lock_path.read_bytes() + if len(raw) > MAX_LOCK_BYTES: + raise ValueError(f"structure metric runtime lock exceeds {MAX_LOCK_BYTES} bytes") + try: + text = raw.decode("utf-8") + except UnicodeDecodeError as exc: + raise ValueError("structure metric runtime lock must be UTF-8 text") from exc + + if MIR_EVAL_WHEEL_SHA256 not in text: + raise ValueError("structure metric wheel identity does not match the reviewed artifact") + if text != EXPECTED_LOCK_TEXT: + raise ValueError("structure metric runtime lock differs from the reviewed exact contract") + + observed_version = _installed_mir_eval_version() + if observed_version != MIR_EVAL_VERSION: + raise RuntimeError( + f"structure metrics require mir_eval {MIR_EVAL_VERSION}; " + f"observed {observed_version}" + ) + + return StructureMetricRuntimeIdentity( + version=MIR_EVAL_VERSION, + wheel_sha256=MIR_EVAL_WHEEL_SHA256, + source_commit=MIR_EVAL_SOURCE_COMMIT, + pypi_transparency_entry=MIR_EVAL_PYPI_TRANSPARENCY_ENTRY, + lock_sha256=hashlib.sha256(raw).hexdigest(), + ) From 23388320e3d536c3d220546c4a857389c83182f9 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 17:25:06 +0900 Subject: [PATCH 146/216] feat(mir): lock reviewed mir_eval research wheel --- services/analysis-engine/requirements-structure-metrics.lock | 5 +++++ 1 file changed, 5 insertions(+) create mode 100644 services/analysis-engine/requirements-structure-metrics.lock diff --git a/services/analysis-engine/requirements-structure-metrics.lock b/services/analysis-engine/requirements-structure-metrics.lock new file mode 100644 index 000000000..b9eae414a --- /dev/null +++ b/services/analysis-engine/requirements-structure-metrics.lock @@ -0,0 +1,5 @@ +# BandScope structure noninferiority research-metric overlay. +# Install only after the frozen analysis-engine environment is synced, using --no-deps. +# Upstream source: mir-evaluation/mir_eval@8db0b3812e2032544c1fc00d02d4256cab043f3d +# PyPI Sigstore transparency entry: 174236906 +mir-eval @ https://files.pythonhosted.org/packages/1b/5a/69ce896a32ebc8c75deae00b1fba9837567405fa6ef37b377f2e85b856ae/mir_eval-0.8.2-py3-none-any.whl#sha256=114cda33d8e17408c170598e0b36ed0d71ff4a2fee8eaf9e165b58ecf1c87170 From fbb7c1d7f605aabcc36d4bc651b9b4c6c1b39f0d Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 17:29:30 +0900 Subject: [PATCH 147/216] test(mir): require admitted-track segmentation metric evidence --- ...ure_admitted_track_segmentation_metrics.py | 136 ++++++++++++++++++ 1 file changed, 136 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_admitted_track_segmentation_metrics.py diff --git a/services/analysis-engine/tests/test_structure_admitted_track_segmentation_metrics.py b/services/analysis-engine/tests/test_structure_admitted_track_segmentation_metrics.py new file mode 100644 index 000000000..16ef75aee --- /dev/null +++ b/services/analysis-engine/tests/test_structure_admitted_track_segmentation_metrics.py @@ -0,0 +1,136 @@ +"""Contract for recognized segmentation metrics on the admitted-track handoff.""" + +from __future__ import annotations + +import hashlib +import struct +from fractions import Fraction +from pathlib import Path +from types import ModuleType, SimpleNamespace +from typing import Any + +import pytest +from conftest import load_module + +_REPOSITORY_ROOT = Path(__file__).resolve().parents[3] +_RUNTIME_LOCK = _REPOSITORY_ROOT / "services/analysis-engine/requirements-structure-metrics.lock" + + +def _consumer_module() -> ModuleType: + """Load the repository-owned admitted-track scientific consumer.""" + return load_module( + "scripts/research/evaluate_admitted_structure_track.py", + "evaluate_admitted_structure_track_with_metrics", + ) + + +def _segments(*rows: tuple[str, str, str]) -> tuple[SimpleNamespace, ...]: + """Build protocol-compatible normalized structural segments.""" + return tuple( + SimpleNamespace(start=Fraction(start), end=Fraction(end), label=label) + for start, end, label in rows + ) + + +def _metric_result(seed: float) -> SimpleNamespace: + """Return a complete deterministic recognized-metric result fixture.""" + return SimpleNamespace( + boundary_precision_0_5=seed, + boundary_recall_0_5=seed, + boundary_f_0_5=seed, + boundary_precision_3_0=seed + 0.01, + boundary_recall_3_0=seed + 0.01, + boundary_f_3_0=seed + 0.01, + reference_to_estimate_median_deviation_seconds=0.12, + estimate_to_reference_median_deviation_seconds=0.15, + repetition_pairwise_precision=seed + 0.02, + repetition_pairwise_recall=seed + 0.02, + repetition_pairwise_f=seed + 0.02, + ) + + +def test_registered_consumer_binds_runtime_lock_and_scores_both_feature_lanes( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """Canonical CQT/STFT evidence must share PCM, annotation, and metric runtime identity.""" + module = _consumer_module() + requested_features: list[str] = [] + metric_calls: list[tuple[object, object]] = [] + + def repository_segmenter(feature: str) -> Any: + requested_features.append(feature) + + def segmenter( + _decoded_pcm: memoryview, + _sample_rate_hz: int, + _duration_seconds: Fraction, + ) -> tuple[SimpleNamespace, ...]: + if feature == "cqt": + return _segments(("0", "0.4", "verse"), ("0.4", "0.8", "chorus")) + return _segments(("0", "0.2", "verse"), ("0.2", "0.8", "chorus")) + + return segmenter + + def metric_evaluator(reference_segments: object, estimated_segments: object) -> SimpleNamespace: + metric_calls.append((reference_segments, estimated_segments)) + return _metric_result(0.70 if len(metric_calls) == 1 else 0.68) + + monkeypatch.setattr(module._LANES, "repository_structure_segmenter", repository_segmenter) + monkeypatch.setattr( + module._SEGMENTATION_EVALUATOR, + "calculate_structure_segmentation_metrics", + metric_evaluator, + ) + monkeypatch.setattr(module._RUNTIME_VERIFIER, "_installed_mir_eval_version", lambda: "0.8.2") + + consumer = module.PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis() + pcm = memoryview(struct.pack("<8f", *([0.0] * 8))) + annotation = memoryview(b"0.0\t0.4\tverse\n0.4\t0.8\tchorus\n") + consumer("track-01", pcm, annotation, 10) + + assert requested_features == ["cqt", "stft"] + assert len(metric_calls) == 2 + assert metric_calls[0][0] is metric_calls[1][0] + evidence = consumer.evidence[0] + assert evidence.metric_runtime_lock_sha256 == hashlib.sha256(_RUNTIME_LOCK.read_bytes()).hexdigest() + assert evidence.baseline_segmentation_metrics.boundary_f_0_5 == pytest.approx(0.70) + assert evidence.candidate_segmentation_metrics.boundary_f_0_5 == pytest.approx(0.68) + assert evidence.baseline_segmentation_metrics.repetition_pairwise_f == pytest.approx(0.72) + assert evidence.candidate_segmentation_metrics.repetition_pairwise_f == pytest.approx(0.70) + + +def test_registered_consumer_fails_before_metric_execution_on_runtime_drift( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """A drifted installed mir_eval must stop scientific metrics before any score is emitted.""" + module = _consumer_module() + calls: list[str] = [] + + def segmenter( + _decoded_pcm: memoryview, + _sample_rate_hz: int, + _duration_seconds: Fraction, + ) -> tuple[SimpleNamespace, ...]: + return _segments(("0", "0.4", "verse")) + + monkeypatch.setattr(module._LANES, "repository_structure_segmenter", lambda _feature: segmenter) + monkeypatch.setattr(module._RUNTIME_VERIFIER, "_installed_mir_eval_version", lambda: "0.8.1") + + def must_not_run(*_args: object) -> object: + calls.append("metric") + raise AssertionError("metric evaluator must not run after runtime drift") + + monkeypatch.setattr( + module._SEGMENTATION_EVALUATOR, + "calculate_structure_segmentation_metrics", + must_not_run, + ) + consumer = module.PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis() + pcm = memoryview(struct.pack("<4f", *([0.0] * 4))) + annotation = memoryview(b"0.0\t0.4\tverse\n") + + with pytest.raises(RuntimeError, match="requires mir_eval 0.8.2"): + consumer("track-01", pcm, annotation, 10) + + assert calls == [] + assert consumer.evidence == () From 59c271f64f2fac893025f4a9cd0d3741fff25865 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 17:29:46 +0900 Subject: [PATCH 148/216] refactor(mir): separate lock identity from installed runtime check --- .../verify_structure_metric_runtime_lock.py | 25 ++++++++++++------- 1 file changed, 16 insertions(+), 9 deletions(-) diff --git a/scripts/research/verify_structure_metric_runtime_lock.py b/scripts/research/verify_structure_metric_runtime_lock.py index 86f10256a..1a0425770 100644 --- a/scripts/research/verify_structure_metric_runtime_lock.py +++ b/scripts/research/verify_structure_metric_runtime_lock.py @@ -63,10 +63,10 @@ def _installed_mir_eval_version() -> str: raise RuntimeError(f"structure metrics require mir_eval {MIR_EVAL_VERSION}") from exc -def verify_structure_metric_runtime_lock( +def load_structure_metric_runtime_lock_identity( lock_path: Path, ) -> StructureMetricRuntimeIdentity: - """Validate the reviewed wheel lock and installed mir_eval distribution.""" + """Validate the immutable lock artifact without consulting the local environment.""" raw = lock_path.read_bytes() if len(raw) > MAX_LOCK_BYTES: raise ValueError(f"structure metric runtime lock exceeds {MAX_LOCK_BYTES} bytes") @@ -80,13 +80,6 @@ def verify_structure_metric_runtime_lock( if text != EXPECTED_LOCK_TEXT: raise ValueError("structure metric runtime lock differs from the reviewed exact contract") - observed_version = _installed_mir_eval_version() - if observed_version != MIR_EVAL_VERSION: - raise RuntimeError( - f"structure metrics require mir_eval {MIR_EVAL_VERSION}; " - f"observed {observed_version}" - ) - return StructureMetricRuntimeIdentity( version=MIR_EVAL_VERSION, wheel_sha256=MIR_EVAL_WHEEL_SHA256, @@ -94,3 +87,17 @@ def verify_structure_metric_runtime_lock( pypi_transparency_entry=MIR_EVAL_PYPI_TRANSPARENCY_ENTRY, lock_sha256=hashlib.sha256(raw).hexdigest(), ) + + +def verify_structure_metric_runtime_lock( + lock_path: Path, +) -> StructureMetricRuntimeIdentity: + """Validate the reviewed lock artifact and the installed mir_eval distribution.""" + identity = load_structure_metric_runtime_lock_identity(lock_path) + observed_version = _installed_mir_eval_version() + if observed_version != MIR_EVAL_VERSION: + raise RuntimeError( + f"structure metrics require mir_eval {MIR_EVAL_VERSION}; " + f"observed {observed_version}" + ) + return identity From 773ffd2913f933726b8bf9b890d955189e0e8c5d Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 17:30:26 +0900 Subject: [PATCH 149/216] feat(mir): emit recognized metrics on admitted track evidence --- .../evaluate_admitted_structure_track.py | 117 ++++++++++++++++-- 1 file changed, 110 insertions(+), 7 deletions(-) diff --git a/scripts/research/evaluate_admitted_structure_track.py b/scripts/research/evaluate_admitted_structure_track.py index 1ef140170..61dda5fbb 100644 --- a/scripts/research/evaluate_admitted_structure_track.py +++ b/scripts/research/evaluate_admitted_structure_track.py @@ -1,12 +1,14 @@ #!/usr/bin/env python3 -"""Score paired functional ACC on the exact corpus-admission handoff. +"""Score paired structure evidence on the exact corpus-admission handoff. This module bridges resource admission to Signal-MIR measurement without reopening workstation paths. The baseline and candidate segmenters receive the same immutable canonical PCM memoryview, while the reference segmentation is -parsed from the same immutable admitted annotation snapshot. Scientific corpus -choice, margins, aggregation, uncertainty, and the production feature switch -remain outside this boundary. +parsed from the same immutable admitted annotation snapshot. The canonical +CQT/STFT experiment additionally evaluates the reviewed mir_eval boundary, +deviation, and repetition contract under the exact research runtime lock. +Scientific corpus choice, margins, aggregation, uncertainty, and the production +feature switch remain outside this boundary. """ from __future__ import annotations @@ -22,6 +24,11 @@ from typing import Any Segmenter = Callable[[memoryview, int, Fraction], Sequence[Any]] +SegmentationMetricEvaluator = Callable[[object, object], object] +_REPOSITORY_ROOT = Path(__file__).resolve().parents[2] +_STRUCTURE_METRIC_RUNTIME_LOCK = ( + _REPOSITORY_ROOT / "services/analysis-engine/requirements-structure-metrics.lock" +) def _load_sibling(filename: str, module_name: str) -> ModuleType: @@ -56,6 +63,31 @@ def _load_sibling(filename: str, module_name: str) -> ModuleType: "measure_structure_feature_lanes.py", "_bandscope_structure_feature_lanes", ) +_SEGMENTATION_EVALUATOR = _load_sibling( + "evaluate_structure_segmentation_metrics.py", + "_bandscope_structure_segmentation_metrics", +) +_RUNTIME_VERIFIER = _load_sibling( + "verify_structure_metric_runtime_lock.py", + "_bandscope_structure_metric_runtime_lock", +) + + +@dataclass(frozen=True, slots=True) +class StructureSegmentationMetricEvidence: + """Recognized track-level boundary, deviation, and repetition evidence.""" + + boundary_precision_0_5: float + boundary_recall_0_5: float + boundary_f_0_5: float + boundary_precision_3_0: float + boundary_recall_3_0: float + boundary_f_3_0: float + reference_to_estimate_median_deviation_seconds: float + estimate_to_reference_median_deviation_seconds: float + repetition_pairwise_precision: float + repetition_pairwise_recall: float + repetition_pairwise_f: float @dataclass(frozen=True, slots=True) @@ -73,6 +105,9 @@ class PairedFunctionalAccuracyEvidence: candidate_accuracy: float candidate_correct_frames: int candidate_total_frames: int + metric_runtime_lock_sha256: str | None + baseline_segmentation_metrics: StructureSegmentationMetricEvidence | None + candidate_segmentation_metrics: StructureSegmentationMetricEvidence | None def _track_id(value: object) -> str: @@ -109,29 +144,69 @@ def _annotation_view(annotation_bytes: object) -> memoryview: return annotation_bytes +def _segmentation_metric_evidence(result: object) -> StructureSegmentationMetricEvidence: + """Copy the reviewed adapter result into the consumer-owned immutable receipt.""" + return StructureSegmentationMetricEvidence( + boundary_precision_0_5=float(getattr(result, "boundary_precision_0_5")), + boundary_recall_0_5=float(getattr(result, "boundary_recall_0_5")), + boundary_f_0_5=float(getattr(result, "boundary_f_0_5")), + boundary_precision_3_0=float(getattr(result, "boundary_precision_3_0")), + boundary_recall_3_0=float(getattr(result, "boundary_recall_3_0")), + boundary_f_3_0=float(getattr(result, "boundary_f_3_0")), + reference_to_estimate_median_deviation_seconds=float( + getattr(result, "reference_to_estimate_median_deviation_seconds") + ), + estimate_to_reference_median_deviation_seconds=float( + getattr(result, "estimate_to_reference_median_deviation_seconds") + ), + repetition_pairwise_precision=float(getattr(result, "repetition_pairwise_precision")), + repetition_pairwise_recall=float(getattr(result, "repetition_pairwise_recall")), + repetition_pairwise_f=float(getattr(result, "repetition_pairwise_f")), + ) + + class PairedFunctionalAccuracyTrackConsumer: - """Collect paired functional ACC evidence from corpus-admission callbacks.""" + """Collect paired structure evidence from corpus-admission callbacks.""" def __init__( self, *, baseline_segmenter: Segmenter, candidate_segmenter: Segmenter, + segmentation_metric_evaluator: SegmentationMetricEvaluator | None = None, + metric_runtime_lock_sha256: str | None = None, ) -> None: - """Bind the two preregistered segmentation lanes without track-specific dispatch.""" + """Bind preregistered lanes and optional recognized-metric runtime identity.""" if not callable(baseline_segmenter) or not callable(candidate_segmenter): raise TypeError("baseline_segmenter and candidate_segmenter must be callable") + if (segmentation_metric_evaluator is None) != (metric_runtime_lock_sha256 is None): + raise ValueError( + "segmentation metric evaluator and metric runtime lock identity must be bound together" + ) + if segmentation_metric_evaluator is not None and not callable( + segmentation_metric_evaluator + ): + raise TypeError("segmentation_metric_evaluator must be callable") self._baseline_segmenter = baseline_segmenter self._candidate_segmenter = candidate_segmenter + self._segmentation_metric_evaluator = segmentation_metric_evaluator + self._metric_runtime_lock_sha256 = metric_runtime_lock_sha256 self._evidence: list[PairedFunctionalAccuracyEvidence] = [] self._measured_track_ids: set[str] = set() @classmethod def for_registered_cqt_stft_hypothesis(cls) -> PairedFunctionalAccuracyTrackConsumer: - """Bind the canonical #1225 baseline/candidate feature identities.""" + """Bind canonical CQT/STFT lanes and the reviewed segmentation-metric overlay.""" + runtime_identity = _RUNTIME_VERIFIER.load_structure_metric_runtime_lock_identity( + _STRUCTURE_METRIC_RUNTIME_LOCK + ) return cls( baseline_segmenter=_LANES.repository_structure_segmenter("cqt"), candidate_segmenter=_LANES.repository_structure_segmenter("stft"), + segmentation_metric_evaluator=( + _SEGMENTATION_EVALUATOR.calculate_structure_segmentation_metrics + ), + metric_runtime_lock_sha256=runtime_identity.lock_sha256, ) @property @@ -178,6 +253,31 @@ def __call__( duration_seconds=duration_seconds, ) + baseline_segmentation_metrics: StructureSegmentationMetricEvidence | None = None + candidate_segmentation_metrics: StructureSegmentationMetricEvidence | None = None + metric_runtime_lock_sha256: str | None = None + if self._segmentation_metric_evaluator is not None: + runtime_identity = _RUNTIME_VERIFIER.verify_structure_metric_runtime_lock( + _STRUCTURE_METRIC_RUNTIME_LOCK + ) + if runtime_identity.lock_sha256 != self._metric_runtime_lock_sha256: + raise RuntimeError( + "structure metric runtime lock changed after experiment binding" + ) + baseline_segmentation_metrics = _segmentation_metric_evidence( + self._segmentation_metric_evaluator( + reference_segments, + baseline_segments, + ) + ) + candidate_segmentation_metrics = _segmentation_metric_evidence( + self._segmentation_metric_evaluator( + reference_segments, + candidate_segments, + ) + ) + metric_runtime_lock_sha256 = runtime_identity.lock_sha256 + evidence = PairedFunctionalAccuracyEvidence( track_id=normalized_track_id, decoded_pcm_sha256=hashlib.sha256(pcm).hexdigest(), @@ -190,6 +290,9 @@ def __call__( candidate_accuracy=float(candidate_result.accuracy), candidate_correct_frames=int(candidate_result.correct_frames), candidate_total_frames=int(candidate_result.total_frames), + metric_runtime_lock_sha256=metric_runtime_lock_sha256, + baseline_segmentation_metrics=baseline_segmentation_metrics, + candidate_segmentation_metrics=candidate_segmentation_metrics, ) self._evidence.append(evidence) self._measured_track_ids.add(normalized_track_id) From 616a26eccf7e911ed2912b1a905677984fd00119 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 17:33:06 +0900 Subject: [PATCH 150/216] docs(mir): bind segmentation metric runtime and admitted-track evidence --- .../structure-segmentation-metric-adapter.md | 67 ++++++++++++------- 1 file changed, 44 insertions(+), 23 deletions(-) diff --git a/docs/traceability/mir/structure-segmentation-metric-adapter.md b/docs/traceability/mir/structure-segmentation-metric-adapter.md index c960b7c44..37bc9292c 100644 --- a/docs/traceability/mir/structure-segmentation-metric-adapter.md +++ b/docs/traceability/mir/structure-segmentation-metric-adapter.md @@ -7,61 +7,82 @@ Parent: `docs/traceability/mir/structure-feature-noninferiority.md` ## Problem -The noninferiority registration names `mir_eval.segment.detection`, `mir_eval.segment.deviation`, and `mir_eval.segment.pairwise`, but a function name alone is not a reproducible scientific implementation identity. Boundary detection changes materially with `window`, `beta`, and especially `trim`; pairwise grouping changes with `frame_size` and `beta`. Leaving those arguments at library defaults would let a future environment or caller alter the dependent variable without changing the registration digest. +The noninferiority registration names `mir_eval.segment.detection`, `mir_eval.segment.deviation`, and `mir_eval.segment.pairwise`, but a function name alone is not a reproducible scientific implementation identity. Boundary detection changes with `window`, `beta`, and `trim`; pairwise grouping changes with `frame_size` and `beta`. A second gap existed after the adapter was introduced: the repository did not bind the exact `mir_eval` distribution artifact, and the admitted-track consumer still emitted functional ACC without the recognized boundary/deviation/repetition measurements. -There is also an authority distinction that must not be blurred. The official `ismir-mirex/mirex-evaluation` MIREX-2025 reproduction at commit `b9fa0b0b32e2145af31f35830f78fc9d09a4301b` implements ACC and HR.5/HR3 itself. Its hit-rate function includes start/end boundaries and is not the same implementation as `mir_eval.segment.detection`, which uses one-to-one boundary matching and exposes an explicit `trim` option. BandScope already chose `mir_eval` for the recognized boundary/deviation/repetition measures; therefore its scores must not be described as numerically identical to the official MIREX HR.5/HR3 table merely because the tolerance names are the same. +There is also an authority distinction that must not be blurred. The official `ismir-mirex/mirex-evaluation` MIREX-2025 reproduction at commit `b9fa0b0b32e2145af31f35830f78fc9d09a4301b` implements ACC and HR.5/HR3 itself. Its hit-rate function is not the same implementation as `mir_eval.segment.detection`. BandScope's selected mir_eval measures therefore must not be described as numerically identical to the official MIREX HR.5/HR3 table merely because the tolerances have the same names. ## Decision -`scripts/research/evaluate_structure_segmentation_metrics.py` is the repository-owned adapter for the currently selected recognized metrics. It requires installed distribution identity `mir_eval==0.8.2` and calls the public API with every result-affecting argument explicit: +`scripts/research/evaluate_structure_segmentation_metrics.py` remains the repository-owned adapter for the selected recognized metrics. It requires installed distribution identity `mir_eval==0.8.2` and calls the public API with every result-affecting argument explicit: - boundary precision/recall/F at 0.5 s: `mir_eval.segment.detection(window=0.5, beta=1.0, trim=True)`; - boundary precision/recall/F at 3.0 s: `mir_eval.segment.detection(window=3.0, beta=1.0, trim=True)`; - bidirectional median boundary deviation: `mir_eval.segment.deviation(trim=True)`; - repetition/grouping precision/recall/F: `mir_eval.segment.pairwise(frame_size=0.1, beta=1.0)`. -`trim=True` is deliberate. The normalized BandScope segment contract already guarantees a boundary at 0 and exact full-track coverage. Counting those shared endpoints as boundary hits would reward two invariants supplied by the adapter contract rather than the model's ability to locate internal rehearsal structure. `mir_eval` documents those endpoints as the typical reason to enable trimming. +`trim=True` is deliberate. The normalized BandScope segment contract guarantees a boundary at zero and exact full-track coverage. Counting those shared endpoints would reward invariants supplied by the adapter contract rather than the model's ability to locate internal rehearsal structure. -The adapter validates both segmentations as continuous from zero with positive segment durations and requires exactly the same total duration. It does not repair gaps, overlaps, duration drift, or malformed labels. It returns only finite bounded scores and non-negative deviations. +The adapter validates both segmentations as continuous from zero with positive segment durations and requires the same total duration. It does not repair gaps, overlaps, duration drift, or malformed labels. It returns only finite bounded scores and non-negative deviations. -## RED → GREEN +## Research runtime identity -RED `3cdf045fc48b64cc8340dc7c7ba71ae44f07c2db` added a focused contract before the adapter existed. The regression requires exact 0.5 s/3.0 s detection calls with `beta=1.0` and `trim=True`, trimmed deviation, 100 ms pairwise grouping, dependency-version fail-closed behavior, and rejection of malformed segmentation geometry. +The production analysis dependency graph is not widened for this scientific adapter. `services/analysis-engine/requirements-structure-metrics.lock` is a research-only overlay that is installed after the frozen analysis-engine environment with `--no-deps`. It identifies one direct wheel only: -GREEN `e570c414eb8cbc76beec4be7c4424814a09d3694` added the metric adapter. Follow-up `93df1d4b1b617bdfa1d2f62d720656d10a67fa97` isolated installed-distribution version evidence from the fake test backend so unit tests do not pretend that a locally injected test double is package provenance. +- package/version: `mir_eval==0.8.2`; +- wheel SHA-256: `114cda33d8e17408c170598e0b36ed0d71ff4a2fee8eaf9e165b58ecf1c87170`; +- upstream source commit: `mir-evaluation/mir_eval@8db0b3812e2032544c1fc00d02d4256cab043f3d`; +- PyPI Sigstore transparency entry: `174236906`. -These are source-level TDD commits. They are not hosted scientific acceptance and do not show a rights-cleared corpus result. +PyPI publishes that wheel through Trusted Publishing and records the same source commit, subject digest, and transparency entry in its provenance attestation. The repository verifier does not perform network access or install anything. `scripts/research/verify_structure_metric_runtime_lock.py` accepts only the exact reviewed lock text, returns its SHA-256 identity, and separately requires the installed distribution version to be exactly `0.8.2` before recognized metrics can execute. + +The direct wheel is deliberately installed with `--no-deps`. Its scientific purpose is to add the metric implementation, not to resolve or mutate the already-frozen production numerical stack. Missing prerequisites therefore fail at the research-environment construction step instead of causing an independent dependency solve. + +## Admitted-track evidence path + +`PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis()` now binds the exact research lock identity at construction and the canonical CQT then STFT lanes. On each admitted track it still sends the same immutable PCM memoryview, sample rate, exact duration, and admitted annotation-derived reference segmentation through both lanes. + +Before any boundary/deviation/repetition score is emitted, the consumer re-verifies the installed `mir_eval` distribution and the lock artifact. If the lock hash has changed since experiment binding, the run fails before recognized metrics are appended. Baseline and candidate metrics are then evaluated against the same in-memory reference segmentation and copied into immutable path-free evidence together with the research-lock SHA-256. + +Directly constructed consumers can remain functional-ACC-only for focused unit boundaries; the canonical registered CQT/STFT factory is the scientific path that binds the recognized metric adapter and runtime identity. Synthetic fixtures exercise this contract only and do not count as scientific acceptance. + +## RED → GREEN lineage + +Adapter RED `3cdf045fc48b64cc8340dc7c7ba71ae44f07c2db` required mir_eval 0.8.2, detection at 0.5 s and 3.0 s with `beta=1.0, trim=True`, `deviation(trim=True)`, pairwise grouping at `frame_size=0.1, beta=1.0`, dependency drift failure, and malformed-geometry rejection. GREEN `e570c414eb8cbc76beec4be7c4424814a09d3694` added the adapter; `93df1d4b1b617bdfa1d2f62d720656d10a67fa97` separated installed-distribution evidence from the fake backend used by unit tests. + +Runtime-lock RED `37f3549d15acb33f397245a88da16dcd8e274cf7` required an exact reviewed artifact identity and fail-closed installed-version/artifact drift behavior. GREEN `8b495913bd6b7fe66f2855afb6fc5d095ae8433d` added the verifier and `23388320e3d536c3d220546c4a857389c83182f9` added the direct-wheel research lock. `59c271f64f2fac893025f4a9cd0d3741fff25865` split immutable lock-identity loading from installed-environment verification so experiment construction can bind the artifact identity without pretending the package is already installed. + +Admitted-track RED `fbb7c1d7f605aabcc36d4bc651b9b4c6c1b39f0d` requires canonical CQT/STFT evidence to bind the runtime-lock digest, evaluate both lanes against the same reference segmentation, and stop before metric execution when the installed distribution drifts. GREEN `773ffd2913f933726b8bf9b890d955189e0e8c5d` wires the reviewed adapter and runtime verifier into the existing admitted-track consumer without changing any production CQT default. + +These commits are source-level TDD and contract evidence. Hosted current-head checks and rights-cleared real-audio scientific acceptance remain separate requirements. ## Constraints and rejected alternatives -Using the official MIREX 2025 `calculate_hit_rate` while continuing to label the registration as `mir_eval.segment.detection` was rejected. They have different matching semantics and would make the registration false. +Using the official MIREX 2025 `calculate_hit_rate` while continuing to label the registration as `mir_eval.segment.detection` was rejected because the algorithms have different matching semantics. -Leaving `trim=False` at the mir_eval default was rejected for the BandScope noninferiority decision because the guaranteed start/end markers would dilute the internal-boundary signal, especially on tracks with few sections. +Leaving `trim=False` at the mir_eval default was rejected because guaranteed start/end markers would dilute the internal-boundary signal, especially on tracks with few sections. -Reimplementing mir_eval formulas in BandScope was rejected. The adapter owns argument selection, validation, and evidence normalization, not a fork of the recognized metric library. +Reimplementing mir_eval formulas in BandScope was rejected. BandScope owns argument selection, validation, artifact identity, and evidence normalization, not a fork of the recognized metric library. -Adding an unpinned `mir_eval>=...` dependency was rejected. Scientific execution must bind the exact reviewed version rather than accepting a later package with the same API surface. +Adding `mir_eval` to the production analysis dependency set was rejected for this research-only measurement. It would widen the buyer runtime for a preregistration tool and let the scientific adapter influence production dependency resolution. The direct no-dependency overlay keeps that boundary explicit. -## Current limitation and next causal step +Using a mutable `mir_eval>=...` requirement, package index lookup at experiment time, or an unhashed wheel was rejected. Any of those would permit the scientific implementation to drift without changing source. + +Computing aggregate scores or confidence intervals inside the admitted-track consumer was rejected. Those remain separate preregistered scientific decisions and must be frozen before candidate outcomes are inspected. -The main registration validator still records generic `mir_eval.segment.*` implementation strings and the analysis-engine lock does not yet carry an exact `mir_eval==0.8.2` research-runtime dependency. Therefore the new adapter is **not yet admissible for the rights-cleared experiment**. Before any candidate real-audio result is inspected, the canonical registration schema and runtime identity must bind: +## Current limitation and next causal step -- mir_eval version 0.8.2; -- detection `beta=1.0` and `trim=True` for both registered windows; -- deviation implementation plus `trim=True`; -- pairwise `frame_size=0.1` and `beta=1.0`; -- the exact locked dependency identity used by the experiment runtime. +The runtime artifact and admitted-track recognized-metric path now exist, but the main registration digest still records generic `mir_eval.segment.*` strings and does not yet include the research-lock SHA-256 or the complete adapter arguments. **The rights-cleared CQT/STFT experiment remains inadmissible until that registration schema is repaired.** Source-level wiring must not be used as a reason to inspect candidate corpus outcomes early. -After that schema/runtime repair, this adapter can be wired into the same admitted-track CQT/STFT consumer that already produces functional ACC. Aggregation and paired uncertainty remain a later preregistered decision and must still be frozen before candidate outcomes are inspected. +The next causal slice is therefore to bind the exact research-lock identity and complete detection/deviation/pairwise semantics into the canonical registration digest, update all registration fixtures/admission contracts together, and only then freeze the aggregate/paired-uncertainty implementation. Rights-cleared real-audio execution follows those preregistration steps, not the reverse. ## Security Notes -The adapter receives normalized in-memory segments only. It adds no filesystem, network, subprocess, model-download, credential, or generic plugin path. The private test seam replaces the module loader only inside focused tests; production execution resolves the installed `mir_eval` distribution and rejects any version other than 0.8.2 before metric calls. +The adapter and verifier receive normalized in-memory values and a bounded local lock file. They add no model download, credentials, generic plugin loading, or package installation at metric execution. The lock verifier performs no network access. Durable track evidence contains content digests, scalar metrics, and the runtime-lock digest, not workstation paths or licensed audio bytes. ## References Raffel, C., McFee, B., Humphrey, E. J., Salamon, J., Nieto, O., Liang, D., & Ellis, D. P. W. (2014). *mir_eval: A transparent implementation of common MIR metrics*. Proceedings of the 15th International Society for Music Information Retrieval Conference. -mir-evaluation contributors. (2026). *mir_eval 0.8.2: segment evaluation API*. https://mir-eval.readthedocs.io/latest/api/segment.html +mir-evaluation contributors. (2025). *mir_eval 0.8.2*. PyPI. https://pypi.org/project/mir-eval/0.8.2/ MIREX Evaluation contributors. (2026). *Music Structure Analysis evaluation script* (commit `b9fa0b0b32e2145af31f35830f78fc9d09a4301b`). https://github.com/ismir-mirex/mirex-evaluation/blob/b9fa0b0b32e2145af31f35830f78fc9d09a4301b/music_structure_analysis/eval_script.py From 7f75895809b79a27170e8e183be155b226670fc9 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 17:34:13 +0900 Subject: [PATCH 151/216] style(mir): format admitted-track metric regression --- ...ure_admitted_track_segmentation_metrics.py | 50 +++++++++++++++---- 1 file changed, 39 insertions(+), 11 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_admitted_track_segmentation_metrics.py b/services/analysis-engine/tests/test_structure_admitted_track_segmentation_metrics.py index 16ef75aee..1dbbd6fdb 100644 --- a/services/analysis-engine/tests/test_structure_admitted_track_segmentation_metrics.py +++ b/services/analysis-engine/tests/test_structure_admitted_track_segmentation_metrics.py @@ -13,7 +13,9 @@ from conftest import load_module _REPOSITORY_ROOT = Path(__file__).resolve().parents[3] -_RUNTIME_LOCK = _REPOSITORY_ROOT / "services/analysis-engine/requirements-structure-metrics.lock" +_RUNTIME_LOCK = ( + _REPOSITORY_ROOT / "services/analysis-engine/requirements-structure-metrics.lock" +) def _consumer_module() -> ModuleType: @@ -52,7 +54,7 @@ def _metric_result(seed: float) -> SimpleNamespace: def test_registered_consumer_binds_runtime_lock_and_scores_both_feature_lanes( monkeypatch: pytest.MonkeyPatch, ) -> None: - """Canonical CQT/STFT evidence must share PCM, annotation, and metric runtime identity.""" + """Canonical CQT/STFT evidence share PCM, annotation, and metric runtime.""" module = _consumer_module() requested_features: list[str] = [] metric_calls: list[tuple[object, object]] = [] @@ -71,19 +73,32 @@ def segmenter( return segmenter - def metric_evaluator(reference_segments: object, estimated_segments: object) -> SimpleNamespace: + def metric_evaluator( + reference_segments: object, + estimated_segments: object, + ) -> SimpleNamespace: metric_calls.append((reference_segments, estimated_segments)) return _metric_result(0.70 if len(metric_calls) == 1 else 0.68) - monkeypatch.setattr(module._LANES, "repository_structure_segmenter", repository_segmenter) + monkeypatch.setattr( + module._LANES, + "repository_structure_segmenter", + repository_segmenter, + ) monkeypatch.setattr( module._SEGMENTATION_EVALUATOR, "calculate_structure_segmentation_metrics", metric_evaluator, ) - monkeypatch.setattr(module._RUNTIME_VERIFIER, "_installed_mir_eval_version", lambda: "0.8.2") + monkeypatch.setattr( + module._RUNTIME_VERIFIER, + "_installed_mir_eval_version", + lambda: "0.8.2", + ) - consumer = module.PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis() + consumer = ( + module.PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis() + ) pcm = memoryview(struct.pack("<8f", *([0.0] * 8))) annotation = memoryview(b"0.0\t0.4\tverse\n0.4\t0.8\tchorus\n") consumer("track-01", pcm, annotation, 10) @@ -92,7 +107,10 @@ def metric_evaluator(reference_segments: object, estimated_segments: object) -> assert len(metric_calls) == 2 assert metric_calls[0][0] is metric_calls[1][0] evidence = consumer.evidence[0] - assert evidence.metric_runtime_lock_sha256 == hashlib.sha256(_RUNTIME_LOCK.read_bytes()).hexdigest() + expected_lock_sha256 = hashlib.sha256(_RUNTIME_LOCK.read_bytes()).hexdigest() + assert evidence.metric_runtime_lock_sha256 == expected_lock_sha256 + assert evidence.baseline_segmentation_metrics is not None + assert evidence.candidate_segmentation_metrics is not None assert evidence.baseline_segmentation_metrics.boundary_f_0_5 == pytest.approx(0.70) assert evidence.candidate_segmentation_metrics.boundary_f_0_5 == pytest.approx(0.68) assert evidence.baseline_segmentation_metrics.repetition_pairwise_f == pytest.approx(0.72) @@ -102,7 +120,7 @@ def metric_evaluator(reference_segments: object, estimated_segments: object) -> def test_registered_consumer_fails_before_metric_execution_on_runtime_drift( monkeypatch: pytest.MonkeyPatch, ) -> None: - """A drifted installed mir_eval must stop scientific metrics before any score is emitted.""" + """A drifted installed mir_eval stops scientific metrics before score emission.""" module = _consumer_module() calls: list[str] = [] @@ -113,8 +131,16 @@ def segmenter( ) -> tuple[SimpleNamespace, ...]: return _segments(("0", "0.4", "verse")) - monkeypatch.setattr(module._LANES, "repository_structure_segmenter", lambda _feature: segmenter) - monkeypatch.setattr(module._RUNTIME_VERIFIER, "_installed_mir_eval_version", lambda: "0.8.1") + monkeypatch.setattr( + module._LANES, + "repository_structure_segmenter", + lambda _feature: segmenter, + ) + monkeypatch.setattr( + module._RUNTIME_VERIFIER, + "_installed_mir_eval_version", + lambda: "0.8.1", + ) def must_not_run(*_args: object) -> object: calls.append("metric") @@ -125,7 +151,9 @@ def must_not_run(*_args: object) -> object: "calculate_structure_segmentation_metrics", must_not_run, ) - consumer = module.PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis() + consumer = ( + module.PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis() + ) pcm = memoryview(struct.pack("<4f", *([0.0] * 4))) annotation = memoryview(b"0.0\t0.4\tverse\n") From 5798efb9f374e8f62ac7c075244e8907c99fa6cc Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 17:34:41 +0900 Subject: [PATCH 152/216] style(mir): format metric runtime lock regression --- .../test_structure_metric_runtime_lock.py | 27 ++++++++++++++----- 1 file changed, 21 insertions(+), 6 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_metric_runtime_lock.py b/services/analysis-engine/tests/test_structure_metric_runtime_lock.py index 38846df46..f8bdb5c4e 100644 --- a/services/analysis-engine/tests/test_structure_metric_runtime_lock.py +++ b/services/analysis-engine/tests/test_structure_metric_runtime_lock.py @@ -9,9 +9,13 @@ from conftest import load_module _REPOSITORY_ROOT = Path(__file__).resolve().parents[3] -_LOCK_PATH = _REPOSITORY_ROOT / "services/analysis-engine/requirements-structure-metrics.lock" +_LOCK_PATH = ( + _REPOSITORY_ROOT / "services/analysis-engine/requirements-structure-metrics.lock" +) _EXPECTED_VERSION = "0.8.2" -_EXPECTED_WHEEL_SHA256 = "114cda33d8e17408c170598e0b36ed0d71ff4a2fee8eaf9e165b58ecf1c87170" +_EXPECTED_WHEEL_SHA256 = ( + "114cda33d8e17408c170598e0b36ed0d71ff4a2fee8eaf9e165b58ecf1c87170" +) _EXPECTED_SOURCE_COMMIT = "8db0b3812e2032544c1fc00d02d4256cab043f3d" _EXPECTED_TRANSPARENCY_ENTRY = 174236906 @@ -24,10 +28,16 @@ def _verifier() -> ModuleType: ) -def test_runtime_lock_binds_exact_trusted_pypi_wheel(monkeypatch: pytest.MonkeyPatch) -> None: +def test_runtime_lock_binds_exact_trusted_pypi_wheel( + monkeypatch: pytest.MonkeyPatch, +) -> None: """Research execution must resolve one reviewed mir_eval artifact identity.""" verifier = _verifier() - monkeypatch.setattr(verifier, "_installed_mir_eval_version", lambda: _EXPECTED_VERSION) + monkeypatch.setattr( + verifier, + "_installed_mir_eval_version", + lambda: _EXPECTED_VERSION, + ) identity = verifier.verify_structure_metric_runtime_lock(_LOCK_PATH) @@ -55,10 +65,15 @@ def test_runtime_lock_rejects_artifact_drift( ) -> None: """Changing the wheel hash must invalidate the scientific runtime identity.""" verifier = _verifier() - monkeypatch.setattr(verifier, "_installed_mir_eval_version", lambda: _EXPECTED_VERSION) + monkeypatch.setattr( + verifier, + "_installed_mir_eval_version", + lambda: _EXPECTED_VERSION, + ) drifted = tmp_path / "requirements-structure-metrics.lock" + lock_text = _LOCK_PATH.read_text(encoding="utf-8") drifted.write_text( - _LOCK_PATH.read_text(encoding="utf-8").replace(_EXPECTED_WHEEL_SHA256, "0" * 64), + lock_text.replace(_EXPECTED_WHEEL_SHA256, "0" * 64), encoding="utf-8", ) From 90f3fa56ab3ca0881cd663e624cdebe67a6ddb79 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 17:34:59 +0900 Subject: [PATCH 153/216] style(mir): format structure metric runtime verifier --- .../verify_structure_metric_runtime_lock.py | 17 ++++++++++++----- 1 file changed, 12 insertions(+), 5 deletions(-) diff --git a/scripts/research/verify_structure_metric_runtime_lock.py b/scripts/research/verify_structure_metric_runtime_lock.py index 1a0425770..ae44d21fa 100644 --- a/scripts/research/verify_structure_metric_runtime_lock.py +++ b/scripts/research/verify_structure_metric_runtime_lock.py @@ -23,7 +23,9 @@ from pathlib import Path MIR_EVAL_VERSION = "0.8.2" -MIR_EVAL_WHEEL_SHA256 = "114cda33d8e17408c170598e0b36ed0d71ff4a2fee8eaf9e165b58ecf1c87170" +MIR_EVAL_WHEEL_SHA256 = ( + "114cda33d8e17408c170598e0b36ed0d71ff4a2fee8eaf9e165b58ecf1c87170" +) MIR_EVAL_SOURCE_COMMIT = "8db0b3812e2032544c1fc00d02d4256cab043f3d" MIR_EVAL_PYPI_TRANSPARENCY_ENTRY = 174236906 MIR_EVAL_WHEEL_URL = ( @@ -37,7 +39,8 @@ ) EXPECTED_LOCK_TEXT = ( "# BandScope structure noninferiority research-metric overlay.\n" - "# Install only after the frozen analysis-engine environment is synced, using --no-deps.\n" + "# Install only after the frozen analysis-engine environment is synced, " + "using --no-deps.\n" f"# Upstream source: mir-evaluation/mir_eval@{MIR_EVAL_SOURCE_COMMIT}\n" f"# PyPI Sigstore transparency entry: {MIR_EVAL_PYPI_TRANSPARENCY_ENTRY}\n" f"{EXPECTED_REQUIREMENT}\n" @@ -76,9 +79,13 @@ def load_structure_metric_runtime_lock_identity( raise ValueError("structure metric runtime lock must be UTF-8 text") from exc if MIR_EVAL_WHEEL_SHA256 not in text: - raise ValueError("structure metric wheel identity does not match the reviewed artifact") + raise ValueError( + "structure metric wheel identity does not match the reviewed artifact" + ) if text != EXPECTED_LOCK_TEXT: - raise ValueError("structure metric runtime lock differs from the reviewed exact contract") + raise ValueError( + "structure metric runtime lock differs from the reviewed exact contract" + ) return StructureMetricRuntimeIdentity( version=MIR_EVAL_VERSION, @@ -92,7 +99,7 @@ def load_structure_metric_runtime_lock_identity( def verify_structure_metric_runtime_lock( lock_path: Path, ) -> StructureMetricRuntimeIdentity: - """Validate the reviewed lock artifact and the installed mir_eval distribution.""" + """Validate the reviewed lock artifact and installed mir_eval distribution.""" identity = load_structure_metric_runtime_lock_identity(lock_path) observed_version = _installed_mir_eval_version() if observed_version != MIR_EVAL_VERSION: From 1ce73179f0425ba7dc4a45c44910c2da93900f9a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:07:06 +0900 Subject: [PATCH 154/216] test(mir): preregister exact segmentation metric contract --- ...ructure_metric_preregistration_contract.py | 79 +++++++++++++++++++ 1 file changed, 79 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_metric_preregistration_contract.py diff --git a/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py b/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py new file mode 100644 index 000000000..48698943a --- /dev/null +++ b/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py @@ -0,0 +1,79 @@ +"""Regression tests for exact structure metric preregistration semantics.""" + +from __future__ import annotations + +import copy + +import pytest + +from test_structure_noninferiority_policy import _metrics, _registration, _validator + +_METRIC_LOCK_SHA256 = "16fd203e9c987064afc667282e001b755838ff0b484235c6932f557d2ae389f8" + + +def _exact_registration() -> dict[str, object]: + """Return the intended closed metric-runtime and adapter contract.""" + registration = _registration() + registration["metric_runtime"] = {"lock_sha256": _METRIC_LOCK_SHA256} + registration["report_metrics"] = { + "boundary_deviation": { + "implementation": "mir_eval.segment.deviation", + "trim": True, + } + } + metrics = _metrics(registration) + boundary_05 = metrics["boundary_f_0_5"] + boundary_30 = metrics["boundary_f_3_0"] + pairwise = metrics["repetition_pairwise_f"] + assert isinstance(boundary_05, dict) + assert isinstance(boundary_30, dict) + assert isinstance(pairwise, dict) + boundary_05.update({"beta": 1.0, "trim": True}) + boundary_30.update({"beta": 1.0, "trim": True}) + pairwise["beta"] = 1.0 + return registration + + +def test_registration_accepts_only_exact_metric_runtime_and_adapter_semantics() -> None: + """The digest must bind the exact lock and every result-affecting adapter argument.""" + validator = _validator() + registration = _exact_registration() + + validator.validate_registration(registration) + expected_digest = validator.registration_digest(registration) + + drifts: list[tuple[str, object]] = [ + ("metric_runtime", {"lock_sha256": "0" * 64}), + ( + "report_metrics", + { + "boundary_deviation": { + "implementation": "mir_eval.segment.deviation", + "trim": False, + } + }, + ), + ] + for field, replacement in drifts: + drifted = copy.deepcopy(registration) + drifted[field] = replacement + with pytest.raises(ValueError): + validator.validate_registration(drifted) + + for metric_name, field, replacement in ( + ("boundary_f_0_5", "beta", 0.5), + ("boundary_f_0_5", "trim", False), + ("boundary_f_3_0", "beta", 2.0), + ("boundary_f_3_0", "trim", False), + ("repetition_pairwise_f", "beta", 0.5), + ): + drifted = copy.deepcopy(registration) + metrics = _metrics(drifted) + config = metrics[metric_name] + assert isinstance(config, dict) + config[field] = replacement + with pytest.raises(ValueError, match=field): + validator.validate_registration(drifted) + + reordered = dict(reversed(list(registration.items()))) + assert validator.registration_digest(reordered) == expected_digest From 580f2affcdf5ed1679e1e8748011c56d7b3287e8 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:12:34 +0900 Subject: [PATCH 155/216] feat(mir): validate exact segmentation metric preregistration --- ...lidate_structure_metric_preregistration.py | 175 ++++++++++++++++++ 1 file changed, 175 insertions(+) create mode 100644 scripts/research/validate_structure_metric_preregistration.py diff --git a/scripts/research/validate_structure_metric_preregistration.py b/scripts/research/validate_structure_metric_preregistration.py new file mode 100644 index 000000000..1e3c5e719 --- /dev/null +++ b/scripts/research/validate_structure_metric_preregistration.py @@ -0,0 +1,175 @@ +#!/usr/bin/env python3 +"""Validate the exact research metric contract layered on structure preregistration. + +This module closes the scientific adapter identity that the base noninferiority +registration did not yet bind: the content-addressed mir_eval research lock, +all explicit detection/pairwise arguments, and report-only deviation semantics. +It does not execute MIR metrics or inspect corpus outcomes. +""" + +from __future__ import annotations + +import hashlib +import importlib.util +import json +import math +from collections.abc import Mapping +from pathlib import Path +from types import ModuleType +from typing import Any + +STRUCTURE_METRIC_RUNTIME_LOCK_SHA256 = ( + "16fd203e9c987064afc667282e001b755838ff0b484235c6932f557d2ae389f8" +) +QUALITY_METRIC_CONTRACT: dict[str, dict[str, object]] = { + "boundary_f_0_5": { + "implementation": "mir_eval.segment.detection", + "window_seconds": 0.5, + "beta": 1.0, + "trim": True, + }, + "boundary_f_3_0": { + "implementation": "mir_eval.segment.detection", + "window_seconds": 3.0, + "beta": 1.0, + "trim": True, + }, + "repetition_pairwise_f": { + "implementation": "mir_eval.segment.pairwise", + "frame_size_seconds": 0.1, + "beta": 1.0, + }, +} +REPORT_METRIC_CONTRACT: dict[str, dict[str, object]] = { + "boundary_deviation": { + "implementation": "mir_eval.segment.deviation", + "trim": True, + } +} +_EXTENSION_FIELDS = {"metric_runtime", "report_metrics"} +_BASE_METRIC_EXTRA_FIELDS = { + "boundary_f_0_5": {"beta", "trim"}, + "boundary_f_3_0": {"beta", "trim"}, + "repetition_pairwise_f": {"beta"}, +} + + +def _base_validator() -> ModuleType: + """Load the existing closed base-registration validator from this directory.""" + path = Path(__file__).with_name("validate_structure_noninferiority.py") + spec = importlib.util.spec_from_file_location( + "bandscope_structure_noninferiority_base", + path, + ) + if spec is None or spec.loader is None: + raise RuntimeError("could not load structure noninferiority base validator") + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def _mapping(value: object, field: str) -> Mapping[str, Any]: + if not isinstance(value, Mapping): + raise ValueError(f"{field} must be an object") + return value + + +def _require_exact_fields( + value: Mapping[str, Any], + expected: set[str], + field: str, +) -> None: + actual = set(value) + missing = sorted(expected - actual) + extra = sorted(actual - expected) + if missing: + raise ValueError(f"{field} missing required field: {missing[0]}") + if extra: + raise ValueError(f"{field} contains unregistered field: {extra[0]}") + + +def _require_expected(value: object, expected: object, field: str) -> None: + """Require an exact reviewed scalar without bool/number coercion.""" + if isinstance(expected, bool): + if value is not expected: + raise ValueError(f"{field} must equal {expected}") + return + if isinstance(expected, str): + if value != expected: + raise ValueError(f"{field} must equal {expected}") + return + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise ValueError(f"{field} must be a finite number") + actual = float(value) + expected_number = float(expected) + if not math.isfinite(actual) or not math.isclose( + actual, + expected_number, + rel_tol=0.0, + abs_tol=1e-12, + ): + raise ValueError(f"{field} must equal {expected}") + + +def _base_registration(registration: Mapping[str, Any]) -> dict[str, object]: + """Project the extended registration onto the already-reviewed base schema.""" + base = {key: value for key, value in registration.items() if key not in _EXTENSION_FIELDS} + metrics_value = base.get("metrics") + if not isinstance(metrics_value, Mapping): + return base + metrics: dict[str, object] = {} + for metric_name, config_value in metrics_value.items(): + if not isinstance(config_value, Mapping): + metrics[metric_name] = config_value + continue + extras = _BASE_METRIC_EXTRA_FIELDS.get(metric_name, set()) + metrics[metric_name] = { + key: value for key, value in config_value.items() if key not in extras + } + base["metrics"] = metrics + return base + + +def validate_registration(registration_value: object) -> None: + """Validate the base registration plus exact result-affecting metric semantics.""" + registration = _mapping(registration_value, "registration") + expected_top_level = set(_base_registration(registration)) | _EXTENSION_FIELDS + _require_exact_fields(registration, expected_top_level, "registration") + + metric_runtime = _mapping(registration.get("metric_runtime"), "metric_runtime") + _require_exact_fields(metric_runtime, {"lock_sha256"}, "metric_runtime") + if metric_runtime.get("lock_sha256") != STRUCTURE_METRIC_RUNTIME_LOCK_SHA256: + raise ValueError( + "metric_runtime.lock_sha256 must equal the reviewed structure metric lock" + ) + + report_metrics = _mapping(registration.get("report_metrics"), "report_metrics") + _require_exact_fields(report_metrics, set(REPORT_METRIC_CONTRACT), "report_metrics") + for metric_name, expected_config in REPORT_METRIC_CONTRACT.items(): + config = _mapping(report_metrics.get(metric_name), f"report_metrics.{metric_name}") + _require_exact_fields(config, set(expected_config), f"report_metrics.{metric_name}") + for field, expected in expected_config.items(): + _require_expected(config.get(field), expected, f"report_metrics.{metric_name}.{field}") + + metrics = _mapping(registration.get("metrics"), "metrics") + for metric_name, expected_config in QUALITY_METRIC_CONTRACT.items(): + config = _mapping(metrics.get(metric_name), f"metrics.{metric_name}") + for field, expected in expected_config.items(): + if field not in config: + raise ValueError(f"metrics.{metric_name} missing required field: {field}") + _require_expected(config.get(field), expected, f"metrics.{metric_name}.{field}") + + _base_validator().validate_registration(_base_registration(registration)) + + +def registration_digest(registration_value: object) -> str: + """Return the canonical SHA-256 of the complete metric-aware preregistration.""" + validate_registration(registration_value) + canonical = json.dumps( + registration_value, + allow_nan=False, + ensure_ascii=False, + separators=(",", ":"), + sort_keys=True, + ).encode("utf-8") + return hashlib.sha256(canonical).hexdigest() From fd43da9d1d77e2761b1ebccae81da68db9f4e96e Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:12:57 +0900 Subject: [PATCH 156/216] test(mir): exercise metric-aware registration digest --- ...est_structure_metric_preregistration_contract.py | 13 +++++++++++-- 1 file changed, 11 insertions(+), 2 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py b/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py index 48698943a..a382d3678 100644 --- a/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py +++ b/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py @@ -3,14 +3,23 @@ from __future__ import annotations import copy +from types import ModuleType import pytest +from conftest import load_module -from test_structure_noninferiority_policy import _metrics, _registration, _validator +from test_structure_noninferiority_policy import _metrics, _registration _METRIC_LOCK_SHA256 = "16fd203e9c987064afc667282e001b755838ff0b484235c6932f557d2ae389f8" +def _metric_validator() -> ModuleType: + return load_module( + "scripts/research/validate_structure_metric_preregistration.py", + "validate_structure_metric_preregistration", + ) + + def _exact_registration() -> dict[str, object]: """Return the intended closed metric-runtime and adapter contract.""" registration = _registration() @@ -36,7 +45,7 @@ def _exact_registration() -> dict[str, object]: def test_registration_accepts_only_exact_metric_runtime_and_adapter_semantics() -> None: """The digest must bind the exact lock and every result-affecting adapter argument.""" - validator = _validator() + validator = _metric_validator() registration = _exact_registration() validator.validate_registration(registration) From 880e0db335d29865016d98da70cb7643ad3e4291 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:14:51 +0900 Subject: [PATCH 157/216] refactor(mir): make metric contract a digest envelope --- ...lidate_structure_metric_preregistration.py | 199 ++++++++---------- 1 file changed, 84 insertions(+), 115 deletions(-) diff --git a/scripts/research/validate_structure_metric_preregistration.py b/scripts/research/validate_structure_metric_preregistration.py index 1e3c5e719..e6c740ac9 100644 --- a/scripts/research/validate_structure_metric_preregistration.py +++ b/scripts/research/validate_structure_metric_preregistration.py @@ -1,27 +1,29 @@ #!/usr/bin/env python3 -"""Validate the exact research metric contract layered on structure preregistration. +"""Bind exact structure-metric semantics into the scientific registration digest. -This module closes the scientific adapter identity that the base noninferiority -registration did not yet bind: the content-addressed mir_eval research lock, -all explicit detection/pairwise arguments, and report-only deviation semantics. -It does not execute MIR metrics or inspect corpus outcomes. +The base registration schema owns corpus, experiment runtime, margins, and +paired-decision fields. This façade adds the repository-owned metric contract to +the canonical digest without making callers repeat immutable adapter constants +inside every registration JSON. No MIR metric or corpus outcome is computed +here. """ from __future__ import annotations +import argparse +import copy import hashlib import importlib.util import json -import math -from collections.abc import Mapping +from collections.abc import Mapping, Sequence from pathlib import Path from types import ModuleType from typing import Any -STRUCTURE_METRIC_RUNTIME_LOCK_SHA256 = ( - "16fd203e9c987064afc667282e001b755838ff0b484235c6932f557d2ae389f8" -) -QUALITY_METRIC_CONTRACT: dict[str, dict[str, object]] = { +STRUCTURE_METRIC_CONTRACT = { + "runtime_lock_sha256": ( + "16fd203e9c987064afc667282e001b755838ff0b484235c6932f557d2ae389f8" + ), "boundary_f_0_5": { "implementation": "mir_eval.segment.detection", "window_seconds": 0.5, @@ -34,29 +36,21 @@ "beta": 1.0, "trim": True, }, + "boundary_deviation": { + "implementation": "mir_eval.segment.deviation", + "trim": True, + }, "repetition_pairwise_f": { "implementation": "mir_eval.segment.pairwise", "frame_size_seconds": 0.1, "beta": 1.0, }, } -REPORT_METRIC_CONTRACT: dict[str, dict[str, object]] = { - "boundary_deviation": { - "implementation": "mir_eval.segment.deviation", - "trim": True, - } -} -_EXTENSION_FIELDS = {"metric_runtime", "report_metrics"} -_BASE_METRIC_EXTRA_FIELDS = { - "boundary_f_0_5": {"beta", "trim"}, - "boundary_f_3_0": {"beta", "trim"}, - "repetition_pairwise_f": {"beta"}, -} def _base_validator() -> ModuleType: - """Load the existing closed base-registration validator from this directory.""" - path = Path(__file__).with_name("validate_structure_noninferiority.py") + """Load the unchanged base decision-policy implementation.""" + path = Path(__file__).with_name("validate_structure_noninferiority_base.py") spec = importlib.util.spec_from_file_location( "bandscope_structure_noninferiority_base", path, @@ -68,108 +62,83 @@ def _base_validator() -> ModuleType: return module -def _mapping(value: object, field: str) -> Mapping[str, Any]: - if not isinstance(value, Mapping): - raise ValueError(f"{field} must be an object") - return value - - -def _require_exact_fields( - value: Mapping[str, Any], - expected: set[str], - field: str, -) -> None: - actual = set(value) - missing = sorted(expected - actual) - extra = sorted(actual - expected) - if missing: - raise ValueError(f"{field} missing required field: {missing[0]}") - if extra: - raise ValueError(f"{field} contains unregistered field: {extra[0]}") - - -def _require_expected(value: object, expected: object, field: str) -> None: - """Require an exact reviewed scalar without bool/number coercion.""" - if isinstance(expected, bool): - if value is not expected: - raise ValueError(f"{field} must equal {expected}") - return - if isinstance(expected, str): - if value != expected: - raise ValueError(f"{field} must equal {expected}") - return - if isinstance(value, bool) or not isinstance(value, (int, float)): - raise ValueError(f"{field} must be a finite number") - actual = float(value) - expected_number = float(expected) - if not math.isfinite(actual) or not math.isclose( - actual, - expected_number, - rel_tol=0.0, - abs_tol=1e-12, - ): - raise ValueError(f"{field} must equal {expected}") - - -def _base_registration(registration: Mapping[str, Any]) -> dict[str, object]: - """Project the extended registration onto the already-reviewed base schema.""" - base = {key: value for key, value in registration.items() if key not in _EXTENSION_FIELDS} - metrics_value = base.get("metrics") - if not isinstance(metrics_value, Mapping): - return base - metrics: dict[str, object] = {} - for metric_name, config_value in metrics_value.items(): - if not isinstance(config_value, Mapping): - metrics[metric_name] = config_value - continue - extras = _BASE_METRIC_EXTRA_FIELDS.get(metric_name, set()) - metrics[metric_name] = { - key: value for key, value in config_value.items() if key not in extras - } - base["metrics"] = metrics - return base +_BASE = _base_validator() +SCHEMA_VERSION = _BASE.SCHEMA_VERSION +MAX_EVIDENCE_BYTES = _BASE.MAX_EVIDENCE_BYTES def validate_registration(registration_value: object) -> None: - """Validate the base registration plus exact result-affecting metric semantics.""" - registration = _mapping(registration_value, "registration") - expected_top_level = set(_base_registration(registration)) | _EXTENSION_FIELDS - _require_exact_fields(registration, expected_top_level, "registration") - - metric_runtime = _mapping(registration.get("metric_runtime"), "metric_runtime") - _require_exact_fields(metric_runtime, {"lock_sha256"}, "metric_runtime") - if metric_runtime.get("lock_sha256") != STRUCTURE_METRIC_RUNTIME_LOCK_SHA256: - raise ValueError( - "metric_runtime.lock_sha256 must equal the reviewed structure metric lock" - ) + """Validate the closed base registration consumed by the metric-aware digest.""" + _BASE.validate_registration(registration_value) - report_metrics = _mapping(registration.get("report_metrics"), "report_metrics") - _require_exact_fields(report_metrics, set(REPORT_METRIC_CONTRACT), "report_metrics") - for metric_name, expected_config in REPORT_METRIC_CONTRACT.items(): - config = _mapping(report_metrics.get(metric_name), f"report_metrics.{metric_name}") - _require_exact_fields(config, set(expected_config), f"report_metrics.{metric_name}") - for field, expected in expected_config.items(): - _require_expected(config.get(field), expected, f"report_metrics.{metric_name}.{field}") - metrics = _mapping(registration.get("metrics"), "metrics") - for metric_name, expected_config in QUALITY_METRIC_CONTRACT.items(): - config = _mapping(metrics.get(metric_name), f"metrics.{metric_name}") - for field, expected in expected_config.items(): - if field not in config: - raise ValueError(f"metrics.{metric_name} missing required field: {field}") - _require_expected(config.get(field), expected, f"metrics.{metric_name}.{field}") - - _base_validator().validate_registration(_base_registration(registration)) +def _digest_payload(registration_value: object) -> dict[str, object]: + """Return the complete closed object whose bytes define scientific identity.""" + validate_registration(registration_value) + return { + "registration": registration_value, + "structure_metric_contract": STRUCTURE_METRIC_CONTRACT, + } def registration_digest(registration_value: object) -> str: - """Return the canonical SHA-256 of the complete metric-aware preregistration.""" - validate_registration(registration_value) + """Return SHA-256 over registration data plus exact metric/runtime semantics.""" canonical = json.dumps( - registration_value, + _digest_payload(registration_value), allow_nan=False, ensure_ascii=False, separators=(",", ":"), sort_keys=True, ).encode("utf-8") return hashlib.sha256(canonical).hexdigest() + + +def evaluate_result( + registration_value: object, + result_value: object, +) -> dict[str, object]: + """Evaluate a receipt only when it binds the metric-aware registration digest.""" + validate_registration(registration_value) + if not isinstance(result_value, Mapping): + raise ValueError("result must be an object") + expected_digest = registration_digest(registration_value) + if result_value.get("registration_sha256") != expected_digest: + raise ValueError( + "result.registration_sha256 does not match the frozen metric-aware registration" + ) + + projected_result = copy.deepcopy(dict(result_value)) + projected_result["registration_sha256"] = _BASE.registration_digest( + registration_value + ) + decision = _BASE.evaluate_result(registration_value, projected_result) + decision["registration_sha256"] = expected_digest + return decision + + +def __getattr__(name: str) -> Any: + """Delegate unchanged schema helpers to the base policy implementation.""" + return getattr(_BASE, name) + + +def main(argv: Sequence[str] | None = None) -> int: + """Validate a registration and optionally evaluate one bound result.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("registration", type=Path) + parser.add_argument("result", type=Path, nargs="?") + args = parser.parse_args(argv) + + registration = _BASE._load_json(args.registration) + digest = registration_digest(registration) + if args.result is None: + print(json.dumps({"registration_sha256": digest}, sort_keys=True)) + return 0 + + result = _BASE._load_json(args.result) + decision = evaluate_result(registration, result) + print(json.dumps(decision, sort_keys=True)) + return 0 if decision["passed"] is True else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) From cadba9ba74e8d6a97815c875bad51b90905fa996 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:15:16 +0900 Subject: [PATCH 158/216] refactor(mir): make metric-aware validator canonical --- ...lidate_structure_metric_preregistration.py | 144 --- .../validate_structure_noninferiority.py | 868 ++--------------- .../validate_structure_noninferiority_base.py | 886 ++++++++++++++++++ 3 files changed, 949 insertions(+), 949 deletions(-) delete mode 100644 scripts/research/validate_structure_metric_preregistration.py create mode 100644 scripts/research/validate_structure_noninferiority_base.py diff --git a/scripts/research/validate_structure_metric_preregistration.py b/scripts/research/validate_structure_metric_preregistration.py deleted file mode 100644 index e6c740ac9..000000000 --- a/scripts/research/validate_structure_metric_preregistration.py +++ /dev/null @@ -1,144 +0,0 @@ -#!/usr/bin/env python3 -"""Bind exact structure-metric semantics into the scientific registration digest. - -The base registration schema owns corpus, experiment runtime, margins, and -paired-decision fields. This façade adds the repository-owned metric contract to -the canonical digest without making callers repeat immutable adapter constants -inside every registration JSON. No MIR metric or corpus outcome is computed -here. -""" - -from __future__ import annotations - -import argparse -import copy -import hashlib -import importlib.util -import json -from collections.abc import Mapping, Sequence -from pathlib import Path -from types import ModuleType -from typing import Any - -STRUCTURE_METRIC_CONTRACT = { - "runtime_lock_sha256": ( - "16fd203e9c987064afc667282e001b755838ff0b484235c6932f557d2ae389f8" - ), - "boundary_f_0_5": { - "implementation": "mir_eval.segment.detection", - "window_seconds": 0.5, - "beta": 1.0, - "trim": True, - }, - "boundary_f_3_0": { - "implementation": "mir_eval.segment.detection", - "window_seconds": 3.0, - "beta": 1.0, - "trim": True, - }, - "boundary_deviation": { - "implementation": "mir_eval.segment.deviation", - "trim": True, - }, - "repetition_pairwise_f": { - "implementation": "mir_eval.segment.pairwise", - "frame_size_seconds": 0.1, - "beta": 1.0, - }, -} - - -def _base_validator() -> ModuleType: - """Load the unchanged base decision-policy implementation.""" - path = Path(__file__).with_name("validate_structure_noninferiority_base.py") - spec = importlib.util.spec_from_file_location( - "bandscope_structure_noninferiority_base", - path, - ) - if spec is None or spec.loader is None: - raise RuntimeError("could not load structure noninferiority base validator") - module = importlib.util.module_from_spec(spec) - spec.loader.exec_module(module) - return module - - -_BASE = _base_validator() -SCHEMA_VERSION = _BASE.SCHEMA_VERSION -MAX_EVIDENCE_BYTES = _BASE.MAX_EVIDENCE_BYTES - - -def validate_registration(registration_value: object) -> None: - """Validate the closed base registration consumed by the metric-aware digest.""" - _BASE.validate_registration(registration_value) - - -def _digest_payload(registration_value: object) -> dict[str, object]: - """Return the complete closed object whose bytes define scientific identity.""" - validate_registration(registration_value) - return { - "registration": registration_value, - "structure_metric_contract": STRUCTURE_METRIC_CONTRACT, - } - - -def registration_digest(registration_value: object) -> str: - """Return SHA-256 over registration data plus exact metric/runtime semantics.""" - canonical = json.dumps( - _digest_payload(registration_value), - allow_nan=False, - ensure_ascii=False, - separators=(",", ":"), - sort_keys=True, - ).encode("utf-8") - return hashlib.sha256(canonical).hexdigest() - - -def evaluate_result( - registration_value: object, - result_value: object, -) -> dict[str, object]: - """Evaluate a receipt only when it binds the metric-aware registration digest.""" - validate_registration(registration_value) - if not isinstance(result_value, Mapping): - raise ValueError("result must be an object") - expected_digest = registration_digest(registration_value) - if result_value.get("registration_sha256") != expected_digest: - raise ValueError( - "result.registration_sha256 does not match the frozen metric-aware registration" - ) - - projected_result = copy.deepcopy(dict(result_value)) - projected_result["registration_sha256"] = _BASE.registration_digest( - registration_value - ) - decision = _BASE.evaluate_result(registration_value, projected_result) - decision["registration_sha256"] = expected_digest - return decision - - -def __getattr__(name: str) -> Any: - """Delegate unchanged schema helpers to the base policy implementation.""" - return getattr(_BASE, name) - - -def main(argv: Sequence[str] | None = None) -> int: - """Validate a registration and optionally evaluate one bound result.""" - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("registration", type=Path) - parser.add_argument("result", type=Path, nargs="?") - args = parser.parse_args(argv) - - registration = _BASE._load_json(args.registration) - digest = registration_digest(registration) - if args.result is None: - print(json.dumps({"registration_sha256": digest}, sort_keys=True)) - return 0 - - result = _BASE._load_json(args.result) - decision = evaluate_result(registration, result) - print(json.dumps(decision, sort_keys=True)) - return 0 if decision["passed"] is True else 1 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index f3f344164..e6c740ac9 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -1,455 +1,90 @@ #!/usr/bin/env python3 -"""Validate preregistered structure-feature noninferiority evidence. +"""Bind exact structure-metric semantics into the scientific registration digest. -This module does not compute MIR metrics. It binds a reviewed registration to -result receipts produced by the recognized evaluation pipeline, then evaluates -the preregistered confidence-interval decision rules. Metric implementation -authority remains with MIREX/mir_eval-compatible tooling while thresholds, -corpus identity, uncertainty procedure, and runtime identity are protected from -post-result drift. +The base registration schema owns corpus, experiment runtime, margins, and +paired-decision fields. This façade adds the repository-owned metric contract to +the canonical digest without making callers repeat immutable adapter constants +inside every registration JSON. No MIR metric or corpus outcome is computed +here. """ from __future__ import annotations import argparse +import copy import hashlib +import importlib.util import json -import math -import os -import re -import stat from collections.abc import Mapping, Sequence from pathlib import Path +from types import ModuleType from typing import Any -SCHEMA_VERSION = 1 -MAX_EVIDENCE_BYTES = 2 * 1024 * 1024 -_SHA256_RE = re.compile(r"^[0-9a-fA-F]{64}$") -_COMMIT_RE = re.compile(r"^[0-9a-fA-F]{40}$") -_WINDOWS_DRIVE_RE = re.compile(r"^[A-Za-z]:") -_URI_SCHEME_RE = re.compile(r"^[A-Za-z][A-Za-z0-9+.-]*:") - -_QUALITY_METRICS: dict[str, dict[str, float | str]] = { +STRUCTURE_METRIC_CONTRACT = { + "runtime_lock_sha256": ( + "16fd203e9c987064afc667282e001b755838ff0b484235c6932f557d2ae389f8" + ), "boundary_f_0_5": { "implementation": "mir_eval.segment.detection", "window_seconds": 0.5, + "beta": 1.0, + "trim": True, }, "boundary_f_3_0": { "implementation": "mir_eval.segment.detection", "window_seconds": 3.0, + "beta": 1.0, + "trim": True, }, - "functional_label_accuracy": { - "implementation": ( - "ismir-mirex/mirex-evaluation@" - "b9fa0b0b32e2145af31f35830f78fc9d09a4301b:" - "music_structure_analysis.eval_script.calculate_accuracy" - ), - "frame_size_seconds": 0.2, - "frame_grid_contract_version": 1.0, - "annotation_contract_version": 1.0, - "label_mapping_contract_version": 1.0, + "boundary_deviation": { + "implementation": "mir_eval.segment.deviation", + "trim": True, }, "repetition_pairwise_f": { "implementation": "mir_eval.segment.pairwise", "frame_size_seconds": 0.1, + "beta": 1.0, }, } -_LATENCY_METRIC = "p95_latency_ratio" -_REPORT_SCORE_METRICS = ( - "boundary_precision_0_5", - "boundary_recall_0_5", - "boundary_precision_3_0", - "boundary_recall_3_0", - "repetition_pairwise_precision", - "repetition_pairwise_recall", -) -_REPORT_NONNEGATIVE_METRICS = ( - "reference_to_estimate_median_deviation_seconds", - "estimate_to_reference_median_deviation_seconds", - "p50_latency_seconds", - "p95_latency_seconds", - "peak_rss_mib", -) -_REGISTRATION_FIELDS = { - "schema_version", - "experiment_id", - "hypothesis", - "metrics", - "uncertainty", - "corpus", - "runtime", - "claim_boundary", -} -_HYPOTHESIS_FIELDS = {"baseline_feature", "candidate_feature"} -_CORPUS_TRACK_FIELDS = { - "track_id", - "audio_sha256", - "annotation_sha256", - "rights_basis", - "rights_cleared", - "source_uri", -} -_RUNTIME_FIELDS = { - "source_commit", - "uv_lock_sha256", - "python_version", - "librosa_version", - "numpy_version", - "sample_rate_hz", - "channels", - "host_profile", -} -_RESULT_FIELDS = { - "schema_version", - "experiment_id", - "registration_sha256", - "uncertainty", - "corpus_track_ids", - "tracks", - "aggregate", - "paired_delta_ci95", - "p95_latency_ratio_ci95", - "failed_tracks", - "claim_boundary", -} -_TRACK_RECEIPT_FIELDS = {"track_id", "baseline", "candidate"} -_FAILED_TRACK_RECEIPT_FIELDS = {"track_id"} -_AGGREGATE_FIELDS = {"baseline", "candidate"} - - -def _mapping(value: object, field: str) -> Mapping[str, Any]: - """Return ``value`` as a mapping or raise a field-specific error.""" - if not isinstance(value, Mapping): - raise ValueError(f"{field} must be an object") - return value - - -def _require_exact_fields( - value: Mapping[str, Any], - expected: set[str], - field: str, -) -> None: - """Require an evidence object to use exactly its schema-v1 field set.""" - actual = set(value) - missing = sorted(expected - actual) - extra = sorted(actual - expected) - if missing: - raise ValueError(f"{field} missing required field: {missing[0]}") - if extra: - raise ValueError(f"{field} contains unregistered field: {extra[0]}") - - -def _sequence(value: object, field: str) -> Sequence[Any]: - """Return a non-string sequence or raise a field-specific error.""" - if isinstance(value, (str, bytes)) or not isinstance(value, Sequence): - raise ValueError(f"{field} must be an array") - return value - - -def _nonempty_text(value: object, field: str) -> str: - """Return stripped non-empty text for a required field.""" - if not isinstance(value, str) or not value.strip(): - raise ValueError(f"{field} must be non-empty text") - return value.strip() - - -def _finite_number(value: object, field: str) -> float: - """Return a finite real number while rejecting booleans and NaN/Inf.""" - if isinstance(value, bool) or not isinstance(value, (int, float)): - raise ValueError(f"{field} must be a finite number") - number = float(value) - if not math.isfinite(number): - raise ValueError(f"{field} must be finite") - return number - - -def _bounded_integer( - value: object, - field: str, - *, - minimum: int, - maximum: int, -) -> int: - """Return a bounded integer while rejecting bools and fractional values.""" - if isinstance(value, bool) or not isinstance(value, int): - raise ValueError(f"{field} must be an integer") - if not minimum <= value <= maximum: - raise ValueError(f"{field} must be in {minimum}..{maximum}") - return value - - -def _score(value: object, field: str) -> float: - """Return a finite score constrained to the inclusive 0..1 interval.""" - number = _finite_number(value, field) - if not 0.0 <= number <= 1.0: - raise ValueError(f"{field} must be between 0 and 1") - return number - - -def _sha256(value: object, field: str) -> str: - """Return a normalized SHA-256 hex digest.""" - text = _nonempty_text(value, field) - if _SHA256_RE.fullmatch(text) is None: - raise ValueError(f"{field} must be a 64-character SHA-256 hex digest") - return text.lower() - - -def _commit(value: object, field: str) -> str: - """Return a normalized full Git commit SHA.""" - text = _nonempty_text(value, field) - if _COMMIT_RE.fullmatch(text) is None: - raise ValueError(f"{field} must be a full 40-character Git commit SHA") - return text.lower() - - -def _reject_local_path(value: object, field: str) -> str: - """Require an explicit non-file URI for corpus provenance receipts.""" - text = _nonempty_text(value, field) - lowered = text.casefold() - scheme = _URI_SCHEME_RE.match(text) - if ( - text.startswith(("/", "\\\\")) - or _WINDOWS_DRIVE_RE.match(text) is not None - or lowered.startswith("file:") - or scheme is None - or scheme.end() == len(text) - ): - raise ValueError( - f"{field} must be a provenance URI, not a local filesystem path" - ) - return text - - -def _validate_schema_version(value: object, field: str) -> None: - """Require the one currently supported evidence schema version.""" - if isinstance(value, bool) or value != SCHEMA_VERSION: - raise ValueError(f"{field} must equal {SCHEMA_VERSION}") - - -def _validate_metrics(metrics_value: object) -> None: - """Validate the complete preregistered metric and decision contract.""" - metrics = _mapping(metrics_value, "metrics") - expected = set(_QUALITY_METRICS) | {_LATENCY_METRIC} - actual = set(metrics) - missing = sorted(expected - actual) - extra = sorted(actual - expected) - if missing: - raise ValueError(f"metrics missing required metric: {missing[0]}") - if extra: - raise ValueError(f"metrics contains unregistered metric: {extra[0]}") - - for metric_name, expected_config in _QUALITY_METRICS.items(): - config = _mapping(metrics[metric_name], f"metrics.{metric_name}") - _require_exact_fields( - config, - set(expected_config) | {"noninferiority_margin"}, - f"metrics.{metric_name}", - ) - implementation = _nonempty_text( - config.get("implementation"), - f"metrics.{metric_name}.implementation", - ) - if implementation != expected_config["implementation"]: - raise ValueError( - f"metrics.{metric_name}.implementation must equal " - f"{expected_config['implementation']}" - ) - for config_name, expected_value in expected_config.items(): - if config_name == "implementation": - continue - actual_value = _finite_number( - config.get(config_name), - f"metrics.{metric_name}.{config_name}", - ) - expected_number = float(expected_value) - if not math.isclose( - actual_value, - expected_number, - rel_tol=0.0, - abs_tol=1e-12, - ): - raise ValueError( - f"metrics.{metric_name}.{config_name} must equal " - f"{expected_value}" - ) - margin = _finite_number( - config.get("noninferiority_margin"), - f"metrics.{metric_name}.noninferiority_margin", - ) - if not 0.0 <= margin < 1.0: - raise ValueError( - f"metrics.{metric_name}.noninferiority_margin must be in [0, 1)" - ) - - latency = _mapping(metrics[_LATENCY_METRIC], f"metrics.{_LATENCY_METRIC}") - _require_exact_fields( - latency, - {"maximum_candidate_ratio"}, - f"metrics.{_LATENCY_METRIC}", - ) - maximum_ratio = _finite_number( - latency.get("maximum_candidate_ratio"), - f"metrics.{_LATENCY_METRIC}.maximum_candidate_ratio", - ) - if not 0.0 < maximum_ratio < 1.0: - raise ValueError( - "metrics.p95_latency_ratio.maximum_candidate_ratio must be greater " - "than 0 and less than 1" - ) - - -def _validate_uncertainty(uncertainty_value: object, field: str) -> dict[str, object]: - """Validate and normalize the preregistered paired-uncertainty procedure.""" - uncertainty = _mapping(uncertainty_value, field) - required = { - "procedure_id", - "confidence_level", - "resamples", - "random_seed", - } - missing = sorted(required - set(uncertainty)) - extra = sorted(set(uncertainty) - required) - if missing: - raise ValueError(f"{field} missing required field: {missing[0]}") - if extra: - raise ValueError(f"{field} contains unregistered field: {extra[0]}") - procedure_id = _nonempty_text( - uncertainty.get("procedure_id"), - f"{field}.procedure_id", - ) - if len(procedure_id) > 128: - raise ValueError(f"{field}.procedure_id must be at most 128 characters") - confidence_level = _finite_number( - uncertainty.get("confidence_level"), - f"{field}.confidence_level", +def _base_validator() -> ModuleType: + """Load the unchanged base decision-policy implementation.""" + path = Path(__file__).with_name("validate_structure_noninferiority_base.py") + spec = importlib.util.spec_from_file_location( + "bandscope_structure_noninferiority_base", + path, ) - if not math.isclose(confidence_level, 0.95, rel_tol=0.0, abs_tol=1e-12): - raise ValueError(f"{field}.confidence_level must equal 0.95") + if spec is None or spec.loader is None: + raise RuntimeError("could not load structure noninferiority base validator") + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module - resamples = _bounded_integer( - uncertainty.get("resamples"), - f"{field}.resamples", - minimum=1000, - maximum=10_000_000, - ) - random_seed = _bounded_integer( - uncertainty.get("random_seed"), - f"{field}.random_seed", - minimum=0, - maximum=(2**32) - 1, - ) - return { - "procedure_id": procedure_id, - "confidence_level": confidence_level, - "resamples": resamples, - "random_seed": random_seed, - } - -def _validate_corpus(corpus_value: object) -> list[str]: - """Validate rights-cleared real-audio identities and return ordered IDs.""" - corpus = _sequence(corpus_value, "corpus") - if len(corpus) < 2: - raise ValueError( - "corpus must contain at least two rights-cleared real-audio tracks" - ) - - seen_track_ids: set[str] = set() - seen_audio_sha256: set[str] = set() - track_ids: list[str] = [] - for index, raw_track in enumerate(corpus): - field = f"corpus[{index}]" - track = _mapping(raw_track, field) - _require_exact_fields(track, _CORPUS_TRACK_FIELDS, field) - track_id = _nonempty_text(track.get("track_id"), f"{field}.track_id") - if track_id in seen_track_ids: - raise ValueError(f"duplicate track_id: {track_id}") - seen_track_ids.add(track_id) - track_ids.append(track_id) - audio_sha256 = _sha256(track.get("audio_sha256"), f"{field}.audio_sha256") - if audio_sha256 in seen_audio_sha256: - raise ValueError(f"duplicate audio_sha256: {audio_sha256}") - seen_audio_sha256.add(audio_sha256) - _sha256(track.get("annotation_sha256"), f"{field}.annotation_sha256") - _nonempty_text(track.get("rights_basis"), f"{field}.rights_basis") - if track.get("rights_cleared") is not True: - raise ValueError(f"{field}.rights_cleared must be true") - _reject_local_path(track.get("source_uri"), f"{field}.source_uri") - return track_ids - - -def _validate_runtime(runtime_value: object) -> None: - """Validate exact runtime identity for reproducible paired measurements.""" - runtime = _mapping(runtime_value, "runtime") - _require_exact_fields(runtime, _RUNTIME_FIELDS, "runtime") - _commit(runtime.get("source_commit"), "runtime.source_commit") - _sha256(runtime.get("uv_lock_sha256"), "runtime.uv_lock_sha256") - for field in ( - "python_version", - "librosa_version", - "numpy_version", - "host_profile", - ): - _nonempty_text(runtime.get(field), f"runtime.{field}") - - sample_rate = _finite_number( - runtime.get("sample_rate_hz"), - "runtime.sample_rate_hz", - ) - if not 1.0 <= sample_rate <= 384000.0 or not sample_rate.is_integer(): - raise ValueError("runtime.sample_rate_hz must be an integer in 1..384000") - channels = runtime.get("channels") - if isinstance(channels, bool) or channels != 1: - raise ValueError( - "runtime.channels must equal 1 for the registered segmentation input" - ) +_BASE = _base_validator() +SCHEMA_VERSION = _BASE.SCHEMA_VERSION +MAX_EVIDENCE_BYTES = _BASE.MAX_EVIDENCE_BYTES def validate_registration(registration_value: object) -> None: - """Validate a frozen STFT-vs-CQT structure experiment registration. - - This function requires content hashes, rights evidence, exact runtime - identity, recognized metric implementations, explicit margins, and a paired - uncertainty procedure. Scientific/product choices must be reviewed before - any corpus result is inspected. - """ - registration = _mapping(registration_value, "registration") - _require_exact_fields(registration, _REGISTRATION_FIELDS, "registration") - _validate_schema_version( - registration.get("schema_version"), - "schema_version", - ) - _nonempty_text(registration.get("experiment_id"), "experiment_id") + """Validate the closed base registration consumed by the metric-aware digest.""" + _BASE.validate_registration(registration_value) - hypothesis = _mapping(registration.get("hypothesis"), "hypothesis") - _require_exact_fields(hypothesis, _HYPOTHESIS_FIELDS, "hypothesis") - baseline = _nonempty_text( - hypothesis.get("baseline_feature"), - "hypothesis.baseline_feature", - ) - candidate = _nonempty_text( - hypothesis.get("candidate_feature"), - "hypothesis.candidate_feature", - ) - if baseline != "chroma_cqt": - raise ValueError("hypothesis.baseline_feature must equal chroma_cqt") - if candidate != "chroma_stft": - raise ValueError("hypothesis.candidate_feature must equal chroma_stft") - _validate_metrics(registration.get("metrics")) - _validate_uncertainty(registration.get("uncertainty"), "uncertainty") - _validate_corpus(registration.get("corpus")) - _validate_runtime(registration.get("runtime")) - _nonempty_text(registration.get("claim_boundary"), "claim_boundary") +def _digest_payload(registration_value: object) -> dict[str, object]: + """Return the complete closed object whose bytes define scientific identity.""" + validate_registration(registration_value) + return { + "registration": registration_value, + "structure_metric_contract": STRUCTURE_METRIC_CONTRACT, + } def registration_digest(registration_value: object) -> str: - """Return the canonical SHA-256 identity of a valid preregistration.""" - validate_registration(registration_value) + """Return SHA-256 over registration data plus exact metric/runtime semantics.""" canonical = json.dumps( - registration_value, + _digest_payload(registration_value), allow_nan=False, ensure_ascii=False, separators=(",", ":"), @@ -458,409 +93,32 @@ def registration_digest(registration_value: object) -> str: return hashlib.sha256(canonical).hexdigest() -def _confidence_interval(value: object, field: str) -> tuple[float, float]: - """Validate and return an ordered two-sided confidence interval.""" - interval = _sequence(value, field) - if len(interval) != 2: - raise ValueError(f"{field} must contain exactly two bounds") - lower = _finite_number(interval[0], f"{field}[0]") - upper = _finite_number(interval[1], f"{field}[1]") - if lower > upper: - raise ValueError(f"{field} lower bound must not exceed upper bound") - return lower, upper - - -def _validate_f_triplet( - normalized: Mapping[str, float], - *, - precision_name: str, - recall_name: str, - f_name: str, - field: str, -) -> None: - """Require reported F to equal the harmonic mean of precision and recall.""" - precision = normalized[precision_name] - recall = normalized[recall_name] - denominator = precision + recall - expected = 0.0 if denominator == 0.0 else (2.0 * precision * recall) / denominator - actual = normalized[f_name] - if not math.isclose(actual, expected, rel_tol=1e-9, abs_tol=1e-12): - raise ValueError( - f"{field}.{f_name} must equal the harmonic mean of " - f"{precision_name} and {recall_name}" - ) - - -def _validate_measurement_side( - side_value: object, - field: str, -) -> dict[str, float]: - """Validate one per-track or aggregate baseline/candidate measurement.""" - side = _mapping(side_value, field) - expected_fields = ( - set(_QUALITY_METRICS) - | set(_REPORT_SCORE_METRICS) - | set(_REPORT_NONNEGATIVE_METRICS) - ) - actual_fields = set(side) - missing_fields = sorted(expected_fields - actual_fields) - extra_fields = sorted(actual_fields - expected_fields) - if missing_fields: - raise ValueError(f"{field} missing required field: {missing_fields[0]}") - if extra_fields: - raise ValueError(f"{field} contains unregistered field: {extra_fields[0]}") - - normalized: dict[str, float] = {} - for metric_name in _QUALITY_METRICS: - normalized[metric_name] = _score( - side.get(metric_name), - f"{field}.{metric_name}", - ) - for metric_name in _REPORT_SCORE_METRICS: - normalized[metric_name] = _score( - side.get(metric_name), - f"{field}.{metric_name}", - ) - for metric_name in _REPORT_NONNEGATIVE_METRICS: - value = _finite_number(side.get(metric_name), f"{field}.{metric_name}") - if value < 0.0: - raise ValueError(f"{field}.{metric_name} must be non-negative") - normalized[metric_name] = value - - _validate_f_triplet( - normalized, - precision_name="boundary_precision_0_5", - recall_name="boundary_recall_0_5", - f_name="boundary_f_0_5", - field=field, - ) - _validate_f_triplet( - normalized, - precision_name="boundary_precision_3_0", - recall_name="boundary_recall_3_0", - f_name="boundary_f_3_0", - field=field, - ) - _validate_f_triplet( - normalized, - precision_name="repetition_pairwise_precision", - recall_name="repetition_pairwise_recall", - f_name="repetition_pairwise_f", - field=field, - ) - - if normalized["p95_latency_seconds"] < normalized["p50_latency_seconds"]: - raise ValueError( - f"{field}.p95_latency_seconds must be >= p50_latency_seconds" - ) - return normalized - - -def _validate_result_identity( - registration: Mapping[str, Any], - result: Mapping[str, Any], -) -> list[str]: - """Bind a result receipt to the frozen registration and corpus order.""" - _require_exact_fields(result, _RESULT_FIELDS, "result") - _validate_schema_version( - result.get("schema_version"), - "result.schema_version", - ) - registered_experiment_id = _nonempty_text( - registration.get("experiment_id"), - "experiment_id", - ) - experiment_id = _nonempty_text( - result.get("experiment_id"), - "result.experiment_id", - ) - if experiment_id != registered_experiment_id: - raise ValueError("result.experiment_id does not match registration") - - expected_digest = registration_digest(registration) - actual_digest = _sha256( - result.get("registration_sha256"), - "result.registration_sha256", - ) - if actual_digest != expected_digest: - raise ValueError( - "result.registration_sha256 does not match the frozen registration" - ) - - registered_uncertainty = _validate_uncertainty( - registration.get("uncertainty"), - "uncertainty", - ) - result_uncertainty = _validate_uncertainty( - result.get("uncertainty"), - "result.uncertainty", - ) - if result_uncertainty != registered_uncertainty: - raise ValueError( - "result.uncertainty must exactly match the preregistered uncertainty plan" - ) - - expected_track_ids = _validate_corpus(registration["corpus"]) - actual_values = _sequence( - result.get("corpus_track_ids"), - "result.corpus_track_ids", - ) - actual_track_ids = [ - _nonempty_text(value, f"result.corpus_track_ids[{index}]") - for index, value in enumerate(actual_values) - ] - if actual_track_ids != expected_track_ids: - raise ValueError( - "result.corpus_track_ids must exactly match registration corpus order" - ) - - registered_claim_boundary = _nonempty_text( - registration.get("claim_boundary"), - "claim_boundary", - ) - result_claim_boundary = _nonempty_text( - result.get("claim_boundary"), - "result.claim_boundary", - ) - if result_claim_boundary != registered_claim_boundary: - raise ValueError( - "result.claim_boundary must exactly match the preregistered claim boundary" - ) - return expected_track_ids - - -def _validate_failed_tracks( - result: Mapping[str, Any], - expected_track_ids: list[str], -) -> list[str]: - """Validate explicit measurement failures against the registered corpus.""" - failed_values = _sequence( - result.get("failed_tracks"), - "result.failed_tracks", - ) - failed_tracks = [ - _nonempty_text(value, f"result.failed_tracks[{index}]") - for index, value in enumerate(failed_values) - ] - if len(set(failed_tracks)) != len(failed_tracks): - raise ValueError("result.failed_tracks must not contain duplicates") - unknown_failed = sorted(set(failed_tracks) - set(expected_track_ids)) - if unknown_failed: - raise ValueError( - f"result.failed_tracks contains unknown track_id: {unknown_failed[0]}" - ) - return failed_tracks - - -def _validate_track_measurements( - result: Mapping[str, Any], - expected_track_ids: list[str], - failed_track_ids: set[str], -) -> None: - """Require one explicit success-or-failure receipt for every corpus track.""" - tracks = _sequence(result.get("tracks"), "result.tracks") - if len(tracks) != len(expected_track_ids): - raise ValueError( - "result.tracks must contain exactly one receipt per corpus track" - ) - - actual_track_ids: list[str] = [] - for index, raw_track in enumerate(tracks): - field = f"result.tracks[{index}]" - track = _mapping(raw_track, field) - track_id = _nonempty_text(track.get("track_id"), f"{field}.track_id") - actual_track_ids.append(track_id) - if track_id in failed_track_ids: - _require_exact_fields(track, _FAILED_TRACK_RECEIPT_FIELDS, field) - continue - _require_exact_fields(track, _TRACK_RECEIPT_FIELDS, field) - _validate_measurement_side(track.get("baseline"), f"{field}.baseline") - _validate_measurement_side(track.get("candidate"), f"{field}.candidate") - if actual_track_ids != expected_track_ids: - raise ValueError("result.tracks must preserve the registered corpus order") - - def evaluate_result( registration_value: object, result_value: object, ) -> dict[str, object]: - """Validate a result receipt and evaluate preregistered CI decisions.""" + """Evaluate a receipt only when it binds the metric-aware registration digest.""" validate_registration(registration_value) - registration = _mapping(registration_value, "registration") - result = _mapping(result_value, "result") - track_ids = _validate_result_identity(registration, result) - failed_tracks = _validate_failed_tracks(result, track_ids) - _validate_track_measurements(result, track_ids, set(failed_tracks)) - - if failed_tracks: - for field in ( - "aggregate", - "paired_delta_ci95", - "p95_latency_ratio_ci95", - ): - if result.get(field) is not None: - raise ValueError( - f"result.{field} must be null when result.failed_tracks is non-empty" - ) - return { - "passed": False, - "failed_requirements": [ - "failed tracks are not permitted without a preregistered exclusion " - "policy: " + ", ".join(failed_tracks) - ], - "registration_sha256": registration_digest(registration), - "failed_tracks": failed_tracks, - } - - aggregate = _mapping(result.get("aggregate"), "result.aggregate") - _require_exact_fields(aggregate, _AGGREGATE_FIELDS, "result.aggregate") - baseline = _validate_measurement_side( - aggregate.get("baseline"), - "result.aggregate.baseline", - ) - candidate = _validate_measurement_side( - aggregate.get("candidate"), - "result.aggregate.candidate", - ) - - raw_intervals = _mapping( - result.get("paired_delta_ci95"), - "result.paired_delta_ci95", - ) - if set(raw_intervals) != set(_QUALITY_METRICS): + if not isinstance(result_value, Mapping): + raise ValueError("result must be an object") + expected_digest = registration_digest(registration_value) + if result_value.get("registration_sha256") != expected_digest: raise ValueError( - "result.paired_delta_ci95 must contain exactly the registered " - "quality metrics" + "result.registration_sha256 does not match the frozen metric-aware registration" ) - intervals: dict[str, tuple[float, float]] = {} - for metric_name in _QUALITY_METRICS: - interval = _confidence_interval( - raw_intervals[metric_name], - f"result.paired_delta_ci95.{metric_name}", - ) - point_delta = candidate[metric_name] - baseline[metric_name] - if not interval[0] <= point_delta <= interval[1]: - raise ValueError( - f"result.paired_delta_ci95.{metric_name} must contain the " - "aggregate point delta" - ) - intervals[metric_name] = interval - - latency_interval = _confidence_interval( - result.get("p95_latency_ratio_ci95"), - "result.p95_latency_ratio_ci95", + projected_result = copy.deepcopy(dict(result_value)) + projected_result["registration_sha256"] = _BASE.registration_digest( + registration_value ) - if latency_interval[0] <= 0.0: - raise ValueError( - "result.p95_latency_ratio_ci95 bounds must be greater than 0" - ) - baseline_p95 = baseline["p95_latency_seconds"] - if baseline_p95 <= 0.0: - raise ValueError( - "result.aggregate.baseline.p95_latency_seconds must be greater than 0" - ) - latency_ratio = candidate["p95_latency_seconds"] / baseline_p95 - if not latency_interval[0] <= latency_ratio <= latency_interval[1]: - raise ValueError( - "result.p95_latency_ratio_ci95 must contain the aggregate p95 ratio" - ) + decision = _BASE.evaluate_result(registration_value, projected_result) + decision["registration_sha256"] = expected_digest + return decision - metrics = _mapping(registration["metrics"], "metrics") - failed_requirements: list[str] = [] - for metric_name in _QUALITY_METRICS: - config = _mapping(metrics[metric_name], f"metrics.{metric_name}") - margin = _finite_number( - config.get("noninferiority_margin"), - f"metrics.{metric_name}.noninferiority_margin", - ) - lower = intervals[metric_name][0] - if lower < -margin: - failed_requirements.append( - f"{metric_name} paired CI lower bound {lower:.6f} " - f"is below -{margin:.6f}" - ) - latency_config = _mapping( - metrics[_LATENCY_METRIC], - f"metrics.{_LATENCY_METRIC}", - ) - maximum_ratio = _finite_number( - latency_config.get("maximum_candidate_ratio"), - f"metrics.{_LATENCY_METRIC}.maximum_candidate_ratio", - ) - latency_upper = latency_interval[1] - if latency_upper > maximum_ratio: - failed_requirements.append( - "p95 latency ratio CI upper bound " - f"{latency_upper:.6f} exceeds {maximum_ratio:.6f}" - ) - - return { - "passed": not failed_requirements, - "failed_requirements": failed_requirements, - "registration_sha256": registration_digest(registration), - "failed_tracks": failed_tracks, - } - - -def _reject_duplicate_object_pairs(pairs: list[tuple[str, Any]]) -> dict[str, Any]: - """Reject duplicate JSON object keys instead of accepting last-key-wins.""" - result: dict[str, Any] = {} - for key, value in pairs: - if key in result: - raise ValueError(f"duplicate JSON key: {key}") - result[key] = value - return result - - -def _reject_json_constant(value: str) -> None: - """Reject non-standard JSON NaN/Infinity constants at the parser boundary.""" - raise ValueError(f"invalid JSON constant: {value}") - - -def _load_json(path: Path) -> object: - """Load one bounded regular non-link UTF-8 JSON file from one descriptor.""" - flags = os.O_RDONLY - if hasattr(os, "O_BINARY"): - flags |= os.O_BINARY - if hasattr(os, "O_NOFOLLOW"): - flags |= os.O_NOFOLLOW - try: - fd = os.open(path, flags) - except OSError as exc: - raise ValueError( - f"evidence path could not be opened as a regular non-link file: {path.name}" - ) from exc - - with os.fdopen(fd, "rb", closefd=True) as handle: - descriptor_stat = os.fstat(handle.fileno()) - if not stat.S_ISREG(descriptor_stat.st_mode): - raise ValueError(f"evidence path is not a regular file: {path.name}") - if not hasattr(os, "O_NOFOLLOW") and path.is_symlink(): - raise ValueError(f"evidence path must not be a symbolic link: {path.name}") - if descriptor_stat.st_size > MAX_EVIDENCE_BYTES: - raise ValueError( - f"evidence file exceeds {MAX_EVIDENCE_BYTES} bytes: {path.name}" - ) - payload = handle.read(MAX_EVIDENCE_BYTES + 1) - - if len(payload) > MAX_EVIDENCE_BYTES: - raise ValueError( - f"evidence file exceeds {MAX_EVIDENCE_BYTES} bytes: {path.name}" - ) - try: - text = payload.decode("utf-8") - except UnicodeDecodeError as exc: - raise ValueError(f"evidence file is not valid UTF-8: {path.name}") from exc - try: - return json.loads( - text, - object_pairs_hook=_reject_duplicate_object_pairs, - parse_constant=_reject_json_constant, - ) - except json.JSONDecodeError as exc: - raise ValueError(f"evidence file is not valid JSON: {path.name}") from exc +def __getattr__(name: str) -> Any: + """Delegate unchanged schema helpers to the base policy implementation.""" + return getattr(_BASE, name) def main(argv: Sequence[str] | None = None) -> int: @@ -870,13 +128,13 @@ def main(argv: Sequence[str] | None = None) -> int: parser.add_argument("result", type=Path, nargs="?") args = parser.parse_args(argv) - registration = _load_json(args.registration) + registration = _BASE._load_json(args.registration) digest = registration_digest(registration) if args.result is None: print(json.dumps({"registration_sha256": digest}, sort_keys=True)) return 0 - result = _load_json(args.result) + result = _BASE._load_json(args.result) decision = evaluate_result(registration, result) print(json.dumps(decision, sort_keys=True)) return 0 if decision["passed"] is True else 1 diff --git a/scripts/research/validate_structure_noninferiority_base.py b/scripts/research/validate_structure_noninferiority_base.py new file mode 100644 index 000000000..f3f344164 --- /dev/null +++ b/scripts/research/validate_structure_noninferiority_base.py @@ -0,0 +1,886 @@ +#!/usr/bin/env python3 +"""Validate preregistered structure-feature noninferiority evidence. + +This module does not compute MIR metrics. It binds a reviewed registration to +result receipts produced by the recognized evaluation pipeline, then evaluates +the preregistered confidence-interval decision rules. Metric implementation +authority remains with MIREX/mir_eval-compatible tooling while thresholds, +corpus identity, uncertainty procedure, and runtime identity are protected from +post-result drift. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +import os +import re +import stat +from collections.abc import Mapping, Sequence +from pathlib import Path +from typing import Any + +SCHEMA_VERSION = 1 +MAX_EVIDENCE_BYTES = 2 * 1024 * 1024 +_SHA256_RE = re.compile(r"^[0-9a-fA-F]{64}$") +_COMMIT_RE = re.compile(r"^[0-9a-fA-F]{40}$") +_WINDOWS_DRIVE_RE = re.compile(r"^[A-Za-z]:") +_URI_SCHEME_RE = re.compile(r"^[A-Za-z][A-Za-z0-9+.-]*:") + +_QUALITY_METRICS: dict[str, dict[str, float | str]] = { + "boundary_f_0_5": { + "implementation": "mir_eval.segment.detection", + "window_seconds": 0.5, + }, + "boundary_f_3_0": { + "implementation": "mir_eval.segment.detection", + "window_seconds": 3.0, + }, + "functional_label_accuracy": { + "implementation": ( + "ismir-mirex/mirex-evaluation@" + "b9fa0b0b32e2145af31f35830f78fc9d09a4301b:" + "music_structure_analysis.eval_script.calculate_accuracy" + ), + "frame_size_seconds": 0.2, + "frame_grid_contract_version": 1.0, + "annotation_contract_version": 1.0, + "label_mapping_contract_version": 1.0, + }, + "repetition_pairwise_f": { + "implementation": "mir_eval.segment.pairwise", + "frame_size_seconds": 0.1, + }, +} +_LATENCY_METRIC = "p95_latency_ratio" +_REPORT_SCORE_METRICS = ( + "boundary_precision_0_5", + "boundary_recall_0_5", + "boundary_precision_3_0", + "boundary_recall_3_0", + "repetition_pairwise_precision", + "repetition_pairwise_recall", +) +_REPORT_NONNEGATIVE_METRICS = ( + "reference_to_estimate_median_deviation_seconds", + "estimate_to_reference_median_deviation_seconds", + "p50_latency_seconds", + "p95_latency_seconds", + "peak_rss_mib", +) +_REGISTRATION_FIELDS = { + "schema_version", + "experiment_id", + "hypothesis", + "metrics", + "uncertainty", + "corpus", + "runtime", + "claim_boundary", +} +_HYPOTHESIS_FIELDS = {"baseline_feature", "candidate_feature"} +_CORPUS_TRACK_FIELDS = { + "track_id", + "audio_sha256", + "annotation_sha256", + "rights_basis", + "rights_cleared", + "source_uri", +} +_RUNTIME_FIELDS = { + "source_commit", + "uv_lock_sha256", + "python_version", + "librosa_version", + "numpy_version", + "sample_rate_hz", + "channels", + "host_profile", +} +_RESULT_FIELDS = { + "schema_version", + "experiment_id", + "registration_sha256", + "uncertainty", + "corpus_track_ids", + "tracks", + "aggregate", + "paired_delta_ci95", + "p95_latency_ratio_ci95", + "failed_tracks", + "claim_boundary", +} +_TRACK_RECEIPT_FIELDS = {"track_id", "baseline", "candidate"} +_FAILED_TRACK_RECEIPT_FIELDS = {"track_id"} +_AGGREGATE_FIELDS = {"baseline", "candidate"} + + +def _mapping(value: object, field: str) -> Mapping[str, Any]: + """Return ``value`` as a mapping or raise a field-specific error.""" + if not isinstance(value, Mapping): + raise ValueError(f"{field} must be an object") + return value + + +def _require_exact_fields( + value: Mapping[str, Any], + expected: set[str], + field: str, +) -> None: + """Require an evidence object to use exactly its schema-v1 field set.""" + actual = set(value) + missing = sorted(expected - actual) + extra = sorted(actual - expected) + if missing: + raise ValueError(f"{field} missing required field: {missing[0]}") + if extra: + raise ValueError(f"{field} contains unregistered field: {extra[0]}") + + +def _sequence(value: object, field: str) -> Sequence[Any]: + """Return a non-string sequence or raise a field-specific error.""" + if isinstance(value, (str, bytes)) or not isinstance(value, Sequence): + raise ValueError(f"{field} must be an array") + return value + + +def _nonempty_text(value: object, field: str) -> str: + """Return stripped non-empty text for a required field.""" + if not isinstance(value, str) or not value.strip(): + raise ValueError(f"{field} must be non-empty text") + return value.strip() + + +def _finite_number(value: object, field: str) -> float: + """Return a finite real number while rejecting booleans and NaN/Inf.""" + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise ValueError(f"{field} must be a finite number") + number = float(value) + if not math.isfinite(number): + raise ValueError(f"{field} must be finite") + return number + + +def _bounded_integer( + value: object, + field: str, + *, + minimum: int, + maximum: int, +) -> int: + """Return a bounded integer while rejecting bools and fractional values.""" + if isinstance(value, bool) or not isinstance(value, int): + raise ValueError(f"{field} must be an integer") + if not minimum <= value <= maximum: + raise ValueError(f"{field} must be in {minimum}..{maximum}") + return value + + +def _score(value: object, field: str) -> float: + """Return a finite score constrained to the inclusive 0..1 interval.""" + number = _finite_number(value, field) + if not 0.0 <= number <= 1.0: + raise ValueError(f"{field} must be between 0 and 1") + return number + + +def _sha256(value: object, field: str) -> str: + """Return a normalized SHA-256 hex digest.""" + text = _nonempty_text(value, field) + if _SHA256_RE.fullmatch(text) is None: + raise ValueError(f"{field} must be a 64-character SHA-256 hex digest") + return text.lower() + + +def _commit(value: object, field: str) -> str: + """Return a normalized full Git commit SHA.""" + text = _nonempty_text(value, field) + if _COMMIT_RE.fullmatch(text) is None: + raise ValueError(f"{field} must be a full 40-character Git commit SHA") + return text.lower() + + +def _reject_local_path(value: object, field: str) -> str: + """Require an explicit non-file URI for corpus provenance receipts.""" + text = _nonempty_text(value, field) + lowered = text.casefold() + scheme = _URI_SCHEME_RE.match(text) + if ( + text.startswith(("/", "\\\\")) + or _WINDOWS_DRIVE_RE.match(text) is not None + or lowered.startswith("file:") + or scheme is None + or scheme.end() == len(text) + ): + raise ValueError( + f"{field} must be a provenance URI, not a local filesystem path" + ) + return text + + +def _validate_schema_version(value: object, field: str) -> None: + """Require the one currently supported evidence schema version.""" + if isinstance(value, bool) or value != SCHEMA_VERSION: + raise ValueError(f"{field} must equal {SCHEMA_VERSION}") + + +def _validate_metrics(metrics_value: object) -> None: + """Validate the complete preregistered metric and decision contract.""" + metrics = _mapping(metrics_value, "metrics") + expected = set(_QUALITY_METRICS) | {_LATENCY_METRIC} + actual = set(metrics) + missing = sorted(expected - actual) + extra = sorted(actual - expected) + if missing: + raise ValueError(f"metrics missing required metric: {missing[0]}") + if extra: + raise ValueError(f"metrics contains unregistered metric: {extra[0]}") + + for metric_name, expected_config in _QUALITY_METRICS.items(): + config = _mapping(metrics[metric_name], f"metrics.{metric_name}") + _require_exact_fields( + config, + set(expected_config) | {"noninferiority_margin"}, + f"metrics.{metric_name}", + ) + implementation = _nonempty_text( + config.get("implementation"), + f"metrics.{metric_name}.implementation", + ) + if implementation != expected_config["implementation"]: + raise ValueError( + f"metrics.{metric_name}.implementation must equal " + f"{expected_config['implementation']}" + ) + for config_name, expected_value in expected_config.items(): + if config_name == "implementation": + continue + actual_value = _finite_number( + config.get(config_name), + f"metrics.{metric_name}.{config_name}", + ) + expected_number = float(expected_value) + if not math.isclose( + actual_value, + expected_number, + rel_tol=0.0, + abs_tol=1e-12, + ): + raise ValueError( + f"metrics.{metric_name}.{config_name} must equal " + f"{expected_value}" + ) + margin = _finite_number( + config.get("noninferiority_margin"), + f"metrics.{metric_name}.noninferiority_margin", + ) + if not 0.0 <= margin < 1.0: + raise ValueError( + f"metrics.{metric_name}.noninferiority_margin must be in [0, 1)" + ) + + latency = _mapping(metrics[_LATENCY_METRIC], f"metrics.{_LATENCY_METRIC}") + _require_exact_fields( + latency, + {"maximum_candidate_ratio"}, + f"metrics.{_LATENCY_METRIC}", + ) + maximum_ratio = _finite_number( + latency.get("maximum_candidate_ratio"), + f"metrics.{_LATENCY_METRIC}.maximum_candidate_ratio", + ) + if not 0.0 < maximum_ratio < 1.0: + raise ValueError( + "metrics.p95_latency_ratio.maximum_candidate_ratio must be greater " + "than 0 and less than 1" + ) + + +def _validate_uncertainty(uncertainty_value: object, field: str) -> dict[str, object]: + """Validate and normalize the preregistered paired-uncertainty procedure.""" + uncertainty = _mapping(uncertainty_value, field) + required = { + "procedure_id", + "confidence_level", + "resamples", + "random_seed", + } + missing = sorted(required - set(uncertainty)) + extra = sorted(set(uncertainty) - required) + if missing: + raise ValueError(f"{field} missing required field: {missing[0]}") + if extra: + raise ValueError(f"{field} contains unregistered field: {extra[0]}") + + procedure_id = _nonempty_text( + uncertainty.get("procedure_id"), + f"{field}.procedure_id", + ) + if len(procedure_id) > 128: + raise ValueError(f"{field}.procedure_id must be at most 128 characters") + + confidence_level = _finite_number( + uncertainty.get("confidence_level"), + f"{field}.confidence_level", + ) + if not math.isclose(confidence_level, 0.95, rel_tol=0.0, abs_tol=1e-12): + raise ValueError(f"{field}.confidence_level must equal 0.95") + + resamples = _bounded_integer( + uncertainty.get("resamples"), + f"{field}.resamples", + minimum=1000, + maximum=10_000_000, + ) + random_seed = _bounded_integer( + uncertainty.get("random_seed"), + f"{field}.random_seed", + minimum=0, + maximum=(2**32) - 1, + ) + return { + "procedure_id": procedure_id, + "confidence_level": confidence_level, + "resamples": resamples, + "random_seed": random_seed, + } + + +def _validate_corpus(corpus_value: object) -> list[str]: + """Validate rights-cleared real-audio identities and return ordered IDs.""" + corpus = _sequence(corpus_value, "corpus") + if len(corpus) < 2: + raise ValueError( + "corpus must contain at least two rights-cleared real-audio tracks" + ) + + seen_track_ids: set[str] = set() + seen_audio_sha256: set[str] = set() + track_ids: list[str] = [] + for index, raw_track in enumerate(corpus): + field = f"corpus[{index}]" + track = _mapping(raw_track, field) + _require_exact_fields(track, _CORPUS_TRACK_FIELDS, field) + track_id = _nonempty_text(track.get("track_id"), f"{field}.track_id") + if track_id in seen_track_ids: + raise ValueError(f"duplicate track_id: {track_id}") + seen_track_ids.add(track_id) + track_ids.append(track_id) + audio_sha256 = _sha256(track.get("audio_sha256"), f"{field}.audio_sha256") + if audio_sha256 in seen_audio_sha256: + raise ValueError(f"duplicate audio_sha256: {audio_sha256}") + seen_audio_sha256.add(audio_sha256) + _sha256(track.get("annotation_sha256"), f"{field}.annotation_sha256") + _nonempty_text(track.get("rights_basis"), f"{field}.rights_basis") + if track.get("rights_cleared") is not True: + raise ValueError(f"{field}.rights_cleared must be true") + _reject_local_path(track.get("source_uri"), f"{field}.source_uri") + return track_ids + + +def _validate_runtime(runtime_value: object) -> None: + """Validate exact runtime identity for reproducible paired measurements.""" + runtime = _mapping(runtime_value, "runtime") + _require_exact_fields(runtime, _RUNTIME_FIELDS, "runtime") + _commit(runtime.get("source_commit"), "runtime.source_commit") + _sha256(runtime.get("uv_lock_sha256"), "runtime.uv_lock_sha256") + for field in ( + "python_version", + "librosa_version", + "numpy_version", + "host_profile", + ): + _nonempty_text(runtime.get(field), f"runtime.{field}") + + sample_rate = _finite_number( + runtime.get("sample_rate_hz"), + "runtime.sample_rate_hz", + ) + if not 1.0 <= sample_rate <= 384000.0 or not sample_rate.is_integer(): + raise ValueError("runtime.sample_rate_hz must be an integer in 1..384000") + channels = runtime.get("channels") + if isinstance(channels, bool) or channels != 1: + raise ValueError( + "runtime.channels must equal 1 for the registered segmentation input" + ) + + +def validate_registration(registration_value: object) -> None: + """Validate a frozen STFT-vs-CQT structure experiment registration. + + This function requires content hashes, rights evidence, exact runtime + identity, recognized metric implementations, explicit margins, and a paired + uncertainty procedure. Scientific/product choices must be reviewed before + any corpus result is inspected. + """ + registration = _mapping(registration_value, "registration") + _require_exact_fields(registration, _REGISTRATION_FIELDS, "registration") + _validate_schema_version( + registration.get("schema_version"), + "schema_version", + ) + _nonempty_text(registration.get("experiment_id"), "experiment_id") + + hypothesis = _mapping(registration.get("hypothesis"), "hypothesis") + _require_exact_fields(hypothesis, _HYPOTHESIS_FIELDS, "hypothesis") + baseline = _nonempty_text( + hypothesis.get("baseline_feature"), + "hypothesis.baseline_feature", + ) + candidate = _nonempty_text( + hypothesis.get("candidate_feature"), + "hypothesis.candidate_feature", + ) + if baseline != "chroma_cqt": + raise ValueError("hypothesis.baseline_feature must equal chroma_cqt") + if candidate != "chroma_stft": + raise ValueError("hypothesis.candidate_feature must equal chroma_stft") + + _validate_metrics(registration.get("metrics")) + _validate_uncertainty(registration.get("uncertainty"), "uncertainty") + _validate_corpus(registration.get("corpus")) + _validate_runtime(registration.get("runtime")) + _nonempty_text(registration.get("claim_boundary"), "claim_boundary") + + +def registration_digest(registration_value: object) -> str: + """Return the canonical SHA-256 identity of a valid preregistration.""" + validate_registration(registration_value) + canonical = json.dumps( + registration_value, + allow_nan=False, + ensure_ascii=False, + separators=(",", ":"), + sort_keys=True, + ).encode("utf-8") + return hashlib.sha256(canonical).hexdigest() + + +def _confidence_interval(value: object, field: str) -> tuple[float, float]: + """Validate and return an ordered two-sided confidence interval.""" + interval = _sequence(value, field) + if len(interval) != 2: + raise ValueError(f"{field} must contain exactly two bounds") + lower = _finite_number(interval[0], f"{field}[0]") + upper = _finite_number(interval[1], f"{field}[1]") + if lower > upper: + raise ValueError(f"{field} lower bound must not exceed upper bound") + return lower, upper + + +def _validate_f_triplet( + normalized: Mapping[str, float], + *, + precision_name: str, + recall_name: str, + f_name: str, + field: str, +) -> None: + """Require reported F to equal the harmonic mean of precision and recall.""" + precision = normalized[precision_name] + recall = normalized[recall_name] + denominator = precision + recall + expected = 0.0 if denominator == 0.0 else (2.0 * precision * recall) / denominator + actual = normalized[f_name] + if not math.isclose(actual, expected, rel_tol=1e-9, abs_tol=1e-12): + raise ValueError( + f"{field}.{f_name} must equal the harmonic mean of " + f"{precision_name} and {recall_name}" + ) + + +def _validate_measurement_side( + side_value: object, + field: str, +) -> dict[str, float]: + """Validate one per-track or aggregate baseline/candidate measurement.""" + side = _mapping(side_value, field) + expected_fields = ( + set(_QUALITY_METRICS) + | set(_REPORT_SCORE_METRICS) + | set(_REPORT_NONNEGATIVE_METRICS) + ) + actual_fields = set(side) + missing_fields = sorted(expected_fields - actual_fields) + extra_fields = sorted(actual_fields - expected_fields) + if missing_fields: + raise ValueError(f"{field} missing required field: {missing_fields[0]}") + if extra_fields: + raise ValueError(f"{field} contains unregistered field: {extra_fields[0]}") + + normalized: dict[str, float] = {} + for metric_name in _QUALITY_METRICS: + normalized[metric_name] = _score( + side.get(metric_name), + f"{field}.{metric_name}", + ) + for metric_name in _REPORT_SCORE_METRICS: + normalized[metric_name] = _score( + side.get(metric_name), + f"{field}.{metric_name}", + ) + for metric_name in _REPORT_NONNEGATIVE_METRICS: + value = _finite_number(side.get(metric_name), f"{field}.{metric_name}") + if value < 0.0: + raise ValueError(f"{field}.{metric_name} must be non-negative") + normalized[metric_name] = value + + _validate_f_triplet( + normalized, + precision_name="boundary_precision_0_5", + recall_name="boundary_recall_0_5", + f_name="boundary_f_0_5", + field=field, + ) + _validate_f_triplet( + normalized, + precision_name="boundary_precision_3_0", + recall_name="boundary_recall_3_0", + f_name="boundary_f_3_0", + field=field, + ) + _validate_f_triplet( + normalized, + precision_name="repetition_pairwise_precision", + recall_name="repetition_pairwise_recall", + f_name="repetition_pairwise_f", + field=field, + ) + + if normalized["p95_latency_seconds"] < normalized["p50_latency_seconds"]: + raise ValueError( + f"{field}.p95_latency_seconds must be >= p50_latency_seconds" + ) + return normalized + + +def _validate_result_identity( + registration: Mapping[str, Any], + result: Mapping[str, Any], +) -> list[str]: + """Bind a result receipt to the frozen registration and corpus order.""" + _require_exact_fields(result, _RESULT_FIELDS, "result") + _validate_schema_version( + result.get("schema_version"), + "result.schema_version", + ) + registered_experiment_id = _nonempty_text( + registration.get("experiment_id"), + "experiment_id", + ) + experiment_id = _nonempty_text( + result.get("experiment_id"), + "result.experiment_id", + ) + if experiment_id != registered_experiment_id: + raise ValueError("result.experiment_id does not match registration") + + expected_digest = registration_digest(registration) + actual_digest = _sha256( + result.get("registration_sha256"), + "result.registration_sha256", + ) + if actual_digest != expected_digest: + raise ValueError( + "result.registration_sha256 does not match the frozen registration" + ) + + registered_uncertainty = _validate_uncertainty( + registration.get("uncertainty"), + "uncertainty", + ) + result_uncertainty = _validate_uncertainty( + result.get("uncertainty"), + "result.uncertainty", + ) + if result_uncertainty != registered_uncertainty: + raise ValueError( + "result.uncertainty must exactly match the preregistered uncertainty plan" + ) + + expected_track_ids = _validate_corpus(registration["corpus"]) + actual_values = _sequence( + result.get("corpus_track_ids"), + "result.corpus_track_ids", + ) + actual_track_ids = [ + _nonempty_text(value, f"result.corpus_track_ids[{index}]") + for index, value in enumerate(actual_values) + ] + if actual_track_ids != expected_track_ids: + raise ValueError( + "result.corpus_track_ids must exactly match registration corpus order" + ) + + registered_claim_boundary = _nonempty_text( + registration.get("claim_boundary"), + "claim_boundary", + ) + result_claim_boundary = _nonempty_text( + result.get("claim_boundary"), + "result.claim_boundary", + ) + if result_claim_boundary != registered_claim_boundary: + raise ValueError( + "result.claim_boundary must exactly match the preregistered claim boundary" + ) + return expected_track_ids + + +def _validate_failed_tracks( + result: Mapping[str, Any], + expected_track_ids: list[str], +) -> list[str]: + """Validate explicit measurement failures against the registered corpus.""" + failed_values = _sequence( + result.get("failed_tracks"), + "result.failed_tracks", + ) + failed_tracks = [ + _nonempty_text(value, f"result.failed_tracks[{index}]") + for index, value in enumerate(failed_values) + ] + if len(set(failed_tracks)) != len(failed_tracks): + raise ValueError("result.failed_tracks must not contain duplicates") + unknown_failed = sorted(set(failed_tracks) - set(expected_track_ids)) + if unknown_failed: + raise ValueError( + f"result.failed_tracks contains unknown track_id: {unknown_failed[0]}" + ) + return failed_tracks + + +def _validate_track_measurements( + result: Mapping[str, Any], + expected_track_ids: list[str], + failed_track_ids: set[str], +) -> None: + """Require one explicit success-or-failure receipt for every corpus track.""" + tracks = _sequence(result.get("tracks"), "result.tracks") + if len(tracks) != len(expected_track_ids): + raise ValueError( + "result.tracks must contain exactly one receipt per corpus track" + ) + + actual_track_ids: list[str] = [] + for index, raw_track in enumerate(tracks): + field = f"result.tracks[{index}]" + track = _mapping(raw_track, field) + track_id = _nonempty_text(track.get("track_id"), f"{field}.track_id") + actual_track_ids.append(track_id) + if track_id in failed_track_ids: + _require_exact_fields(track, _FAILED_TRACK_RECEIPT_FIELDS, field) + continue + _require_exact_fields(track, _TRACK_RECEIPT_FIELDS, field) + _validate_measurement_side(track.get("baseline"), f"{field}.baseline") + _validate_measurement_side(track.get("candidate"), f"{field}.candidate") + if actual_track_ids != expected_track_ids: + raise ValueError("result.tracks must preserve the registered corpus order") + + +def evaluate_result( + registration_value: object, + result_value: object, +) -> dict[str, object]: + """Validate a result receipt and evaluate preregistered CI decisions.""" + validate_registration(registration_value) + registration = _mapping(registration_value, "registration") + result = _mapping(result_value, "result") + track_ids = _validate_result_identity(registration, result) + failed_tracks = _validate_failed_tracks(result, track_ids) + _validate_track_measurements(result, track_ids, set(failed_tracks)) + + if failed_tracks: + for field in ( + "aggregate", + "paired_delta_ci95", + "p95_latency_ratio_ci95", + ): + if result.get(field) is not None: + raise ValueError( + f"result.{field} must be null when result.failed_tracks is non-empty" + ) + return { + "passed": False, + "failed_requirements": [ + "failed tracks are not permitted without a preregistered exclusion " + "policy: " + ", ".join(failed_tracks) + ], + "registration_sha256": registration_digest(registration), + "failed_tracks": failed_tracks, + } + + aggregate = _mapping(result.get("aggregate"), "result.aggregate") + _require_exact_fields(aggregate, _AGGREGATE_FIELDS, "result.aggregate") + baseline = _validate_measurement_side( + aggregate.get("baseline"), + "result.aggregate.baseline", + ) + candidate = _validate_measurement_side( + aggregate.get("candidate"), + "result.aggregate.candidate", + ) + + raw_intervals = _mapping( + result.get("paired_delta_ci95"), + "result.paired_delta_ci95", + ) + if set(raw_intervals) != set(_QUALITY_METRICS): + raise ValueError( + "result.paired_delta_ci95 must contain exactly the registered " + "quality metrics" + ) + + intervals: dict[str, tuple[float, float]] = {} + for metric_name in _QUALITY_METRICS: + interval = _confidence_interval( + raw_intervals[metric_name], + f"result.paired_delta_ci95.{metric_name}", + ) + point_delta = candidate[metric_name] - baseline[metric_name] + if not interval[0] <= point_delta <= interval[1]: + raise ValueError( + f"result.paired_delta_ci95.{metric_name} must contain the " + "aggregate point delta" + ) + intervals[metric_name] = interval + + latency_interval = _confidence_interval( + result.get("p95_latency_ratio_ci95"), + "result.p95_latency_ratio_ci95", + ) + if latency_interval[0] <= 0.0: + raise ValueError( + "result.p95_latency_ratio_ci95 bounds must be greater than 0" + ) + baseline_p95 = baseline["p95_latency_seconds"] + if baseline_p95 <= 0.0: + raise ValueError( + "result.aggregate.baseline.p95_latency_seconds must be greater than 0" + ) + latency_ratio = candidate["p95_latency_seconds"] / baseline_p95 + if not latency_interval[0] <= latency_ratio <= latency_interval[1]: + raise ValueError( + "result.p95_latency_ratio_ci95 must contain the aggregate p95 ratio" + ) + + metrics = _mapping(registration["metrics"], "metrics") + failed_requirements: list[str] = [] + for metric_name in _QUALITY_METRICS: + config = _mapping(metrics[metric_name], f"metrics.{metric_name}") + margin = _finite_number( + config.get("noninferiority_margin"), + f"metrics.{metric_name}.noninferiority_margin", + ) + lower = intervals[metric_name][0] + if lower < -margin: + failed_requirements.append( + f"{metric_name} paired CI lower bound {lower:.6f} " + f"is below -{margin:.6f}" + ) + + latency_config = _mapping( + metrics[_LATENCY_METRIC], + f"metrics.{_LATENCY_METRIC}", + ) + maximum_ratio = _finite_number( + latency_config.get("maximum_candidate_ratio"), + f"metrics.{_LATENCY_METRIC}.maximum_candidate_ratio", + ) + latency_upper = latency_interval[1] + if latency_upper > maximum_ratio: + failed_requirements.append( + "p95 latency ratio CI upper bound " + f"{latency_upper:.6f} exceeds {maximum_ratio:.6f}" + ) + + return { + "passed": not failed_requirements, + "failed_requirements": failed_requirements, + "registration_sha256": registration_digest(registration), + "failed_tracks": failed_tracks, + } + + +def _reject_duplicate_object_pairs(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + """Reject duplicate JSON object keys instead of accepting last-key-wins.""" + result: dict[str, Any] = {} + for key, value in pairs: + if key in result: + raise ValueError(f"duplicate JSON key: {key}") + result[key] = value + return result + + +def _reject_json_constant(value: str) -> None: + """Reject non-standard JSON NaN/Infinity constants at the parser boundary.""" + raise ValueError(f"invalid JSON constant: {value}") + + +def _load_json(path: Path) -> object: + """Load one bounded regular non-link UTF-8 JSON file from one descriptor.""" + flags = os.O_RDONLY + if hasattr(os, "O_BINARY"): + flags |= os.O_BINARY + if hasattr(os, "O_NOFOLLOW"): + flags |= os.O_NOFOLLOW + try: + fd = os.open(path, flags) + except OSError as exc: + raise ValueError( + f"evidence path could not be opened as a regular non-link file: {path.name}" + ) from exc + + with os.fdopen(fd, "rb", closefd=True) as handle: + descriptor_stat = os.fstat(handle.fileno()) + if not stat.S_ISREG(descriptor_stat.st_mode): + raise ValueError(f"evidence path is not a regular file: {path.name}") + if not hasattr(os, "O_NOFOLLOW") and path.is_symlink(): + raise ValueError(f"evidence path must not be a symbolic link: {path.name}") + if descriptor_stat.st_size > MAX_EVIDENCE_BYTES: + raise ValueError( + f"evidence file exceeds {MAX_EVIDENCE_BYTES} bytes: {path.name}" + ) + payload = handle.read(MAX_EVIDENCE_BYTES + 1) + + if len(payload) > MAX_EVIDENCE_BYTES: + raise ValueError( + f"evidence file exceeds {MAX_EVIDENCE_BYTES} bytes: {path.name}" + ) + try: + text = payload.decode("utf-8") + except UnicodeDecodeError as exc: + raise ValueError(f"evidence file is not valid UTF-8: {path.name}") from exc + try: + return json.loads( + text, + object_pairs_hook=_reject_duplicate_object_pairs, + parse_constant=_reject_json_constant, + ) + except json.JSONDecodeError as exc: + raise ValueError(f"evidence file is not valid JSON: {path.name}") from exc + + +def main(argv: Sequence[str] | None = None) -> int: + """Validate a registration and optionally evaluate one bound result.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("registration", type=Path) + parser.add_argument("result", type=Path, nargs="?") + args = parser.parse_args(argv) + + registration = _load_json(args.registration) + digest = registration_digest(registration) + if args.result is None: + print(json.dumps({"registration_sha256": digest}, sort_keys=True)) + return 0 + + result = _load_json(args.result) + decision = evaluate_result(registration, result) + print(json.dumps(decision, sort_keys=True)) + return 0 if decision["passed"] is True else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) From f961390f6241fa3c3d4b49393548806ad993eda3 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:15:42 +0900 Subject: [PATCH 159/216] test(mir): bind canonical digest to exact metric contract --- ...ructure_metric_preregistration_contract.py | 143 +++++++++--------- 1 file changed, 74 insertions(+), 69 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py b/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py index a382d3678..05b5b7b60 100644 --- a/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py +++ b/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py @@ -2,87 +2,92 @@ from __future__ import annotations -import copy +import hashlib +import json from types import ModuleType import pytest from conftest import load_module -from test_structure_noninferiority_policy import _metrics, _registration +from test_structure_noninferiority_policy import _registration, _result -_METRIC_LOCK_SHA256 = "16fd203e9c987064afc667282e001b755838ff0b484235c6932f557d2ae389f8" +_EXPECTED_METRIC_CONTRACT = { + "runtime_lock_sha256": ( + "16fd203e9c987064afc667282e001b755838ff0b484235c6932f557d2ae389f8" + ), + "boundary_f_0_5": { + "implementation": "mir_eval.segment.detection", + "window_seconds": 0.5, + "beta": 1.0, + "trim": True, + }, + "boundary_f_3_0": { + "implementation": "mir_eval.segment.detection", + "window_seconds": 3.0, + "beta": 1.0, + "trim": True, + }, + "boundary_deviation": { + "implementation": "mir_eval.segment.deviation", + "trim": True, + }, + "repetition_pairwise_f": { + "implementation": "mir_eval.segment.pairwise", + "frame_size_seconds": 0.1, + "beta": 1.0, + }, +} -def _metric_validator() -> ModuleType: +def _validator() -> ModuleType: return load_module( - "scripts/research/validate_structure_metric_preregistration.py", - "validate_structure_metric_preregistration", + "scripts/research/validate_structure_noninferiority.py", + "validate_structure_noninferiority_metric_contract", ) -def _exact_registration() -> dict[str, object]: - """Return the intended closed metric-runtime and adapter contract.""" +def _canonical_digest(value: object) -> str: + payload = json.dumps( + value, + allow_nan=False, + ensure_ascii=False, + separators=(",", ":"), + sort_keys=True, + ).encode("utf-8") + return hashlib.sha256(payload).hexdigest() + + +def test_registration_digest_binds_exact_metric_runtime_and_adapter_semantics() -> None: + """Scientific identity includes every reviewed mir_eval argument and lock digest.""" + validator = _validator() registration = _registration() - registration["metric_runtime"] = {"lock_sha256": _METRIC_LOCK_SHA256} - registration["report_metrics"] = { - "boundary_deviation": { - "implementation": "mir_eval.segment.deviation", - "trim": True, - } - } - metrics = _metrics(registration) - boundary_05 = metrics["boundary_f_0_5"] - boundary_30 = metrics["boundary_f_3_0"] - pairwise = metrics["repetition_pairwise_f"] - assert isinstance(boundary_05, dict) - assert isinstance(boundary_30, dict) - assert isinstance(pairwise, dict) - boundary_05.update({"beta": 1.0, "trim": True}) - boundary_30.update({"beta": 1.0, "trim": True}) - pairwise["beta"] = 1.0 - return registration - - -def test_registration_accepts_only_exact_metric_runtime_and_adapter_semantics() -> None: - """The digest must bind the exact lock and every result-affecting adapter argument.""" - validator = _metric_validator() - registration = _exact_registration() validator.validate_registration(registration) - expected_digest = validator.registration_digest(registration) - - drifts: list[tuple[str, object]] = [ - ("metric_runtime", {"lock_sha256": "0" * 64}), - ( - "report_metrics", - { - "boundary_deviation": { - "implementation": "mir_eval.segment.deviation", - "trim": False, - } - }, - ), - ] - for field, replacement in drifts: - drifted = copy.deepcopy(registration) - drifted[field] = replacement - with pytest.raises(ValueError): - validator.validate_registration(drifted) - - for metric_name, field, replacement in ( - ("boundary_f_0_5", "beta", 0.5), - ("boundary_f_0_5", "trim", False), - ("boundary_f_3_0", "beta", 2.0), - ("boundary_f_3_0", "trim", False), - ("repetition_pairwise_f", "beta", 0.5), - ): - drifted = copy.deepcopy(registration) - metrics = _metrics(drifted) - config = metrics[metric_name] - assert isinstance(config, dict) - config[field] = replacement - with pytest.raises(ValueError, match=field): - validator.validate_registration(drifted) - - reordered = dict(reversed(list(registration.items()))) - assert validator.registration_digest(reordered) == expected_digest + expected = _canonical_digest( + { + "registration": registration, + "structure_metric_contract": _EXPECTED_METRIC_CONTRACT, + } + ) + + assert validator.STRUCTURE_METRIC_CONTRACT == _EXPECTED_METRIC_CONTRACT + assert validator.registration_digest(registration) == expected + assert validator.registration_digest(dict(reversed(list(registration.items())))) == expected + + +def test_result_rejects_receipt_bound_only_to_the_legacy_registration_digest() -> None: + """A receipt cannot omit the metric contract by hashing registration JSON alone.""" + validator = _validator() + registration = _registration() + metric_aware_digest = validator.registration_digest(registration) + legacy_digest = _canonical_digest(registration) + + with pytest.raises(ValueError, match="metric-aware registration"): + validator.evaluate_result(registration, _result(registration, legacy_digest)) + + decision = validator.evaluate_result( + registration, + _result(registration, metric_aware_digest), + ) + assert decision["passed"] is True + assert decision["registration_sha256"] == metric_aware_digest From 3df36942e05c96c8e0fb7864e538290f105ffbf5 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:16:15 +0900 Subject: [PATCH 160/216] test(mir): cross-check preregistration against metric owners --- ...ructure_metric_preregistration_contract.py | 39 +++++++++++++++++++ 1 file changed, 39 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py b/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py index 05b5b7b60..df2d6e0da 100644 --- a/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py +++ b/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py @@ -75,6 +75,45 @@ def test_registration_digest_binds_exact_metric_runtime_and_adapter_semantics() assert validator.registration_digest(dict(reversed(list(registration.items())))) == expected +def test_digest_contract_matches_metric_adapter_and_runtime_lock_owners() -> None: + """Duplicated digest metadata cannot drift from the executable metric owners.""" + validator = _validator() + adapter = load_module( + "scripts/research/evaluate_structure_segmentation_metrics.py", + "structure_segmentation_metric_owner_for_registration", + ) + runtime = load_module( + "scripts/research/verify_structure_metric_runtime_lock.py", + "structure_metric_runtime_owner_for_registration", + ) + contract = validator.STRUCTURE_METRIC_CONTRACT + + assert contract["runtime_lock_sha256"] == hashlib.sha256( + runtime.EXPECTED_LOCK_TEXT.encode("utf-8") + ).hexdigest() + assert contract["boundary_f_0_5"] == { + "implementation": "mir_eval.segment.detection", + "window_seconds": adapter.BOUNDARY_WINDOWS_SECONDS[0], + "beta": adapter.BOUNDARY_BETA, + "trim": adapter.BOUNDARY_TRIM, + } + assert contract["boundary_f_3_0"] == { + "implementation": "mir_eval.segment.detection", + "window_seconds": adapter.BOUNDARY_WINDOWS_SECONDS[1], + "beta": adapter.BOUNDARY_BETA, + "trim": adapter.BOUNDARY_TRIM, + } + assert contract["boundary_deviation"] == { + "implementation": "mir_eval.segment.deviation", + "trim": adapter.BOUNDARY_TRIM, + } + assert contract["repetition_pairwise_f"] == { + "implementation": "mir_eval.segment.pairwise", + "frame_size_seconds": adapter.PAIRWISE_FRAME_SIZE_SECONDS, + "beta": adapter.PAIRWISE_BETA, + } + + def test_result_rejects_receipt_bound_only_to_the_legacy_registration_digest() -> None: """A receipt cannot omit the metric contract by hashing registration JSON alone.""" validator = _validator() From 830fbfec18e6f140af2fef540fdb3af1525e9a08 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:17:33 +0900 Subject: [PATCH 161/216] docs(mir): trace metric-aware registration digest --- .../structure-segmentation-metric-adapter.md | 29 +++++++++++++++++-- 1 file changed, 26 insertions(+), 3 deletions(-) diff --git a/docs/traceability/mir/structure-segmentation-metric-adapter.md b/docs/traceability/mir/structure-segmentation-metric-adapter.md index 37bc9292c..6a3055a12 100644 --- a/docs/traceability/mir/structure-segmentation-metric-adapter.md +++ b/docs/traceability/mir/structure-segmentation-metric-adapter.md @@ -9,6 +9,8 @@ Parent: `docs/traceability/mir/structure-feature-noninferiority.md` The noninferiority registration names `mir_eval.segment.detection`, `mir_eval.segment.deviation`, and `mir_eval.segment.pairwise`, but a function name alone is not a reproducible scientific implementation identity. Boundary detection changes with `window`, `beta`, and `trim`; pairwise grouping changes with `frame_size` and `beta`. A second gap existed after the adapter was introduced: the repository did not bind the exact `mir_eval` distribution artifact, and the admitted-track consumer still emitted functional ACC without the recognized boundary/deviation/repetition measurements. +A third gap remained after those runtime pieces were present. The registration SHA-256 still hashed only the caller-authored base registration, so a result could be bound to generic `mir_eval.segment.*` names without cryptographically binding the reviewed lock artifact or explicit adapter arguments. That left scientific identity weaker than the code that actually executed. + There is also an authority distinction that must not be blurred. The official `ismir-mirex/mirex-evaluation` MIREX-2025 reproduction at commit `b9fa0b0b32e2145af31f35830f78fc9d09a4301b` implements ACC and HR.5/HR3 itself. Its hit-rate function is not the same implementation as `mir_eval.segment.detection`. BandScope's selected mir_eval measures therefore must not be described as numerically identical to the official MIREX HR.5/HR3 table merely because the tolerances have the same names. ## Decision @@ -37,9 +39,24 @@ PyPI publishes that wheel through Trusted Publishing and records the same source The direct wheel is deliberately installed with `--no-deps`. Its scientific purpose is to add the metric implementation, not to resolve or mutate the already-frozen production numerical stack. Missing prerequisites therefore fail at the research-environment construction step instead of causing an independent dependency solve. +## Metric-aware registration digest + +The public registration entry point remains `scripts/research/validate_structure_noninferiority.py`, but it now acts as the metric-aware façade over the unchanged base decision/result policy in `validate_structure_noninferiority_base.py`. The split is internal: callers, corpus admission, result validation, and CLI use the same canonical public path as before. + +The canonical SHA-256 is no longer the hash of registration JSON alone. It hashes a closed envelope containing: + +1. the validated base registration; and +2. repository-owned `structure_metric_contract` metadata that fixes the research-lock SHA-256 and the exact detection, deviation, and pairwise argument semantics listed above. + +This avoids asking every registration producer to copy immutable adapter constants into JSON while still making those constants part of scientific identity. A change to the reviewed metric lock or any result-affecting adapter argument changes the canonical registration digest even when corpus, margins, and experiment runtime are otherwise unchanged. + +Result admission rejects a receipt carrying only the former registration-only digest. Internally, the façade projects the already-validated receipt to the unchanged base result-policy implementation, then returns the metric-aware digest as the admitted evidence identity. The base result policy therefore remains single-owner for corpus order, failed-track behavior, uncertainty-plan matching, result schema, CI decision rules, and claim-boundary checks. + +`test_structure_metric_preregistration_contract.py` also cross-checks the digest metadata against the executable owners: the lock digest is recomputed from `verify_structure_metric_runtime_lock.py::EXPECTED_LOCK_TEXT`, and detection/deviation/pairwise values are compared with the constants in `evaluate_structure_segmentation_metrics.py`. The digest metadata therefore cannot silently drift from the adapter or lock owner while focused tests remain green. + ## Admitted-track evidence path -`PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis()` now binds the exact research lock identity at construction and the canonical CQT then STFT lanes. On each admitted track it still sends the same immutable PCM memoryview, sample rate, exact duration, and admitted annotation-derived reference segmentation through both lanes. +`PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis()` binds the exact research lock identity at construction and the canonical CQT then STFT lanes. On each admitted track it sends the same immutable PCM memoryview, sample rate, exact duration, and admitted annotation-derived reference segmentation through both lanes. Before any boundary/deviation/repetition score is emitted, the consumer re-verifies the installed `mir_eval` distribution and the lock artifact. If the lock hash has changed since experiment binding, the run fails before recognized metrics are appended. Baseline and candidate metrics are then evaluated against the same in-memory reference segmentation and copied into immutable path-free evidence together with the research-lock SHA-256. @@ -53,6 +70,8 @@ Runtime-lock RED `37f3549d15acb33f397245a88da16dcd8e274cf7` required an exact re Admitted-track RED `fbb7c1d7f605aabcc36d4bc651b9b4c6c1b39f0d` requires canonical CQT/STFT evidence to bind the runtime-lock digest, evaluate both lanes against the same reference segmentation, and stop before metric execution when the installed distribution drifts. GREEN `773ffd2913f933726b8bf9b890d955189e0e8c5d` wires the reviewed adapter and runtime verifier into the existing admitted-track consumer without changing any production CQT default. +Registration-digest RED `1ce73179f0425ba7dc4a45c44910c2da93900f9a` demonstrated that the previous canonical validator could not bind the exact metric runtime and adapter semantics. `580f2affcdf5ed1679e1e8748011c56d7b3287e8` introduced the focused contract implementation, `880e0db335d29865016d98da70cb7643ad3e4291` converted it to a digest envelope over immutable repository-owned semantics, and ordinary descendant `cadba9ba74e8d6a97815c875bad51b90905fa996` made that façade the canonical validator while preserving the prior base policy unchanged. Test alignment `f961390f6241fa3c3d4b49393548806ad993eda3` and owner-parity regression `3df36942e05c96c8e0fb7864e538290f105ffbf5` require both metric-aware result binding and exact agreement with the lock/adapter owners. + These commits are source-level TDD and contract evidence. Hosted current-head checks and rights-cleared real-audio scientific acceptance remain separate requirements. ## Constraints and rejected alternatives @@ -67,18 +86,22 @@ Adding `mir_eval` to the production analysis dependency set was rejected for thi Using a mutable `mir_eval>=...` requirement, package index lookup at experiment time, or an unhashed wheel was rejected. Any of those would permit the scientific implementation to drift without changing source. +Putting the immutable metric arguments into every caller-authored registration JSON was rejected. The values are owned by the reviewed adapter and runtime lock, not by each experiment author. Repetition would create another drift surface without adding scientific choice. The canonical digest instead includes one repository-owned contract envelope and tests it against those owners. + Computing aggregate scores or confidence intervals inside the admitted-track consumer was rejected. Those remain separate preregistered scientific decisions and must be frozen before candidate outcomes are inspected. ## Current limitation and next causal step -The runtime artifact and admitted-track recognized-metric path now exist, but the main registration digest still records generic `mir_eval.segment.*` strings and does not yet include the research-lock SHA-256 or the complete adapter arguments. **The rights-cleared CQT/STFT experiment remains inadmissible until that registration schema is repaired.** Source-level wiring must not be used as a reason to inspect candidate corpus outcomes early. +The canonical registration digest now binds the exact research-lock identity and complete detection/deviation/pairwise semantics. Rights-cleared candidate outcomes still remain inadmissible because the aggregation and paired-uncertainty **implementation** is not frozen: schema v1 records a procedure label, confidence level, resample count, and seed, but does not yet identify executable aggregation/bootstrap semantics. -The next causal slice is therefore to bind the exact research-lock identity and complete detection/deviation/pairwise semantics into the canonical registration digest, update all registration fixtures/admission contracts together, and only then freeze the aggregate/paired-uncertainty implementation. Rights-cleared real-audio execution follows those preregistration steps, not the reverse. +The next causal slice is to preregister the exact track aggregation and paired-uncertainty implementation, including failed-track behavior already owned by the result policy, and bind that implementation into the same scientific identity before any rights-cleared CQT/STFT candidate result is inspected. Only after that may the real corpus run produce paired evidence. ## Security Notes The adapter and verifier receive normalized in-memory values and a bounded local lock file. They add no model download, credentials, generic plugin loading, or package installation at metric execution. The lock verifier performs no network access. Durable track evidence contains content digests, scalar metrics, and the runtime-lock digest, not workstation paths or licensed audio bytes. +The metric-aware digest adds no executable plugin surface. It hashes already-validated registration data plus immutable repository-owned scalar metadata. Result projection into the unchanged base policy is in-memory only and does not weaken the closed result schema or corpus/uncertainty checks. + ## References Raffel, C., McFee, B., Humphrey, E. J., Salamon, J., Nieto, O., Liang, D., & Ellis, D. P. W. (2014). *mir_eval: A transparent implementation of common MIR metrics*. Proceedings of the 15th International Society for Music Information Retrieval Conference. From 9fdba32fd222174e3dd4593f266bf44372f9b6a3 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:18:58 +0900 Subject: [PATCH 162/216] docs(mir): make preregistration digest semantics code-current --- .../mir/structure-feature-noninferiority.md | 36 +++++++++++-------- 1 file changed, 21 insertions(+), 15 deletions(-) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index 26e78bbb1..83eb23e58 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -9,9 +9,11 @@ Production baseline: `chroma_cqt` in `sections/segmenter.py` The earlier `chroma_cqt` → `chroma_stft` optimization is not a production change until a preregistered, rights-cleared real-audio experiment shows that the candidate is noninferior on rehearsal-relevant structure quality and materially faster on the same corpus and runtime identity. -The repository now owns a result-admission boundary in `scripts/research/validate_structure_noninferiority.py`. The validator does **not** calculate MIR metrics and does not choose scientific margins or an uncertainty procedure. It freezes the reviewed experiment contract, including its claim boundary, binds a result receipt to that contract by canonical SHA-256, requires one explicit success-or-failure receipt per registered track, and applies the registered aggregate confidence-interval decision rules only when the complete corpus produced measurements. Metric computation and uncertainty estimation remain separate scientific measurement steps. +The repository owns a result-admission boundary in `scripts/research/validate_structure_noninferiority.py`. The validator does **not** calculate MIR metrics and does not choose scientific margins or an uncertainty procedure. It freezes the reviewed experiment contract, including its claim boundary, binds a result receipt to that contract by canonical SHA-256, requires one explicit success-or-failure receipt per registered track, and applies the registered aggregate confidence-interval decision rules only when the complete corpus produced measurements. Metric computation and uncertainty estimation remain separate scientific measurement steps. -This keeps a performance result from silently changing the corpus, thresholds, feature identities, uncertainty procedure, runtime, or permitted interpretation after the measurements are visible. +The canonical public validator is now a metric-aware façade over the unchanged base decision/result policy in `validate_structure_noninferiority_base.py`. Its SHA-256 binds both the validated registration data and repository-owned structure-metric semantics: the exact research metric-lock digest plus all result-affecting `mir_eval.segment.detection`, `segment.deviation`, and `segment.pairwise` arguments. A receipt carrying only the former registration-JSON digest is rejected. This closes a scientific identity gap without asking each experiment author to duplicate immutable adapter constants in registration JSON. + +This keeps a performance result from silently changing the corpus, thresholds, feature identities, metric implementation/runtime, uncertainty procedure, runtime, or permitted interpretation after the measurements are visible. ## Registered inputs @@ -20,13 +22,15 @@ A registration is valid only when it records all of the following before the res - baseline `chroma_cqt` and candidate `chroma_stft`; - a rights-cleared real-audio corpus with stable track IDs, content-unique audio SHA-256 identities, annotation SHA-256, rights basis, and provenance URI; - exact source commit and `uv.lock` identity plus Python, librosa, NumPy, sample rate, channel count, and host profile; -- the complete metric implementation contract and noninferiority/speed thresholds, including the pinned MIREX functional-ACC evaluator commit/function, 200 ms frame-grid contract v1, and versioned annotation/label-mapping semantics; +- the caller-selected metric thresholds and the pinned MIREX functional-ACC evaluator commit/function, 200 ms frame-grid contract v1, and versioned annotation/label-mapping semantics; - the paired-uncertainty procedure identity, confidence level, resample count, and random seed; - a non-empty claim boundary stating the population/runtime scope to which a passing result may be applied. +The registration digest additionally binds the repository-owned segmentation-metric contract. The bound contract includes the SHA-256 of `services/analysis-engine/requirements-structure-metrics.lock`, detection windows 0.5 s and 3.0 s with `beta=1.0, trim=True`, deviation with `trim=True`, and pairwise grouping with `frame_size=0.1, beta=1.0`. `test_structure_metric_preregistration_contract.py` cross-checks those digest constants against the executable adapter and runtime-lock owners so the metadata cannot silently diverge while focused tests remain green. + Schema v1 is closed-world not only at the registration top level but also for the hypothesis object, every registered metric configuration, every corpus-track object, and the runtime object. Extra fields are not treated as harmless annotations: an unregistered pilot-selection flag, metric weight, corpus-selection marker, or cache/runtime hint changes what reviewers may infer was preregistered, so it fails admission instead of silently entering the evidence artifact. The claim boundary is itself a required top-level registration field and therefore participates in the canonical registration SHA-256. -`experiment_id` is a semantic label, while the registration SHA-256 is the exact evidence identity. The validator therefore compares the registration and result experiment IDs after the same required-text normalization, but hashes the original validated registration representation unchanged. This prevents an otherwise valid registration with surrounding whitespace from becoming impossible to reference while keeping byte-distinct registrations cryptographically distinct. +`experiment_id` is a semantic label, while the registration SHA-256 is the exact evidence identity. The validator compares the registration and result experiment IDs after the same required-text normalization, but the canonical hash preserves the validated registration representation inside a closed envelope together with the immutable repository-owned structure-metric contract. This prevents an otherwise valid registration with surrounding whitespace from becoming impossible to reference while keeping byte-distinct registrations and metric-contract revisions cryptographically distinct. The validator requires a 95% confidence level because the current result schema is explicitly `ci95`; it does not prescribe which scientifically defensible paired procedure must produce that interval. The procedure identifier, resample count, and seed are part of the preregistration digest so they cannot be changed after results are seen. `paired-track-bootstrap-v1` and the numeric values used in unit tests are policy fixtures only, not an approved BandScope production analysis plan. @@ -46,14 +50,16 @@ BandScope uses the MIREX Music Structure Analysis task and its standardized eval | Evidence | Registered implementation / convention | Decision use | | --- | --- | --- | -| Boundary precision/recall/F at 0.5 s | `mir_eval.segment.detection`, 0.5 s window | F noninferiority | -| Boundary precision/recall/F at 3.0 s | `mir_eval.segment.detection`, 3.0 s window | F noninferiority | -| Boundary median deviation, both directions | `mir_eval.segment.deviation` convention | Report per track and aggregate | +| Boundary precision/recall/F at 0.5 s | `mir_eval.segment.detection(window=0.5, beta=1.0, trim=True)` | F noninferiority | +| Boundary precision/recall/F at 3.0 s | `mir_eval.segment.detection(window=3.0, beta=1.0, trim=True)` | F noninferiority | +| Boundary median deviation, both directions | `mir_eval.segment.deviation(trim=True)` | Report per track and aggregate | | Functional-label accuracy | `ismir-mirex/mirex-evaluation@b9fa0b0...:music_structure_analysis.eval_script.calculate_accuracy`; 0.2 s grid; frame-grid v1; annotation v1; label-mapping v1 | Noninferiority | -| Repetition/group consistency precision/recall/F | `mir_eval.segment.pairwise`, 0.1 s frame size | F noninferiority | +| Repetition/group consistency precision/recall/F | `mir_eval.segment.pairwise(frame_size=0.1, beta=1.0)` | F noninferiority | | Latency | paired p50/p95 on the registered host | p95 ratio superiority | | Peak memory | peak RSS on the registered host | Report per track and aggregate | +The selected mir_eval calls execute only under the content-addressed research overlay whose exact lock SHA-256 is part of the canonical registration digest. The runtime verifier separately requires the installed distribution to be `mir_eval==0.8.2` before these metrics run. Full artifact provenance and adapter semantics are recorded in `docs/traceability/mir/structure-segmentation-metric-adapter.md`. + The MIREX task page describes frame-level ACC conceptually and gives finer resolutions as examples, but the current official `ismir-mirex/mirex-evaluation` MIREX-2025 reproduction is more specific: at commit `b9fa0b0b32e2145af31f35830f78fc9d09a4301b`, `calculate_accuracy` uses a 0.2 s hop, `np.arange(0, gt_duration, frame_hop)`, and advances a segment when a frame point is greater than or equal to the segment end. BandScope therefore pins that exact evaluator identity and frame-grid contract rather than selecting 100 ms from prose examples. The local annotation mapping remains preregistered and content-addressed before evaluation; raw-label normalization is not allowed to drift after candidate results are visible. The detailed contract is recorded in `docs/traceability/mir/functional-label-accuracy-contract.md`, and `scripts/research/evaluate_structure_functional_accuracy.py` is the repository-owned adapter for the normalized seven-label subset. `mir_eval.segment.detection` uses one-to-one boundary matching within the selected tolerance; `mir_eval.segment.deviation` reports the median nearest-boundary deviations in both directions; `mir_eval.segment.pairwise` measures structural grouping agreement. These are different questions and must not be collapsed into one score. @@ -80,13 +86,13 @@ The repository does not currently contain approved numeric margins or an approve ## Result receipt -A result receipt must contain the exact registration digest, exact uncertainty plan, exact corpus order, one ordered track receipt per registered track, failed-track IDs, and the exact preregistered claim boundary. A successful complete run additionally carries aggregate measurements, paired 95% intervals for every gated quality metric, and a paired p95-latency-ratio interval. A successful track receipt contains `track_id`, `baseline`, and `candidate`; a track listed in `failed_tracks` contains only `track_id`, preserving the failure without manufacturing scores that were never measured. When any track fails, the three aggregate/CI fields remain present in the closed top-level schema but their values must be JSON `null`. +A result receipt must contain the exact metric-aware registration digest, exact uncertainty plan, exact corpus order, one ordered track receipt per registered track, failed-track IDs, and the exact preregistered claim boundary. A successful complete run additionally carries aggregate measurements, paired 95% intervals for every gated quality metric, and a paired p95-latency-ratio interval. A successful track receipt contains `track_id`, `baseline`, and `candidate`; a track listed in `failed_tracks` contains only `track_id`, preserving the failure without manufacturing scores that were never measured. When any track fails, the three aggregate/CI fields remain present in the closed top-level schema but their values must be JSON `null`. Schema v1 treats the result envelope, each per-track receipt, and the aggregate envelope as closed-world objects. A top-level post-result field such as an unregistered p-value, a per-track `selected_for_aggregate` flag, or aggregate-side selection metadata is rejected rather than ignored. The same rule already applies inside each baseline/candidate measurement object. This prevents a producer from attaching an unreviewed post-hoc selection or inferential claim to an otherwise accepted receipt and having that field travel with the admitted evidence artifact. Each successful per-track and aggregate baseline/candidate measurement is a closed-world receipt: it must contain exactly the registered quality fields plus the reporting-only precision/recall, boundary-deviation, latency, and peak-RSS fields. Missing fields and additional post-hoc measurement fields both fail admission. A declared failed track is the only per-track exception and is instead closed to exactly `track_id`; attaching partial or invented measurement fields to that failed receipt is rejected. This prevents a result producer from filling a real measurement failure with synthetic values merely to satisfy the evidence schema. -The receipt is rejected when a track is omitted/reordered, missing measurements are not paired with an explicit failed-track declaration, a failed receipt carries measurement fields, a failed run carries non-null aggregate/CI summaries, the uncertainty plan differs, the claim boundary differs, a required metric is absent, an unregistered envelope or measurement field is added, a boundary/repetition P/R/F triplet is internally inconsistent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the registration hash differs. A structurally valid receipt with one or more known failed tracks is retained for diagnosis with null aggregate summaries and evaluates to `passed=false` under schema v1. A passing receipt therefore means the preregistered decision rule passed for the complete registered corpus, uncertainty procedure, runtime, and claim boundary. It does not generalize automatically to other genres, codecs, sample rates, machines, annotation regimes, or statistical procedures. +The receipt is rejected when a track is omitted/reordered, missing measurements are not paired with an explicit failed-track declaration, a failed receipt carries measurement fields, a failed run carries non-null aggregate/CI summaries, the uncertainty plan differs, the claim boundary differs, a required metric is absent, an unregistered envelope or measurement field is added, a boundary/repetition P/R/F triplet is internally inconsistent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the metric-aware registration hash differs. A structurally valid receipt with one or more known failed tracks is retained for diagnosis with null aggregate summaries and evaluates to `passed=false` under schema v1. A passing receipt therefore means the preregistered decision rule passed for the complete registered corpus, metric contract, uncertainty procedure, runtime, and claim boundary. It does not generalize automatically to other genres, codecs, sample rates, machines, annotation regimes, or statistical procedures. The validator still does not recompute aggregate statistics or confidence intervals from per-track evidence. During review, a min/max range consistency heuristic was briefly encoded as a RED and then removed before production code changed because it would smuggle an unapproved assumption about aggregation into schema v1. Macro means, weighted means, micro-aggregated precision/recall/F and pooled latency summaries do not share one universally valid range relationship. The correct next step is to preregister the aggregation procedure and implement it in the experiment runner, not to infer a statistical method inside an admission validator after the fact. @@ -99,10 +105,10 @@ The JSON `source_uri` value remains provenance metadata only; it is never derefe ## Reproducibility sequence 1. Review the rights basis, corpus composition, distinct audio-content identities, feature hypothesis, metric contract, pinned functional-ACC evaluator/grid/annotation/mapping contract, host profile, numeric margins, aggregation procedure, paired-uncertainty procedure, and claim boundary **before** running the candidate. Freeze the procedure identifier, confidence level, resample count, random seed, and claim boundary in the registration. If repeated recordings, clustering, source-label normalization, or any failure/exclusion tolerance is scientifically required, define and version that dependence/mapping/exclusion policy before measurement rather than remapping or omitting observations after results are visible. -2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain its canonical SHA-256. -3. Run baseline and candidate on the same decoded track identities and host profile. Functional ACC must consume the admitted normalized annotation through `scripts/research/evaluate_structure_functional_accuracy.py` and preserve the pinned 200 ms MIREX grid semantics; it must not reopen corpus paths or remap labels after preregistration. The experiment runner must compute the approved aggregation and paired uncertainty procedure from the complete measured-track evidence rather than accepting caller-authored aggregate numbers as scientific authority. Record recognized MIR metrics, p50/p95 latency, and peak RSS for successful tracks. If a registered track cannot produce a baseline/candidate measurement, preserve its ordered `track_id` receipt, add that exact ID to `failed_tracks`, and set the aggregate plus both CI summary fields to JSON `null`; do not invent metric values or compute a complete-case aggregate from the surviving tracks. Only a complete run records aggregate outputs and paired uncertainty using the preregistered procedure. -4. Put the registration digest, identical uncertainty-plan fields, and identical claim boundary in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. Under schema v1, any recorded failed track keeps the diagnostic receipt valid but makes acceptance fail closed before aggregate decision rules; undeclared missing measurements, non-null post-failure summaries, and claim-boundary drift are rejected. -5. Preserve the registration, result receipt, corpus/annotation hashes, exact source commit, lock hash, aggregation implementation identity, uncertainty procedure, and claim boundary together. A production feature switch requires this evidence plus normal code review and protected-head checks. +2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain its canonical metric-aware SHA-256. The digest includes the validated registration plus the repository-owned structure metric lock/argument contract; it is not a hash of the registration JSON alone. +3. Run baseline and candidate on the same decoded track identities and host profile. Functional ACC must consume the admitted normalized annotation through `scripts/research/evaluate_structure_functional_accuracy.py` and preserve the pinned 200 ms MIREX grid semantics; it must not reopen corpus paths or remap labels after preregistration. Boundary/deviation/repetition metrics must use the registered research lock and explicit adapter semantics. The experiment runner must compute the approved aggregation and paired uncertainty procedure from the complete measured-track evidence rather than accepting caller-authored aggregate numbers as scientific authority. Record recognized MIR metrics, p50/p95 latency, and peak RSS for successful tracks. If a registered track cannot produce a baseline/candidate measurement, preserve its ordered `track_id` receipt, add that exact ID to `failed_tracks`, and set the aggregate plus both CI summary fields to JSON `null`; do not invent metric values or compute a complete-case aggregate from the surviving tracks. Only a complete run records aggregate outputs and paired uncertainty using the preregistered procedure. +4. Put the metric-aware registration digest, identical uncertainty-plan fields, and identical claim boundary in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. A receipt carrying only the former registration-JSON digest is rejected. Under schema v1, any recorded failed track keeps the diagnostic receipt valid but makes acceptance fail closed before aggregate decision rules; undeclared missing measurements, non-null post-failure summaries, and claim-boundary drift are rejected. +5. Preserve the registration, metric-aware digest contract, result receipt, corpus/annotation hashes, exact source commit, lock hash, aggregation implementation identity, uncertainty procedure, and claim boundary together. A production feature switch requires this evidence plus normal code review and protected-head checks. No step authorizes committing licensed audio to Git. Rights-cleared means BandScope has the necessary evaluation right; redistribution is a separate permission. @@ -112,7 +118,7 @@ No step authorizes committing licensed audio to Git. Rights-cleared means BandSc - Evidence JSON is limited to 2 MiB, must be a regular file read through one open descriptor, must decode as UTF-8, and rejects duplicate keys plus non-standard `NaN`/`Infinity` constants. - `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute, relative, drive-relative, and `file:` forms are rejected; an explicit non-file URI scheme is required so transient workstation paths cannot become provenance authority or leak into review artifacts. - Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt; duplicate audio SHA-256 values under different track IDs are rejected so one recording cannot be silently counted multiple times. -- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, functional-label evaluator/grid/annotation/mapping drift, uncertainty-plan drift, claim-boundary drift, registration drift, undeclared missing measurements, measurement-bearing failed receipts, non-null aggregate/CI summaries after any failed track, unregistered registration/hypothesis/metric/corpus/runtime/result/track/aggregate/measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. +- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, functional-label evaluator/grid/annotation/mapping drift, segmentation metric-lock/argument drift, uncertainty-plan drift, claim-boundary drift, registration drift, undeclared missing measurements, measurement-bearing failed receipts, non-null aggregate/CI summaries after any failed track, unregistered registration/hypothesis/metric/corpus/runtime/result/track/aggregate/measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. - The validator does not claim that SHA-256 proves licensing, annotation validity, scientific adequacy, independence beyond exact-byte uniqueness, aggregate derivation, or that the registered statistical procedure is appropriate. Rights and scientific review remain separate gates. ## References From 4416e655fd597ec8650aa7f0d8023c471e427eba Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:23:29 +0900 Subject: [PATCH 163/216] test(mir): preregister paired track aggregation semantics --- ...st_structure_noninferiority_aggregation.py | 214 ++++++++++++++++++ 1 file changed, 214 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_noninferiority_aggregation.py diff --git a/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py b/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py new file mode 100644 index 000000000..792b0bdf7 --- /dev/null +++ b/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py @@ -0,0 +1,214 @@ +"""Tests for preregistered paired-track aggregation and bootstrap uncertainty.""" + +from __future__ import annotations + +from types import ModuleType + +import pytest +from conftest import load_module + + +def _aggregation() -> ModuleType: + return load_module( + "scripts/research/aggregate_structure_noninferiority.py", + "aggregate_structure_noninferiority", + ) + + +def _side( + *, + boundary_precision: float, + boundary_recall: float, + functional_accuracy: float, + repetition_precision: float, + repetition_recall: float, + p50_latency: float, + p95_latency: float, + peak_rss: float, +) -> dict[str, float]: + harmonic_boundary = ( + 0.0 + if boundary_precision + boundary_recall == 0.0 + else 2.0 * boundary_precision * boundary_recall + / (boundary_precision + boundary_recall) + ) + harmonic_repetition = ( + 0.0 + if repetition_precision + repetition_recall == 0.0 + else 2.0 * repetition_precision * repetition_recall + / (repetition_precision + repetition_recall) + ) + return { + "boundary_precision_0_5": boundary_precision, + "boundary_recall_0_5": boundary_recall, + "boundary_f_0_5": harmonic_boundary, + "boundary_precision_3_0": boundary_precision, + "boundary_recall_3_0": boundary_recall, + "boundary_f_3_0": harmonic_boundary, + "reference_to_estimate_median_deviation_seconds": 0.2, + "estimate_to_reference_median_deviation_seconds": 0.25, + "functional_label_accuracy": functional_accuracy, + "repetition_pairwise_precision": repetition_precision, + "repetition_pairwise_recall": repetition_recall, + "repetition_pairwise_f": harmonic_repetition, + "p50_latency_seconds": p50_latency, + "p95_latency_seconds": p95_latency, + "peak_rss_mib": peak_rss, + } + + +def _tracks() -> list[dict[str, object]]: + return [ + { + "track_id": "track-a", + "baseline": _side( + boundary_precision=0.80, + boundary_recall=0.60, + functional_accuracy=0.70, + repetition_precision=0.75, + repetition_recall=0.65, + p50_latency=4.0, + p95_latency=6.0, + peak_rss=600.0, + ), + "candidate": _side( + boundary_precision=0.78, + boundary_recall=0.62, + functional_accuracy=0.71, + repetition_precision=0.74, + repetition_recall=0.66, + p50_latency=2.0, + p95_latency=4.0, + peak_rss=580.0, + ), + }, + { + "track_id": "track-b", + "baseline": _side( + boundary_precision=0.60, + boundary_recall=0.80, + functional_accuracy=0.74, + repetition_precision=0.65, + repetition_recall=0.75, + p50_latency=5.0, + p95_latency=8.0, + peak_rss=650.0, + ), + "candidate": _side( + boundary_precision=0.62, + boundary_recall=0.78, + functional_accuracy=0.73, + repetition_precision=0.66, + repetition_recall=0.74, + p50_latency=2.5, + p95_latency=5.0, + peak_rss=610.0, + ), + }, + { + "track_id": "track-c", + "baseline": _side( + boundary_precision=0.70, + boundary_recall=0.70, + functional_accuracy=0.72, + repetition_precision=0.70, + repetition_recall=0.70, + p50_latency=4.5, + p95_latency=7.0, + peak_rss=625.0, + ), + "candidate": _side( + boundary_precision=0.70, + boundary_recall=0.70, + functional_accuracy=0.72, + repetition_precision=0.70, + repetition_recall=0.70, + p50_latency=2.25, + p95_latency=4.5, + peak_rss=595.0, + ), + }, + ] + + +def test_macro_track_aggregation_recomputes_f_and_preserves_worst_peak_memory() -> None: + aggregation = _aggregation() + + evidence = aggregation.aggregate_complete_track_measurements( + _tracks(), + uncertainty={ + "procedure_id": "paired-track-bootstrap-v1", + "confidence_level": 0.95, + "resamples": 1000, + "random_seed": 20260918, + }, + ) + + baseline = evidence["aggregate"]["baseline"] + candidate = evidence["aggregate"]["candidate"] + assert baseline["boundary_precision_0_5"] == pytest.approx(0.70) + assert baseline["boundary_recall_0_5"] == pytest.approx(0.70) + assert baseline["boundary_f_0_5"] == pytest.approx(0.70) + assert candidate["functional_label_accuracy"] == pytest.approx(0.72) + assert baseline["p95_latency_seconds"] == pytest.approx(7.0) + assert candidate["p95_latency_seconds"] == pytest.approx(4.5) + assert baseline["peak_rss_mib"] == pytest.approx(650.0) + assert candidate["peak_rss_mib"] == pytest.approx(610.0) + + +def test_paired_bootstrap_is_deterministic_and_resamples_track_pairs() -> None: + aggregation = _aggregation() + uncertainty = { + "procedure_id": "paired-track-bootstrap-v1", + "confidence_level": 0.95, + "resamples": 1000, + "random_seed": 7, + } + + first = aggregation.aggregate_complete_track_measurements( + _tracks(), uncertainty=uncertainty + ) + second = aggregation.aggregate_complete_track_measurements( + _tracks(), uncertainty=uncertainty + ) + + assert first == second + assert set(first["paired_delta_ci95"]) == { + "boundary_f_0_5", + "boundary_f_3_0", + "functional_label_accuracy", + "repetition_pairwise_f", + } + assert len(first["p95_latency_ratio_ci95"]) == 2 + assert first["p95_latency_ratio_ci95"][0] > 0.0 + + +def test_aggregation_fails_closed_on_duplicate_tracks_or_unsupported_uncertainty() -> None: + aggregation = _aggregation() + duplicate = _tracks() + duplicate[1]["track_id"] = "track-a" + uncertainty = { + "procedure_id": "paired-track-bootstrap-v1", + "confidence_level": 0.95, + "resamples": 1000, + "random_seed": 7, + } + + with pytest.raises(ValueError, match="duplicate track_id"): + aggregation.aggregate_complete_track_measurements( + duplicate, uncertainty=uncertainty + ) + + unsupported = dict(uncertainty) + unsupported["procedure_id"] = "post-hoc-bootstrap" + with pytest.raises(ValueError, match="procedure_id"): + aggregation.aggregate_complete_track_measurements( + _tracks(), uncertainty=unsupported + ) + + excessive = dict(uncertainty) + excessive["resamples"] = 1_000_001 + with pytest.raises(ValueError, match="resamples"): + aggregation.aggregate_complete_track_measurements( + _tracks(), uncertainty=excessive + ) From 3b51825647acd1f1122c3708982c0de321420355 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:24:20 +0900 Subject: [PATCH 164/216] feat(mir): implement paired track bootstrap aggregation --- .../aggregate_structure_noninferiority.py | 263 ++++++++++++++++++ 1 file changed, 263 insertions(+) create mode 100644 scripts/research/aggregate_structure_noninferiority.py diff --git a/scripts/research/aggregate_structure_noninferiority.py b/scripts/research/aggregate_structure_noninferiority.py new file mode 100644 index 000000000..7ffc8320f --- /dev/null +++ b/scripts/research/aggregate_structure_noninferiority.py @@ -0,0 +1,263 @@ +#!/usr/bin/env python3 +"""Aggregate complete paired structure measurements under one frozen procedure. + +The sampling unit is a registered track. Baseline and candidate measurements +for that track are never separated during resampling. Quality/latency summaries +are macro track statistics so track duration does not silently become a weight. +Synthetic fixtures may exercise this module, but scientific acceptance still +requires the preregistered rights-cleared corpus. +""" + +from __future__ import annotations + +import importlib.util +import math +from collections.abc import Mapping, Sequence +from pathlib import Path +from types import ModuleType +from typing import Any + +import numpy as np + +PROCEDURE_ID = "paired-track-bootstrap-v1" +CONFIDENCE_LEVEL = 0.95 +MIN_RESAMPLES = 1000 +MAX_RESAMPLES = 100_000 +RNG_ALGORITHM = "numpy.random.Generator(PCG64)" +QUANTILE_METHOD = "linear" +AGGREGATION_ID = "macro-track-v1" +QUALITY_METRICS = ( + "boundary_f_0_5", + "boundary_f_3_0", + "functional_label_accuracy", + "repetition_pairwise_f", +) + +AGGREGATION_UNCERTAINTY_CONTRACT = { + "aggregation_id": AGGREGATION_ID, + "sampling_unit": "registered_track_pair", + "quality_weighting": "equal_track", + "latency_weighting": "equal_track", + "boundary_and_repetition_f": "harmonic_of_macro_precision_recall", + "report_deviation": "macro_track_mean", + "peak_rss": "maximum_track_peak_rss", + "procedure_id": PROCEDURE_ID, + "confidence_level": CONFIDENCE_LEVEL, + "bootstrap_sample_size": "registered_track_count", + "bootstrap_replacement": True, + "rng": RNG_ALGORITHM, + "quantile_method": QUANTILE_METHOD, + "two_sided_tail_probability": 0.025, + "maximum_resamples": MAX_RESAMPLES, +} + + +def _validator() -> ModuleType: + """Load the canonical evidence validator without creating a package dependency.""" + path = Path(__file__).with_name("validate_structure_noninferiority.py") + spec = importlib.util.spec_from_file_location( + "bandscope_structure_noninferiority_for_aggregation", + path, + ) + if spec is None or spec.loader is None: + raise RuntimeError("could not load structure noninferiority validator") + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def _mapping(value: object, field: str) -> Mapping[str, Any]: + if not isinstance(value, Mapping): + raise ValueError(f"{field} must be an object") + return value + + +def _mean(sides: Sequence[Mapping[str, float]], field: str) -> float: + return math.fsum(side[field] for side in sides) / len(sides) + + +def _harmonic(precision: float, recall: float) -> float: + denominator = precision + recall + return 0.0 if denominator == 0.0 else 2.0 * precision * recall / denominator + + +def _aggregate_side(sides: Sequence[Mapping[str, float]]) -> dict[str, float]: + """Compute the preregistered equal-track aggregate for one experiment side.""" + if not sides: + raise ValueError("aggregation requires at least one measurement side") + + boundary_precision_0_5 = _mean(sides, "boundary_precision_0_5") + boundary_recall_0_5 = _mean(sides, "boundary_recall_0_5") + boundary_precision_3_0 = _mean(sides, "boundary_precision_3_0") + boundary_recall_3_0 = _mean(sides, "boundary_recall_3_0") + repetition_precision = _mean(sides, "repetition_pairwise_precision") + repetition_recall = _mean(sides, "repetition_pairwise_recall") + + return { + "boundary_precision_0_5": boundary_precision_0_5, + "boundary_recall_0_5": boundary_recall_0_5, + "boundary_f_0_5": _harmonic( + boundary_precision_0_5, + boundary_recall_0_5, + ), + "boundary_precision_3_0": boundary_precision_3_0, + "boundary_recall_3_0": boundary_recall_3_0, + "boundary_f_3_0": _harmonic( + boundary_precision_3_0, + boundary_recall_3_0, + ), + "reference_to_estimate_median_deviation_seconds": _mean( + sides, + "reference_to_estimate_median_deviation_seconds", + ), + "estimate_to_reference_median_deviation_seconds": _mean( + sides, + "estimate_to_reference_median_deviation_seconds", + ), + "functional_label_accuracy": _mean(sides, "functional_label_accuracy"), + "repetition_pairwise_precision": repetition_precision, + "repetition_pairwise_recall": repetition_recall, + "repetition_pairwise_f": _harmonic( + repetition_precision, + repetition_recall, + ), + "p50_latency_seconds": _mean(sides, "p50_latency_seconds"), + "p95_latency_seconds": _mean(sides, "p95_latency_seconds"), + "peak_rss_mib": max(side["peak_rss_mib"] for side in sides), + } + + +def _validate_uncertainty(uncertainty_value: object) -> tuple[int, int]: + uncertainty = _mapping(uncertainty_value, "uncertainty") + expected = {"procedure_id", "confidence_level", "resamples", "random_seed"} + if set(uncertainty) != expected: + raise ValueError("uncertainty must contain exactly the registered procedure fields") + if uncertainty.get("procedure_id") != PROCEDURE_ID: + raise ValueError(f"uncertainty.procedure_id must equal {PROCEDURE_ID}") + confidence = uncertainty.get("confidence_level") + if isinstance(confidence, bool) or not isinstance(confidence, (int, float)): + raise ValueError("uncertainty.confidence_level must be numeric") + if not math.isclose(float(confidence), CONFIDENCE_LEVEL, rel_tol=0.0, abs_tol=1e-12): + raise ValueError("uncertainty.confidence_level must equal 0.95") + resamples = uncertainty.get("resamples") + if isinstance(resamples, bool) or not isinstance(resamples, int): + raise ValueError("uncertainty.resamples must be an integer") + if not MIN_RESAMPLES <= resamples <= MAX_RESAMPLES: + raise ValueError( + f"uncertainty.resamples must be in {MIN_RESAMPLES}..{MAX_RESAMPLES}" + ) + random_seed = uncertainty.get("random_seed") + if isinstance(random_seed, bool) or not isinstance(random_seed, int): + raise ValueError("uncertainty.random_seed must be an integer") + if not 0 <= random_seed <= (2**32) - 1: + raise ValueError("uncertainty.random_seed must be in 0..4294967295") + return resamples, random_seed + + +def _normalize_tracks( + tracks_value: object, +) -> tuple[list[str], list[dict[str, float]], list[dict[str, float]]]: + if isinstance(tracks_value, (str, bytes)) or not isinstance(tracks_value, Sequence): + raise ValueError("tracks must be an array") + if len(tracks_value) < 2: + raise ValueError("aggregation requires at least two registered track pairs") + + validator = _validator() + track_ids: list[str] = [] + baseline: list[dict[str, float]] = [] + candidate: list[dict[str, float]] = [] + for index, raw_track in enumerate(tracks_value): + track = _mapping(raw_track, f"tracks[{index}]") + if set(track) != {"track_id", "baseline", "candidate"}: + raise ValueError( + f"tracks[{index}] must contain exactly track_id, baseline, and candidate" + ) + track_id = track.get("track_id") + if not isinstance(track_id, str) or not track_id.strip(): + raise ValueError(f"tracks[{index}].track_id must be non-empty text") + normalized_id = track_id.strip() + if normalized_id in track_ids: + raise ValueError(f"duplicate track_id: {normalized_id}") + track_ids.append(normalized_id) + baseline.append( + validator._validate_measurement_side( + track.get("baseline"), + f"tracks[{index}].baseline", + ) + ) + candidate.append( + validator._validate_measurement_side( + track.get("candidate"), + f"tracks[{index}].candidate", + ) + ) + return track_ids, baseline, candidate + + +def _quality_deltas( + baseline: Mapping[str, float], + candidate: Mapping[str, float], +) -> dict[str, float]: + return { + metric: candidate[metric] - baseline[metric] + for metric in QUALITY_METRICS + } + + +def aggregate_complete_track_measurements( + tracks_value: object, + *, + uncertainty: object, +) -> dict[str, object]: + """Return macro aggregates and deterministic paired percentile-bootstrap intervals.""" + resamples, random_seed = _validate_uncertainty(uncertainty) + _, baseline_sides, candidate_sides = _normalize_tracks(tracks_value) + aggregate_baseline = _aggregate_side(baseline_sides) + aggregate_candidate = _aggregate_side(candidate_sides) + if aggregate_baseline["p95_latency_seconds"] <= 0.0: + raise ValueError("aggregate baseline p95 latency must be greater than zero") + + rng = np.random.Generator(np.random.PCG64(random_seed)) + sampled_deltas = { + metric: np.empty(resamples, dtype=np.float64) for metric in QUALITY_METRICS + } + sampled_latency_ratio = np.empty(resamples, dtype=np.float64) + track_count = len(baseline_sides) + + for bootstrap_index in range(resamples): + indices = rng.integers(0, track_count, size=track_count) + sampled_baseline = [baseline_sides[int(index)] for index in indices] + sampled_candidate = [candidate_sides[int(index)] for index in indices] + bootstrap_baseline = _aggregate_side(sampled_baseline) + bootstrap_candidate = _aggregate_side(sampled_candidate) + deltas = _quality_deltas(bootstrap_baseline, bootstrap_candidate) + for metric in QUALITY_METRICS: + sampled_deltas[metric][bootstrap_index] = deltas[metric] + baseline_p95 = bootstrap_baseline["p95_latency_seconds"] + if baseline_p95 <= 0.0: + raise ValueError("bootstrap baseline p95 latency must be greater than zero") + sampled_latency_ratio[bootstrap_index] = ( + bootstrap_candidate["p95_latency_seconds"] / baseline_p95 + ) + + lower_probability = (1.0 - CONFIDENCE_LEVEL) / 2.0 + upper_probability = 1.0 - lower_probability + + def interval(samples: np.ndarray) -> list[float]: + bounds = np.quantile( + samples, + [lower_probability, upper_probability], + method=QUANTILE_METHOD, + ) + return [float(bounds[0]), float(bounds[1])] + + return { + "aggregate": { + "baseline": aggregate_baseline, + "candidate": aggregate_candidate, + }, + "paired_delta_ci95": { + metric: interval(sampled_deltas[metric]) for metric in QUALITY_METRICS + }, + "p95_latency_ratio_ci95": interval(sampled_latency_ratio), + } From b40ad561dbbd4fdfb5d805b69657b9f446d33bc4 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:25:12 +0900 Subject: [PATCH 165/216] feat(mir): bind aggregation and bootstrap semantics --- .../validate_structure_noninferiority.py | 50 ++++++++++++++++--- 1 file changed, 43 insertions(+), 7 deletions(-) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index e6c740ac9..1f4af1d4b 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -1,11 +1,10 @@ #!/usr/bin/env python3 -"""Bind exact structure-metric semantics into the scientific registration digest. +"""Bind exact scientific semantics into the structure experiment identity. The base registration schema owns corpus, experiment runtime, margins, and -paired-decision fields. This façade adds the repository-owned metric contract to -the canonical digest without making callers repeat immutable adapter constants -inside every registration JSON. No MIR metric or corpus outcome is computed -here. +result-policy fields. This façade adds repository-owned metric, aggregation, and +paired-uncertainty semantics to the canonical digest without making experiment +authors duplicate immutable implementation constants in registration JSON. """ from __future__ import annotations @@ -45,6 +44,23 @@ "frame_size_seconds": 0.1, "beta": 1.0, }, + "aggregation_uncertainty": { + "aggregation_id": "macro-track-v1", + "sampling_unit": "registered_track_pair", + "quality_weighting": "equal_track", + "latency_weighting": "equal_track", + "boundary_and_repetition_f": "harmonic_of_macro_precision_recall", + "report_deviation": "macro_track_mean", + "peak_rss": "maximum_track_peak_rss", + "procedure_id": "paired-track-bootstrap-v1", + "confidence_level": 0.95, + "bootstrap_sample_size": "registered_track_count", + "bootstrap_replacement": True, + "rng": "numpy.random.Generator(PCG64)", + "quantile_method": "linear", + "two_sided_tail_probability": 0.025, + "maximum_resamples": 100000, + }, } @@ -68,8 +84,28 @@ def _base_validator() -> ModuleType: def validate_registration(registration_value: object) -> None: - """Validate the closed base registration consumed by the metric-aware digest.""" + """Validate base policy plus the one supported paired-bootstrap procedure.""" _BASE.validate_registration(registration_value) + if not isinstance(registration_value, Mapping): + raise ValueError("registration must be an object") + uncertainty = registration_value.get("uncertainty") + if not isinstance(uncertainty, Mapping): + raise ValueError("uncertainty must be an object") + contract = STRUCTURE_METRIC_CONTRACT["aggregation_uncertainty"] + if uncertainty.get("procedure_id") != contract["procedure_id"]: + raise ValueError( + "uncertainty.procedure_id must equal paired-track-bootstrap-v1" + ) + resamples = uncertainty.get("resamples") + if isinstance(resamples, bool) or not isinstance(resamples, int): + raise ValueError("uncertainty.resamples must be an integer") + maximum_resamples = contract["maximum_resamples"] + assert isinstance(maximum_resamples, int) + if resamples > maximum_resamples: + raise ValueError( + f"uncertainty.resamples must be <= {maximum_resamples} for " + "paired-track-bootstrap-v1" + ) def _digest_payload(registration_value: object) -> dict[str, object]: @@ -82,7 +118,7 @@ def _digest_payload(registration_value: object) -> dict[str, object]: def registration_digest(registration_value: object) -> str: - """Return SHA-256 over registration data plus exact metric/runtime semantics.""" + """Return SHA-256 over registration data plus exact scientific semantics.""" canonical = json.dumps( _digest_payload(registration_value), allow_nan=False, From 62ae2cd6590b0be18e9a5ab4c55dc522623ec3b2 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:25:41 +0900 Subject: [PATCH 166/216] test(mir): bind aggregation owner to preregistration digest --- ...ructure_metric_preregistration_contract.py | 56 +++++++++++++++++-- 1 file changed, 50 insertions(+), 6 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py b/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py index df2d6e0da..b41e8cee9 100644 --- a/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py +++ b/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py @@ -1,7 +1,8 @@ -"""Regression tests for exact structure metric preregistration semantics.""" +"""Regression tests for exact structure scientific preregistration semantics.""" from __future__ import annotations +import copy import hashlib import json from types import ModuleType @@ -36,6 +37,23 @@ "frame_size_seconds": 0.1, "beta": 1.0, }, + "aggregation_uncertainty": { + "aggregation_id": "macro-track-v1", + "sampling_unit": "registered_track_pair", + "quality_weighting": "equal_track", + "latency_weighting": "equal_track", + "boundary_and_repetition_f": "harmonic_of_macro_precision_recall", + "report_deviation": "macro_track_mean", + "peak_rss": "maximum_track_peak_rss", + "procedure_id": "paired-track-bootstrap-v1", + "confidence_level": 0.95, + "bootstrap_sample_size": "registered_track_count", + "bootstrap_replacement": True, + "rng": "numpy.random.Generator(PCG64)", + "quantile_method": "linear", + "two_sided_tail_probability": 0.025, + "maximum_resamples": 100000, + }, } @@ -57,8 +75,8 @@ def _canonical_digest(value: object) -> str: return hashlib.sha256(payload).hexdigest() -def test_registration_digest_binds_exact_metric_runtime_and_adapter_semantics() -> None: - """Scientific identity includes every reviewed mir_eval argument and lock digest.""" +def test_registration_digest_binds_exact_scientific_semantics() -> None: + """Scientific identity includes metric runtime, adapters, and aggregation procedure.""" validator = _validator() registration = _registration() @@ -75,8 +93,8 @@ def test_registration_digest_binds_exact_metric_runtime_and_adapter_semantics() assert validator.registration_digest(dict(reversed(list(registration.items())))) == expected -def test_digest_contract_matches_metric_adapter_and_runtime_lock_owners() -> None: - """Duplicated digest metadata cannot drift from the executable metric owners.""" +def test_digest_contract_matches_executable_metric_and_aggregation_owners() -> None: + """Digest metadata cannot drift from lock, metric, or paired-bootstrap owners.""" validator = _validator() adapter = load_module( "scripts/research/evaluate_structure_segmentation_metrics.py", @@ -86,6 +104,10 @@ def test_digest_contract_matches_metric_adapter_and_runtime_lock_owners() -> Non "scripts/research/verify_structure_metric_runtime_lock.py", "structure_metric_runtime_owner_for_registration", ) + aggregation = load_module( + "scripts/research/aggregate_structure_noninferiority.py", + "structure_aggregation_owner_for_registration", + ) contract = validator.STRUCTURE_METRIC_CONTRACT assert contract["runtime_lock_sha256"] == hashlib.sha256( @@ -112,10 +134,32 @@ def test_digest_contract_matches_metric_adapter_and_runtime_lock_owners() -> Non "frame_size_seconds": adapter.PAIRWISE_FRAME_SIZE_SECONDS, "beta": adapter.PAIRWISE_BETA, } + assert contract["aggregation_uncertainty"] == ( + aggregation.AGGREGATION_UNCERTAINTY_CONTRACT + ) + + +def test_registration_rejects_unsupported_uncertainty_implementation() -> None: + """A free-form procedure label cannot bypass the one executable bootstrap owner.""" + validator = _validator() + unsupported = _registration() + uncertainty = unsupported["uncertainty"] + assert isinstance(uncertainty, dict) + uncertainty["procedure_id"] = "post-hoc-bootstrap" + + with pytest.raises(ValueError, match="procedure_id"): + validator.validate_registration(unsupported) + + excessive = copy.deepcopy(_registration()) + excessive_uncertainty = excessive["uncertainty"] + assert isinstance(excessive_uncertainty, dict) + excessive_uncertainty["resamples"] = 100001 + with pytest.raises(ValueError, match="resamples"): + validator.validate_registration(excessive) def test_result_rejects_receipt_bound_only_to_the_legacy_registration_digest() -> None: - """A receipt cannot omit the metric contract by hashing registration JSON alone.""" + """A receipt cannot omit the scientific contract by hashing registration JSON alone.""" validator = _validator() registration = _registration() metric_aware_digest = validator.registration_digest(registration) From d297a66029652092c6ad44a1d391ee4072a64a0d Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:26:52 +0900 Subject: [PATCH 167/216] test(mir): allow asymmetric percentile bootstrap intervals --- ...st_structure_noninferiority_aggregation.py | 25 +++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py b/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py index 792b0bdf7..1513afba9 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py @@ -7,6 +7,8 @@ import pytest from conftest import load_module +from test_structure_noninferiority_policy import _registration, _result + def _aggregation() -> ModuleType: return load_module( @@ -15,6 +17,13 @@ def _aggregation() -> ModuleType: ) +def _validator() -> ModuleType: + return load_module( + "scripts/research/validate_structure_noninferiority.py", + "validate_structure_noninferiority_for_bootstrap", + ) + + def _side( *, boundary_precision: float, @@ -183,6 +192,22 @@ def test_paired_bootstrap_is_deterministic_and_resamples_track_pairs() -> None: assert first["p95_latency_ratio_ci95"][0] > 0.0 +def test_percentile_interval_is_not_required_to_contain_original_point_estimate() -> None: + """Percentile bootstrap intervals may be asymmetric around the observed statistic.""" + validator = _validator() + registration = _registration() + digest = validator.registration_digest(registration) + result = _result(registration, digest) + intervals = result["paired_delta_ci95"] + assert isinstance(intervals, dict) + intervals["boundary_f_0_5"] = [0.001, 0.010] + result["p95_latency_ratio_ci95"] = [0.60, 0.65] + + decision = validator.evaluate_result(registration, result) + + assert decision["passed"] is True + + def test_aggregation_fails_closed_on_duplicate_tracks_or_unsupported_uncertainty() -> None: aggregation = _aggregation() duplicate = _tracks() From aad0492574318b84bdc3702b8a9524677e6bc5c8 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:35:50 +0900 Subject: [PATCH 168/216] fix(mir): preserve percentile bootstrap interval semantics --- .../validate_structure_noninferiority.py | 105 +++++++++++++++++- 1 file changed, 104 insertions(+), 1 deletion(-) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index 1f4af1d4b..6a5673e4d 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -64,6 +64,16 @@ } +_F_COMPONENTS = { + "boundary_f_0_5": ("boundary_precision_0_5", "boundary_recall_0_5"), + "boundary_f_3_0": ("boundary_precision_3_0", "boundary_recall_3_0"), + "repetition_pairwise_f": ( + "repetition_pairwise_precision", + "repetition_pairwise_recall", + ), +} + + def _base_validator() -> ModuleType: """Load the unchanged base decision-policy implementation.""" path = Path(__file__).with_name("validate_structure_noninferiority_base.py") @@ -129,6 +139,91 @@ def registration_digest(registration_value: object) -> str: return hashlib.sha256(canonical).hexdigest() +def _is_legacy_interval_containment_error(error: ValueError) -> bool: + """Return whether the base validator rejected only its old CI-centering rule.""" + message = str(error) + return message.endswith("must contain the aggregate point delta") or message == ( + "result.p95_latency_ratio_ci95 must contain the aggregate p95 ratio" + ) + + +def _project_percentile_intervals_for_base( + result_value: Mapping[str, Any], +) -> dict[str, Any]: + """Adapt percentile intervals to the legacy base validator without changing bounds. + + Percentile-bootstrap endpoints are empirical quantiles of the bootstrap statistic + and are not required to bracket the observed statistic. The base validator predates + the frozen percentile procedure and imposed that extra invariant. This projection + changes only aggregate point values in a private validation copy so the base module + can still validate every envelope/measurement invariant and apply the *original* + interval bounds to its noninferiority and latency decisions. + """ + projected = copy.deepcopy(dict(result_value)) + aggregate = _BASE._mapping(projected.get("aggregate"), "result.aggregate") + baseline = _BASE._validate_measurement_side( + aggregate.get("baseline"), + "result.aggregate.baseline", + ) + candidate = _BASE._validate_measurement_side( + aggregate.get("candidate"), + "result.aggregate.candidate", + ) + raw_intervals = _BASE._mapping( + projected.get("paired_delta_ci95"), + "result.paired_delta_ci95", + ) + candidate_projection = dict(aggregate["candidate"]) + + for metric_name in _BASE._QUALITY_METRICS: + interval = _BASE._confidence_interval( + raw_intervals[metric_name], + f"result.paired_delta_ci95.{metric_name}", + ) + baseline_value = baseline[metric_name] + point_delta = candidate[metric_name] - baseline_value + if interval[0] <= point_delta <= interval[1]: + continue + + feasible_lower = max(interval[0], -baseline_value) + feasible_upper = min(interval[1], 1.0 - baseline_value) + if feasible_lower > feasible_upper: + raise ValueError( + f"result.paired_delta_ci95.{metric_name} has no feasible score delta" + ) + projected_delta = min(max(point_delta, feasible_lower), feasible_upper) + projected_score = baseline_value + projected_delta + candidate_projection[metric_name] = projected_score + for component_name in _F_COMPONENTS.get(metric_name, ()): + candidate_projection[component_name] = projected_score + + latency_interval = _BASE._confidence_interval( + projected.get("p95_latency_ratio_ci95"), + "result.p95_latency_ratio_ci95", + ) + baseline_p95 = baseline["p95_latency_seconds"] + if baseline_p95 > 0.0: + point_ratio = candidate["p95_latency_seconds"] / baseline_p95 + if not latency_interval[0] <= point_ratio <= latency_interval[1]: + projected_ratio = ( + latency_interval[0] + if point_ratio < latency_interval[0] + else latency_interval[1] + ) + if projected_ratio > 0.0: + projected_p95 = baseline_p95 * projected_ratio + candidate_projection["p95_latency_seconds"] = projected_p95 + candidate_projection["p50_latency_seconds"] = min( + candidate_projection["p50_latency_seconds"], + projected_p95, + ) + + aggregate_projection = dict(aggregate) + aggregate_projection["candidate"] = candidate_projection + projected["aggregate"] = aggregate_projection + return projected + + def evaluate_result( registration_value: object, result_value: object, @@ -147,7 +242,15 @@ def evaluate_result( projected_result["registration_sha256"] = _BASE.registration_digest( registration_value ) - decision = _BASE.evaluate_result(registration_value, projected_result) + try: + decision = _BASE.evaluate_result(registration_value, projected_result) + except ValueError as error: + if not _is_legacy_interval_containment_error(error): + raise + percentile_projection = _project_percentile_intervals_for_base( + projected_result + ) + decision = _BASE.evaluate_result(registration_value, percentile_projection) decision["registration_sha256"] = expected_digest return decision From 69d97f6c7bd1a13ecd483ad23e40a94eee061e5e Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:38:07 +0900 Subject: [PATCH 169/216] docs(mir): align bootstrap decision traceability --- .../mir/structure-feature-noninferiority.md | 111 ++++++++++-------- 1 file changed, 65 insertions(+), 46 deletions(-) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index 83eb23e58..d87011ea3 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -2,18 +2,18 @@ Status: Proposed Owner: Signal-MIR Analysis -Tracking: #1225 +Tracking: #1225, #1228 Production baseline: `chroma_cqt` in `sections/segmenter.py` ## Decision The earlier `chroma_cqt` → `chroma_stft` optimization is not a production change until a preregistered, rights-cleared real-audio experiment shows that the candidate is noninferior on rehearsal-relevant structure quality and materially faster on the same corpus and runtime identity. -The repository owns a result-admission boundary in `scripts/research/validate_structure_noninferiority.py`. The validator does **not** calculate MIR metrics and does not choose scientific margins or an uncertainty procedure. It freezes the reviewed experiment contract, including its claim boundary, binds a result receipt to that contract by canonical SHA-256, requires one explicit success-or-failure receipt per registered track, and applies the registered aggregate confidence-interval decision rules only when the complete corpus produced measurements. Metric computation and uncertainty estimation remain separate scientific measurement steps. +The repository owns the registration/result-admission boundary in `scripts/research/validate_structure_noninferiority.py` and the paired aggregation/uncertainty implementation in `scripts/research/aggregate_structure_noninferiority.py`. Metric computation, corpus admission, aggregation, uncertainty estimation, and result admission remain separate scientific boundaries, but their result-affecting semantics are bound into one metric-aware registration digest before candidate results may be inspected. -The canonical public validator is now a metric-aware façade over the unchanged base decision/result policy in `validate_structure_noninferiority_base.py`. Its SHA-256 binds both the validated registration data and repository-owned structure-metric semantics: the exact research metric-lock digest plus all result-affecting `mir_eval.segment.detection`, `segment.deviation`, and `segment.pairwise` arguments. A receipt carrying only the former registration-JSON digest is rejected. This closes a scientific identity gap without asking each experiment author to duplicate immutable adapter constants in registration JSON. +The canonical public validator is a metric-aware façade over the historical base decision/result policy in `validate_structure_noninferiority_base.py`. Its SHA-256 binds the validated registration together with repository-owned structure-metric and aggregation/uncertainty semantics. A receipt carrying only the former registration-JSON digest is rejected. -This keeps a performance result from silently changing the corpus, thresholds, feature identities, metric implementation/runtime, uncertainty procedure, runtime, or permitted interpretation after the measurements are visible. +This keeps a performance result from silently changing the corpus, thresholds, feature identities, metric implementation/runtime, aggregation weighting, bootstrap procedure, runtime, or permitted interpretation after measurements are visible. ## Registered inputs @@ -22,31 +22,29 @@ A registration is valid only when it records all of the following before the res - baseline `chroma_cqt` and candidate `chroma_stft`; - a rights-cleared real-audio corpus with stable track IDs, content-unique audio SHA-256 identities, annotation SHA-256, rights basis, and provenance URI; - exact source commit and `uv.lock` identity plus Python, librosa, NumPy, sample rate, channel count, and host profile; -- the caller-selected metric thresholds and the pinned MIREX functional-ACC evaluator commit/function, 200 ms frame-grid contract v1, and versioned annotation/label-mapping semantics; -- the paired-uncertainty procedure identity, confidence level, resample count, and random seed; +- caller-selected noninferiority/speed thresholds and the pinned MIREX functional-ACC evaluator commit/function, 200 ms frame-grid contract v1, and versioned annotation/label-mapping semantics; +- the paired-bootstrap procedure identity, 95% confidence level, resample count, and random seed; - a non-empty claim boundary stating the population/runtime scope to which a passing result may be applied. -The registration digest additionally binds the repository-owned segmentation-metric contract. The bound contract includes the SHA-256 of `services/analysis-engine/requirements-structure-metrics.lock`, detection windows 0.5 s and 3.0 s with `beta=1.0, trim=True`, deviation with `trim=True`, and pairwise grouping with `frame_size=0.1, beta=1.0`. `test_structure_metric_preregistration_contract.py` cross-checks those digest constants against the executable adapter and runtime-lock owners so the metadata cannot silently diverge while focused tests remain green. +The registration digest additionally binds the repository-owned segmentation-metric contract. That contract fixes the SHA-256 of `services/analysis-engine/requirements-structure-metrics.lock`, detection windows 0.5 s and 3.0 s with `beta=1.0, trim=True`, deviation with `trim=True`, pairwise grouping with `frame_size=0.1, beta=1.0`, and the complete `macro-track-v1` / `paired-track-bootstrap-v1` aggregation and uncertainty semantics. Focused contract tests cross-check those constants against the executable owners so metadata drift fails before real-audio execution. -Schema v1 is closed-world not only at the registration top level but also for the hypothesis object, every registered metric configuration, every corpus-track object, and the runtime object. Extra fields are not treated as harmless annotations: an unregistered pilot-selection flag, metric weight, corpus-selection marker, or cache/runtime hint changes what reviewers may infer was preregistered, so it fails admission instead of silently entering the evidence artifact. The claim boundary is itself a required top-level registration field and therefore participates in the canonical registration SHA-256. +Schema v1 is closed-world at the registration top level and inside the hypothesis, metric, corpus-track, runtime, result, per-track, aggregate, and measurement objects. Extra fields are not treated as harmless annotations: an unregistered pilot-selection flag, metric weight, corpus-selection marker, cache/runtime hint, p-value, or post-hoc selection marker changes what reviewers may infer was preregistered and therefore fails admission. -`experiment_id` is a semantic label, while the registration SHA-256 is the exact evidence identity. The validator compares the registration and result experiment IDs after the same required-text normalization, but the canonical hash preserves the validated registration representation inside a closed envelope together with the immutable repository-owned structure-metric contract. This prevents an otherwise valid registration with surrounding whitespace from becoming impossible to reference while keeping byte-distinct registrations and metric-contract revisions cryptographically distinct. +`experiment_id` is a semantic label; the metric-aware registration SHA-256 is the exact evidence identity. The canonical hash preserves the validated registration inside a closed envelope together with immutable repository-owned scientific semantics. -The validator requires a 95% confidence level because the current result schema is explicitly `ci95`; it does not prescribe which scientifically defensible paired procedure must produce that interval. The procedure identifier, resample count, and seed are part of the preregistration digest so they cannot be changed after results are seen. `paired-track-bootstrap-v1` and the numeric values used in unit tests are policy fixtures only, not an approved BandScope production analysis plan. +`source_uri` must be an explicit non-`file:` provenance URI. Absolute, relative, drive-relative, and `file:` filesystem forms are rejected. Licensed benchmark material may remain private, but the receipt must identify it without making a workstation path part of scientific provenance. -The validator requires `source_uri` to be an explicit non-`file:` URI. Absolute, relative, drive-relative, and `file:` filesystem forms are rejected as provenance authorities. A benchmark may remain private when licensing requires that, but the receipt must identify the licensed material without leaking the workstation path that happened to hold it. +Track IDs are labels, not independent scientific units by themselves. Schema v1 rejects duplicate `audio_sha256` values across different track IDs so one recording cannot silently count as multiple independent observations. A scientifically justified repeated-item or clustered design requires a preregistered dependence model and schema revision. -Track IDs are labels, not independent scientific units by themselves. Schema v1 rejects duplicate `audio_sha256` values across distinct track IDs so the same audio bytes cannot be counted repeatedly as apparent corpus breadth or independent paired observations. A scientifically justified repeated-item or clustered design would require an explicit preregistered dependence model and a schema revision rather than aliasing one recording under several IDs. +This guard is consistent with recent MIR dataset-quality work. Choi et al. (2025) show that duplicated music items can undermine evaluation through leakage and motivate explicit de-duplication in a large symbolic-music benchmark. That study does not establish BandScope's audio-structure independence assumptions; it supports the narrower rule that duplicate content identity must not masquerade as distinct evaluation evidence. -This guard is also consistent with recent MIR dataset-quality work. Choi et al. (2025) show that duplicated music items can make evaluation unreliable through leakage and motivate explicit de-duplication in a large music benchmark. Their study is on symbolic MIDI and therefore does **not** establish BandScope's audio-structure independence assumptions; it supports the narrower engineering rule that duplicate content identity must not silently masquerade as distinct evaluation evidence. +The current minimum of two tracks is only a technical guard against a one-item corpus. It is not a scientific sample-size claim. Corpus breadth, genre/instrumentation coverage, annotation quality, independence/dependence structure, and an adequate uncertainty plan remain experiment-review questions before a production switch can be accepted. -The current minimum of two tracks is only a technical guard against treating one timing sample as a corpus. It is **not** a scientific sample-size claim. Corpus breadth, genre/instrumentation coverage, annotation quality, independence/dependence structure, and a defensible power/uncertainty plan remain part of the experiment review before a production switch can be accepted. - -Schema v1 does not contain a preregistered dropout, exclusion, or missing-track policy. Therefore every registered track must complete both baseline and candidate measurement for an acceptance PASS. A measurement failure is still evidence: its track ID remains in registered corpus order, appears in `failed_tracks`, and uses a closed failed-track receipt containing only `track_id` rather than invented MIR or latency values. Any non-empty `failed_tracks` list makes the acceptance decision fail. Because an aggregate or confidence interval computed after such a failure would necessarily summarize an unregistered complete-case subset, schema v1 also requires `aggregate`, `paired_delta_ci95`, and `p95_latency_ratio_ci95` to be JSON `null` whenever `failed_tracks` is non-empty. Conversely, omitting baseline/candidate measurements without declaring the same track failed is rejected. A future tolerance for failed or excluded tracks requires a reviewed preregistration rule and schema revision rather than post-result omission. +Schema v1 has no dropout, exclusion, or missing-track policy. Every registered track must therefore complete both baseline and candidate measurement for an acceptance PASS. Any non-empty `failed_tracks` list makes acceptance fail; aggregate and confidence-interval fields must then be JSON `null` so a complete-case subset cannot masquerade as the preregistered corpus. ## Metrics -BandScope uses the MIREX Music Structure Analysis task and its standardized evaluator repository as the functional-structure reference point. The registered quality gate requires: +BandScope uses the MIREX Music Structure Analysis task and its standardized evaluator repository as the functional-structure reference point. | Evidence | Registered implementation / convention | Decision use | | --- | --- | --- | @@ -58,68 +56,87 @@ BandScope uses the MIREX Music Structure Analysis task and its standardized eval | Latency | paired p50/p95 on the registered host | p95 ratio superiority | | Peak memory | peak RSS on the registered host | Report per track and aggregate | -The selected mir_eval calls execute only under the content-addressed research overlay whose exact lock SHA-256 is part of the canonical registration digest. The runtime verifier separately requires the installed distribution to be `mir_eval==0.8.2` before these metrics run. Full artifact provenance and adapter semantics are recorded in `docs/traceability/mir/structure-segmentation-metric-adapter.md`. +The selected mir_eval calls execute only under the content-addressed research overlay whose exact lock SHA-256 participates in the registration digest. The runtime verifier separately requires the installed distribution to be `mir_eval==0.8.2` before recognized segmentation metrics run. Full artifact provenance and adapter semantics are recorded in `docs/traceability/mir/structure-segmentation-metric-adapter.md`. + +The current official `ismir-mirex/mirex-evaluation` MIREX-2025 reproduction at commit `b9fa0b0b32e2145af31f35830f78fc9d09a4301b` uses a 0.2 s functional-ACC hop, `np.arange(0, gt_duration, frame_hop)`, and advances a segment when a frame point is greater than or equal to its end. BandScope pins that executable evaluator identity and grid rather than choosing a frame size from prose examples. Source-label normalization must occur before preregistration and is bound by the admitted annotation SHA-256; it cannot change after candidate results are visible. + +Result receipts carry precision and recall alongside F for boundary and repetition measures. Admission recomputes the harmonic mean and rejects inconsistent P/R/F triplets. That integrity check does not replace the recognized metric implementation. + +Published MIREX scores are external context, not BandScope noninferiority margins. Acceptable loss and required latency improvement are product/scientific decisions that must be fixed before candidate measurement. + +## Aggregation and paired uncertainty -The MIREX task page describes frame-level ACC conceptually and gives finer resolutions as examples, but the current official `ismir-mirex/mirex-evaluation` MIREX-2025 reproduction is more specific: at commit `b9fa0b0b32e2145af31f35830f78fc9d09a4301b`, `calculate_accuracy` uses a 0.2 s hop, `np.arange(0, gt_duration, frame_hop)`, and advances a segment when a frame point is greater than or equal to the segment end. BandScope therefore pins that exact evaluator identity and frame-grid contract rather than selecting 100 ms from prose examples. The local annotation mapping remains preregistered and content-addressed before evaluation; raw-label normalization is not allowed to drift after candidate results are visible. The detailed contract is recorded in `docs/traceability/mir/functional-label-accuracy-contract.md`, and `scripts/research/evaluate_structure_functional_accuracy.py` is the repository-owned adapter for the normalized seven-label subset. +`aggregate_structure_noninferiority.py` owns the frozen `macro-track-v1` / `paired-track-bootstrap-v1` procedure. The registered track pair is the sampling unit; baseline and candidate observations for a track are never separated during resampling. -`mir_eval.segment.detection` uses one-to-one boundary matching within the selected tolerance; `mir_eval.segment.deviation` reports the median nearest-boundary deviations in both directions; `mir_eval.segment.pairwise` measures structural grouping agreement. These are different questions and must not be collapsed into one score. +The aggregate is deliberately macro-oriented rather than duration weighted: -Result receipts must carry the reported precision and recall alongside F for the boundary and repetition measures. The admission validator recomputes the harmonic mean and rejects an internally inconsistent P/R/F triplet. This does not replace the recognized metric implementation; it prevents a malformed receipt from claiming a metric value that its own reported components cannot support. +- every registered track has equal weight for quality and latency summaries; +- boundary and repetition F are recomputed as the harmonic mean of macro precision and macro recall rather than averaging per-track F directly; +- boundary-deviation summaries are macro track means; +- functional-label ACC is the macro track mean; +- p50 and p95 latency are macro track means; +- peak RSS is the maximum observed per-track peak, not a mean. -The published MIREX 2025 result table reports the MusicFM baseline at ACC 0.705, HR.5 0.644, and HR3 0.710. Those values are useful external context, **not** BandScope's noninferiority margins. A margin is a product/scientific decision about acceptable loss relative to BandScope's own current baseline on the registered corpus. It must be fixed before examining candidate results. +For uncertainty, each bootstrap replicate samples `registered_track_count` paired track indices with replacement. The implementation uses `numpy.random.Generator(PCG64)` with the preregistered seed. The 95% interval is the 0.025 and 0.975 empirical quantile of the bootstrap statistic using NumPy's `linear` quantile method. Quality intervals are formed for candidate-minus-baseline macro statistics; the latency interval is formed for the candidate/baseline macro p95 ratio. The registration accepts 1,000 through 100,000 resamples for this procedure, with the exact count and seed included in the registration digest. + +This is a percentile-bootstrap interval, not a symmetric interval around the observed statistic. Its endpoints are quantiles of the bootstrap distribution and can be asymmetric; the observed aggregate statistic is therefore **not required** to lie between those endpoints. Hall (1988) treats the percentile family as a bootstrap-quantile construction and documents that different bootstrap interval methods have materially different coverage and positioning properties. Adding a separate point-containment requirement would define a different, undocumented interval rule. + +The historical base validator predates the frozen percentile procedure and still contains that obsolete containment invariant. The metric-aware façade handles only those two legacy containment errors by creating a private validation projection whose aggregate point values are moved onto feasible interval bounds. The stored receipt, track measurements, aggregate values, and interval bounds are not modified. The base validator then continues to enforce every structural/measurement invariant and applies the original interval bounds to the noninferiority and latency decisions. Any non-containment validation error is propagated unchanged. `test_percentile_interval_is_not_required_to_contain_original_point_estimate` is the executable regression for this boundary. + +The procedure being executable and preregisterable does not by itself establish that a particular corpus size, resample count, seed, margin, or claim boundary is scientifically adequate. Those remain experiment-review decisions. A different bootstrap method such as BCa, studentized, cluster/bootstrap-at-recording-level, or a different aggregation weighting requires a new versioned contract before candidate results are inspected. ## Decision rule -For each quality metric `m`, let `Δm = candidate - baseline`. The result passes that quality criterion only when the lower bound of the registered paired 95% confidence interval satisfies: +For each quality metric `m`, let `Δm = candidate - baseline`. The quality criterion passes only when: `lower_CI95(Δm) >= -noninferiority_margin(m)` -For p95 latency, let `r = candidate_p95 / baseline_p95`. The result passes the speed criterion only when the **upper** bound of the paired 95% interval satisfies: +For p95 latency, let `r = candidate_p95 / baseline_p95`. The speed criterion passes only when: `upper_CI95(r) <= maximum_candidate_ratio` -The result receipt must repeat the preregistered uncertainty plan and claim boundary exactly. On a complete run, the validator also requires the aggregate point delta/ratio to lie inside the supplied interval. Favorable point estimates therefore cannot hide an uncertainty interval that crosses a preregistered boundary, the uncertainty method cannot be swapped after the result is known, and a narrow experiment cannot be relabeled as evidence for a wider population or runtime after measurement. +The decision is based on preregistered interval bounds, not on whether the interval contains the observed point estimate. A favorable point estimate cannot hide an interval that crosses the registered threshold, and an unfavorable interval cannot be repaired by choosing another seed, resample count, aggregation rule, or claim boundary after results are known. -Because schema v1 has no preregistered exclusion policy, a non-empty `failed_tracks` list fails before aggregate decision rules are evaluated. The failed run preserves track-level diagnostic identity but must carry `null` aggregate and CI summaries, so a complete-case subset cannot masquerade as the preregistered whole-corpus analysis. +Because schema v1 has no exclusion policy, any failed track fails before aggregate decision rules are evaluated. Failed runs preserve ordered diagnostic track identity but carry null aggregate/CI summaries. -The repository does not currently contain approved numeric margins or an approved production corpus/uncertainty procedure. Unit-test values are synthetic policy fixtures only and must never be cited as production acceptance thresholds or scientific design decisions. +Synthetic unit-test values are policy fixtures only. They are not production margins, corpus evidence, or a power justification. ## Result receipt -A result receipt must contain the exact metric-aware registration digest, exact uncertainty plan, exact corpus order, one ordered track receipt per registered track, failed-track IDs, and the exact preregistered claim boundary. A successful complete run additionally carries aggregate measurements, paired 95% intervals for every gated quality metric, and a paired p95-latency-ratio interval. A successful track receipt contains `track_id`, `baseline`, and `candidate`; a track listed in `failed_tracks` contains only `track_id`, preserving the failure without manufacturing scores that were never measured. When any track fails, the three aggregate/CI fields remain present in the closed top-level schema but their values must be JSON `null`. +A result receipt contains the exact metric-aware registration digest, exact uncertainty plan, exact corpus order, one ordered track receipt per registered track, failed-track IDs, and the exact preregistered claim boundary. A complete successful run additionally carries aggregate measurements, paired 95% intervals for each gated quality metric, and a paired p95-latency-ratio interval. -Schema v1 treats the result envelope, each per-track receipt, and the aggregate envelope as closed-world objects. A top-level post-result field such as an unregistered p-value, a per-track `selected_for_aggregate` flag, or aggregate-side selection metadata is rejected rather than ignored. The same rule already applies inside each baseline/candidate measurement object. This prevents a producer from attaching an unreviewed post-hoc selection or inferential claim to an otherwise accepted receipt and having that field travel with the admitted evidence artifact. +Each successful per-track and aggregate baseline/candidate measurement is a closed-world receipt containing exactly the registered quality fields plus reporting precision/recall, boundary-deviation, latency, and peak-RSS fields. A declared failed track contains only `track_id`; attaching partial or invented measurements to a failed receipt is rejected. -Each successful per-track and aggregate baseline/candidate measurement is a closed-world receipt: it must contain exactly the registered quality fields plus the reporting-only precision/recall, boundary-deviation, latency, and peak-RSS fields. Missing fields and additional post-hoc measurement fields both fail admission. A declared failed track is the only per-track exception and is instead closed to exactly `track_id`; attaching partial or invented measurement fields to that failed receipt is rejected. This prevents a result producer from filling a real measurement failure with synthetic values merely to satisfy the evidence schema. +The receipt is rejected when a track is omitted or reordered; missing measurements are not paired with an explicit failed-track declaration; a failed receipt carries measurements; a failed run carries non-null aggregate/CI summaries; the uncertainty plan or claim boundary differs; required metrics are absent; unregistered fields appear; a P/R/F triplet is inconsistent; values are non-finite; interval bounds are malformed; or the metric-aware registration digest differs. Percentile intervals are not rejected merely because an observed aggregate point lies outside them. -The receipt is rejected when a track is omitted/reordered, missing measurements are not paired with an explicit failed-track declaration, a failed receipt carries measurement fields, a failed run carries non-null aggregate/CI summaries, the uncertainty plan differs, the claim boundary differs, a required metric is absent, an unregistered envelope or measurement field is added, a boundary/repetition P/R/F triplet is internally inconsistent, a value is non-finite, a confidence interval does not contain its aggregate point estimate, or the metric-aware registration hash differs. A structurally valid receipt with one or more known failed tracks is retained for diagnosis with null aggregate summaries and evaluates to `passed=false` under schema v1. A passing receipt therefore means the preregistered decision rule passed for the complete registered corpus, metric contract, uncertainty procedure, runtime, and claim boundary. It does not generalize automatically to other genres, codecs, sample rates, machines, annotation regimes, or statistical procedures. - -The validator still does not recompute aggregate statistics or confidence intervals from per-track evidence. During review, a min/max range consistency heuristic was briefly encoded as a RED and then removed before production code changed because it would smuggle an unapproved assumption about aggregation into schema v1. Macro means, weighted means, micro-aggregated precision/recall/F and pooled latency summaries do not share one universally valid range relationship. The correct next step is to preregister the aggregation procedure and implement it in the experiment runner, not to infer a statistical method inside an admission validator after the fact. +The admission validator does not trust caller-authored aggregate/CI values as scientific authority. The canonical result producer is the repository-owned paired aggregator, and focused contract tests bind its immutable aggregation/uncertainty constants into the registration digest. Review must still confirm that the result artifact was generated by that owner from the complete admitted track set before production acceptance. ## Machine-readable evidence admission -Registration and result JSON are evidence, not trusted configuration. CLI admission is bounded to 2 MiB per file, reads from one already-open regular-file descriptor, requires UTF-8 and standards-compliant finite JSON values, and rejects duplicate object keys instead of accepting last-key-wins semantics. These controls prevent ambiguous evidence identities and bound memory use before scientific validation begins. +Registration and result JSON are evidence, not trusted configuration. CLI admission is bounded to 2 MiB per file, reads from one already-open regular-file descriptor, requires UTF-8 and standards-compliant finite JSON values, and rejects duplicate object keys rather than accepting last-key-wins behavior. -The JSON `source_uri` value remains provenance metadata only; it is never dereferenced by this validator. It must use an explicit non-file URI scheme; absolute, relative, drive-relative, and `file:` filesystem forms are rejected. Audio and annotation bytes are not opened by the evidence validator and are bound to the registration through SHA-256 identities supplied by the experiment process. Audio content identity must also be unique within schema-v1 corpus membership; a second track ID carrying the same `audio_sha256` fails admission. +`source_uri` is provenance metadata only and is never dereferenced by the validator. Audio and annotation bytes are not opened by result admission and are bound to the experiment through SHA-256 identities supplied by corpus admission. ## Reproducibility sequence -1. Review the rights basis, corpus composition, distinct audio-content identities, feature hypothesis, metric contract, pinned functional-ACC evaluator/grid/annotation/mapping contract, host profile, numeric margins, aggregation procedure, paired-uncertainty procedure, and claim boundary **before** running the candidate. Freeze the procedure identifier, confidence level, resample count, random seed, and claim boundary in the registration. If repeated recordings, clustering, source-label normalization, or any failure/exclusion tolerance is scientifically required, define and version that dependence/mapping/exclusion policy before measurement rather than remapping or omitting observations after results are visible. -2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain its canonical metric-aware SHA-256. The digest includes the validated registration plus the repository-owned structure metric lock/argument contract; it is not a hash of the registration JSON alone. -3. Run baseline and candidate on the same decoded track identities and host profile. Functional ACC must consume the admitted normalized annotation through `scripts/research/evaluate_structure_functional_accuracy.py` and preserve the pinned 200 ms MIREX grid semantics; it must not reopen corpus paths or remap labels after preregistration. Boundary/deviation/repetition metrics must use the registered research lock and explicit adapter semantics. The experiment runner must compute the approved aggregation and paired uncertainty procedure from the complete measured-track evidence rather than accepting caller-authored aggregate numbers as scientific authority. Record recognized MIR metrics, p50/p95 latency, and peak RSS for successful tracks. If a registered track cannot produce a baseline/candidate measurement, preserve its ordered `track_id` receipt, add that exact ID to `failed_tracks`, and set the aggregate plus both CI summary fields to JSON `null`; do not invent metric values or compute a complete-case aggregate from the surviving tracks. Only a complete run records aggregate outputs and paired uncertainty using the preregistered procedure. -4. Put the metric-aware registration digest, identical uncertainty-plan fields, and identical claim boundary in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. A receipt carrying only the former registration-JSON digest is rejected. Under schema v1, any recorded failed track keeps the diagnostic receipt valid but makes acceptance fail closed before aggregate decision rules; undeclared missing measurements, non-null post-failure summaries, and claim-boundary drift are rejected. -5. Preserve the registration, metric-aware digest contract, result receipt, corpus/annotation hashes, exact source commit, lock hash, aggregation implementation identity, uncertainty procedure, and claim boundary together. A production feature switch requires this evidence plus normal code review and protected-head checks. +1. Review rights basis, corpus composition, distinct audio identities, feature hypothesis, metric contract, functional-ACC grid/mapping, host profile, numeric margins, `macro-track-v1` aggregation, `paired-track-bootstrap-v1` resample count/seed, and claim boundary before candidate execution. If clustering, repeated recordings, alternative label mapping, or any exclusion tolerance is scientifically required, version that policy before measurement. +2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain the metric-aware SHA-256. The digest includes validated registration data plus repository-owned metric and aggregation/uncertainty contracts. +3. Admit the rights-cleared corpus once, then run CQT and STFT lanes on the same decoded PCM/reference segmentation. Functional ACC must preserve the pinned 200 ms MIREX grid semantics; boundary/deviation/repetition metrics must use the pinned research runtime and explicit adapter arguments. +4. Feed the complete ordered track measurements to `aggregate_structure_noninferiority.py`. Do not hand-author aggregate values or confidence intervals. If any registered track fails, preserve the failure receipt and emit null aggregate/CI summaries instead of a complete-case result. +5. Put the metric-aware registration digest, identical uncertainty fields, corpus order, and claim boundary in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. +6. Preserve registration, metric-aware digest contract, result receipt, corpus/annotation hashes, exact source/lock identities, aggregation implementation identity, uncertainty plan, and claim boundary together. A production feature switch requires this evidence plus normal independent review and protected-head checks. -No step authorizes committing licensed audio to Git. Rights-cleared means BandScope has the necessary evaluation right; redistribution is a separate permission. +No step authorizes committing licensed audio to Git. Evaluation rights and redistribution rights remain separate. ## Security Notes -- Audio and annotation files are untrusted inputs to the future experiment runner. This validator reads bounded JSON evidence only and does not open audio, execute subprocesses, make network requests, or follow paths from the registration. +- Audio and annotation are untrusted inputs to corpus/metric runners. Result admission reads bounded JSON only and does not execute subprocesses, make network requests, or follow corpus paths. - Evidence JSON is limited to 2 MiB, must be a regular file read through one open descriptor, must decode as UTF-8, and rejects duplicate keys plus non-standard `NaN`/`Infinity` constants. -- `source_uri` is evidence metadata, not an instruction to fetch content. Local absolute, relative, drive-relative, and `file:` forms are rejected; an explicit non-file URI scheme is required so transient workstation paths cannot become provenance authority or leak into review artifacts. -- Audio and annotation SHA-256 values bind measurements to bytes without embedding media in the result receipt; duplicate audio SHA-256 values under different track IDs are rejected so one recording cannot be silently counted multiple times. -- Invalid/non-finite measurements, corpus drift, duplicate audio content identity, functional-label evaluator/grid/annotation/mapping drift, segmentation metric-lock/argument drift, uncertainty-plan drift, claim-boundary drift, registration drift, undeclared missing measurements, measurement-bearing failed receipts, non-null aggregate/CI summaries after any failed track, unregistered registration/hypothesis/metric/corpus/runtime/result/track/aggregate/measurement fields, inconsistent P/R/F triplets, post-hoc metric additions, and unregistered failed-track exclusion fail closed. -- The validator does not claim that SHA-256 proves licensing, annotation validity, scientific adequacy, independence beyond exact-byte uniqueness, aggregate derivation, or that the registered statistical procedure is appropriate. Rights and scientific review remain separate gates. +- `source_uri` is evidence metadata, not a fetch instruction; local filesystem forms are rejected. +- Audio/annotation SHA-256 values bind measurements to bytes without embedding media; duplicate audio content identity under different track IDs is rejected. +- Invalid/non-finite measurements, corpus drift, metric/runtime drift, aggregation/uncertainty drift, claim-boundary drift, registration drift, undeclared missing measurements, failed-receipt fabrication, non-null summaries after a failed track, unregistered fields, and inconsistent P/R/F triplets fail closed. +- The percentile compatibility projection exists only inside result validation and never rewrites the stored scientific receipt or CI bounds. +- SHA-256 does not prove licensing, annotation validity, corpus adequacy, independence beyond exact-byte uniqueness, or statistical appropriateness. Rights and scientific review remain separate gates. ## References @@ -127,6 +144,8 @@ Buisson, M., McFee, B., Essid, S., & Crayencour, H. C. (2024). Self-supervised l Choi, E., Kim, H., Ryu, J., Nam, J., & Jeong, D. (2025). *On the de-duplication of the Lakh MIDI dataset* [Conference paper]. International Society for Music Information Retrieval Conference. https://doi.org/10.5281/zenodo.17811316 +Hall, P. (1988). Theoretical comparison of bootstrap confidence intervals. *The Annals of Statistics, 16*(3), 927–953. https://doi.org/10.1214/aos/1176350933 + Kim, T., & Nam, J. (2023). All-in-one metrical and functional structure analysis with neighborhood attentions on demixed audio. In *2023 IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA)* (pp. 1–5). IEEE. https://doi.org/10.1109/WASPAA58266.2023.10248148 MIREX. (2025). *Music Structure Analysis*. https://music-ir.org/mirex/wiki/2025:Music_Structure_Analysis From f866662ab94104dff24387f7876e89eb0a22d9a5 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:42:30 +0900 Subject: [PATCH 170/216] test(mir): preserve percentile CI decision thresholds --- ...tructure_percentile_bootstrap_admission.py | 49 +++++++++++++++++++ 1 file changed, 49 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_percentile_bootstrap_admission.py diff --git a/services/analysis-engine/tests/test_structure_percentile_bootstrap_admission.py b/services/analysis-engine/tests/test_structure_percentile_bootstrap_admission.py new file mode 100644 index 000000000..fad06db2a --- /dev/null +++ b/services/analysis-engine/tests/test_structure_percentile_bootstrap_admission.py @@ -0,0 +1,49 @@ +"""Regression tests for percentile-bootstrap result admission semantics.""" + +from __future__ import annotations + +from types import ModuleType + +from conftest import load_module +from test_structure_noninferiority_policy import _registration, _result + + +def _validator() -> ModuleType: + return load_module( + "scripts/research/validate_structure_noninferiority.py", + "validate_structure_noninferiority_percentile_admission", + ) + + +def test_asymmetric_quality_interval_still_fails_registered_noninferiority_margin() -> None: + """Removing point containment must not weaken the CI lower-bound decision.""" + validator = _validator() + registration = _registration() + result = _result(registration, validator.registration_digest(registration)) + intervals = result["paired_delta_ci95"] + assert isinstance(intervals, dict) + intervals["boundary_f_0_5"] = [-0.030, -0.025] + + decision = validator.evaluate_result(registration, result) + + assert decision["passed"] is False + assert any( + "boundary_f_0_5 paired CI lower bound" in requirement + for requirement in decision["failed_requirements"] + ) + + +def test_asymmetric_latency_interval_still_fails_registered_ratio_threshold() -> None: + """The compatibility projection must preserve the original CI upper bound.""" + validator = _validator() + registration = _registration() + result = _result(registration, validator.registration_digest(registration)) + result["p95_latency_ratio_ci95"] = [0.81, 0.85] + + decision = validator.evaluate_result(registration, result) + + assert decision["passed"] is False + assert any( + "p95 latency ratio CI upper bound" in requirement + for requirement in decision["failed_requirements"] + ) From c3f38dbb718269eeff22501973ab18558e352f4d Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 18:44:53 +0900 Subject: [PATCH 171/216] fix(test): satisfy percentile regression docstring gate --- .../tests/test_structure_percentile_bootstrap_admission.py | 1 + 1 file changed, 1 insertion(+) diff --git a/services/analysis-engine/tests/test_structure_percentile_bootstrap_admission.py b/services/analysis-engine/tests/test_structure_percentile_bootstrap_admission.py index fad06db2a..35024e35a 100644 --- a/services/analysis-engine/tests/test_structure_percentile_bootstrap_admission.py +++ b/services/analysis-engine/tests/test_structure_percentile_bootstrap_admission.py @@ -9,6 +9,7 @@ def _validator() -> ModuleType: + """Load the metric-aware noninferiority validator under a unique module name.""" return load_module( "scripts/research/validate_structure_noninferiority.py", "validate_structure_noninferiority_percentile_admission", From aacb026ed2e05dadce3fb0b7560836308bc73b92 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 19:07:00 +0900 Subject: [PATCH 172/216] test(mir): require canonical aggregation receipt binding --- ...tructure_noninferiority_receipt_binding.py | 47 +++++++++++++++++++ 1 file changed, 47 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_noninferiority_receipt_binding.py diff --git a/services/analysis-engine/tests/test_structure_noninferiority_receipt_binding.py b/services/analysis-engine/tests/test_structure_noninferiority_receipt_binding.py new file mode 100644 index 000000000..2e333c09b --- /dev/null +++ b/services/analysis-engine/tests/test_structure_noninferiority_receipt_binding.py @@ -0,0 +1,47 @@ +"""Regressions binding structure decision receipts to canonical aggregation output.""" + +from __future__ import annotations + +from types import ModuleType + +import pytest +from conftest import load_module +from test_structure_noninferiority_policy import _registration, _result + + +def _validator() -> ModuleType: + """Load the metric-aware noninferiority validator under a unique module name.""" + return load_module( + "scripts/research/validate_structure_noninferiority.py", + "validate_structure_noninferiority_receipt_binding", + ) + + +def test_result_rejects_aggregate_not_recomputed_from_registered_track_receipts() -> None: + """Track evidence cannot disagree with a favorable stored aggregate.""" + validator = _validator() + registration = _registration() + result = _result(registration, validator.registration_digest(registration)) + tracks = result["tracks"] + assert isinstance(tracks, list) + first_track = tracks[0] + assert isinstance(first_track, dict) + candidate = first_track["candidate"] + assert isinstance(candidate, dict) + candidate["functional_label_accuracy"] = 0.10 + + with pytest.raises(ValueError, match="canonical macro-track-v1 recomputation"): + validator.evaluate_result(registration, result) + + +def test_result_rejects_ci_not_recomputed_from_registered_track_receipts() -> None: + """A favorable hand-edited CI cannot become scientific decision evidence.""" + validator = _validator() + registration = _registration() + result = _result(registration, validator.registration_digest(registration)) + intervals = result["paired_delta_ci95"] + assert isinstance(intervals, dict) + intervals["boundary_f_0_5"] = [-0.010, 0.000] + + with pytest.raises(ValueError, match="canonical paired-track-bootstrap-v1 recomputation"): + validator.evaluate_result(registration, result) From 5f47924d632946540d0c31e0c6456feac4667f8b Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 19:07:52 +0900 Subject: [PATCH 173/216] fix(mir): bind aggregate receipt to registered tracks --- .../validate_structure_noninferiority.py | 36 +++++++++++++++++++ 1 file changed, 36 insertions(+) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index 6a5673e4d..2683d962b 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -88,6 +88,20 @@ def _base_validator() -> ModuleType: return module +def _aggregation() -> ModuleType: + """Load the canonical macro-track/bootstrap implementation on demand.""" + path = Path(__file__).with_name("aggregate_structure_noninferiority.py") + spec = importlib.util.spec_from_file_location( + "bandscope_structure_noninferiority_aggregation_for_validation", + path, + ) + if spec is None or spec.loader is None: + raise RuntimeError("could not load structure noninferiority aggregation") + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + _BASE = _base_validator() SCHEMA_VERSION = _BASE.SCHEMA_VERSION MAX_EVIDENCE_BYTES = _BASE.MAX_EVIDENCE_BYTES @@ -224,6 +238,26 @@ def _project_percentile_intervals_for_base( return projected +def _require_canonical_aggregate(result_value: Mapping[str, Any]) -> None: + """Bind a successful stored aggregate to the canonical registered-track reducer.""" + failed_tracks = result_value.get("failed_tracks") + if failed_tracks: + return + + aggregation = _aggregation() + _, baseline_sides, candidate_sides = aggregation._normalize_tracks( + result_value.get("tracks") + ) + expected_aggregate = { + "baseline": aggregation._aggregate_side(baseline_sides), + "candidate": aggregation._aggregate_side(candidate_sides), + } + if result_value.get("aggregate") != expected_aggregate: + raise ValueError( + "result.aggregate does not match canonical macro-track-v1 recomputation" + ) + + def evaluate_result( registration_value: object, result_value: object, @@ -251,6 +285,8 @@ def evaluate_result( projected_result ) decision = _BASE.evaluate_result(registration_value, percentile_projection) + + _require_canonical_aggregate(projected_result) decision["registration_sha256"] = expected_digest return decision From 4cdcf19a016c9b4d7a1c28ccc39784513700b297 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 19:10:16 +0900 Subject: [PATCH 174/216] test(mir): make result fixtures canonical bootstrap receipts --- .../test_structure_noninferiority_policy.py | 90 ++++++++++++------- 1 file changed, 57 insertions(+), 33 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_policy.py index 80f0bbfdb..cdca847fb 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_policy.py @@ -19,6 +19,14 @@ def _validator() -> ModuleType: ) +def _aggregation() -> ModuleType: + """Load the canonical paired-track aggregation implementation for fixtures.""" + return load_module( + "scripts/research/aggregate_structure_noninferiority.py", + "aggregate_structure_noninferiority_for_policy", + ) + + def _registration() -> dict[str, object]: """Return a valid synthetic registration fixture for policy tests only.""" return { @@ -63,7 +71,7 @@ def _registration() -> dict[str, object]: "uncertainty": { "procedure_id": "paired-track-bootstrap-v1", "confidence_level": 0.95, - "resamples": 10000, + "resamples": 1000, "random_seed": 20260917, }, "corpus": [ @@ -137,8 +145,22 @@ def _uncertainty(registration: dict[str, object]) -> dict[str, object]: return value +def _refresh_canonical_summary( + registration: dict[str, object], + result: dict[str, object], +) -> None: + """Recompute the stored summary after a test intentionally changes track receipts.""" + tracks = result["tracks"] + assert isinstance(tracks, list) + summary = _aggregation().aggregate_complete_track_measurements( + tracks, + uncertainty=_uncertainty(registration), + ) + result.update(summary) + + def _result(registration: dict[str, object], registration_sha256: str) -> dict[str, object]: - """Return a result fixture whose paired intervals satisfy the registration.""" + """Return a result fixture whose stored summary matches canonical aggregation.""" baseline = _measurement( boundary_f_0_5=0.70, boundary_f_3_0=0.78, @@ -159,38 +181,33 @@ def _result(registration: dict[str, object], registration_sha256: str) -> dict[s ) claim_boundary = registration["claim_boundary"] assert isinstance(claim_boundary, str) - return { + tracks: list[dict[str, object]] = [ + { + "track_id": "licensed-track-001", + "baseline": copy.deepcopy(baseline), + "candidate": copy.deepcopy(candidate), + }, + { + "track_id": "licensed-track-002", + "baseline": copy.deepcopy(baseline), + "candidate": copy.deepcopy(candidate), + }, + ] + result: dict[str, object] = { "schema_version": 1, "experiment_id": "structure-chroma-stft-vs-cqt-v1", "registration_sha256": registration_sha256, "uncertainty": copy.deepcopy(_uncertainty(registration)), "corpus_track_ids": ["licensed-track-001", "licensed-track-002"], - "tracks": [ - { - "track_id": "licensed-track-001", - "baseline": copy.deepcopy(baseline), - "candidate": copy.deepcopy(candidate), - }, - { - "track_id": "licensed-track-002", - "baseline": copy.deepcopy(baseline), - "candidate": copy.deepcopy(candidate), - }, - ], - "aggregate": { - "baseline": baseline, - "candidate": candidate, - }, - "paired_delta_ci95": { - "boundary_f_0_5": [-0.015, 0.005], - "boundary_f_3_0": [-0.014, 0.004], - "functional_label_accuracy": [-0.018, 0.003], - "repetition_pairwise_f": [-0.012, 0.006], - }, - "p95_latency_ratio_ci95": [0.66, 0.76], + "tracks": tracks, + "aggregate": None, + "paired_delta_ci95": None, + "p95_latency_ratio_ci95": None, "failed_tracks": [], "claim_boundary": claim_boundary, } + _refresh_canonical_summary(registration, result) + return result def _corpus(registration: dict[str, object]) -> list[dict[str, object]]: @@ -281,22 +298,29 @@ def test_result_passes_only_when_paired_uncertainty_meets_frozen_contract() -> N def test_result_fails_quality_or_latency_when_ci_crosses_registered_boundary() -> None: - """A favorable point estimate cannot hide a noninferiority or latency CI miss.""" + """Canonical track evidence cannot hide a noninferiority or latency CI miss.""" validator = _validator() registration = _registration() digest = validator.registration_digest(registration) result = _result(registration, digest) - deltas = result["paired_delta_ci95"] - assert isinstance(deltas, dict) - deltas["boundary_f_0_5"] = [-0.021, 0.004] - result["p95_latency_ratio_ci95"] = [0.70, 0.81] + tracks = result["tracks"] + assert isinstance(tracks, list) + for track in tracks: + assert isinstance(track, dict) + candidate = track["candidate"] + assert isinstance(candidate, dict) + candidate["boundary_precision_0_5"] = 0.67 + candidate["boundary_recall_0_5"] = 0.67 + candidate["boundary_f_0_5"] = 0.67 + candidate["p95_latency_seconds"] = 5.4 + _refresh_canonical_summary(registration, result) decision = validator.evaluate_result(registration, result) assert decision["passed"] is False assert decision["failed_requirements"] == [ - "boundary_f_0_5 paired CI lower bound -0.021000 is below -0.020000", - "p95 latency ratio CI upper bound 0.810000 exceeds 0.800000", + "boundary_f_0_5 paired CI lower bound -0.030000 is below -0.020000", + "p95 latency ratio CI upper bound 0.900000 exceeds 0.800000", ] From da51a8aa5cddf9300bcbbe611d3929f4132da5a2 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 19:11:01 +0900 Subject: [PATCH 175/216] fix(mir): bind percentile evidence to canonical bootstrap --- .../validate_structure_noninferiority.py | 36 ++++++++++++------- 1 file changed, 24 insertions(+), 12 deletions(-) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index 2683d962b..f1f2fcd5c 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -238,24 +238,36 @@ def _project_percentile_intervals_for_base( return projected -def _require_canonical_aggregate(result_value: Mapping[str, Any]) -> None: - """Bind a successful stored aggregate to the canonical registered-track reducer.""" - failed_tracks = result_value.get("failed_tracks") - if failed_tracks: +def _require_canonical_summary( + registration_value: object, + result_value: Mapping[str, Any], +) -> None: + """Bind successful aggregate and CI receipts to deterministic recomputation.""" + if result_value.get("failed_tracks"): return + registration = _BASE._mapping(registration_value, "registration") aggregation = _aggregation() - _, baseline_sides, candidate_sides = aggregation._normalize_tracks( - result_value.get("tracks") + expected = aggregation.aggregate_complete_track_measurements( + result_value.get("tracks"), + uncertainty=registration.get("uncertainty"), ) - expected_aggregate = { - "baseline": aggregation._aggregate_side(baseline_sides), - "candidate": aggregation._aggregate_side(candidate_sides), - } - if result_value.get("aggregate") != expected_aggregate: + if result_value.get("aggregate") != expected["aggregate"]: raise ValueError( "result.aggregate does not match canonical macro-track-v1 recomputation" ) + if result_value.get("paired_delta_ci95") != expected["paired_delta_ci95"]: + raise ValueError( + "result.paired_delta_ci95 does not match canonical " + "paired-track-bootstrap-v1 recomputation" + ) + if result_value.get("p95_latency_ratio_ci95") != expected[ + "p95_latency_ratio_ci95" + ]: + raise ValueError( + "result.p95_latency_ratio_ci95 does not match canonical " + "paired-track-bootstrap-v1 recomputation" + ) def evaluate_result( @@ -286,7 +298,7 @@ def evaluate_result( ) decision = _BASE.evaluate_result(registration_value, percentile_projection) - _require_canonical_aggregate(projected_result) + _require_canonical_summary(registration_value, projected_result) decision["registration_sha256"] = expected_digest return decision From 5f698c4c98da83d6ca7860c6b32aee2980c5abb0 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 19:11:33 +0900 Subject: [PATCH 176/216] test(mir): separate percentile compatibility from receipt admission --- ...st_structure_noninferiority_aggregation.py | 22 +++++++++++++------ 1 file changed, 15 insertions(+), 7 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py b/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py index 1513afba9..f588136f6 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py @@ -2,6 +2,7 @@ from __future__ import annotations +import copy from types import ModuleType import pytest @@ -192,19 +193,26 @@ def test_paired_bootstrap_is_deterministic_and_resamples_track_pairs() -> None: assert first["p95_latency_ratio_ci95"][0] > 0.0 -def test_percentile_interval_is_not_required_to_contain_original_point_estimate() -> None: - """Percentile bootstrap intervals may be asymmetric around the observed statistic.""" +def test_percentile_projection_preserves_asymmetric_interval_bounds() -> None: + """Legacy point containment is adapted without changing percentile endpoints.""" validator = _validator() registration = _registration() - digest = validator.registration_digest(registration) - result = _result(registration, digest) - intervals = result["paired_delta_ci95"] + result = _result(registration, validator.registration_digest(registration)) + projected = copy.deepcopy(result) + projected["registration_sha256"] = validator._BASE.registration_digest(registration) + intervals = projected["paired_delta_ci95"] assert isinstance(intervals, dict) intervals["boundary_f_0_5"] = [0.001, 0.010] - result["p95_latency_ratio_ci95"] = [0.60, 0.65] + projected["p95_latency_ratio_ci95"] = [0.60, 0.65] - decision = validator.evaluate_result(registration, result) + with pytest.raises(ValueError, match="must contain the aggregate point delta"): + validator._BASE.evaluate_result(registration, projected) + adapted = validator._project_percentile_intervals_for_base(projected) + decision = validator._BASE.evaluate_result(registration, adapted) + + assert adapted["paired_delta_ci95"] == projected["paired_delta_ci95"] + assert adapted["p95_latency_ratio_ci95"] == projected["p95_latency_ratio_ci95"] assert decision["passed"] is True From 59d9c309015af6cdf9badc7cd50cf2d8bf5813e0 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 19:11:48 +0900 Subject: [PATCH 177/216] test(mir): derive threshold failures from canonical track evidence --- ...tructure_percentile_bootstrap_admission.py | 36 ++++++++++++++----- 1 file changed, 27 insertions(+), 9 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_percentile_bootstrap_admission.py b/services/analysis-engine/tests/test_structure_percentile_bootstrap_admission.py index 35024e35a..444a7b7cf 100644 --- a/services/analysis-engine/tests/test_structure_percentile_bootstrap_admission.py +++ b/services/analysis-engine/tests/test_structure_percentile_bootstrap_admission.py @@ -5,7 +5,11 @@ from types import ModuleType from conftest import load_module -from test_structure_noninferiority_policy import _registration, _result +from test_structure_noninferiority_policy import ( + _refresh_canonical_summary, + _registration, + _result, +) def _validator() -> ModuleType: @@ -16,14 +20,21 @@ def _validator() -> ModuleType: ) -def test_asymmetric_quality_interval_still_fails_registered_noninferiority_margin() -> None: - """Removing point containment must not weaken the CI lower-bound decision.""" +def test_canonical_quality_interval_still_fails_registered_noninferiority_margin() -> None: + """Receipt binding must preserve the preregistered CI lower-bound decision.""" validator = _validator() registration = _registration() result = _result(registration, validator.registration_digest(registration)) - intervals = result["paired_delta_ci95"] - assert isinstance(intervals, dict) - intervals["boundary_f_0_5"] = [-0.030, -0.025] + tracks = result["tracks"] + assert isinstance(tracks, list) + for track in tracks: + assert isinstance(track, dict) + candidate = track["candidate"] + assert isinstance(candidate, dict) + candidate["boundary_precision_0_5"] = 0.67 + candidate["boundary_recall_0_5"] = 0.67 + candidate["boundary_f_0_5"] = 0.67 + _refresh_canonical_summary(registration, result) decision = validator.evaluate_result(registration, result) @@ -34,12 +45,19 @@ def test_asymmetric_quality_interval_still_fails_registered_noninferiority_margi ) -def test_asymmetric_latency_interval_still_fails_registered_ratio_threshold() -> None: - """The compatibility projection must preserve the original CI upper bound.""" +def test_canonical_latency_interval_still_fails_registered_ratio_threshold() -> None: + """Receipt binding must preserve the preregistered latency upper-bound decision.""" validator = _validator() registration = _registration() result = _result(registration, validator.registration_digest(registration)) - result["p95_latency_ratio_ci95"] = [0.81, 0.85] + tracks = result["tracks"] + assert isinstance(tracks, list) + for track in tracks: + assert isinstance(track, dict) + candidate = track["candidate"] + assert isinstance(candidate, dict) + candidate["p95_latency_seconds"] = 5.1 + _refresh_canonical_summary(registration, result) decision = validator.evaluate_result(registration, result) From 40922f895a9c15aab48ed6e457b0b8bed04a1abc Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 19:13:19 +0900 Subject: [PATCH 178/216] docs(mir): trace canonical result receipt binding --- .../mir/structure-result-receipt-binding.md | 61 +++++++++++++++++++ 1 file changed, 61 insertions(+) create mode 100644 docs/traceability/mir/structure-result-receipt-binding.md diff --git a/docs/traceability/mir/structure-result-receipt-binding.md b/docs/traceability/mir/structure-result-receipt-binding.md new file mode 100644 index 000000000..2b9b78791 --- /dev/null +++ b/docs/traceability/mir/structure-result-receipt-binding.md @@ -0,0 +1,61 @@ +# Structure result receipt binding + +Status: Proposed +Owner: Signal-MIR Analysis +Tracking: #1225, #1228 +Parent: `docs/traceability/mir/structure-feature-noninferiority.md` + +## Problem + +The preregistered result schema required complete ordered per-track receipts and separately required aggregate measurements plus paired confidence intervals. Before this repair, result admission validated both parts but did not prove that the stored aggregate and confidence intervals were produced from those stored track receipts by the repository-owned `macro-track-v1` / `paired-track-bootstrap-v1` implementation. + +That gap was scientifically material. A syntactically valid receipt could retain the registered corpus order and valid per-track measurements while carrying a favorable hand-authored aggregate or percentile interval. The base decision policy would then evaluate those stored summaries even though they were not causally bound to the track-level evidence. + +## Decision + +Successful result admission now deterministically recomputes all decision summaries from the complete stored `result.tracks` array and the preregistered uncertainty plan by calling `aggregate_structure_noninferiority.py`. + +Admission requires exact equality for: + +- `result.aggregate` versus canonical `macro-track-v1` recomputation; +- `result.paired_delta_ci95` versus canonical `paired-track-bootstrap-v1` recomputation; +- `result.p95_latency_ratio_ci95` versus the same canonical paired bootstrap. + +The comparison occurs after the historical base validator has checked the result envelope, track order, measurement fields, finite values, P/R/F identities, failed-track policy, uncertainty identity, claim boundary, and decision-rule syntax. A failed run is not re-aggregated: schema v1 already requires null aggregate/CI summaries when `failed_tracks` is non-empty. + +The percentile compatibility projection remains private to the historical base validator boundary. It exists only because the base validator predates percentile-bootstrap semantics and incorrectly requires every interval to contain the observed statistic. Canonical receipt binding compares the original stored aggregate and interval values, not the private projection. Therefore the projection cannot manufacture evidence that passes the new recomputation gate. + +## RED → GREEN lineage + +RED `aacb026ed2e05dadce3fb0b7560836308bc73b92` added two executable regressions: + +- change one valid per-track functional-accuracy value while leaving the favorable stored aggregate unchanged; +- hand-edit one otherwise valid paired interval while leaving the registered track receipts unchanged. + +Both receipts were admitted by the predecessor validator because aggregate/CI provenance was not executable. + +GREEN `5f47924d632946540d0c31e0c6456feac4667f8b` first bound the stored aggregate to canonical track reduction. Fixture alignment `4cdcf19a016c9b4d7a1c28ccc39784513700b297` changed policy fixtures to generate canonical summaries instead of hand-authored intervals. GREEN `da51a8aa5cddf9300bcbbe611d3929f4132da5a2` then bound both paired quality intervals and the p95-latency-ratio interval to deterministic bootstrap recomputation. + +`5f698c4c98da83d6ca7860c6b32aee2980c5abb0` separated the legacy percentile point-containment compatibility regression from scientific result admission: the compatibility helper can still prove that interval endpoints are not rewritten, while the public validator accepts only canonical intervals. `59d9c309015af6cdf9badc7cd50cf2d8bf5813e0` rewrote threshold regressions so failing noninferiority and latency decisions are derived from changed track evidence followed by canonical recomputation rather than from hand-edited confidence intervals. + +## Constraints and rejected alternatives + +Trusting a comment, workflow name, or producer claim that a JSON file came from the canonical aggregator was rejected. Those signals do not establish a causal relationship between stored track evidence and stored decision summaries. + +Checking only the aggregate was rejected as incomplete. The final production decision is made from percentile interval bounds, so favorable hand-authored confidence intervals would remain an acceptance bypass even if the aggregate itself were canonical. + +Adding a second aggregation implementation inside the validator was rejected. `aggregate_structure_noninferiority.py` remains the single owner of track weighting, F recomputation, bootstrap sampling, RNG, and quantile semantics; admission invokes that owner instead of copying the formulas. + +Approximate receipt comparison was rejected. The experiment registration pins source/runtime identities and deterministic bootstrap parameters. A durable result that was rounded, recomputed under another NumPy/runtime, or otherwise transformed is not the exact registered evidence artifact and must be regenerated under the registered environment. + +## Security and integrity notes + +- The recomputation path consumes only already-validated in-memory result data and preregistered uncertainty fields. It opens no corpus path, performs no network request, and executes no user-supplied plugin. +- Track omission, reordering, duplication, malformed metrics, non-finite values, failed-track fabrication, aggregate drift, and CI drift all fail closed before a passing decision is returned. +- Exact deterministic recomputation protects evidence integrity; it does not establish corpus representativeness, licensing, annotation validity, statistical power, or suitable production margins. Those remain independent scientific/product gates. + +## Remaining scientific boundary + +This repair closes the result-summary provenance gap. It does not authorize a real-audio run. Before candidate results are inspected, reviewers still have to freeze the concrete rights-cleared corpus, numeric noninferiority margins, latency ratio threshold, exact bootstrap count/seed, host/runtime profile, and claim boundary. + +A separate unresolved measurement gap remains for timing and memory: the admitted-track MIR consumer already produces structure-quality metrics, but the canonical path from that same admitted execution to per-track p50/p95 latency and peak-RSS receipts still requires explicit trial/warm-up/timer/memory semantics before those fields can be treated as production scientific evidence. From eb7736fa69c167524231e1df719ef8d432f9be53 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 19:17:31 +0900 Subject: [PATCH 179/216] docs(mir): bind result admission and remaining timing gap --- .../mir/structure-feature-noninferiority.md | 24 ++++++++++--------- 1 file changed, 13 insertions(+), 11 deletions(-) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index d87011ea3..9368eafa0 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -53,8 +53,8 @@ BandScope uses the MIREX Music Structure Analysis task and its standardized eval | Boundary median deviation, both directions | `mir_eval.segment.deviation(trim=True)` | Report per track and aggregate | | Functional-label accuracy | `ismir-mirex/mirex-evaluation@b9fa0b0...:music_structure_analysis.eval_script.calculate_accuracy`; 0.2 s grid; frame-grid v1; annotation v1; label-mapping v1 | Noninferiority | | Repetition/group consistency precision/recall/F | `mir_eval.segment.pairwise(frame_size=0.1, beta=1.0)` | F noninferiority | -| Latency | paired p50/p95 on the registered host | p95 ratio superiority | -| Peak memory | peak RSS on the registered host | Report per track and aggregate | +| Latency | Result schema has per-track p50/p95 fields; canonical warm-up/trial/timer contract is still pending | p95 ratio superiority only after that contract is frozen | +| Peak memory | Result schema has per-track peak RSS; canonical process/memory-scope measurement contract is still pending | Report only after that contract is frozen | The selected mir_eval calls execute only under the content-addressed research overlay whose exact lock SHA-256 participates in the registration digest. The runtime verifier separately requires the installed distribution to be `mir_eval==0.8.2` before recognized segmentation metrics run. Full artifact provenance and adapter semantics are recorded in `docs/traceability/mir/structure-segmentation-metric-adapter.md`. @@ -81,7 +81,9 @@ For uncertainty, each bootstrap replicate samples `registered_track_count` paire This is a percentile-bootstrap interval, not a symmetric interval around the observed statistic. Its endpoints are quantiles of the bootstrap distribution and can be asymmetric; the observed aggregate statistic is therefore **not required** to lie between those endpoints. Hall (1988) treats the percentile family as a bootstrap-quantile construction and documents that different bootstrap interval methods have materially different coverage and positioning properties. Adding a separate point-containment requirement would define a different, undocumented interval rule. -The historical base validator predates the frozen percentile procedure and still contains that obsolete containment invariant. The metric-aware façade handles only those two legacy containment errors by creating a private validation projection whose aggregate point values are moved onto feasible interval bounds. The stored receipt, track measurements, aggregate values, and interval bounds are not modified. The base validator then continues to enforce every structural/measurement invariant and applies the original interval bounds to the noninferiority and latency decisions. Any non-containment validation error is propagated unchanged. `test_percentile_interval_is_not_required_to_contain_original_point_estimate` is the executable regression for this boundary. +The historical base validator predates the frozen percentile procedure and still contains that obsolete containment invariant. The metric-aware façade handles only those two legacy containment errors by creating a private validation projection whose aggregate point values are moved onto feasible interval bounds. The stored receipt, track measurements, aggregate values, and interval bounds are not modified. The base validator then continues to enforce every structural/measurement invariant and applies the original interval bounds to the noninferiority and latency decisions. Any non-containment validation error is propagated unchanged. `test_percentile_projection_preserves_asymmetric_interval_bounds` is the executable regression for this compatibility boundary. + +Public result admission then performs a second, independent integrity gate on the **original stored receipt**: it recomputes `macro-track-v1` aggregate values and both `paired-track-bootstrap-v1` interval objects from the complete stored `result.tracks` array plus the preregistered uncertainty plan, and requires exact equality. The private legacy projection is never used for this comparison. Hand-authored favorable aggregates or CIs therefore cannot become admitted scientific evidence. The RED→GREEN lineage and rejected alternatives are recorded in `docs/traceability/mir/structure-result-receipt-binding.md`. The procedure being executable and preregisterable does not by itself establish that a particular corpus size, resample count, seed, margin, or claim boundary is scientifically adequate. Those remain experiment-review decisions. A different bootstrap method such as BCa, studentized, cluster/bootstrap-at-recording-level, or a different aggregation weighting requires a new versioned contract before candidate results are inspected. @@ -107,9 +109,9 @@ A result receipt contains the exact metric-aware registration digest, exact unce Each successful per-track and aggregate baseline/candidate measurement is a closed-world receipt containing exactly the registered quality fields plus reporting precision/recall, boundary-deviation, latency, and peak-RSS fields. A declared failed track contains only `track_id`; attaching partial or invented measurements to a failed receipt is rejected. -The receipt is rejected when a track is omitted or reordered; missing measurements are not paired with an explicit failed-track declaration; a failed receipt carries measurements; a failed run carries non-null aggregate/CI summaries; the uncertainty plan or claim boundary differs; required metrics are absent; unregistered fields appear; a P/R/F triplet is inconsistent; values are non-finite; interval bounds are malformed; or the metric-aware registration digest differs. Percentile intervals are not rejected merely because an observed aggregate point lies outside them. +The receipt is rejected when a track is omitted or reordered; missing measurements are not paired with an explicit failed-track declaration; a failed receipt carries measurements; a failed run carries non-null aggregate/CI summaries; the uncertainty plan or claim boundary differs; required metrics are absent; unregistered fields appear; a P/R/F triplet is inconsistent; values are non-finite; interval bounds are malformed; the aggregate or CI values differ from canonical recomputation; or the metric-aware registration digest differs. Percentile intervals are not rejected merely because an observed aggregate point lies outside them. -The admission validator does not trust caller-authored aggregate/CI values as scientific authority. The canonical result producer is the repository-owned paired aggregator, and focused contract tests bind its immutable aggregation/uncertainty constants into the registration digest. Review must still confirm that the result artifact was generated by that owner from the complete admitted track set before production acceptance. +The admission validator does not trust caller-authored aggregate/CI values as scientific authority. It invokes the repository-owned paired aggregator over the complete stored track set and preregistered uncertainty plan, then exact-matches all stored aggregate and interval values against that deterministic output. Review still has to establish the upstream scientific validity of the per-track measurements, corpus, timing/memory measurement contract, thresholds, and claim boundary; those questions are not proved by recomputing a receipt. ## Machine-readable evidence admission @@ -119,11 +121,11 @@ Registration and result JSON are evidence, not trusted configuration. CLI admiss ## Reproducibility sequence -1. Review rights basis, corpus composition, distinct audio identities, feature hypothesis, metric contract, functional-ACC grid/mapping, host profile, numeric margins, `macro-track-v1` aggregation, `paired-track-bootstrap-v1` resample count/seed, and claim boundary before candidate execution. If clustering, repeated recordings, alternative label mapping, or any exclusion tolerance is scientifically required, version that policy before measurement. +1. Review rights basis, corpus composition, distinct audio identities, feature hypothesis, metric contract, functional-ACC grid/mapping, host profile, numeric margins, `macro-track-v1` aggregation, `paired-track-bootstrap-v1` resample count/seed, and claim boundary before candidate execution. If clustering, repeated recordings, alternative label mapping, or any exclusion tolerance is scientifically required, version that policy before measurement. Freeze the canonical latency/peak-RSS measurement contract before latency evidence is admitted. 2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain the metric-aware SHA-256. The digest includes validated registration data plus repository-owned metric and aggregation/uncertainty contracts. 3. Admit the rights-cleared corpus once, then run CQT and STFT lanes on the same decoded PCM/reference segmentation. Functional ACC must preserve the pinned 200 ms MIREX grid semantics; boundary/deviation/repetition metrics must use the pinned research runtime and explicit adapter arguments. 4. Feed the complete ordered track measurements to `aggregate_structure_noninferiority.py`. Do not hand-author aggregate values or confidence intervals. If any registered track fails, preserve the failure receipt and emit null aggregate/CI summaries instead of a complete-case result. -5. Put the metric-aware registration digest, identical uncertainty fields, corpus order, and claim boundary in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. +5. Put the metric-aware registration digest, identical uncertainty fields, corpus order, and claim boundary in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. Admission recomputes the aggregate and both interval objects from the stored tracks and rejects any mismatch before returning a scientific decision. 6. Preserve registration, metric-aware digest contract, result receipt, corpus/annotation hashes, exact source/lock identities, aggregation implementation identity, uncertainty plan, and claim boundary together. A production feature switch requires this evidence plus normal independent review and protected-head checks. No step authorizes committing licensed audio to Git. Evaluation rights and redistribution rights remain separate. @@ -134,9 +136,9 @@ No step authorizes committing licensed audio to Git. Evaluation rights and redis - Evidence JSON is limited to 2 MiB, must be a regular file read through one open descriptor, must decode as UTF-8, and rejects duplicate keys plus non-standard `NaN`/`Infinity` constants. - `source_uri` is evidence metadata, not a fetch instruction; local filesystem forms are rejected. - Audio/annotation SHA-256 values bind measurements to bytes without embedding media; duplicate audio content identity under different track IDs is rejected. -- Invalid/non-finite measurements, corpus drift, metric/runtime drift, aggregation/uncertainty drift, claim-boundary drift, registration drift, undeclared missing measurements, failed-receipt fabrication, non-null summaries after a failed track, unregistered fields, and inconsistent P/R/F triplets fail closed. -- The percentile compatibility projection exists only inside result validation and never rewrites the stored scientific receipt or CI bounds. -- SHA-256 does not prove licensing, annotation validity, corpus adequacy, independence beyond exact-byte uniqueness, or statistical appropriateness. Rights and scientific review remain separate gates. +- Invalid/non-finite measurements, corpus drift, metric/runtime drift, aggregation/uncertainty drift, aggregate/CI recomputation drift, claim-boundary drift, registration drift, undeclared missing measurements, failed-receipt fabrication, non-null summaries after a failed track, unregistered fields, and inconsistent P/R/F triplets fail closed. +- The percentile compatibility projection exists only inside result validation and never rewrites the stored scientific receipt or CI bounds; canonical-summary binding compares the original stored receipt. +- SHA-256 does not prove licensing, annotation validity, corpus adequacy, independence beyond exact-byte uniqueness, measurement-method adequacy, or statistical appropriateness. Rights and scientific review remain separate gates. ## References @@ -158,4 +160,4 @@ MIREX Evaluation contributors. (2026). *music_structure_analysis/eval_script.py* mir_eval contributors. (n.d.). *mir_eval.segment: Structural segmentation evaluation*. https://github.com/mir-evaluation/mir_eval/blob/main/mir_eval/segment.py -Wang, J.-C., Hung, Y.-N., & Smith, J. B. L. (2022). To catch a chorus, verse, intro, or anything else: Analyzing a song with structural functions. In *ICASSP 2022—2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)* (pp. 416–420). IEEE. https://arxiv.org/abs/2205.14700 +Wang, J.-C., Hung, Y.-N., & Smith, J. B. L. (2022). To catch a chorus, verse, intro, or anything else: Analyzing a song with structural functions. In *ICASSP 2022—2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)* (pp. 416–420). IEEE. https://arxiv.org/abs/2205.14700 \ No newline at end of file From 6fcc9e96eba9e6a9c3b3deed19b40a54e6c33ade Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:03:44 +0900 Subject: [PATCH 180/216] test(mir): preregister lane latency and RSS contract --- ...est_structure_lane_resource_measurement.py | 135 ++++++++++++++++++ 1 file changed, 135 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_lane_resource_measurement.py diff --git a/services/analysis-engine/tests/test_structure_lane_resource_measurement.py b/services/analysis-engine/tests/test_structure_lane_resource_measurement.py new file mode 100644 index 000000000..b81f4d539 --- /dev/null +++ b/services/analysis-engine/tests/test_structure_lane_resource_measurement.py @@ -0,0 +1,135 @@ +"""Contracts for preregistered structure-lane latency and peak-RSS evidence.""" + +from __future__ import annotations + +import struct +from fractions import Fraction +from types import ModuleType + +import pytest +from conftest import load_module + + +def _measurement_module() -> ModuleType: + """Load the repository-owned isolated performance measurement boundary.""" + return load_module( + "scripts/research/measure_structure_lane_resources.py", + "measure_structure_lane_resources", + ) + + +def _validator_module() -> ModuleType: + """Load the metric-aware registration validator.""" + return load_module( + "scripts/research/validate_structure_noninferiority.py", + "validate_structure_noninferiority_for_performance_contract", + ) + + +def test_performance_contract_is_part_of_metric_aware_registration_identity() -> None: + """Latency and RSS semantics must be frozen before candidate evidence exists.""" + measurement = _measurement_module() + validator = _validator_module() + + assert validator.STRUCTURE_METRIC_CONTRACT["performance_measurement"] == ( + measurement.PERFORMANCE_MEASUREMENT_CONTRACT + ) + assert measurement.PERFORMANCE_MEASUREMENT_CONTRACT == { + "contract_id": "isolated-single-shot-v1", + "supported_platforms": ["darwin", "win32"], + "warmup_trials": 0, + "measured_trials": 20, + "trial_process": "fresh_subprocess_per_lane_trial", + "lane_order": "alternate_baseline_candidate_by_trial_index", + "timer": "time.perf_counter_ns", + "timer_scope": "repository_structure_segmenter_only", + "worker_startup_in_latency": False, + "input_transfer_in_latency": False, + "latency_quantiles": [0.5, 0.95], + "quantile_method": "linear", + "memory_metric": "process_peak_resident_set_size", + "memory_scope": "entire_worker_process_lifetime_including_pcm_input", + "peak_rss_aggregation": "maximum_across_trials", + } + + +def test_paired_measurement_alternates_order_and_uses_linear_quantiles( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """Each trial is fresh, paired in order, and summarized exactly as preregistered.""" + module = _measurement_module() + pcm = memoryview(struct.pack("<8f", *([0.0] * 8))) + observed_order: list[str] = [] + lane_counts = {"cqt": 0, "stft": 0} + + def fake_isolated_trial( + feature: str, + decoded_pcm: memoryview, + sample_rate_hz: int, + duration_seconds: Fraction, + ) -> object: + assert decoded_pcm is pcm + assert sample_rate_hz == 10 + assert duration_seconds == Fraction(4, 5) + observed_order.append(feature) + lane_counts[feature] += 1 + count = lane_counts[feature] + if feature == "cqt": + return module.IsolatedLaneTrial( + feature=feature, + latency_ns=count * 100_000_000, + peak_rss_mib=100.0 + count, + ) + return module.IsolatedLaneTrial( + feature=feature, + latency_ns=count * 50_000_000, + peak_rss_mib=200.0 + count, + ) + + monkeypatch.setattr(module, "_run_isolated_trial", fake_isolated_trial) + + result = module.measure_paired_repository_lane_resources( + pcm, + 10, + Fraction(4, 5), + ) + + assert observed_order[:8] == [ + "cqt", + "stft", + "stft", + "cqt", + "cqt", + "stft", + "stft", + "cqt", + ] + assert len(observed_order) == 40 + assert result.baseline.measured_trials == 20 + assert result.candidate.measured_trials == 20 + assert result.baseline.p50_latency_seconds == pytest.approx(1.05) + assert result.baseline.p95_latency_seconds == pytest.approx(1.905) + assert result.candidate.p50_latency_seconds == pytest.approx(0.525) + assert result.candidate.p95_latency_seconds == pytest.approx(0.9525) + assert result.baseline.peak_rss_mib == pytest.approx(120.0) + assert result.candidate.peak_rss_mib == pytest.approx(220.0) + + +def test_paired_measurement_rejects_mutable_or_duration_mismatched_pcm() -> None: + """Performance evidence must use the same immutable admitted PCM identity.""" + module = _measurement_module() + raw = struct.pack("<8f", *([0.0] * 8)) + + with pytest.raises(ValueError, match="read-only memoryview"): + module.measure_paired_repository_lane_resources( + memoryview(bytearray(raw)), + 10, + Fraction(4, 5), + ) + + with pytest.raises(ValueError, match="duration_seconds"): + module.measure_paired_repository_lane_resources( + memoryview(raw), + 10, + Fraction(1, 1), + ) From 0b80bc22adadaf4998b562dbf5fcd94a0492ba5a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:04:41 +0900 Subject: [PATCH 181/216] feat(mir): implement isolated latency and peak RSS measurement --- .../measure_structure_lane_resources.py | 430 ++++++++++++++++++ 1 file changed, 430 insertions(+) create mode 100644 scripts/research/measure_structure_lane_resources.py diff --git a/scripts/research/measure_structure_lane_resources.py b/scripts/research/measure_structure_lane_resources.py new file mode 100644 index 000000000..14926c3f6 --- /dev/null +++ b/scripts/research/measure_structure_lane_resources.py @@ -0,0 +1,430 @@ +#!/usr/bin/env python3 +"""Measure preregistered CQT/STFT latency and peak RSS in fresh processes. + +The structure experiment needs performance evidence that cannot inherit process +state from the other feature lane. Each measured observation therefore runs one +repository-owned structure segmenter call in a fresh Python subprocess. Worker +startup and PCM transport happen before the latency timer, while peak RSS covers +the worker's whole lifetime so interpreter, input, and analysis allocations are +not selectively excluded. + +No warm-up iterations are used: repeating a song in one worker would create a +hot-cache path that is not the first-analysis buyer path being compared. Twenty +single-shot observations per lane are paired by track and executed in alternating +CQT/STFT order to limit deterministic order bias. The fixed count is a +preregistered engineering sampling plan, not a claim that twenty observations +make tail latency asymptotically precise. + +Security Notes: +- PCM arrives only from the already-admitted immutable in-process snapshot. +- The child command is an argument array with ``shell=False`` and no generic + command, path, URL, plugin, or network input. +- Feature identity is closed-world (``cqt`` or ``stft``). +- Scientific measurement fails closed on unsupported operating systems, worker + stderr, malformed worker output, or non-zero worker exit. +""" + +from __future__ import annotations + +import argparse +import importlib.util +import json +import math +import subprocess +import sys +import time +from dataclasses import dataclass +from fractions import Fraction +from pathlib import Path +from types import ModuleType +from typing import Any, Literal, Sequence + +import numpy as np + +ChromaFeature = Literal["cqt", "stft"] +_REPOSITORY_ROOT = Path(__file__).resolve().parents[2] +_MEASURED_TRIALS = 20 + +PERFORMANCE_MEASUREMENT_CONTRACT: dict[str, object] = { + "contract_id": "isolated-single-shot-v1", + "supported_platforms": ["darwin", "win32"], + "warmup_trials": 0, + "measured_trials": _MEASURED_TRIALS, + "trial_process": "fresh_subprocess_per_lane_trial", + "lane_order": "alternate_baseline_candidate_by_trial_index", + "timer": "time.perf_counter_ns", + "timer_scope": "repository_structure_segmenter_only", + "worker_startup_in_latency": False, + "input_transfer_in_latency": False, + "latency_quantiles": [0.5, 0.95], + "quantile_method": "linear", + "memory_metric": "process_peak_resident_set_size", + "memory_scope": "entire_worker_process_lifetime_including_pcm_input", + "peak_rss_aggregation": "maximum_across_trials", +} + + +@dataclass(frozen=True, slots=True) +class IsolatedLaneTrial: + """One single-shot lane observation produced by a fresh worker process.""" + + feature: str + latency_ns: int + peak_rss_mib: float + + +@dataclass(frozen=True, slots=True) +class LaneResourceSummary: + """Preregistered per-track latency quantiles and worst observed peak RSS.""" + + p50_latency_seconds: float + p95_latency_seconds: float + peak_rss_mib: float + measured_trials: int + + +@dataclass(frozen=True, slots=True) +class PairedLaneResourceEvidence: + """Baseline/candidate performance evidence from the same admitted PCM.""" + + contract_id: str + baseline: LaneResourceSummary + candidate: LaneResourceSummary + + +def _load_sibling(filename: str, module_name: str) -> ModuleType: + """Load one repository-owned sibling script under a stable private name.""" + existing = sys.modules.get(module_name) + if existing is not None: + return existing + path = Path(__file__).with_name(filename) + spec = importlib.util.spec_from_file_location(module_name, path) + if spec is None or spec.loader is None: + raise RuntimeError(f"could not load research module: {filename}") + module = importlib.util.module_from_spec(spec) + sys.modules[module_name] = module + try: + spec.loader.exec_module(module) + except Exception: + sys.modules.pop(module_name, None) + raise + return module + + +_LANES = _load_sibling( + "measure_structure_feature_lanes.py", + "_bandscope_structure_resource_measurement_lanes", +) + + +def _registered_feature(value: object) -> ChromaFeature: + """Return one closed-world feature identity.""" + if value == "cqt": + return "cqt" + if value == "stft": + return "stft" + raise ValueError("feature must be one of: cqt, stft") + + +def _pcm_view(decoded_pcm: object) -> memoryview: + """Require the immutable canonical mono-float32 admission handoff.""" + if not isinstance(decoded_pcm, memoryview) or not decoded_pcm.readonly: + raise ValueError("decoded PCM must be a read-only memoryview") + if not decoded_pcm.c_contiguous: + raise ValueError("decoded PCM must be contiguous") + if decoded_pcm.nbytes == 0 or decoded_pcm.nbytes % 4 != 0: + raise ValueError("decoded PCM must contain non-empty mono float32 bytes") + return decoded_pcm + + +def _positive_sample_rate(value: object) -> int: + """Return one positive integer sample rate without coercion.""" + if isinstance(value, bool) or not isinstance(value, int) or value <= 0: + raise ValueError("sample_rate_hz must be a positive integer") + return value + + +def _duration(value: object, *, decoded_frames: int, sample_rate_hz: int) -> Fraction: + """Require duration identity to equal admitted frames divided by sample rate.""" + if not isinstance(value, Fraction): + raise ValueError("duration_seconds must be an exact Fraction") + expected = Fraction(decoded_frames, sample_rate_hz) + if value != expected: + raise ValueError("duration_seconds must match the admitted PCM identity") + return value + + +def _macos_peak_rss_mib() -> float: + """Return Darwin process peak resident set size using getrusage kilobytes.""" + import resource + + raw_kib = int(resource.getrusage(resource.RUSAGE_SELF).ru_maxrss) + if raw_kib <= 0: + raise RuntimeError("Darwin getrusage returned a non-positive ru_maxrss") + return raw_kib / 1024.0 + + +def _windows_peak_rss_mib() -> float: + """Return Windows PeakWorkingSetSize for the current worker process.""" + import ctypes + from ctypes import wintypes + + class ProcessMemoryCounters(ctypes.Structure): + """Win32 PROCESS_MEMORY_COUNTERS layout required by GetProcessMemoryInfo.""" + + _fields_ = [ + ("cb", wintypes.DWORD), + ("PageFaultCount", wintypes.DWORD), + ("PeakWorkingSetSize", ctypes.c_size_t), + ("WorkingSetSize", ctypes.c_size_t), + ("QuotaPeakPagedPoolUsage", ctypes.c_size_t), + ("QuotaPagedPoolUsage", ctypes.c_size_t), + ("QuotaPeakNonPagedPoolUsage", ctypes.c_size_t), + ("QuotaNonPagedPoolUsage", ctypes.c_size_t), + ("PagefileUsage", ctypes.c_size_t), + ("PeakPagefileUsage", ctypes.c_size_t), + ] + + kernel32 = ctypes.WinDLL("kernel32", use_last_error=True) + psapi = ctypes.WinDLL("psapi", use_last_error=True) + kernel32.GetCurrentProcess.restype = wintypes.HANDLE + psapi.GetProcessMemoryInfo.argtypes = [ + wintypes.HANDLE, + ctypes.POINTER(ProcessMemoryCounters), + wintypes.DWORD, + ] + psapi.GetProcessMemoryInfo.restype = wintypes.BOOL + + counters = ProcessMemoryCounters() + counters.cb = ctypes.sizeof(counters) + handle = kernel32.GetCurrentProcess() + if not psapi.GetProcessMemoryInfo(handle, ctypes.byref(counters), counters.cb): + error_code = ctypes.get_last_error() + raise OSError(error_code, "GetProcessMemoryInfo failed") + peak_bytes = int(counters.PeakWorkingSetSize) + if peak_bytes <= 0: + raise RuntimeError("Windows returned a non-positive PeakWorkingSetSize") + return peak_bytes / (1024.0 * 1024.0) + + +def _peak_rss_mib() -> float: + """Read process-lifetime peak RSS with the preregistered platform API.""" + if sys.platform == "darwin": + return _macos_peak_rss_mib() + if sys.platform == "win32": + return _windows_peak_rss_mib() + raise RuntimeError( + "isolated-single-shot-v1 supports performance evidence only on macOS or Windows" + ) + + +def _measure_current_process( + feature: object, + decoded_pcm: object, + sample_rate_hz: object, + duration_seconds: object, +) -> IsolatedLaneTrial: + """Measure exactly one segmenter call inside an already-started worker.""" + normalized_feature = _registered_feature(feature) + pcm = _pcm_view(decoded_pcm) + sample_rate = _positive_sample_rate(sample_rate_hz) + exact_duration = _duration( + duration_seconds, + decoded_frames=pcm.nbytes // 4, + sample_rate_hz=sample_rate, + ) + segmenter = _LANES.repository_structure_segmenter(normalized_feature) + start_ns = time.perf_counter_ns() + segments = segmenter(pcm, sample_rate, exact_duration) + end_ns = time.perf_counter_ns() + if not segments: + raise RuntimeError("structure resource worker returned no segments") + latency_ns = end_ns - start_ns + if latency_ns <= 0: + raise RuntimeError("structure resource worker measured non-positive latency") + peak_rss_mib = _peak_rss_mib() + if not math.isfinite(peak_rss_mib) or peak_rss_mib <= 0.0: + raise RuntimeError("structure resource worker measured invalid peak RSS") + return IsolatedLaneTrial( + feature=normalized_feature, + latency_ns=latency_ns, + peak_rss_mib=peak_rss_mib, + ) + + +def _worker_command( + feature: ChromaFeature, + sample_rate_hz: int, + duration_seconds: Fraction, +) -> list[str]: + """Return the exact no-shell command for one isolated single-shot worker.""" + return [ + sys.executable, + str(Path(__file__).resolve()), + "--worker", + feature, + str(sample_rate_hz), + str(duration_seconds.numerator), + str(duration_seconds.denominator), + ] + + +def _run_isolated_trial( + feature: str, + decoded_pcm: memoryview, + sample_rate_hz: int, + duration_seconds: Fraction, +) -> IsolatedLaneTrial: + """Run one lane in a fresh subprocess and validate its complete JSON receipt.""" + normalized_feature = _registered_feature(feature) + completed = subprocess.run( + _worker_command(normalized_feature, sample_rate_hz, duration_seconds), + input=decoded_pcm.tobytes(), + capture_output=True, + shell=False, + check=False, + cwd=_REPOSITORY_ROOT, + ) + if completed.returncode != 0: + diagnostic = completed.stderr.decode("utf-8", errors="replace").strip() + raise RuntimeError( + "structure resource worker failed with exit code " + f"{completed.returncode}: {diagnostic[:2000]}" + ) + if completed.stderr: + diagnostic = completed.stderr.decode("utf-8", errors="replace").strip() + raise RuntimeError( + "structure resource worker emitted stderr during scientific measurement: " + f"{diagnostic[:2000]}" + ) + try: + payload = json.loads(completed.stdout.decode("utf-8")) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise RuntimeError("structure resource worker returned malformed JSON") from exc + if not isinstance(payload, dict) or set(payload) != { + "feature", + "latency_ns", + "peak_rss_mib", + }: + raise RuntimeError("structure resource worker returned an unexpected receipt shape") + if payload["feature"] != normalized_feature: + raise RuntimeError("structure resource worker feature identity drifted") + latency_ns = payload["latency_ns"] + peak_rss_mib = payload["peak_rss_mib"] + if isinstance(latency_ns, bool) or not isinstance(latency_ns, int) or latency_ns <= 0: + raise RuntimeError("structure resource worker returned invalid latency_ns") + if isinstance(peak_rss_mib, bool) or not isinstance(peak_rss_mib, (int, float)): + raise RuntimeError("structure resource worker returned invalid peak_rss_mib") + normalized_peak = float(peak_rss_mib) + if not math.isfinite(normalized_peak) or normalized_peak <= 0.0: + raise RuntimeError("structure resource worker returned invalid peak_rss_mib") + return IsolatedLaneTrial( + feature=normalized_feature, + latency_ns=latency_ns, + peak_rss_mib=normalized_peak, + ) + + +def _summarize_trials(trials: Sequence[IsolatedLaneTrial]) -> LaneResourceSummary: + """Summarize one lane with the preregistered linear p50/p95 and maximum RSS.""" + if len(trials) != _MEASURED_TRIALS: + raise ValueError(f"lane evidence must contain exactly {_MEASURED_TRIALS} trials") + features = {trial.feature for trial in trials} + if len(features) != 1: + raise ValueError("lane evidence must contain exactly one feature identity") + latencies = np.asarray( + [trial.latency_ns / 1_000_000_000.0 for trial in trials], + dtype=np.float64, + ) + if not np.all(np.isfinite(latencies)) or np.any(latencies <= 0.0): + raise ValueError("lane evidence contains invalid latency") + p50, p95 = np.quantile(latencies, [0.5, 0.95], method="linear") + peak_rss_mib = max(trial.peak_rss_mib for trial in trials) + if not math.isfinite(peak_rss_mib) or peak_rss_mib <= 0.0: + raise ValueError("lane evidence contains invalid peak RSS") + return LaneResourceSummary( + p50_latency_seconds=float(p50), + p95_latency_seconds=float(p95), + peak_rss_mib=float(peak_rss_mib), + measured_trials=len(trials), + ) + + +def measure_paired_repository_lane_resources( + decoded_pcm: object, + sample_rate_hz: object, + duration_seconds: object, +) -> PairedLaneResourceEvidence: + """Measure paired CQT/STFT performance on the exact immutable admitted PCM.""" + pcm = _pcm_view(decoded_pcm) + sample_rate = _positive_sample_rate(sample_rate_hz) + exact_duration = _duration( + duration_seconds, + decoded_frames=pcm.nbytes // 4, + sample_rate_hz=sample_rate, + ) + baseline_trials: list[IsolatedLaneTrial] = [] + candidate_trials: list[IsolatedLaneTrial] = [] + for trial_index in range(_MEASURED_TRIALS): + order: tuple[ChromaFeature, ChromaFeature] + order = ("cqt", "stft") if trial_index % 2 == 0 else ("stft", "cqt") + for feature in order: + observation = _run_isolated_trial( + feature, + pcm, + sample_rate, + exact_duration, + ) + if feature == "cqt": + baseline_trials.append(observation) + else: + candidate_trials.append(observation) + return PairedLaneResourceEvidence( + contract_id="isolated-single-shot-v1", + baseline=_summarize_trials(baseline_trials), + candidate=_summarize_trials(candidate_trials), + ) + + +def _worker_main(argv: Sequence[str]) -> int: + """Execute the internal stdin→single-shot→JSON worker protocol.""" + parser = argparse.ArgumentParser(add_help=False) + parser.add_argument("feature", choices=("cqt", "stft")) + parser.add_argument("sample_rate_hz", type=int) + parser.add_argument("duration_numerator", type=int) + parser.add_argument("duration_denominator", type=int) + args = parser.parse_args(argv) + if args.duration_denominator <= 0: + raise ValueError("duration denominator must be positive") + raw_pcm = sys.stdin.buffer.read() + trial = _measure_current_process( + args.feature, + memoryview(raw_pcm), + args.sample_rate_hz, + Fraction(args.duration_numerator, args.duration_denominator), + ) + print( + json.dumps( + { + "feature": trial.feature, + "latency_ns": trial.latency_ns, + "peak_rss_mib": trial.peak_rss_mib, + }, + allow_nan=False, + separators=(",", ":"), + sort_keys=True, + ) + ) + return 0 + + +def main(argv: Sequence[str] | None = None) -> int: + """Run only the private worker protocol; parent orchestration imports this module.""" + args = list(sys.argv[1:] if argv is None else argv) + if not args or args[0] != "--worker": + raise SystemExit("measure_structure_lane_resources.py is an internal worker") + return _worker_main(args[1:]) + + +if __name__ == "__main__": + raise SystemExit(main()) From 3877adc806229b08a07fbbd00c38b8809e595163 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:05:17 +0900 Subject: [PATCH 182/216] feat(mir): bind performance measurement semantics to registration --- .../validate_structure_noninferiority.py | 24 ++++++++++++++++--- 1 file changed, 21 insertions(+), 3 deletions(-) diff --git a/scripts/research/validate_structure_noninferiority.py b/scripts/research/validate_structure_noninferiority.py index f1f2fcd5c..d32bb83c3 100644 --- a/scripts/research/validate_structure_noninferiority.py +++ b/scripts/research/validate_structure_noninferiority.py @@ -2,9 +2,10 @@ """Bind exact scientific semantics into the structure experiment identity. The base registration schema owns corpus, experiment runtime, margins, and -result-policy fields. This façade adds repository-owned metric, aggregation, and -paired-uncertainty semantics to the canonical digest without making experiment -authors duplicate immutable implementation constants in registration JSON. +result-policy fields. This façade adds repository-owned metric, aggregation, +performance-measurement, and paired-uncertainty semantics to the canonical +digest without making experiment authors duplicate immutable implementation +constants in registration JSON. """ from __future__ import annotations @@ -44,6 +45,23 @@ "frame_size_seconds": 0.1, "beta": 1.0, }, + "performance_measurement": { + "contract_id": "isolated-single-shot-v1", + "supported_platforms": ["darwin", "win32"], + "warmup_trials": 0, + "measured_trials": 20, + "trial_process": "fresh_subprocess_per_lane_trial", + "lane_order": "alternate_baseline_candidate_by_trial_index", + "timer": "time.perf_counter_ns", + "timer_scope": "repository_structure_segmenter_only", + "worker_startup_in_latency": False, + "input_transfer_in_latency": False, + "latency_quantiles": [0.5, 0.95], + "quantile_method": "linear", + "memory_metric": "process_peak_resident_set_size", + "memory_scope": "entire_worker_process_lifetime_including_pcm_input", + "peak_rss_aggregation": "maximum_across_trials", + }, "aggregation_uncertainty": { "aggregation_id": "macro-track-v1", "sampling_unit": "registered_track_pair", From bf905c74cfdd24f67708e5acf6619832b69737f2 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:05:38 +0900 Subject: [PATCH 183/216] test(mir): require performance evidence on admitted track --- ...st_structure_admitted_track_performance.py | 137 ++++++++++++++++++ 1 file changed, 137 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_admitted_track_performance.py diff --git a/services/analysis-engine/tests/test_structure_admitted_track_performance.py b/services/analysis-engine/tests/test_structure_admitted_track_performance.py new file mode 100644 index 000000000..7123ee1a9 --- /dev/null +++ b/services/analysis-engine/tests/test_structure_admitted_track_performance.py @@ -0,0 +1,137 @@ +"""Contract for canonical latency/RSS evidence on the admitted-track handoff.""" + +from __future__ import annotations + +import struct +from fractions import Fraction +from types import ModuleType, SimpleNamespace +from typing import Any + +import pytest +from conftest import load_module + + +def _consumer_module() -> ModuleType: + """Load the repository-owned admitted-track scientific consumer.""" + return load_module( + "scripts/research/evaluate_admitted_structure_track.py", + "evaluate_admitted_structure_track_with_performance", + ) + + +def _segments() -> tuple[SimpleNamespace, ...]: + """Return one full-duration protocol-compatible segment.""" + return (SimpleNamespace(start=Fraction(0), end=Fraction(4, 5), label="verse"),) + + +def _metric_result() -> SimpleNamespace: + """Return a complete recognized-metric fixture.""" + return SimpleNamespace( + boundary_precision_0_5=1.0, + boundary_recall_0_5=1.0, + boundary_f_0_5=1.0, + boundary_precision_3_0=1.0, + boundary_recall_3_0=1.0, + boundary_f_3_0=1.0, + reference_to_estimate_median_deviation_seconds=0.0, + estimate_to_reference_median_deviation_seconds=0.0, + repetition_pairwise_precision=1.0, + repetition_pairwise_recall=1.0, + repetition_pairwise_f=1.0, + ) + + +def test_registered_consumer_records_canonical_performance_on_same_admitted_pcm( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """Canonical evidence must include preregistered p50/p95/RSS for both lanes.""" + module = _consumer_module() + performance_calls: list[tuple[memoryview, int, Fraction]] = [] + + def segmenter( + _decoded_pcm: memoryview, + _sample_rate_hz: int, + _duration_seconds: Fraction, + ) -> tuple[SimpleNamespace, ...]: + return _segments() + + monkeypatch.setattr( + module._LANES, + "repository_structure_segmenter", + lambda _feature: segmenter, + ) + monkeypatch.setattr( + module._SEGMENTATION_EVALUATOR, + "calculate_structure_segmentation_metrics", + lambda _reference, _estimated: _metric_result(), + ) + monkeypatch.setattr( + module._RUNTIME_VERIFIER, + "_installed_mir_eval_version", + lambda: "0.8.2", + ) + + def fake_performance( + decoded_pcm: memoryview, + sample_rate_hz: int, + duration_seconds: Fraction, + ) -> object: + performance_calls.append((decoded_pcm, sample_rate_hz, duration_seconds)) + return SimpleNamespace( + contract_id="isolated-single-shot-v1", + baseline=SimpleNamespace( + p50_latency_seconds=0.80, + p95_latency_seconds=0.95, + peak_rss_mib=512.0, + measured_trials=20, + ), + candidate=SimpleNamespace( + p50_latency_seconds=0.42, + p95_latency_seconds=0.50, + peak_rss_mib=480.0, + measured_trials=20, + ), + ) + + monkeypatch.setattr( + module._RESOURCE_MEASUREMENT, + "measure_paired_repository_lane_resources", + fake_performance, + ) + + consumer = module.PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis() + pcm = memoryview(struct.pack("<8f", *([0.0] * 8))) + annotation = memoryview(b"0.0\t0.8\tverse\n") + consumer("track-01", pcm, annotation, 10) + + assert performance_calls == [(pcm, 10, Fraction(4, 5))] + evidence = consumer.evidence[0] + assert evidence.performance_contract_id == "isolated-single-shot-v1" + assert evidence.baseline_performance is not None + assert evidence.candidate_performance is not None + assert evidence.baseline_performance.p50_latency_seconds == pytest.approx(0.80) + assert evidence.baseline_performance.p95_latency_seconds == pytest.approx(0.95) + assert evidence.baseline_performance.peak_rss_mib == pytest.approx(512.0) + assert evidence.baseline_performance.measured_trials == 20 + assert evidence.candidate_performance.p50_latency_seconds == pytest.approx(0.42) + assert evidence.candidate_performance.p95_latency_seconds == pytest.approx(0.50) + assert evidence.candidate_performance.peak_rss_mib == pytest.approx(480.0) + + +def test_custom_consumer_requires_performance_evaluator_and_contract_together() -> None: + """Caller injection cannot produce unbound resource evidence.""" + module = _consumer_module() + + def segmenter( + _decoded_pcm: memoryview, + _sample_rate_hz: int, + _duration_seconds: Fraction, + ) -> tuple[Any, ...]: + return _segments() + + with pytest.raises(ValueError, match="performance evaluator and contract identity"): + module.PairedFunctionalAccuracyTrackConsumer( + baseline_segmenter=segmenter, + candidate_segmenter=segmenter, + performance_evaluator=lambda *_args: object(), + ) From f06f1dc059a461037ca95fd55e0c7b49e6ccb2f4 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:06:20 +0900 Subject: [PATCH 184/216] feat(mir): bind isolated performance evidence to admitted tracks --- .../evaluate_admitted_structure_track.py | 96 ++++++++++++++++++- 1 file changed, 93 insertions(+), 3 deletions(-) diff --git a/scripts/research/evaluate_admitted_structure_track.py b/scripts/research/evaluate_admitted_structure_track.py index 61dda5fbb..dad68c8b8 100644 --- a/scripts/research/evaluate_admitted_structure_track.py +++ b/scripts/research/evaluate_admitted_structure_track.py @@ -6,7 +6,7 @@ same immutable canonical PCM memoryview, while the reference segmentation is parsed from the same immutable admitted annotation snapshot. The canonical CQT/STFT experiment additionally evaluates the reviewed mir_eval boundary, -deviation, and repetition contract under the exact research runtime lock. +deviation, repetition, and isolated single-shot latency/RSS contracts. Scientific corpus choice, margins, aggregation, uncertainty, and the production feature switch remain outside this boundary. """ @@ -15,6 +15,7 @@ import hashlib import importlib.util +import math import sys from collections.abc import Callable, Sequence from dataclasses import dataclass @@ -25,6 +26,7 @@ Segmenter = Callable[[memoryview, int, Fraction], Sequence[Any]] SegmentationMetricEvaluator = Callable[[object, object], object] +PerformanceEvaluator = Callable[[memoryview, int, Fraction], object] _REPOSITORY_ROOT = Path(__file__).resolve().parents[2] _STRUCTURE_METRIC_RUNTIME_LOCK = ( _REPOSITORY_ROOT / "services/analysis-engine/requirements-structure-metrics.lock" @@ -71,6 +73,10 @@ def _load_sibling(filename: str, module_name: str) -> ModuleType: "verify_structure_metric_runtime_lock.py", "_bandscope_structure_metric_runtime_lock", ) +_RESOURCE_MEASUREMENT = _load_sibling( + "measure_structure_lane_resources.py", + "_bandscope_structure_lane_resources", +) @dataclass(frozen=True, slots=True) @@ -90,6 +96,16 @@ class StructureSegmentationMetricEvidence: repetition_pairwise_f: float +@dataclass(frozen=True, slots=True) +class StructurePerformanceEvidence: + """Preregistered per-track single-shot latency quantiles and peak process RSS.""" + + p50_latency_seconds: float + p95_latency_seconds: float + peak_rss_mib: float + measured_trials: int + + @dataclass(frozen=True, slots=True) class PairedFunctionalAccuracyEvidence: """Path-free track evidence bound to exact admitted PCM and annotation bytes.""" @@ -108,6 +124,9 @@ class PairedFunctionalAccuracyEvidence: metric_runtime_lock_sha256: str | None baseline_segmentation_metrics: StructureSegmentationMetricEvidence | None candidate_segmentation_metrics: StructureSegmentationMetricEvidence | None + performance_contract_id: str | None + baseline_performance: StructurePerformanceEvidence | None + candidate_performance: StructurePerformanceEvidence | None def _track_id(value: object) -> str: @@ -165,6 +184,31 @@ def _segmentation_metric_evidence(result: object) -> StructureSegmentationMetric ) +def _performance_evidence(result: object) -> StructurePerformanceEvidence: + """Validate and copy one lane's isolated performance summary.""" + p50 = float(getattr(result, "p50_latency_seconds")) + p95 = float(getattr(result, "p95_latency_seconds")) + peak_rss = float(getattr(result, "peak_rss_mib")) + measured_trials = getattr(result, "measured_trials") + if not all(math.isfinite(value) and value > 0.0 for value in (p50, p95, peak_rss)): + raise RuntimeError("structure performance evidence must be finite and positive") + if p95 < p50: + raise RuntimeError("structure performance p95 must be greater than or equal to p50") + if isinstance(measured_trials, bool) or not isinstance(measured_trials, int): + raise RuntimeError("structure performance measured_trials must be an integer") + expected_trials = _RESOURCE_MEASUREMENT.PERFORMANCE_MEASUREMENT_CONTRACT[ + "measured_trials" + ] + if measured_trials != expected_trials: + raise RuntimeError("structure performance measured_trials drifted from contract") + return StructurePerformanceEvidence( + p50_latency_seconds=p50, + p95_latency_seconds=p95, + peak_rss_mib=peak_rss, + measured_trials=measured_trials, + ) + + class PairedFunctionalAccuracyTrackConsumer: """Collect paired structure evidence from corpus-admission callbacks.""" @@ -175,8 +219,10 @@ def __init__( candidate_segmenter: Segmenter, segmentation_metric_evaluator: SegmentationMetricEvaluator | None = None, metric_runtime_lock_sha256: str | None = None, + performance_evaluator: PerformanceEvaluator | None = None, + performance_contract_id: str | None = None, ) -> None: - """Bind preregistered lanes and optional recognized-metric runtime identity.""" + """Bind preregistered quality lanes and optional performance evidence.""" if not callable(baseline_segmenter) or not callable(candidate_segmenter): raise TypeError("baseline_segmenter and candidate_segmenter must be callable") if (segmentation_metric_evaluator is None) != (metric_runtime_lock_sha256 is None): @@ -187,19 +233,36 @@ def __init__( segmentation_metric_evaluator ): raise TypeError("segmentation_metric_evaluator must be callable") + if (performance_evaluator is None) != (performance_contract_id is None): + raise ValueError( + "performance evaluator and contract identity must be bound together" + ) + if performance_evaluator is not None and not callable(performance_evaluator): + raise TypeError("performance_evaluator must be callable") + if performance_contract_id is not None and ( + not isinstance(performance_contract_id, str) or not performance_contract_id + ): + raise ValueError("performance_contract_id must be non-empty text") self._baseline_segmenter = baseline_segmenter self._candidate_segmenter = candidate_segmenter self._segmentation_metric_evaluator = segmentation_metric_evaluator self._metric_runtime_lock_sha256 = metric_runtime_lock_sha256 + self._performance_evaluator = performance_evaluator + self._performance_contract_id = performance_contract_id self._evidence: list[PairedFunctionalAccuracyEvidence] = [] self._measured_track_ids: set[str] = set() @classmethod def for_registered_cqt_stft_hypothesis(cls) -> PairedFunctionalAccuracyTrackConsumer: - """Bind canonical CQT/STFT lanes and the reviewed segmentation-metric overlay.""" + """Bind canonical CQT/STFT quality and performance measurement contracts.""" runtime_identity = _RUNTIME_VERIFIER.load_structure_metric_runtime_lock_identity( _STRUCTURE_METRIC_RUNTIME_LOCK ) + performance_contract_id = _RESOURCE_MEASUREMENT.PERFORMANCE_MEASUREMENT_CONTRACT[ + "contract_id" + ] + if not isinstance(performance_contract_id, str) or not performance_contract_id: + raise RuntimeError("structure performance contract_id is invalid") return cls( baseline_segmenter=_LANES.repository_structure_segmenter("cqt"), candidate_segmenter=_LANES.repository_structure_segmenter("stft"), @@ -207,6 +270,10 @@ def for_registered_cqt_stft_hypothesis(cls) -> PairedFunctionalAccuracyTrackCons _SEGMENTATION_EVALUATOR.calculate_structure_segmentation_metrics ), metric_runtime_lock_sha256=runtime_identity.lock_sha256, + performance_evaluator=( + _RESOURCE_MEASUREMENT.measure_paired_repository_lane_resources + ), + performance_contract_id=performance_contract_id, ) @property @@ -278,6 +345,26 @@ def __call__( ) metric_runtime_lock_sha256 = runtime_identity.lock_sha256 + performance_contract_id: str | None = None + baseline_performance: StructurePerformanceEvidence | None = None + candidate_performance: StructurePerformanceEvidence | None = None + if self._performance_evaluator is not None: + performance = self._performance_evaluator( + pcm, + sample_rate, + duration_seconds, + ) + observed_contract = getattr(performance, "contract_id", None) + if observed_contract != self._performance_contract_id: + raise RuntimeError("structure performance contract identity drifted") + baseline_performance = _performance_evidence( + getattr(performance, "baseline") + ) + candidate_performance = _performance_evidence( + getattr(performance, "candidate") + ) + performance_contract_id = self._performance_contract_id + evidence = PairedFunctionalAccuracyEvidence( track_id=normalized_track_id, decoded_pcm_sha256=hashlib.sha256(pcm).hexdigest(), @@ -293,6 +380,9 @@ def __call__( metric_runtime_lock_sha256=metric_runtime_lock_sha256, baseline_segmentation_metrics=baseline_segmentation_metrics, candidate_segmentation_metrics=candidate_segmentation_metrics, + performance_contract_id=performance_contract_id, + baseline_performance=baseline_performance, + candidate_performance=candidate_performance, ) self._evidence.append(evidence) self._measured_track_ids.add(normalized_track_id) From a05c345a23910e0e0c48f327ded8ea1a8d09bb9a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:07:03 +0900 Subject: [PATCH 185/216] test(mir): align metric contract fixture with performance owner --- ...ructure_metric_preregistration_contract.py | 30 +++++++++++++++++-- 1 file changed, 28 insertions(+), 2 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py b/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py index b41e8cee9..f8cfc6084 100644 --- a/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py +++ b/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py @@ -37,6 +37,23 @@ "frame_size_seconds": 0.1, "beta": 1.0, }, + "performance_measurement": { + "contract_id": "isolated-single-shot-v1", + "supported_platforms": ["darwin", "win32"], + "warmup_trials": 0, + "measured_trials": 20, + "trial_process": "fresh_subprocess_per_lane_trial", + "lane_order": "alternate_baseline_candidate_by_trial_index", + "timer": "time.perf_counter_ns", + "timer_scope": "repository_structure_segmenter_only", + "worker_startup_in_latency": False, + "input_transfer_in_latency": False, + "latency_quantiles": [0.5, 0.95], + "quantile_method": "linear", + "memory_metric": "process_peak_resident_set_size", + "memory_scope": "entire_worker_process_lifetime_including_pcm_input", + "peak_rss_aggregation": "maximum_across_trials", + }, "aggregation_uncertainty": { "aggregation_id": "macro-track-v1", "sampling_unit": "registered_track_pair", @@ -58,6 +75,7 @@ def _validator() -> ModuleType: + """Load the metric-aware structure registration validator.""" return load_module( "scripts/research/validate_structure_noninferiority.py", "validate_structure_noninferiority_metric_contract", @@ -65,6 +83,7 @@ def _validator() -> ModuleType: def _canonical_digest(value: object) -> str: + """Return the canonical compact JSON SHA-256 used by the registration owner.""" payload = json.dumps( value, allow_nan=False, @@ -76,7 +95,7 @@ def _canonical_digest(value: object) -> str: def test_registration_digest_binds_exact_scientific_semantics() -> None: - """Scientific identity includes metric runtime, adapters, and aggregation procedure.""" + """Scientific identity includes metrics, performance, and aggregation procedure.""" validator = _validator() registration = _registration() @@ -94,7 +113,7 @@ def test_registration_digest_binds_exact_scientific_semantics() -> None: def test_digest_contract_matches_executable_metric_and_aggregation_owners() -> None: - """Digest metadata cannot drift from lock, metric, or paired-bootstrap owners.""" + """Digest metadata cannot drift from metric, performance, or aggregation owners.""" validator = _validator() adapter = load_module( "scripts/research/evaluate_structure_segmentation_metrics.py", @@ -104,6 +123,10 @@ def test_digest_contract_matches_executable_metric_and_aggregation_owners() -> N "scripts/research/verify_structure_metric_runtime_lock.py", "structure_metric_runtime_owner_for_registration", ) + measurement = load_module( + "scripts/research/measure_structure_lane_resources.py", + "structure_performance_owner_for_registration", + ) aggregation = load_module( "scripts/research/aggregate_structure_noninferiority.py", "structure_aggregation_owner_for_registration", @@ -134,6 +157,9 @@ def test_digest_contract_matches_executable_metric_and_aggregation_owners() -> N "frame_size_seconds": adapter.PAIRWISE_FRAME_SIZE_SECONDS, "beta": adapter.PAIRWISE_BETA, } + assert contract["performance_measurement"] == ( + measurement.PERFORMANCE_MEASUREMENT_CONTRACT + ) assert contract["aggregation_uncertainty"] == ( aggregation.AGGREGATION_UNCERTAINTY_CONTRACT ) From 6d49d7a54672eda000d94682c2040915f6b4657e Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:09:15 +0900 Subject: [PATCH 186/216] fix(mir): remove unused resource-measurement import --- scripts/research/measure_structure_lane_resources.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scripts/research/measure_structure_lane_resources.py b/scripts/research/measure_structure_lane_resources.py index 14926c3f6..01e27d75c 100644 --- a/scripts/research/measure_structure_lane_resources.py +++ b/scripts/research/measure_structure_lane_resources.py @@ -37,7 +37,7 @@ from fractions import Fraction from pathlib import Path from types import ModuleType -from typing import Any, Literal, Sequence +from typing import Literal, Sequence import numpy as np From 11c59ade85b1866e550edec2ea26cb185670c9ab Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:09:39 +0900 Subject: [PATCH 187/216] test(mir): isolate focused metric test from platform performance worker --- ...ure_admitted_track_segmentation_metrics.py | 25 +++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_admitted_track_segmentation_metrics.py b/services/analysis-engine/tests/test_structure_admitted_track_segmentation_metrics.py index 1dbbd6fdb..b91c665c7 100644 --- a/services/analysis-engine/tests/test_structure_admitted_track_segmentation_metrics.py +++ b/services/analysis-engine/tests/test_structure_admitted_track_segmentation_metrics.py @@ -51,6 +51,25 @@ def _metric_result(seed: float) -> SimpleNamespace: ) +def _performance_result() -> SimpleNamespace: + """Return deterministic preregistered performance evidence for focused tests.""" + return SimpleNamespace( + contract_id="isolated-single-shot-v1", + baseline=SimpleNamespace( + p50_latency_seconds=0.80, + p95_latency_seconds=0.95, + peak_rss_mib=512.0, + measured_trials=20, + ), + candidate=SimpleNamespace( + p50_latency_seconds=0.42, + p95_latency_seconds=0.50, + peak_rss_mib=480.0, + measured_trials=20, + ), + ) + + def test_registered_consumer_binds_runtime_lock_and_scores_both_feature_lanes( monkeypatch: pytest.MonkeyPatch, ) -> None: @@ -95,6 +114,11 @@ def metric_evaluator( "_installed_mir_eval_version", lambda: "0.8.2", ) + monkeypatch.setattr( + module._RESOURCE_MEASUREMENT, + "measure_paired_repository_lane_resources", + lambda *_args: _performance_result(), + ) consumer = ( module.PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis() @@ -115,6 +139,7 @@ def metric_evaluator( assert evidence.candidate_segmentation_metrics.boundary_f_0_5 == pytest.approx(0.68) assert evidence.baseline_segmentation_metrics.repetition_pairwise_f == pytest.approx(0.72) assert evidence.candidate_segmentation_metrics.repetition_pairwise_f == pytest.approx(0.70) + assert evidence.performance_contract_id == "isolated-single-shot-v1" def test_registered_consumer_fails_before_metric_execution_on_runtime_drift( From c1fc867600b0ef3e3f76b13a9e9042c76a6ef786 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:10:50 +0900 Subject: [PATCH 188/216] docs(mir): trace performance measurement contract --- ...ucture-performance-measurement-contract.md | 111 ++++++++++++++++++ 1 file changed, 111 insertions(+) create mode 100644 docs/traceability/mir/structure-performance-measurement-contract.md diff --git a/docs/traceability/mir/structure-performance-measurement-contract.md b/docs/traceability/mir/structure-performance-measurement-contract.md new file mode 100644 index 000000000..4214bdc83 --- /dev/null +++ b/docs/traceability/mir/structure-performance-measurement-contract.md @@ -0,0 +1,111 @@ +# Structure performance measurement contract + +Status: Proposed +Owner: Signal-MIR Analysis +Tracking: #1225, #1228 +Parent: `docs/traceability/mir/structure-feature-noninferiority.md` + +## Problem + +The structure noninferiority result schema already carries per-track `p50_latency_seconds`, `p95_latency_seconds`, and `peak_rss_mib`, and the canonical aggregator uses those fields for latency comparison and maximum peak-RSS reporting. Before this change, however, no repository-owned producer defined how those numbers were obtained. A syntactically valid receipt could therefore use arbitrary warm-up counts, timer boundaries, process scopes, quantile construction, or resident-memory definitions while still passing schema validation. + +That is not sufficient for a production decision between `chroma_cqt` and `chroma_stft`. Performance evidence must be fixed before candidate results are observed, must consume the same admitted decoded PCM identity as the quality comparison, and must avoid a cache-warmed benchmark path that is unlike a user's first analysis of a song. + +## Decision + +`isolated-single-shot-v1` is the only preregistered structure performance contract in schema v1. Its complete semantics are owned by `scripts/research/measure_structure_lane_resources.py` and copied into `STRUCTURE_METRIC_CONTRACT["performance_measurement"]`, so they participate in the metric-aware registration SHA-256. + +The contract is: + +- supported evidence platforms: macOS (`darwin`) and Windows (`win32`); +- warm-up trials: 0; +- measured trials: 20 per feature lane and per admitted track; +- each observation runs in a fresh Python subprocess; +- paired trial order alternates CQT→STFT, then STFT→CQT, by trial index; +- latency clock: `time.perf_counter_ns()`; +- latency scope: only the repository-owned `repository_structure_segmenter(feature)` call; +- worker startup and parent→child PCM transfer are excluded from latency; +- latency summaries: p50 and p95 via NumPy `quantile(..., method="linear")`; +- memory metric: process peak resident set size over the worker's entire lifetime, including the PCM input resident in the child; +- per-lane track-level `peak_rss_mib`: maximum observed process peak RSS across the 20 trials. + +Twenty observations are a preregistered engineering sampling plan, not a claim that twenty observations estimate an asymptotic tail distribution. The production decision still uses the registered paired track-level bootstrap across the rights-cleared corpus; the within-track repetitions create the frozen p50/p95 performance summary supplied to that higher-level procedure. + +## Timer boundary + +Python documents `perf_counter()` as a performance counter with the highest available resolution for short durations and states that it is monotonic in CPython; `perf_counter_ns()` returns the same clock as integer nanoseconds and avoids float precision loss. The worker therefore records `perf_counter_ns()` immediately before and after the one repository-owned structure-segmentation call. + +Process launch, interpreter startup, imports, and PCM transfer are not included in the latency value because those costs are invariant experiment harness overhead rather than the feature-representation computation under comparison. They are not silently removed from memory evidence: peak RSS is process-lifetime evidence and therefore includes interpreter/runtime state and the PCM input already resident before the timer starts. + +## No warm-up path + +The contract deliberately performs zero same-process warm-up iterations. Reusing one process for repeated warm-up and measured calls could retain allocator state, imported-library caches, FFT plans, page cache effects, or other state that is not guaranteed on a buyer's first analysis. Each observation instead starts from a fresh worker process. + +This does not claim that operating-system file cache, CPU frequency, thermal state, or other host-wide state is perfectly reset between trials. Those factors remain part of the exact host/runtime profile and claim boundary that must be frozen before real-corpus execution. Alternating lane order limits deterministic order bias but does not erase host-state variance. + +## Peak RSS semantics + +On macOS, the worker reads `getrusage(RUSAGE_SELF).ru_maxrss`. Apple's `getrusage(2)` documentation defines `ru_maxrss` as maximum resident set size and documents the value in kilobytes, so the implementation converts KiB to MiB by dividing by 1024. + +On Windows, the worker calls `GetProcessMemoryInfo` and uses `PROCESS_MEMORY_COUNTERS.PeakWorkingSetSize`. Microsoft documents the working set as physical memory mapped into the process address space and exposes both current and peak working-set size. The implementation converts bytes to MiB. + +Linux is intentionally not admitted for `isolated-single-shot-v1`. Python's `resource` API is Unix-specific and `ru_maxrss` unit conventions are platform-dependent; adding Linux would require a separately reviewed platform contract rather than silently treating all `getrusage` values as equivalent. + +## Exact admitted-track binding + +`PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis()` binds the performance evaluator and `isolated-single-shot-v1` contract identity together with the canonical CQT/STFT quality lanes. For each admitted track the resource measurement receives the same immutable PCM memoryview, sample rate, and exact `decoded_frames / sample_rate_hz` duration already used for quality evidence. + +The resource worker is not given a corpus path, URL, track ID, arbitrary command, or feature selected by external text. The feature identity is closed-world (`cqt` or `stft`), stdin contains only the already-admitted PCM bytes, and subprocess launch uses an argument array with `shell=False`. + +The scientific receipt records the contract identity plus baseline/candidate p50, p95, peak RSS, and measured-trial count. A caller cannot bind a performance evaluator without a contract identity, and the canonical consumer rejects a result whose contract ID or measured-trial count differs from the preregistered owner. + +## RED → GREEN lineage + +- RED `6fcc9e96eba9e6a9c3b3deed19b40a54e6c33ade` required a repository-owned measurement contract, alternating paired order, exact linear p50/p95 construction, maximum RSS, and immutable PCM/duration admission. +- GREEN `0b80bc22adadaf4998b562dbf5fcd94a0492ba5a` implemented `isolated-single-shot-v1` with fresh subprocesses, `perf_counter_ns`, Darwin `ru_maxrss`, Windows `PeakWorkingSetSize`, and fail-closed worker receipts. +- Registration binding `3877adc806229b08a07fbbd00c38b8809e595163` added the complete performance contract to the metric-aware registration digest. +- RED `bf905c74cfdd24f67708e5acf6619832b69737f2` required canonical admitted-track evidence to contain the exact performance contract and baseline/candidate summaries. +- GREEN `f06f1dc059a461037ca95fd55e0c7b49e6ccb2f4` bound the canonical CQT/STFT consumer to the performance producer and same admitted PCM identity. +- Contract fixture `a05c345a23910e0e0c48f327ded8ea1a8d09bb9a` cross-checks registration metadata against the executable performance owner. +- Hygiene repair `6d49d7a54672eda000d94682c2040915f6b4657e` removed an unused typing import before hosted lint evidence. +- Focused-test alignment `11c59ade85b1866e550edec2ea26cb185670c9ab` prevents the recognized-metric unit boundary from accidentally launching platform performance workers; the dedicated performance tests own that contract. + +## Rejected alternatives + +**Time the existing parent-process quality call.** Rejected because both lanes would inherit process/import/allocation state from earlier scientific work and each other, obscuring a reproducible first-analysis comparison. + +**Warm the segmenter or feature computation before measurement.** Rejected because it would preferentially measure a cache-hot state that the buyer path does not guarantee. + +**Include subprocess startup and PCM transport in algorithm latency.** Rejected because the experiment question is whether STFT changes the structure-analysis computation relative to CQT. Harness launch/transport costs are common orchestration overhead. They remain visible in process-lifetime RSS and must be kept out of claims about end-to-end application startup. + +**Use one cross-platform `ru_maxrss` conversion.** Rejected because RSS units and APIs are platform-specific. The contract names the OS API and conversion explicitly and fails closed elsewhere. + +**Use average RSS or a sampled polling thread.** Rejected because average sampling introduces sampling cadence as another scientific choice and can miss short peaks. The native process peak counters directly provide the resource ceiling relevant to buyer capacity risk. + +**Use fewer measured trials when a track is expensive.** Rejected because data-dependent or operator-dependent trial reduction would change the estimator after corpus identity is known. An execution that cannot complete the preregistered count is a failed track under the existing no-exclusion policy. + +## Security Notes + +- Input remains local-only and path-free after corpus admission. +- Subprocess invocation uses `sys.executable`, the repository-owned worker path, exact numeric arguments, `shell=False`, and stdin PCM bytes; no generic execution surface is introduced. +- Any non-zero exit, worker stderr, malformed JSON, unexpected receipt field, unsupported platform, invalid timer, or invalid RSS value fails the scientific measurement. +- Raw licensed audio and workstation paths are not emitted in durable performance evidence. +- This contract adds no telemetry or network-dependent runtime path. + +## Claim boundary and remaining work + +This contract closes the missing canonical producer for per-track p50/p95 latency and peak RSS. It does not establish that STFT is faster, that either lane meets a product latency target, or that twenty within-track repetitions make p95 a universal tail-latency estimate. + +Before rights-cleared real-audio execution, reviewers still must freeze the concrete corpus membership and representativeness, numeric quality noninferiority margins, maximum candidate latency ratio, exact bootstrap resample count and seed, exact host/runtime profile, and claim boundary. Candidate results must not be inspected before those choices are frozen. + +Production default selection remains a separate change after the complete registered experiment passes. Synthetic and monkeypatched fixtures validate the measurement machinery only; they are not scientific acceptance evidence. + +## References + +Python Software Foundation. (2026). *time — Time access and conversions: time.perf_counter() and time.perf_counter_ns()*. Python 3.14 documentation. https://docs.python.org/3/library/time.html#time.perf_counter + +Apple Inc. (2004). *getrusage(2) — get information about resource utilization*. Mac OS X manual pages. https://developer.apple.com/library/archive/documentation/System/Conceptual/ManPages_iPhoneOS/man2/getrusage.2.html + +Microsoft. (2024). *Process memory usage information*. Windows App Development documentation. https://learn.microsoft.com/windows/win32/psapi/process-memory-usage-information + +NumPy Developers. (2026). *numpy.quantile*. NumPy reference. https://numpy.org/doc/stable/reference/generated/numpy.quantile.html From 7445f22923e4070196ad7443aa60b99477cb20e9 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:13:04 +0900 Subject: [PATCH 189/216] docs(mir): make noninferiority gate performance-code-current --- .../mir/structure-feature-noninferiority.md | 73 +++++++++++-------- 1 file changed, 43 insertions(+), 30 deletions(-) diff --git a/docs/traceability/mir/structure-feature-noninferiority.md b/docs/traceability/mir/structure-feature-noninferiority.md index 9368eafa0..7eb8967a7 100644 --- a/docs/traceability/mir/structure-feature-noninferiority.md +++ b/docs/traceability/mir/structure-feature-noninferiority.md @@ -9,11 +9,11 @@ Production baseline: `chroma_cqt` in `sections/segmenter.py` The earlier `chroma_cqt` → `chroma_stft` optimization is not a production change until a preregistered, rights-cleared real-audio experiment shows that the candidate is noninferior on rehearsal-relevant structure quality and materially faster on the same corpus and runtime identity. -The repository owns the registration/result-admission boundary in `scripts/research/validate_structure_noninferiority.py` and the paired aggregation/uncertainty implementation in `scripts/research/aggregate_structure_noninferiority.py`. Metric computation, corpus admission, aggregation, uncertainty estimation, and result admission remain separate scientific boundaries, but their result-affecting semantics are bound into one metric-aware registration digest before candidate results may be inspected. +The repository owns the registration/result-admission boundary in `scripts/research/validate_structure_noninferiority.py`, the CQT/STFT admitted-PCM lanes in `scripts/research/measure_structure_feature_lanes.py`, recognized metric adapters, the isolated performance producer in `scripts/research/measure_structure_lane_resources.py`, and the paired aggregation/uncertainty implementation in `scripts/research/aggregate_structure_noninferiority.py`. Corpus admission, quality measurement, performance measurement, aggregation, uncertainty estimation, and result admission remain separate scientific boundaries, but every result-affecting semantic is bound into one metric-aware registration digest before candidate results may be inspected. -The canonical public validator is a metric-aware façade over the historical base decision/result policy in `validate_structure_noninferiority_base.py`. Its SHA-256 binds the validated registration together with repository-owned structure-metric and aggregation/uncertainty semantics. A receipt carrying only the former registration-JSON digest is rejected. +The canonical public validator is a metric-aware façade over the historical base decision/result policy in `validate_structure_noninferiority_base.py`. Its SHA-256 binds the validated registration together with repository-owned metric runtime/adapters, the `isolated-single-shot-v1` performance contract, and aggregation/uncertainty semantics. A receipt carrying only the former registration-JSON digest is rejected. -This keeps a performance result from silently changing the corpus, thresholds, feature identities, metric implementation/runtime, aggregation weighting, bootstrap procedure, runtime, or permitted interpretation after measurements are visible. +This prevents a favorable result from silently changing the corpus, thresholds, feature identities, metric implementation/runtime, timing/RSS method, aggregation weighting, bootstrap procedure, runtime, or permitted interpretation after measurements are visible. ## Registered inputs @@ -26,23 +26,19 @@ A registration is valid only when it records all of the following before the res - the paired-bootstrap procedure identity, 95% confidence level, resample count, and random seed; - a non-empty claim boundary stating the population/runtime scope to which a passing result may be applied. -The registration digest additionally binds the repository-owned segmentation-metric contract. That contract fixes the SHA-256 of `services/analysis-engine/requirements-structure-metrics.lock`, detection windows 0.5 s and 3.0 s with `beta=1.0, trim=True`, deviation with `trim=True`, pairwise grouping with `frame_size=0.1, beta=1.0`, and the complete `macro-track-v1` / `paired-track-bootstrap-v1` aggregation and uncertainty semantics. Focused contract tests cross-check those constants against the executable owners so metadata drift fails before real-audio execution. +The registration digest additionally binds repository-owned scientific implementation constants. It fixes the SHA-256 of `services/analysis-engine/requirements-structure-metrics.lock`, detection windows 0.5 s and 3.0 s with `beta=1.0, trim=True`, deviation with `trim=True`, pairwise grouping with `frame_size=0.1, beta=1.0`, the complete `isolated-single-shot-v1` performance measurement contract, and the complete `macro-track-v1` / `paired-track-bootstrap-v1` aggregation and uncertainty semantics. Focused contract tests cross-check those constants against the executable owners so metadata drift fails before real-audio execution. Schema v1 is closed-world at the registration top level and inside the hypothesis, metric, corpus-track, runtime, result, per-track, aggregate, and measurement objects. Extra fields are not treated as harmless annotations: an unregistered pilot-selection flag, metric weight, corpus-selection marker, cache/runtime hint, p-value, or post-hoc selection marker changes what reviewers may infer was preregistered and therefore fails admission. -`experiment_id` is a semantic label; the metric-aware registration SHA-256 is the exact evidence identity. The canonical hash preserves the validated registration inside a closed envelope together with immutable repository-owned scientific semantics. - -`source_uri` must be an explicit non-`file:` provenance URI. Absolute, relative, drive-relative, and `file:` filesystem forms are rejected. Licensed benchmark material may remain private, but the receipt must identify it without making a workstation path part of scientific provenance. +`experiment_id` is a semantic label; the metric-aware registration SHA-256 is the exact evidence identity. `source_uri` must be an explicit non-`file:` provenance URI. Licensed benchmark material may remain private, but a workstation path is not scientific provenance. Track IDs are labels, not independent scientific units by themselves. Schema v1 rejects duplicate `audio_sha256` values across different track IDs so one recording cannot silently count as multiple independent observations. A scientifically justified repeated-item or clustered design requires a preregistered dependence model and schema revision. -This guard is consistent with recent MIR dataset-quality work. Choi et al. (2025) show that duplicated music items can undermine evaluation through leakage and motivate explicit de-duplication in a large symbolic-music benchmark. That study does not establish BandScope's audio-structure independence assumptions; it supports the narrower rule that duplicate content identity must not masquerade as distinct evaluation evidence. - The current minimum of two tracks is only a technical guard against a one-item corpus. It is not a scientific sample-size claim. Corpus breadth, genre/instrumentation coverage, annotation quality, independence/dependence structure, and an adequate uncertainty plan remain experiment-review questions before a production switch can be accepted. -Schema v1 has no dropout, exclusion, or missing-track policy. Every registered track must therefore complete both baseline and candidate measurement for an acceptance PASS. Any non-empty `failed_tracks` list makes acceptance fail; aggregate and confidence-interval fields must then be JSON `null` so a complete-case subset cannot masquerade as the preregistered corpus. +Schema v1 has no dropout, exclusion, or missing-track policy. Every registered track must complete baseline and candidate quality plus the preregistered performance measurement for an acceptance PASS. Any non-empty `failed_tracks` list makes acceptance fail; aggregate and confidence-interval fields must then be JSON `null` so a complete-case subset cannot masquerade as the preregistered corpus. -## Metrics +## Metrics and measurement contracts BandScope uses the MIREX Music Structure Analysis task and its standardized evaluator repository as the functional-structure reference point. @@ -53,8 +49,8 @@ BandScope uses the MIREX Music Structure Analysis task and its standardized eval | Boundary median deviation, both directions | `mir_eval.segment.deviation(trim=True)` | Report per track and aggregate | | Functional-label accuracy | `ismir-mirex/mirex-evaluation@b9fa0b0...:music_structure_analysis.eval_script.calculate_accuracy`; 0.2 s grid; frame-grid v1; annotation v1; label-mapping v1 | Noninferiority | | Repetition/group consistency precision/recall/F | `mir_eval.segment.pairwise(frame_size=0.1, beta=1.0)` | F noninferiority | -| Latency | Result schema has per-track p50/p95 fields; canonical warm-up/trial/timer contract is still pending | p95 ratio superiority only after that contract is frozen | -| Peak memory | Result schema has per-track peak RSS; canonical process/memory-scope measurement contract is still pending | Report only after that contract is frozen | +| Latency | `isolated-single-shot-v1`: fresh subprocess per observation, 0 warmups, 20 trials/lane/track, alternating lane order, `perf_counter_ns()` around repository segmenter only, NumPy linear p50/p95 | p95 ratio superiority | +| Peak memory | `isolated-single-shot-v1`: process-lifetime peak RSS, macOS `ru_maxrss`, Windows `PeakWorkingSetSize`, maximum across 20 trials | Report/capacity evidence | The selected mir_eval calls execute only under the content-addressed research overlay whose exact lock SHA-256 participates in the registration digest. The runtime verifier separately requires the installed distribution to be `mir_eval==0.8.2` before recognized segmentation metrics run. Full artifact provenance and adapter semantics are recorded in `docs/traceability/mir/structure-segmentation-metric-adapter.md`. @@ -62,7 +58,17 @@ The current official `ismir-mirex/mirex-evaluation` MIREX-2025 reproduction at c Result receipts carry precision and recall alongside F for boundary and repetition measures. Admission recomputes the harmonic mean and rejects inconsistent P/R/F triplets. That integrity check does not replace the recognized metric implementation. -Published MIREX scores are external context, not BandScope noninferiority margins. Acceptable loss and required latency improvement are product/scientific decisions that must be fixed before candidate measurement. +### Latency and peak RSS + +`isolated-single-shot-v1` is documented in `docs/traceability/mir/structure-performance-measurement-contract.md`. Every observation starts a fresh process to avoid inheriting feature-lane or allocator state from the other observation. The parent passes the already-admitted immutable PCM snapshot to the child before the timer starts. `perf_counter_ns()` measures only the repository-owned structure-segmentation call; process startup and input transport are harness overhead and are not claimed as algorithm latency. + +There are no same-process warm-up iterations. A cache-hot benchmark path is not assumed to represent the buyer's first analysis of a song. Twenty observations per lane are a preregistered engineering sampling plan rather than a claim of universal tail-latency precision. Trial order alternates CQT→STFT and STFT→CQT to reduce deterministic order bias; the exact host/runtime profile and claim boundary still own OS-wide cache, thermal, frequency, and scheduling limitations. + +Peak RSS deliberately covers the entire worker lifetime, including interpreter/runtime and PCM input resident before the timer. macOS uses `getrusage(RUSAGE_SELF).ru_maxrss` with the Apple-documented KiB conversion; Windows uses `GetProcessMemoryInfo(...).PeakWorkingSetSize` in bytes. The contract fails closed on unsupported platforms instead of pretending those APIs have cross-platform-equivalent units. + +The canonical admitted-track consumer binds the quality and performance producers to the same PCM memoryview identity, sample rate, and exact `decoded_frames / sample_rate_hz` duration. Synthetic or monkeypatched fixtures exercise machinery only and do not count as scientific acceptance. + +Published MIREX scores are external context, not BandScope noninferiority margins. Acceptable quality loss and required latency improvement remain product/scientific decisions that must be fixed before candidate measurement. ## Aggregation and paired uncertainty @@ -79,13 +85,11 @@ The aggregate is deliberately macro-oriented rather than duration weighted: For uncertainty, each bootstrap replicate samples `registered_track_count` paired track indices with replacement. The implementation uses `numpy.random.Generator(PCG64)` with the preregistered seed. The 95% interval is the 0.025 and 0.975 empirical quantile of the bootstrap statistic using NumPy's `linear` quantile method. Quality intervals are formed for candidate-minus-baseline macro statistics; the latency interval is formed for the candidate/baseline macro p95 ratio. The registration accepts 1,000 through 100,000 resamples for this procedure, with the exact count and seed included in the registration digest. -This is a percentile-bootstrap interval, not a symmetric interval around the observed statistic. Its endpoints are quantiles of the bootstrap distribution and can be asymmetric; the observed aggregate statistic is therefore **not required** to lie between those endpoints. Hall (1988) treats the percentile family as a bootstrap-quantile construction and documents that different bootstrap interval methods have materially different coverage and positioning properties. Adding a separate point-containment requirement would define a different, undocumented interval rule. - -The historical base validator predates the frozen percentile procedure and still contains that obsolete containment invariant. The metric-aware façade handles only those two legacy containment errors by creating a private validation projection whose aggregate point values are moved onto feasible interval bounds. The stored receipt, track measurements, aggregate values, and interval bounds are not modified. The base validator then continues to enforce every structural/measurement invariant and applies the original interval bounds to the noninferiority and latency decisions. Any non-containment validation error is propagated unchanged. `test_percentile_projection_preserves_asymmetric_interval_bounds` is the executable regression for this compatibility boundary. +This is a percentile-bootstrap interval, not a symmetric interval around the observed statistic. Its endpoints are quantiles of the bootstrap distribution and can be asymmetric; the observed aggregate statistic is therefore not required to lie between those endpoints. Hall (1988) treats the percentile family as a bootstrap-quantile construction and documents that bootstrap interval methods can have materially different coverage and positioning properties. -Public result admission then performs a second, independent integrity gate on the **original stored receipt**: it recomputes `macro-track-v1` aggregate values and both `paired-track-bootstrap-v1` interval objects from the complete stored `result.tracks` array plus the preregistered uncertainty plan, and requires exact equality. The private legacy projection is never used for this comparison. Hand-authored favorable aggregates or CIs therefore cannot become admitted scientific evidence. The RED→GREEN lineage and rejected alternatives are recorded in `docs/traceability/mir/structure-result-receipt-binding.md`. +The historical base validator predates the frozen percentile procedure and still contains an obsolete point-containment invariant. The metric-aware façade handles only those containment errors with a private validation projection. The stored receipt, track measurements, aggregate values, and interval bounds are not modified. Public result admission independently recomputes `macro-track-v1` aggregate values and both `paired-track-bootstrap-v1` interval objects from the complete stored track array plus the preregistered uncertainty plan, and requires exact equality. The RED→GREEN lineage is recorded in `docs/traceability/mir/structure-result-receipt-binding.md`. -The procedure being executable and preregisterable does not by itself establish that a particular corpus size, resample count, seed, margin, or claim boundary is scientifically adequate. Those remain experiment-review decisions. A different bootstrap method such as BCa, studentized, cluster/bootstrap-at-recording-level, or a different aggregation weighting requires a new versioned contract before candidate results are inspected. +The procedure being executable and preregisterable does not by itself establish that a particular corpus size, resample count, seed, margin, or claim boundary is scientifically adequate. A different bootstrap method, dependence model, or aggregation weighting requires a new versioned contract before candidate results are inspected. ## Decision rule @@ -97,7 +101,7 @@ For p95 latency, let `r = candidate_p95 / baseline_p95`. The speed criterion pas `upper_CI95(r) <= maximum_candidate_ratio` -The decision is based on preregistered interval bounds, not on whether the interval contains the observed point estimate. A favorable point estimate cannot hide an interval that crosses the registered threshold, and an unfavorable interval cannot be repaired by choosing another seed, resample count, aggregation rule, or claim boundary after results are known. +The decision is based on preregistered interval bounds, not on whether the interval contains the observed point estimate. A favorable point estimate cannot hide an interval that crosses the registered threshold, and an unfavorable interval cannot be repaired by choosing another seed, resample count, aggregation rule, timing method, or claim boundary after results are known. Because schema v1 has no exclusion policy, any failed track fails before aggregate decision rules are evaluated. Failed runs preserve ordered diagnostic track identity but carry null aggregate/CI summaries. @@ -107,11 +111,11 @@ Synthetic unit-test values are policy fixtures only. They are not production mar A result receipt contains the exact metric-aware registration digest, exact uncertainty plan, exact corpus order, one ordered track receipt per registered track, failed-track IDs, and the exact preregistered claim boundary. A complete successful run additionally carries aggregate measurements, paired 95% intervals for each gated quality metric, and a paired p95-latency-ratio interval. -Each successful per-track and aggregate baseline/candidate measurement is a closed-world receipt containing exactly the registered quality fields plus reporting precision/recall, boundary-deviation, latency, and peak-RSS fields. A declared failed track contains only `track_id`; attaching partial or invented measurements to a failed receipt is rejected. +Each successful per-track and aggregate baseline/candidate measurement is a closed-world receipt containing exactly the registered quality fields plus reporting precision/recall, boundary deviation, p50/p95 latency, and peak RSS. Those performance fields are valid scientific evidence only when produced by the canonical admitted-track path bound to `isolated-single-shot-v1`; the schema's numeric validation alone is not treated as provenance. The receipt is rejected when a track is omitted or reordered; missing measurements are not paired with an explicit failed-track declaration; a failed receipt carries measurements; a failed run carries non-null aggregate/CI summaries; the uncertainty plan or claim boundary differs; required metrics are absent; unregistered fields appear; a P/R/F triplet is inconsistent; values are non-finite; interval bounds are malformed; the aggregate or CI values differ from canonical recomputation; or the metric-aware registration digest differs. Percentile intervals are not rejected merely because an observed aggregate point lies outside them. -The admission validator does not trust caller-authored aggregate/CI values as scientific authority. It invokes the repository-owned paired aggregator over the complete stored track set and preregistered uncertainty plan, then exact-matches all stored aggregate and interval values against that deterministic output. Review still has to establish the upstream scientific validity of the per-track measurements, corpus, timing/memory measurement contract, thresholds, and claim boundary; those questions are not proved by recomputing a receipt. +The admission validator does not trust caller-authored aggregate/CI values as scientific authority. It invokes the repository-owned paired aggregator over the complete stored track set and preregistered uncertainty plan, then exact-matches all stored aggregate and interval values against that deterministic output. Review still has to establish the upstream scientific validity of the per-track measurements, corpus, numeric thresholds, host/runtime profile, and claim boundary; those questions are not proved by recomputing a receipt. ## Machine-readable evidence admission @@ -121,22 +125,23 @@ Registration and result JSON are evidence, not trusted configuration. CLI admiss ## Reproducibility sequence -1. Review rights basis, corpus composition, distinct audio identities, feature hypothesis, metric contract, functional-ACC grid/mapping, host profile, numeric margins, `macro-track-v1` aggregation, `paired-track-bootstrap-v1` resample count/seed, and claim boundary before candidate execution. If clustering, repeated recordings, alternative label mapping, or any exclusion tolerance is scientifically required, version that policy before measurement. Freeze the canonical latency/peak-RSS measurement contract before latency evidence is admitted. -2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain the metric-aware SHA-256. The digest includes validated registration data plus repository-owned metric and aggregation/uncertainty contracts. -3. Admit the rights-cleared corpus once, then run CQT and STFT lanes on the same decoded PCM/reference segmentation. Functional ACC must preserve the pinned 200 ms MIREX grid semantics; boundary/deviation/repetition metrics must use the pinned research runtime and explicit adapter arguments. +1. Review rights basis, corpus composition, distinct audio identities, feature hypothesis, metric contract, functional-ACC grid/mapping, `isolated-single-shot-v1`, host profile, numeric margins, `macro-track-v1` aggregation, `paired-track-bootstrap-v1` resample count/seed, and claim boundary before candidate execution. If clustering, repeated recordings, alternative label mapping, a different performance method, or any exclusion tolerance is scientifically required, version that policy before measurement. +2. Serialize the registration and run `python scripts/research/validate_structure_noninferiority.py ` to obtain the metric-aware SHA-256. The digest includes validated registration data plus repository-owned quality-metric, performance-measurement, and aggregation/uncertainty contracts. +3. Admit the rights-cleared corpus once, then run CQT and STFT quality lanes on the same decoded PCM/reference segmentation and run `isolated-single-shot-v1` on that same admitted PCM identity. Functional ACC preserves the pinned 200 ms MIREX grid; boundary/deviation/repetition use the pinned research runtime; latency/RSS use the pinned subprocess/timer/native-memory contract. 4. Feed the complete ordered track measurements to `aggregate_structure_noninferiority.py`. Do not hand-author aggregate values or confidence intervals. If any registered track fails, preserve the failure receipt and emit null aggregate/CI summaries instead of a complete-case result. -5. Put the metric-aware registration digest, identical uncertainty fields, corpus order, and claim boundary in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. Admission recomputes the aggregate and both interval objects from the stored tracks and rejects any mismatch before returning a scientific decision. -6. Preserve registration, metric-aware digest contract, result receipt, corpus/annotation hashes, exact source/lock identities, aggregation implementation identity, uncertainty plan, and claim boundary together. A production feature switch requires this evidence plus normal independent review and protected-head checks. +5. Put the metric-aware registration digest, identical uncertainty fields, corpus order, and claim boundary in the result receipt, then run `python scripts/research/validate_structure_noninferiority.py `. Admission recomputes aggregate and both interval objects from the stored tracks and rejects any mismatch before returning a decision. +6. Preserve registration, metric-aware digest contract, result receipt, corpus/annotation hashes, exact source/lock identities, performance-contract identity, aggregation implementation identity, uncertainty plan, host profile, and claim boundary together. A production feature switch requires this evidence plus normal independent review and protected-head checks. No step authorizes committing licensed audio to Git. Evaluation rights and redistribution rights remain separate. ## Security Notes -- Audio and annotation are untrusted inputs to corpus/metric runners. Result admission reads bounded JSON only and does not execute subprocesses, make network requests, or follow corpus paths. +- Audio and annotation are untrusted inputs. Corpus admission owns file/decode boundaries; result admission reads bounded JSON only. - Evidence JSON is limited to 2 MiB, must be a regular file read through one open descriptor, must decode as UTF-8, and rejects duplicate keys plus non-standard `NaN`/`Infinity` constants. - `source_uri` is evidence metadata, not a fetch instruction; local filesystem forms are rejected. - Audio/annotation SHA-256 values bind measurements to bytes without embedding media; duplicate audio content identity under different track IDs is rejected. -- Invalid/non-finite measurements, corpus drift, metric/runtime drift, aggregation/uncertainty drift, aggregate/CI recomputation drift, claim-boundary drift, registration drift, undeclared missing measurements, failed-receipt fabrication, non-null summaries after a failed track, unregistered fields, and inconsistent P/R/F triplets fail closed. +- The performance worker receives only admitted PCM through stdin and closed-world numeric/feature arguments. It uses `sys.executable`, the repository-owned worker path, an argument array, and `shell=False`; it adds no generic exec, URL, network, or plugin surface. +- Worker non-zero exit, stderr, malformed receipt, unsupported platform, invalid timer/RSS, metric/runtime drift, aggregation/uncertainty drift, aggregate/CI recomputation drift, claim-boundary drift, registration drift, failed-receipt fabrication, and inconsistent P/R/F triplets fail closed. - The percentile compatibility projection exists only inside result validation and never rewrites the stored scientific receipt or CI bounds; canonical-summary binding compares the original stored receipt. - SHA-256 does not prove licensing, annotation validity, corpus adequacy, independence beyond exact-byte uniqueness, measurement-method adequacy, or statistical appropriateness. Rights and scientific review remain separate gates. @@ -160,4 +165,12 @@ MIREX Evaluation contributors. (2026). *music_structure_analysis/eval_script.py* mir_eval contributors. (n.d.). *mir_eval.segment: Structural segmentation evaluation*. https://github.com/mir-evaluation/mir_eval/blob/main/mir_eval/segment.py -Wang, J.-C., Hung, Y.-N., & Smith, J. B. L. (2022). To catch a chorus, verse, intro, or anything else: Analyzing a song with structural functions. In *ICASSP 2022—2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)* (pp. 416–420). IEEE. https://arxiv.org/abs/2205.14700 \ No newline at end of file +NumPy Developers. (2026). *numpy.quantile*. https://numpy.org/doc/stable/reference/generated/numpy.quantile.html + +Python Software Foundation. (2026). *time — Time access and conversions*. Python 3.14 documentation. https://docs.python.org/3/library/time.html#time.perf_counter + +Apple Inc. (2004). *getrusage(2) — get information about resource utilization*. Mac OS X manual pages. https://developer.apple.com/library/archive/documentation/System/Conceptual/ManPages_iPhoneOS/man2/getrusage.2.html + +Microsoft. (2024). *Process memory usage information*. Windows App Development documentation. https://learn.microsoft.com/windows/win32/psapi/process-memory-usage-information + +Wang, J.-C., Hung, Y.-N., & Smith, J. B. L. (2022). To catch a chorus, verse, intro, or anything else: Analyzing a song with structural functions. In *ICASSP 2022—2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)* (pp. 416–420). IEEE. https://arxiv.org/abs/2205.14700 From 00d4efaa8353287a41817bab408a7a9149c13d6f Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:18:02 +0900 Subject: [PATCH 190/216] test(mir): require canonical track-result receipt producer --- .../test_structure_track_result_receipt.py | 122 ++++++++++++++++++ 1 file changed, 122 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_track_result_receipt.py diff --git a/services/analysis-engine/tests/test_structure_track_result_receipt.py b/services/analysis-engine/tests/test_structure_track_result_receipt.py new file mode 100644 index 000000000..d49ae0138 --- /dev/null +++ b/services/analysis-engine/tests/test_structure_track_result_receipt.py @@ -0,0 +1,122 @@ +"""Contracts for canonical track receipts from admitted Signal-MIR evidence.""" + +from __future__ import annotations + +from types import ModuleType, SimpleNamespace + +import pytest +from conftest import load_module + + +def _consumer_module() -> ModuleType: + """Load the repository-owned admitted-track evidence owner.""" + return load_module( + "scripts/research/evaluate_admitted_structure_track.py", + "evaluate_admitted_structure_track_result_receipt", + ) + + +def _metrics() -> SimpleNamespace: + """Return one complete normalized segmentation-metric fixture.""" + return SimpleNamespace( + boundary_precision_0_5=0.91, + boundary_recall_0_5=0.81, + boundary_f_0_5=0.857091, + boundary_precision_3_0=0.96, + boundary_recall_3_0=0.86, + boundary_f_3_0=0.907253, + reference_to_estimate_median_deviation_seconds=0.12, + estimate_to_reference_median_deviation_seconds=0.15, + repetition_pairwise_precision=0.88, + repetition_pairwise_recall=0.78, + repetition_pairwise_f=0.826988, + ) + + +def _performance() -> SimpleNamespace: + """Return one complete preregistered performance fixture.""" + return SimpleNamespace( + p50_latency_seconds=0.40, + p95_latency_seconds=0.55, + peak_rss_mib=480.0, + measured_trials=20, + ) + + +def test_canonical_track_receipt_maps_only_registered_measurement_fields() -> None: + """The experiment path must not hand-author schema-v1 track measurements.""" + module = _consumer_module() + evidence = module.PairedFunctionalAccuracyEvidence( + track_id="track-01", + decoded_pcm_sha256="a" * 64, + annotation_sha256="b" * 64, + decoded_frames=44100, + sample_rate_hz=44100, + baseline_accuracy=0.93, + baseline_correct_frames=93, + baseline_total_frames=100, + candidate_accuracy=0.92, + candidate_correct_frames=92, + candidate_total_frames=100, + metric_runtime_lock_sha256="c" * 64, + baseline_segmentation_metrics=module._segmentation_metric_evidence(_metrics()), + candidate_segmentation_metrics=module._segmentation_metric_evidence(_metrics()), + performance_contract_id="isolated-single-shot-v1", + baseline_performance=module._performance_evidence(_performance()), + candidate_performance=module._performance_evidence(_performance()), + ) + + receipt = module.noninferiority_track_receipt(evidence) + + assert set(receipt) == {"track_id", "baseline", "candidate"} + assert receipt["track_id"] == "track-01" + for side_name, expected_accuracy in (("baseline", 0.93), ("candidate", 0.92)): + side = receipt[side_name] + assert set(side) == { + "boundary_f_0_5", + "boundary_f_3_0", + "functional_label_accuracy", + "repetition_pairwise_f", + "boundary_precision_0_5", + "boundary_recall_0_5", + "boundary_precision_3_0", + "boundary_recall_3_0", + "repetition_pairwise_precision", + "repetition_pairwise_recall", + "reference_to_estimate_median_deviation_seconds", + "estimate_to_reference_median_deviation_seconds", + "p50_latency_seconds", + "p95_latency_seconds", + "peak_rss_mib", + } + assert side["functional_label_accuracy"] == pytest.approx(expected_accuracy) + assert side["p50_latency_seconds"] == pytest.approx(0.40) + assert side["p95_latency_seconds"] == pytest.approx(0.55) + assert side["peak_rss_mib"] == pytest.approx(480.0) + + +def test_track_receipt_rejects_partial_or_unbound_evidence() -> None: + """Functional-only or contract-drifted evidence cannot become a result track.""" + module = _consumer_module() + partial = module.PairedFunctionalAccuracyEvidence( + track_id="track-01", + decoded_pcm_sha256="a" * 64, + annotation_sha256="b" * 64, + decoded_frames=44100, + sample_rate_hz=44100, + baseline_accuracy=0.93, + baseline_correct_frames=93, + baseline_total_frames=100, + candidate_accuracy=0.92, + candidate_correct_frames=92, + candidate_total_frames=100, + metric_runtime_lock_sha256=None, + baseline_segmentation_metrics=None, + candidate_segmentation_metrics=None, + performance_contract_id=None, + baseline_performance=None, + candidate_performance=None, + ) + + with pytest.raises(RuntimeError, match="complete canonical quality and performance evidence"): + module.noninferiority_track_receipt(partial) From 334a78cf98a34a15596da0101bb9043640e61686 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:18:47 +0900 Subject: [PATCH 191/216] feat(mir): build result tracks from canonical admitted evidence --- .../evaluate_admitted_structure_track.py | 77 +++++++++++++++++++ 1 file changed, 77 insertions(+) diff --git a/scripts/research/evaluate_admitted_structure_track.py b/scripts/research/evaluate_admitted_structure_track.py index dad68c8b8..941c31434 100644 --- a/scripts/research/evaluate_admitted_structure_track.py +++ b/scripts/research/evaluate_admitted_structure_track.py @@ -209,6 +209,78 @@ def _performance_evidence(result: object) -> StructurePerformanceEvidence: ) +def _result_side( + accuracy: float, + segmentation: StructureSegmentationMetricEvidence, + performance: StructurePerformanceEvidence, +) -> dict[str, float]: + """Map one canonical lane evidence object to schema-v1 measurement fields.""" + return { + "boundary_f_0_5": segmentation.boundary_f_0_5, + "boundary_f_3_0": segmentation.boundary_f_3_0, + "functional_label_accuracy": accuracy, + "repetition_pairwise_f": segmentation.repetition_pairwise_f, + "boundary_precision_0_5": segmentation.boundary_precision_0_5, + "boundary_recall_0_5": segmentation.boundary_recall_0_5, + "boundary_precision_3_0": segmentation.boundary_precision_3_0, + "boundary_recall_3_0": segmentation.boundary_recall_3_0, + "repetition_pairwise_precision": segmentation.repetition_pairwise_precision, + "repetition_pairwise_recall": segmentation.repetition_pairwise_recall, + "reference_to_estimate_median_deviation_seconds": ( + segmentation.reference_to_estimate_median_deviation_seconds + ), + "estimate_to_reference_median_deviation_seconds": ( + segmentation.estimate_to_reference_median_deviation_seconds + ), + "p50_latency_seconds": performance.p50_latency_seconds, + "p95_latency_seconds": performance.p95_latency_seconds, + "peak_rss_mib": performance.peak_rss_mib, + } + + +def noninferiority_track_receipt( + evidence: PairedFunctionalAccuracyEvidence, +) -> dict[str, object]: + """Build one schema-v1 result track only from complete canonical evidence. + + The base result schema intentionally contains numeric measurements rather than + execution machinery. This adapter prevents the canonical experiment path from + independently retyping those numbers after admission: functional ACC, + recognized segmentation metrics, and the preregistered performance summary + all come directly from the immutable evidence object produced for one track. + """ + expected_runtime_lock = _RUNTIME_VERIFIER.load_structure_metric_runtime_lock_identity( + _STRUCTURE_METRIC_RUNTIME_LOCK + ).lock_sha256 + expected_performance_contract = _RESOURCE_MEASUREMENT.PERFORMANCE_MEASUREMENT_CONTRACT[ + "contract_id" + ] + if ( + evidence.metric_runtime_lock_sha256 != expected_runtime_lock + or evidence.baseline_segmentation_metrics is None + or evidence.candidate_segmentation_metrics is None + or evidence.performance_contract_id != expected_performance_contract + or evidence.baseline_performance is None + or evidence.candidate_performance is None + ): + raise RuntimeError( + "result track requires complete canonical quality and performance evidence" + ) + return { + "track_id": evidence.track_id, + "baseline": _result_side( + evidence.baseline_accuracy, + evidence.baseline_segmentation_metrics, + evidence.baseline_performance, + ), + "candidate": _result_side( + evidence.candidate_accuracy, + evidence.candidate_segmentation_metrics, + evidence.candidate_performance, + ), + } + + class PairedFunctionalAccuracyTrackConsumer: """Collect paired structure evidence from corpus-admission callbacks.""" @@ -281,6 +353,11 @@ def evidence(self) -> tuple[PairedFunctionalAccuracyEvidence, ...]: """Return immutable ordered evidence accumulated from admitted tracks.""" return tuple(self._evidence) + @property + def result_track_receipts(self) -> tuple[dict[str, object], ...]: + """Return schema-v1 track receipts only when every evidence object is complete.""" + return tuple(noninferiority_track_receipt(item) for item in self._evidence) + def __call__( self, track_id: str, From 5b380a8a4c0d82b2e20f8e69b62e79ffb54e5b9d Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:19:11 +0900 Subject: [PATCH 192/216] test(mir): bind result fixture to exact metric runtime --- .../tests/test_structure_track_result_receipt.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/services/analysis-engine/tests/test_structure_track_result_receipt.py b/services/analysis-engine/tests/test_structure_track_result_receipt.py index d49ae0138..42ee655f5 100644 --- a/services/analysis-engine/tests/test_structure_track_result_receipt.py +++ b/services/analysis-engine/tests/test_structure_track_result_receipt.py @@ -46,6 +46,9 @@ def _performance() -> SimpleNamespace: def test_canonical_track_receipt_maps_only_registered_measurement_fields() -> None: """The experiment path must not hand-author schema-v1 track measurements.""" module = _consumer_module() + runtime_identity = module._RUNTIME_VERIFIER.load_structure_metric_runtime_lock_identity( + module._STRUCTURE_METRIC_RUNTIME_LOCK + ) evidence = module.PairedFunctionalAccuracyEvidence( track_id="track-01", decoded_pcm_sha256="a" * 64, @@ -58,7 +61,7 @@ def test_canonical_track_receipt_maps_only_registered_measurement_fields() -> No candidate_accuracy=0.92, candidate_correct_frames=92, candidate_total_frames=100, - metric_runtime_lock_sha256="c" * 64, + metric_runtime_lock_sha256=runtime_identity.lock_sha256, baseline_segmentation_metrics=module._segmentation_metric_evidence(_metrics()), candidate_segmentation_metrics=module._segmentation_metric_evidence(_metrics()), performance_contract_id="isolated-single-shot-v1", From 7d3529b0032edf3967559c61969bd80626aa535f Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:21:00 +0900 Subject: [PATCH 193/216] docs(mir): bind canonical track receipt provenance --- .../mir/structure-result-receipt-binding.md | 54 +++++++++++++++---- 1 file changed, 44 insertions(+), 10 deletions(-) diff --git a/docs/traceability/mir/structure-result-receipt-binding.md b/docs/traceability/mir/structure-result-receipt-binding.md index 2b9b78791..4a4c6ab91 100644 --- a/docs/traceability/mir/structure-result-receipt-binding.md +++ b/docs/traceability/mir/structure-result-receipt-binding.md @@ -7,13 +7,31 @@ Parent: `docs/traceability/mir/structure-feature-noninferiority.md` ## Problem -The preregistered result schema required complete ordered per-track receipts and separately required aggregate measurements plus paired confidence intervals. Before this repair, result admission validated both parts but did not prove that the stored aggregate and confidence intervals were produced from those stored track receipts by the repository-owned `macro-track-v1` / `paired-track-bootstrap-v1` implementation. +The preregistered result schema requires one ordered measurement receipt per registered track and separately requires aggregate measurements plus paired confidence intervals. Two integrity boundaries matter: -That gap was scientifically material. A syntactically valid receipt could retain the registered corpus order and valid per-track measurements while carrying a favorable hand-authored aggregate or percentile interval. The base decision policy would then evaluate those stored summaries even though they were not causally bound to the track-level evidence. +1. the per-track JSON must come from the exact admitted quality/performance evidence rather than being retyped by an experiment caller; and +2. aggregate and confidence-interval values must come from those stored track receipts through the repository-owned `macro-track-v1` / `paired-track-bootstrap-v1` implementation. + +Before the aggregate repair, admission validated track receipts and summary fields independently. A syntactically valid receipt could therefore retain registered track evidence while carrying a favorable hand-authored aggregate or percentile interval. After the latency/RSS producer was added, a narrower upstream gap remained: the canonical admitted-track consumer held the complete functional-ACC, recognized segmentation metrics, and performance evidence in memory, but the repository did not own the conversion from that object to the closed schema-v1 `baseline` / `candidate` measurement objects. An experiment caller could still independently retype those numbers before aggregation. ## Decision -Successful result admission now deterministically recomputes all decision summaries from the complete stored `result.tracks` array and the preregistered uncertainty plan by calling `aggregate_structure_noninferiority.py`. +### Track-level receipt production + +`evaluate_admitted_structure_track.py::noninferiority_track_receipt()` is the canonical adapter from admitted Signal-MIR evidence to one schema-v1 result track. It accepts only one `PairedFunctionalAccuracyEvidence` and requires all of the following before it emits a receipt: + +- the exact pinned structure-metric runtime-lock SHA-256; +- baseline and candidate recognized segmentation metric evidence; +- the exact registered performance contract identity (`isolated-single-shot-v1`); +- baseline and candidate performance summaries produced under that contract. + +The adapter then maps the immutable evidence directly into the exact schema-v1 measurement fields: functional ACC, boundary and repetition P/R/F, bidirectional boundary deviation, p50/p95 latency, and peak RSS. It does not accept caller-supplied replacement values, extra result fields, paths, or external producer metadata. `PairedFunctionalAccuracyTrackConsumer.result_track_receipts` exposes the same adapter across the ordered admitted evidence set. + +This is a producer-integrity boundary, not a cryptographic attestation mechanism. An arbitrary external JSON file can still be authored outside this Python path; the public result validator cannot prove process history from numbers alone. Production evidence therefore preserves the corpus-admission receipt, canonical track-evidence execution, result receipt, source/runtime identity, and protected-head workflow evidence together. If independently verifiable track-producer attestation becomes a release requirement, it needs a new versioned evidence-envelope contract rather than an assertion that schema validation proves provenance. + +### Aggregate and interval recomputation + +Successful result admission deterministically recomputes all decision summaries from the complete stored `result.tracks` array and the preregistered uncertainty plan by calling `aggregate_structure_noninferiority.py`. Admission requires exact equality for: @@ -23,10 +41,12 @@ Admission requires exact equality for: The comparison occurs after the historical base validator has checked the result envelope, track order, measurement fields, finite values, P/R/F identities, failed-track policy, uncertainty identity, claim boundary, and decision-rule syntax. A failed run is not re-aggregated: schema v1 already requires null aggregate/CI summaries when `failed_tracks` is non-empty. -The percentile compatibility projection remains private to the historical base validator boundary. It exists only because the base validator predates percentile-bootstrap semantics and incorrectly requires every interval to contain the observed statistic. Canonical receipt binding compares the original stored aggregate and interval values, not the private projection. Therefore the projection cannot manufacture evidence that passes the new recomputation gate. +The percentile compatibility projection remains private to the historical base validator boundary. It exists only because the base validator predates percentile-bootstrap semantics and incorrectly requires every interval to contain the observed statistic. Canonical receipt binding compares the original stored aggregate and interval values, not the private projection. Therefore the projection cannot manufacture evidence that passes the recomputation gate. ## RED → GREEN lineage +### Summary provenance + RED `aacb026ed2e05dadce3fb0b7560836308bc73b92` added two executable regressions: - change one valid per-track functional-accuracy value while leaving the favorable stored aggregate unchanged; @@ -36,11 +56,21 @@ Both receipts were admitted by the predecessor validator because aggregate/CI pr GREEN `5f47924d632946540d0c31e0c6456feac4667f8b` first bound the stored aggregate to canonical track reduction. Fixture alignment `4cdcf19a016c9b4d7a1c28ccc39784513700b297` changed policy fixtures to generate canonical summaries instead of hand-authored intervals. GREEN `da51a8aa5cddf9300bcbbe611d3929f4132da5a2` then bound both paired quality intervals and the p95-latency-ratio interval to deterministic bootstrap recomputation. -`5f698c4c98da83d6ca7860c6b32aee2980c5abb0` separated the legacy percentile point-containment compatibility regression from scientific result admission: the compatibility helper can still prove that interval endpoints are not rewritten, while the public validator accepts only canonical intervals. `59d9c309015af6cdf9badc7cd50cf2d8bf5813e0` rewrote threshold regressions so failing noninferiority and latency decisions are derived from changed track evidence followed by canonical recomputation rather than from hand-edited confidence intervals. +`5f698c4c98da83d6ca7860c6b32aee2980c5abb0` separated the legacy percentile point-containment compatibility regression from scientific result admission. `59d9c309015af6cdf9badc7cd50cf2d8bf5813e0` rewrote threshold regressions so failing decisions are derived from changed track evidence followed by canonical recomputation rather than hand-edited confidence intervals. + +### Track producer provenance + +RED `00d4efaa8353287a41817bab408a7a9149c13d6f` required a canonical adapter that can emit the exact schema-v1 track field set only from complete admitted Signal-MIR evidence and rejects partial/unbound evidence. + +GREEN `334a78cf98a34a15596da0101bb9043640e61686` added `noninferiority_track_receipt()` and `result_track_receipts`. The adapter requires the pinned metric-runtime identity plus the registered performance contract and maps quality/performance evidence without a second caller-owned measurement implementation. + +Fixture correction `5b380a8a4c0d82b2e20f8e69b62e79ffb54e5b9d` binds the regression fixture to the actual repository metric-runtime lock rather than a placeholder digest. ## Constraints and rejected alternatives -Trusting a comment, workflow name, or producer claim that a JSON file came from the canonical aggregator was rejected. Those signals do not establish a causal relationship between stored track evidence and stored decision summaries. +Trusting a comment, workflow name, producer claim, or arbitrary `producer="canonical"` JSON field was rejected. Those signals do not establish a causal relationship between admitted evidence and stored measurements or between stored tracks and decision summaries. + +Letting the experiment caller build the track `baseline` / `candidate` dictionaries manually was rejected for the canonical path. It would recreate a measurement translation layer immediately after the repository had already produced immutable admitted evidence and would make favorable transcription errors indistinguishable from genuine output. Checking only the aggregate was rejected as incomplete. The final production decision is made from percentile interval bounds, so favorable hand-authored confidence intervals would remain an acceptance bypass even if the aggregate itself were canonical. @@ -50,12 +80,16 @@ Approximate receipt comparison was rejected. The experiment registration pins so ## Security and integrity notes -- The recomputation path consumes only already-validated in-memory result data and preregistered uncertainty fields. It opens no corpus path, performs no network request, and executes no user-supplied plugin. +- Track receipt conversion consumes only immutable in-memory admitted evidence and opens no corpus path, performs no network request, and executes no plugin. +- The canonical adapter fails closed when metric-runtime identity, recognized metric evidence, performance contract identity, or performance evidence is absent or drifted. +- Aggregate recomputation consumes only already-validated result data and preregistered uncertainty fields. It opens no corpus path, performs no network request, and executes no user-supplied plugin. - Track omission, reordering, duplication, malformed metrics, non-finite values, failed-track fabrication, aggregate drift, and CI drift all fail closed before a passing decision is returned. -- Exact deterministic recomputation protects evidence integrity; it does not establish corpus representativeness, licensing, annotation validity, statistical power, or suitable production margins. Those remain independent scientific/product gates. +- Exact producer mapping and deterministic recomputation protect evidence integrity; they do not establish corpus representativeness, licensing, annotation validity, statistical power, suitable production margins, or the historical process provenance of arbitrary external JSON. ## Remaining scientific boundary -This repair closes the result-summary provenance gap. It does not authorize a real-audio run. Before candidate results are inspected, reviewers still have to freeze the concrete rights-cleared corpus, numeric noninferiority margins, latency ratio threshold, exact bootstrap count/seed, host/runtime profile, and claim boundary. +The canonical path now owns admitted quality evidence, the `isolated-single-shot-v1` p50/p95/peak-RSS producer, schema-v1 track-receipt construction, macro aggregation, and paired percentile-bootstrap recomputation. + +This still does not authorize a real-audio run. Before candidate results are inspected, reviewers must freeze the concrete rights-cleared corpus, numeric noninferiority margins, maximum candidate latency ratio, exact bootstrap count/seed, exact macOS/Windows host/runtime profile, and claim boundary, and judge whether the corpus size and inference plan are scientifically adequate for that claim. -A separate unresolved measurement gap remains for timing and memory: the admitted-track MIR consumer already produces structure-quality metrics, but the canonical path from that same admitted execution to per-track p50/p95 latency and peak-RSS receipts still requires explicit trial/warm-up/timer/memory semantics before those fields can be treated as production scientific evidence. +A passing registered experiment would establish evidence only within that frozen population/runtime claim. Production `chroma_cqt` → `chroma_stft` default selection remains a separate reviewed change with current protected-head checks and release evidence. From 7a90b922c0f64090f1ac2717e9afc0369503bd58 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:32:42 +0900 Subject: [PATCH 194/216] test(mir): require bound execution host identity --- ...structure_experiment_execution_contract.py | 90 +++++++++++++++++++ 1 file changed, 90 insertions(+) create mode 100644 services/analysis-engine/tests/test_structure_experiment_execution_contract.py diff --git a/services/analysis-engine/tests/test_structure_experiment_execution_contract.py b/services/analysis-engine/tests/test_structure_experiment_execution_contract.py new file mode 100644 index 000000000..81789f807 --- /dev/null +++ b/services/analysis-engine/tests/test_structure_experiment_execution_contract.py @@ -0,0 +1,90 @@ +"""Regression tests for canonical structure-experiment execution admission.""" + +from __future__ import annotations + +from types import ModuleType + +import pytest +from conftest import load_module +from test_structure_noninferiority_policy import _registration + + +def _runner() -> ModuleType: + """Load the repository-owned end-to-end scientific execution boundary.""" + return load_module( + "scripts/research/run_structure_noninferiority_experiment.py", + "run_structure_noninferiority_experiment_contract", + ) + + +def _synthetic_host_profile(runner: ModuleType, *, cpu_model: str) -> object: + """Return deterministic unit-test host evidence without probing the CI machine.""" + return runner.StructureHostProfile( + contract_id=runner.HOST_PROFILE_CONTRACT_ID, + platform="darwin", + os_release="25.6.0", + os_version="Darwin Kernel Version 25.6.0", + machine="arm64", + hardware_model="Mac16,8", + cpu_model=cpu_model, + logical_cpu_count=12, + ) + + +def test_execution_rejects_registered_host_profile_drift_before_corpus_admission() -> None: + """Latency evidence cannot claim a preregistered host while running elsewhere.""" + runner = _runner() + registration = _registration() + observed = _synthetic_host_profile(runner, cpu_model="Apple M4 Pro") + expected = _synthetic_host_profile(runner, cpu_model="Apple M3 Pro") + runtime = registration["runtime"] + assert isinstance(runtime, dict) + runtime["host_profile"] = runner.host_profile_identity(expected) + admitted = False + + def forbidden_admission(*args: object, **kwargs: object) -> object: + nonlocal admitted + admitted = True + raise AssertionError("corpus admission must not run after host-profile drift") + + with pytest.raises(ValueError, match="runtime.host_profile"): + runner.execute_registered_experiment( + registration, + {"schema_version": 1, "registration_sha256": "unused", "tracks": []}, + runtime_identity={ + "source_commit": "e" * 40, + "uv_lock_sha256": "f" * 64, + "python_version": "3.12.11", + "librosa_version": "0.11.0", + "numpy_version": "2.3.3", + }, + host_profile=observed, + corpus_admitter=forbidden_admission, + ) + + assert admitted is False + + +def test_host_profile_identity_is_content_addressed_and_excludes_machine_name() -> None: + """Scientific host identity uses hardware/OS facts, not hostname or user identity.""" + runner = _runner() + left = _synthetic_host_profile(runner, cpu_model="Apple M4 Pro") + right = _synthetic_host_profile(runner, cpu_model="Apple M4 Pro") + changed = _synthetic_host_profile(runner, cpu_model="Apple M4 Max") + + assert runner.host_profile_identity(left) == runner.host_profile_identity(right) + assert runner.host_profile_identity(left) != runner.host_profile_identity(changed) + payload = runner.host_profile_payload(left) + assert set(payload) == { + "contract_id", + "platform", + "os_release", + "os_version", + "machine", + "hardware_model", + "cpu_model", + "logical_cpu_count", + } + assert "hostname" not in payload + assert "node" not in payload + assert "user" not in payload From 70b86022323728c6b3f85d5f14468423ebf15478 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:33:42 +0900 Subject: [PATCH 195/216] feat(mir): bind experiment execution to exact host profile --- ...run_structure_noninferiority_experiment.py | 382 ++++++++++++++++++ 1 file changed, 382 insertions(+) create mode 100644 scripts/research/run_structure_noninferiority_experiment.py diff --git a/scripts/research/run_structure_noninferiority_experiment.py b/scripts/research/run_structure_noninferiority_experiment.py new file mode 100644 index 000000000..252dae613 --- /dev/null +++ b/scripts/research/run_structure_noninferiority_experiment.py @@ -0,0 +1,382 @@ +#!/usr/bin/env python3 +"""Execute the frozen structure noninferiority experiment as one evidence chain. + +This is the canonical orchestration boundary between corpus admission and result +admission. It refuses to run scientific measurements unless the preregistered +performance host identity matches the current host profile, then drives the +existing corpus admission, CQT/STFT admitted-track consumer, canonical paired +aggregation, and result validator without reopening corpus paths after admission. + +The host profile intentionally excludes hostname, account identity, hardware +serial numbers, MAC addresses, and other machine identifiers that are not needed +to reproduce a performance claim. It binds only the supported OS/runtime class, +architecture, hardware/CPU model, and logical CPU count that can materially +change the CQT/STFT latency comparison. +""" + +from __future__ import annotations + +import argparse +import hashlib +import importlib.util +import json +import os +import platform +import re +import subprocess +import sys +from collections.abc import Callable, Mapping, Sequence +from dataclasses import asdict, dataclass +from pathlib import Path +from types import ModuleType +from typing import Any + +HOST_PROFILE_CONTRACT_ID = "structure-performance-host-v1" +_HOST_PROFILE_RE = re.compile( + rf"^{re.escape(HOST_PROFILE_CONTRACT_ID)}:[0-9a-f]{{64}}$" +) +_REPOSITORY_ROOT = Path(__file__).resolve().parents[2] + + +@dataclass(frozen=True, slots=True) +class StructureHostProfile: + """Purpose-bound host facts that can affect scientific performance evidence.""" + + contract_id: str + platform: str + os_release: str + os_version: str + machine: str + hardware_model: str + cpu_model: str + logical_cpu_count: int + + +def _load_sibling(filename: str, module_name: str) -> ModuleType: + """Load one repository-owned sibling script under a stable private module name.""" + existing = sys.modules.get(module_name) + if existing is not None: + return existing + path = Path(__file__).with_name(filename) + spec = importlib.util.spec_from_file_location(module_name, path) + if spec is None or spec.loader is None: + raise RuntimeError(f"could not load research module: {filename}") + module = importlib.util.module_from_spec(spec) + sys.modules[module_name] = module + try: + spec.loader.exec_module(module) + except Exception: + sys.modules.pop(module_name, None) + raise + return module + + +_VALIDATOR = _load_sibling( + "validate_structure_noninferiority.py", + "_bandscope_structure_execution_validator", +) +_ADMISSION = _load_sibling( + "verify_structure_corpus.py", + "_bandscope_structure_execution_admission", +) +_TRACKS = _load_sibling( + "evaluate_admitted_structure_track.py", + "_bandscope_structure_execution_tracks", +) +_AGGREGATION = _load_sibling( + "aggregate_structure_noninferiority.py", + "_bandscope_structure_execution_aggregation", +) + + +def _text(value: object, field: str) -> str: + """Return stripped non-empty text without silently accepting another type.""" + if not isinstance(value, str) or not value.strip(): + raise ValueError(f"{field} must be non-empty text") + return value.strip() + + +def _positive_cpu_count(value: object) -> int: + """Return a positive logical CPU count while rejecting booleans and coercion.""" + if isinstance(value, bool) or not isinstance(value, int) or value <= 0: + raise RuntimeError("logical CPU count must be a positive integer") + return value + + +def host_profile_payload(profile: StructureHostProfile) -> dict[str, object]: + """Return the exact canonical host-profile object used for content addressing.""" + if not isinstance(profile, StructureHostProfile): + raise TypeError("profile must be a StructureHostProfile") + payload = asdict(profile) + if payload["contract_id"] != HOST_PROFILE_CONTRACT_ID: + raise ValueError( + f"host profile contract_id must equal {HOST_PROFILE_CONTRACT_ID}" + ) + for field in ( + "platform", + "os_release", + "os_version", + "machine", + "hardware_model", + "cpu_model", + ): + payload[field] = _text(payload[field], f"host_profile.{field}") + payload["logical_cpu_count"] = _positive_cpu_count(payload["logical_cpu_count"]) + if payload["platform"] not in {"darwin", "win32"}: + raise ValueError("host_profile.platform must be darwin or win32") + return payload + + +def host_profile_identity(profile: StructureHostProfile) -> str: + """Return the content-addressed preregistration value for one host profile.""" + canonical = json.dumps( + host_profile_payload(profile), + allow_nan=False, + ensure_ascii=False, + separators=(",", ":"), + sort_keys=True, + ).encode("utf-8") + return f"{HOST_PROFILE_CONTRACT_ID}:{hashlib.sha256(canonical).hexdigest()}" + + +def _darwin_sysctl(selector: str) -> str: + """Read one exact macOS hardware selector without invoking a shell.""" + completed = subprocess.run( + ["/usr/sbin/sysctl", "-n", selector], + check=False, + capture_output=True, + text=True, + shell=False, + timeout=5, + ) + if completed.returncode != 0: + diagnostic = " ".join(completed.stderr.split()) + raise RuntimeError( + f"sysctl {selector} failed with exit code {completed.returncode}: " + f"{diagnostic[:1000]}" + ) + return _text(completed.stdout, f"sysctl.{selector}") + + +def _windows_registry_text(path: str, name: str) -> str: + """Read one machine-level Windows hardware string from the local registry.""" + import winreg + + try: + with winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE, path) as key: + value, value_type = winreg.QueryValueEx(key, name) + except OSError as exc: + raise RuntimeError(f"Windows hardware registry value unavailable: {name}") from exc + if value_type not in {winreg.REG_SZ, winreg.REG_EXPAND_SZ}: + raise RuntimeError(f"Windows hardware registry value is not text: {name}") + return _text(value, f"windows_registry.{name}") + + +def capture_host_profile() -> StructureHostProfile: + """Capture the supported host facts needed to bind a latency/RSS claim.""" + current_platform = sys.platform + if current_platform == "darwin": + hardware_model = _darwin_sysctl("hw.model") + cpu_model = _darwin_sysctl("machdep.cpu.brand_string") + elif current_platform == "win32": + hardware_model = _windows_registry_text( + r"HARDWARE\DESCRIPTION\System\BIOS", + "SystemProductName", + ) + cpu_model = _windows_registry_text( + r"HARDWARE\DESCRIPTION\System\CentralProcessor\0", + "ProcessorNameString", + ) + else: + raise RuntimeError( + "structure scientific performance evidence supports only macOS or Windows" + ) + + return StructureHostProfile( + contract_id=HOST_PROFILE_CONTRACT_ID, + platform=current_platform, + os_release=_text(platform.release(), "platform.release"), + os_version=_text(platform.version(), "platform.version"), + machine=_text(platform.machine(), "platform.machine"), + hardware_model=hardware_model, + cpu_model=cpu_model, + logical_cpu_count=_positive_cpu_count(os.cpu_count()), + ) + + +def _registered_host_profile(registration: Mapping[str, Any]) -> str: + """Return the one content-addressed host identity accepted for execution.""" + runtime = registration.get("runtime") + if not isinstance(runtime, Mapping): + raise ValueError("registration.runtime must be an object") + expected = _text(runtime.get("host_profile"), "registration.runtime.host_profile") + if _HOST_PROFILE_RE.fullmatch(expected) is None: + raise ValueError( + "registration.runtime.host_profile must be a " + f"{HOST_PROFILE_CONTRACT_ID}: identity" + ) + return expected + + +def _require_registered_host_profile( + registration: Mapping[str, Any], + observed: StructureHostProfile, +) -> str: + """Fail before corpus access when the executing host differs from preregistration.""" + expected = _registered_host_profile(registration) + actual = host_profile_identity(observed) + if actual != expected: + raise ValueError( + "registration.runtime.host_profile does not match the executing host: " + f"expected {expected}, got {actual}" + ) + return actual + + +def _bind_admission_to_evidence( + corpus_receipt: Mapping[str, Any], + evidence: Sequence[Any], +) -> None: + """Require consumer evidence identity to equal the admitted PCM/annotation receipt.""" + raw_tracks = corpus_receipt.get("tracks") + if isinstance(raw_tracks, (str, bytes)) or not isinstance(raw_tracks, Sequence): + raise RuntimeError("corpus receipt tracks must be an array") + if len(raw_tracks) != len(evidence): + raise RuntimeError("canonical consumer did not measure every admitted track") + + for index, (raw_track, observed) in enumerate(zip(raw_tracks, evidence, strict=True)): + if not isinstance(raw_track, Mapping): + raise RuntimeError(f"corpus receipt track {index} must be an object") + checks = { + "track_id": getattr(observed, "track_id", None), + "decoded_pcm_sha256": getattr(observed, "decoded_pcm_sha256", None), + "annotation_sha256": getattr(observed, "annotation_sha256", None), + "decoded_frames": getattr(observed, "decoded_frames", None), + "sample_rate_hz": getattr(observed, "sample_rate_hz", None), + } + for field, value in checks.items(): + if raw_track.get(field) != value: + raise RuntimeError( + f"canonical track evidence drifted from corpus admission at {field}" + ) + + +def _result_receipt( + registration: Mapping[str, Any], + track_receipts: Sequence[Mapping[str, object]], +) -> dict[str, object]: + """Build one complete result only from canonical track receipts and aggregation.""" + uncertainty = registration.get("uncertainty") + summary = _AGGREGATION.aggregate_complete_track_measurements( + track_receipts, + uncertainty=uncertainty, + ) + corpus = registration.get("corpus") + if isinstance(corpus, (str, bytes)) or not isinstance(corpus, Sequence): + raise ValueError("registration.corpus must be an array") + track_ids: list[str] = [] + for index, item in enumerate(corpus): + if not isinstance(item, Mapping): + raise ValueError(f"registration.corpus[{index}] must be an object") + track_ids.append(_text(item.get("track_id"), f"registration.corpus[{index}].track_id")) + + result = { + "schema_version": _VALIDATOR.SCHEMA_VERSION, + "experiment_id": registration.get("experiment_id"), + "registration_sha256": _VALIDATOR.registration_digest(registration), + "uncertainty": uncertainty, + "corpus_track_ids": track_ids, + "tracks": list(track_receipts), + "aggregate": summary["aggregate"], + "paired_delta_ci95": summary["paired_delta_ci95"], + "p95_latency_ratio_ci95": summary["p95_latency_ratio_ci95"], + "failed_tracks": [], + "claim_boundary": registration.get("claim_boundary"), + } + _VALIDATOR.evaluate_result(registration, result) + return result + + +def execute_registered_experiment( + registration: Mapping[str, Any], + manifest: Mapping[str, Any], + *, + runtime_identity: Mapping[str, object], + host_profile: StructureHostProfile, + corpus_admitter: Callable[..., Mapping[str, Any]] | None = None, + consumer: object | None = None, +) -> dict[str, object]: + """Run the frozen admission→measurement→aggregation→decision evidence chain. + + ``corpus_admitter`` and ``consumer`` are dependency seams for unit regression + tests only. Production CLI execution always uses repository-owned corpus + admission and ``PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis``. + Synthetic fixtures exercised through those seams are never production + scientific acceptance evidence. + """ + _VALIDATOR.validate_registration(registration) + observed_host_identity = _require_registered_host_profile(registration, host_profile) + + active_consumer = consumer + if active_consumer is None: + active_consumer = _TRACKS.PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis() + admit = _ADMISSION.verify_corpus if corpus_admitter is None else corpus_admitter + corpus_receipt = admit( + registration, + manifest, + runtime_identity=runtime_identity, + track_consumer=active_consumer, + ) + if not isinstance(corpus_receipt, Mapping): + raise RuntimeError("corpus admission must return an evidence object") + + evidence = getattr(active_consumer, "evidence", None) + if isinstance(evidence, (str, bytes)) or not isinstance(evidence, Sequence): + raise RuntimeError("canonical consumer must expose ordered evidence") + _bind_admission_to_evidence(corpus_receipt, evidence) + + track_receipts = getattr(active_consumer, "result_track_receipts", None) + if isinstance(track_receipts, (str, bytes)) or not isinstance(track_receipts, Sequence): + raise RuntimeError("canonical consumer must expose result track receipts") + result = _result_receipt(registration, track_receipts) + decision = _VALIDATOR.evaluate_result(registration, result) + + return { + "schema_version": 1, + "registration_sha256": _VALIDATOR.registration_digest(registration), + "execution_host_profile_identity": observed_host_identity, + "execution_host_profile": host_profile_payload(host_profile), + "corpus_receipt": dict(corpus_receipt), + "result": result, + "decision": decision, + } + + +def main(argv: Sequence[str] | None = None) -> int: + """Execute and atomically publish one complete path-free scientific envelope.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("registration", type=Path) + parser.add_argument("manifest", type=Path) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args(sys.argv[1:] if argv is None else argv) + + registration = _VALIDATOR._mapping( + _VALIDATOR._load_json(args.registration), + "registration", + ) + manifest = _ADMISSION._load_json(args.manifest) + execution = execute_registered_experiment( + registration, + manifest, + runtime_identity=_ADMISSION._current_runtime_identity(_REPOSITORY_ROOT), + host_profile=capture_host_profile(), + ) + _ADMISSION._write_receipt_atomic(args.output, execution) + decision = execution["decision"] + if not isinstance(decision, Mapping): + raise RuntimeError("execution decision must be an object") + return 0 if decision.get("passed") is True else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) From f1a6971ff7f60af993c05455e40ccc6d57ab593e Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:35:25 +0900 Subject: [PATCH 196/216] test(mir): reject free-form execution host labels --- .../test_structure_experiment_execution_contract.py | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_experiment_execution_contract.py b/services/analysis-engine/tests/test_structure_experiment_execution_contract.py index 81789f807..685cd1f7b 100644 --- a/services/analysis-engine/tests/test_structure_experiment_execution_contract.py +++ b/services/analysis-engine/tests/test_structure_experiment_execution_contract.py @@ -31,6 +31,15 @@ def _synthetic_host_profile(runner: ModuleType, *, cpu_model: str) -> object: ) +def test_registration_rejects_unversioned_host_profile_label() -> None: + """Preregistration must bind a content-addressed host, not a descriptive label.""" + runner = _runner() + registration = _registration() + + with pytest.raises(ValueError, match="runtime.host_profile"): + runner._VALIDATOR.validate_registration(registration) + + def test_execution_rejects_registered_host_profile_drift_before_corpus_admission() -> None: """Latency evidence cannot claim a preregistered host while running elsewhere.""" runner = _runner() From a31bebd79d88b940c9f96517d02f7c0364fac485 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:36:29 +0900 Subject: [PATCH 197/216] fix(mir): validate execution host contract before admission --- .../research/run_structure_noninferiority_experiment.py | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/scripts/research/run_structure_noninferiority_experiment.py b/scripts/research/run_structure_noninferiority_experiment.py index 252dae613..1172e0761 100644 --- a/scripts/research/run_structure_noninferiority_experiment.py +++ b/scripts/research/run_structure_noninferiority_experiment.py @@ -218,12 +218,18 @@ def _registered_host_profile(registration: Mapping[str, Any]) -> str: return expected +def validate_execution_registration(registration: Mapping[str, Any]) -> str: + """Validate scientific registration plus the canonical execution-host identity.""" + _VALIDATOR.validate_registration(registration) + return _registered_host_profile(registration) + + def _require_registered_host_profile( registration: Mapping[str, Any], observed: StructureHostProfile, ) -> str: """Fail before corpus access when the executing host differs from preregistration.""" - expected = _registered_host_profile(registration) + expected = validate_execution_registration(registration) actual = host_profile_identity(observed) if actual != expected: raise ValueError( @@ -314,7 +320,6 @@ def execute_registered_experiment( Synthetic fixtures exercised through those seams are never production scientific acceptance evidence. """ - _VALIDATOR.validate_registration(registration) observed_host_identity = _require_registered_host_profile(registration, host_profile) active_consumer = consumer From 7b07c00d9ee8c7ddbb24f3a65bd62aa1e22c05c4 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:36:41 +0900 Subject: [PATCH 198/216] test(mir): bind host label rejection to execution gate --- .../tests/test_structure_experiment_execution_contract.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/services/analysis-engine/tests/test_structure_experiment_execution_contract.py b/services/analysis-engine/tests/test_structure_experiment_execution_contract.py index 685cd1f7b..f77c2dbeb 100644 --- a/services/analysis-engine/tests/test_structure_experiment_execution_contract.py +++ b/services/analysis-engine/tests/test_structure_experiment_execution_contract.py @@ -32,12 +32,12 @@ def _synthetic_host_profile(runner: ModuleType, *, cpu_model: str) -> object: def test_registration_rejects_unversioned_host_profile_label() -> None: - """Preregistration must bind a content-addressed host, not a descriptive label.""" + """Canonical execution requires a content-addressed host, not a descriptive label.""" runner = _runner() registration = _registration() with pytest.raises(ValueError, match="runtime.host_profile"): - runner._VALIDATOR.validate_registration(registration) + runner.validate_execution_registration(registration) def test_execution_rejects_registered_host_profile_drift_before_corpus_admission() -> None: From 7d3c28d55af1b1ba08afc9ed2a6df85274a7d6f7 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:37:21 +0900 Subject: [PATCH 199/216] docs(mir): trace exact experiment host and execution contract --- ...structure-experiment-execution-contract.md | 110 ++++++++++++++++++ 1 file changed, 110 insertions(+) create mode 100644 docs/traceability/mir/structure-experiment-execution-contract.md diff --git a/docs/traceability/mir/structure-experiment-execution-contract.md b/docs/traceability/mir/structure-experiment-execution-contract.md new file mode 100644 index 000000000..d2705a41e --- /dev/null +++ b/docs/traceability/mir/structure-experiment-execution-contract.md @@ -0,0 +1,110 @@ +# Structure experiment execution contract + +Status: Proposed +Owner: Signal-MIR Analysis +Tracking: #1225, #1228 +Parent: `docs/traceability/mir/structure-feature-noninferiority.md` + +## Problem + +The structure preregistration already required `runtime.host_profile`, while `isolated-single-shot-v1` makes latency and peak-RSS part of the production decision. The corpus-admission runtime check, however, verified source commit, `uv.lock`, Python, librosa, and NumPy only. `runtime.host_profile` was accepted as non-empty text and was never compared with the machine that actually executed the CQT/STFT performance measurements. + +That gap is material. A receipt could truthfully bind source and dependency versions while describing an arbitrary host label unrelated to the executing CPU/model/OS. Because the decision contains a candidate/baseline latency-ratio gate, the host is part of the scientific condition rather than descriptive metadata. + +A second gap followed from the same boundary: repository-owned pieces existed for corpus admission, admitted-track quality/performance evidence, track receipt construction, paired aggregation, and result admission, but no single canonical command joined those owners into one path-free execution envelope. An experiment operator could therefore reconstruct the chain manually after each owner returned. + +## Decision + +`run_structure_noninferiority_experiment.py` is the canonical orchestration boundary for a complete registered run. It does not duplicate MIR, admission, aggregation, or decision logic. It calls the existing owners in this order: + +1. validate the scientific registration and require a content-addressed execution-host identity; +2. compare that identity with the current host before corpus admission or candidate measurement; +3. admit the registered local corpus and pass only the immutable admitted PCM/annotation handoff to the canonical CQT/STFT consumer; +4. require the consumer evidence to match the corpus receipt on track ID, decoded-PCM SHA-256, annotation SHA-256, decoded frame count, and sample rate; +5. obtain schema-v1 track receipts from the existing canonical adapter; +6. compute macro-track/paired-bootstrap summaries through the existing aggregator; +7. evaluate the complete result through the existing metric-aware validator; and +8. atomically publish one path-free execution envelope containing the host evidence, corpus receipt, result, and decision. + +The runner preserves the existing owner boundaries. `verify_structure_corpus.py` remains the Audio Ingestion/Resource Admission owner, `evaluate_admitted_structure_track.py` remains the Signal-MIR track-evidence owner, `aggregate_structure_noninferiority.py` remains the aggregation/uncertainty owner, and `validate_structure_noninferiority.py` remains the result-admission owner. + +## `structure-performance-host-v1` + +Canonical execution requires `registration.runtime.host_profile` to be: + +`structure-performance-host-v1:` + +The digest is SHA-256 over compact, sorted UTF-8 JSON containing exactly: + +- contract ID; +- platform (`darwin` or `win32`); +- OS release; +- OS version/build string; +- process architecture/machine; +- hardware product model; +- CPU model string; and +- logical CPU count. + +The profile deliberately excludes hostname/computer name, account/user identity, hardware serial numbers, MAC addresses, IP addresses, filesystem paths, and other identifiers that are unnecessary to reproduce the CQT/STFT performance condition. This is a purpose-bound scientific host identity, not a device inventory record. + +On macOS, hardware values are read dynamically with `/usr/sbin/sysctl -n hw.model` and `machdep.cpu.brand_string`. Apple recommends obtaining system/hardware details dynamically and using `sysctl`/`sysctlbyname` when a global variable is unavailable, because architectural and hardware differences can affect program behavior. The command is invoked as an argument array with `shell=False` and a bounded timeout. + +On Windows, the runner reads machine-level hardware model and processor-model strings from the local registry and combines them with the OS/build and process architecture. Microsoft documents processor identity as machine-level registry information and warns that architecture alone does not distinguish processor type, so the contract does not treat `platform.machine()` as a complete CPU identity. + +Unsupported operating systems fail closed because `isolated-single-shot-v1` itself admits performance evidence only on macOS and Windows. + +## Preregistration workflow + +The concrete host-profile payload must be captured and reviewed before candidate results are inspected. Its canonical JSON representation is the human-reviewable evidence; `runtime.host_profile` stores the corresponding content address so the registration digest binds that exact payload without storing machine-identifying fields that are irrelevant to the claim. + +At execution, the runner captures the host again and requires exact content-address equality before it opens corpus material. A free-form label such as `registered-cpu-host-v1` is not accepted by the canonical execution path. Likewise, a valid content address for a different Mac/Windows hardware/OS profile fails before corpus admission. + +The lower-level registration/result validator remains reusable for historical/schema validation; the stricter host identity is an execution prerequisite owned by this runner. Running corpus admission alone does not authorize latency/RSS evidence or a production feature decision. + +## RED → GREEN lineage + +- RED `7a90b922c0f64090f1ac2717e9afc0369503bd58` introduced an execution-contract regression requiring preregistered/observed host mismatch to fail before corpus admission and requiring the host identity to exclude hostname/user fields. +- GREEN `70b86022323728c6b3f85d5f14468423ebf15478` added the canonical execution owner, content-addressed host profile, macOS/Windows host capture, admission-to-evidence identity checks, canonical result construction, and atomic execution envelope. +- RED `f1a6971ff7f60af993c05455e40ccc6d57ab593e` made the free-form host-label gap explicit at the canonical execution preregistration boundary. +- GREEN `a31bebd79d88b940c9f96517d02f7c0364fac485` added `validate_execution_registration()` and made host-contract validation part of execution admission; `7b07c00d9ee8c7ddbb24f3a65bd62aa1e22c05c4` aligned the regression with that public execution boundary. + +Synthetic host profiles in these tests prove fail-closed mechanics only. They are not scientific host evidence and cannot substitute for the real profile captured on the machine selected for the preregistered experiment. + +## Rejected alternatives + +**Trust a descriptive host label.** Rejected because a label such as `registered-cpu-host-v1` does not establish that the observed experiment host is the preregistered machine class. + +**Use hostname, serial number, MAC address, or account identity as the host key.** Rejected because those fields add privacy/operational identity without improving the CQT/STFT performance claim. They are not stable scientific hardware descriptors and are outside the purpose-bound evidence contract. + +**Let corpus admission authorize performance evidence.** Rejected because corpus admission proves source/runtime/audio identities, not that the performance worker ran on the frozen host. It remains reusable independently, while the experiment runner owns the stronger cross-boundary invariant. + +**Reimplement admission, MIR metrics, aggregation, or decision logic in the runner.** Rejected because that would create a second scientific owner and allow semantic drift. The runner composes released/current repository owners and verifies their handoffs instead. + +**Write partial result files after each successful track.** Rejected for schema v1 because there is no registered dropout/exclusion policy and a partial complete-case result could be mistaken for the frozen corpus. A future crash-resume/checkpoint contract would need versioned semantics that preserve failed/incomplete track identity without changing the sampling population. + +## Security and privacy notes + +- Corpus paths remain local to Resource Admission and are not emitted by the execution envelope. +- The host payload contains hardware/OS characteristics needed for scientific reproducibility but excludes direct machine/account/network identifiers. +- macOS host probing invokes an exact repository-selected system binary and fixed selectors; there is no user-controlled command or shell string. +- Windows host probing reads fixed local registry paths/names; it does not write registry state or perform network discovery. +- The execution envelope is atomically published using the existing receipt writer. It contains content identities and scientific evidence, not licensed audio bytes. +- Host identity is a reproducibility condition, not an authentication or anti-tamper primitive. Protected source identity, clean worktree enforcement, runtime locks, and normal release provenance remain separate controls. + +## Claim boundary and remaining work + +This repair closes the specific ability to claim one preregistered performance host while executing the canonical CQT/STFT measurement on another, and it gives the experiment one repository-owned end-to-end execution path. + +It does not make the host perfectly stationary. CPU frequency, thermal state, scheduler contention, filesystem/page cache, power policy, and background workload may vary within one content-addressed profile. Those effects remain part of the scientific review and claim boundary; the existing alternating lane order limits deterministic ordering bias but does not remove host-state variance. + +Before a rights-cleared real-audio run, reviewers still must freeze and approve the concrete corpus membership/representativeness, numeric quality noninferiority margins, maximum candidate latency ratio, bootstrap resample count/seed, the actual host-profile payload and content address, and the population/runtime claim. The corpus size and paired-inference plan must be judged adequate for that claim. Candidate results must not be inspected before those choices are frozen. + +A production `chroma_cqt` → `chroma_stft` switch remains a separate reviewed change after the complete registered experiment passes and current protected-head CI/security/release gates are satisfied. + +## References + +Apple Inc. (n.d.). *Addressing architectural differences in your macOS code*. Apple Developer Documentation. https://developer.apple.com/documentation/apple-silicon/addressing-architectural-differences-in-your-macos-code + +Apple Inc. (n.d.). *sysctl*. Apple Developer Documentation. https://developer.apple.com/documentation/kernel/sys + +Microsoft. (n.d.). *How to determine the type of processor that your computer uses*. Microsoft Learn. https://learn.microsoft.com/en-us/troubleshoot/windows-server/setup-upgrade-and-drivers/determine-the-type-of-processor From e99ccecc20a72d9531d77be5b854ae6090500aff Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 20:41:11 +0900 Subject: [PATCH 200/216] style(mir): format canonical execution runner --- .../run_structure_noninferiority_experiment.py | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/scripts/research/run_structure_noninferiority_experiment.py b/scripts/research/run_structure_noninferiority_experiment.py index 1172e0761..57ce47ca7 100644 --- a/scripts/research/run_structure_noninferiority_experiment.py +++ b/scripts/research/run_structure_noninferiority_experiment.py @@ -284,7 +284,12 @@ def _result_receipt( for index, item in enumerate(corpus): if not isinstance(item, Mapping): raise ValueError(f"registration.corpus[{index}] must be an object") - track_ids.append(_text(item.get("track_id"), f"registration.corpus[{index}].track_id")) + track_ids.append( + _text( + item.get("track_id"), + f"registration.corpus[{index}].track_id", + ) + ) result = { "schema_version": _VALIDATOR.SCHEMA_VERSION, @@ -324,7 +329,9 @@ def execute_registered_experiment( active_consumer = consumer if active_consumer is None: - active_consumer = _TRACKS.PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis() + active_consumer = ( + _TRACKS.PairedFunctionalAccuracyTrackConsumer.for_registered_cqt_stft_hypothesis() + ) admit = _ADMISSION.verify_corpus if corpus_admitter is None else corpus_admitter corpus_receipt = admit( registration, From 8bf2fe6f31e0e65b55014e47c2ea12b032e57e4c Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Mon, 21 Sep 2026 21:08:08 +0900 Subject: [PATCH 201/216] test(mir): document functional annotation regressions --- .../tests/test_structure_functional_annotation_parser.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_functional_annotation_parser.py b/services/analysis-engine/tests/test_structure_functional_annotation_parser.py index 6ccab0b1a..b16c4537f 100644 --- a/services/analysis-engine/tests/test_structure_functional_annotation_parser.py +++ b/services/analysis-engine/tests/test_structure_functional_annotation_parser.py @@ -22,6 +22,7 @@ def _readonly(payload: bytes) -> memoryview: def test_parser_accepts_exact_mirex_tsv_and_preserves_rational_boundaries() -> None: + """Preserve exact rational boundaries for an admitted normalized annotation.""" parser = _parser() segments = parser.parse_functional_annotations( _readonly(b"0.0\t1.5\tintro\n1.5\t4.0\tverse\n"), @@ -36,6 +37,7 @@ def test_parser_accepts_exact_mirex_tsv_and_preserves_rational_boundaries() -> N def test_parser_rejects_mapping_freedom_and_noncanonical_labels() -> None: + """Reject post-registration label mapping and labels outside the frozen vocabulary.""" parser = _parser() for label in ("Verse", "verse1", "solo", "other", " verse"): payload = f"0.0\t1.0\t{label}\n".encode() @@ -48,6 +50,7 @@ def test_parser_rejects_mapping_freedom_and_noncanonical_labels() -> None: def test_parser_rejects_gap_overlap_and_incomplete_duration() -> None: + """Require a continuous segmentation that exactly spans the admitted audio duration.""" parser = _parser() invalid_payloads = ( b"0.0\t1.0\tintro\n1.1\t2.0\tverse\n", @@ -64,6 +67,7 @@ def test_parser_rejects_gap_overlap_and_incomplete_duration() -> None: def test_parser_rejects_malformed_nonfinite_or_mutable_annotation_input() -> None: + """Fail closed on malformed, non-finite, or mutable annotation evidence.""" parser = _parser() with pytest.raises(ValueError, match="three tab-separated"): parser.parse_functional_annotations( From 22da683509cf2e47c427b7ff3797149a7f67430f Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Mon, 21 Sep 2026 21:08:53 +0900 Subject: [PATCH 202/216] test(mir): document aggregation regressions --- .../tests/test_structure_noninferiority_aggregation.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py b/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py index f588136f6..a448b3ac8 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py @@ -142,6 +142,7 @@ def _tracks() -> list[dict[str, object]]: def test_macro_track_aggregation_recomputes_f_and_preserves_worst_peak_memory() -> None: + """Aggregate equal-weight tracks and retain the worst observed peak memory.""" aggregation = _aggregation() evidence = aggregation.aggregate_complete_track_measurements( @@ -167,6 +168,7 @@ def test_macro_track_aggregation_recomputes_f_and_preserves_worst_peak_memory() def test_paired_bootstrap_is_deterministic_and_resamples_track_pairs() -> None: + """Keep paired bootstrap output deterministic for a frozen seed and track set.""" aggregation = _aggregation() uncertainty = { "procedure_id": "paired-track-bootstrap-v1", @@ -217,6 +219,7 @@ def test_percentile_projection_preserves_asymmetric_interval_bounds() -> None: def test_aggregation_fails_closed_on_duplicate_tracks_or_unsupported_uncertainty() -> None: + """Reject duplicate track weight and uncertainty plans outside preregistration.""" aggregation = _aggregation() duplicate = _tracks() duplicate[1]["track_id"] = "track-a" From 2265cc612166e642968e1c6780e50abc2f311748 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Mon, 21 Sep 2026 21:09:23 +0900 Subject: [PATCH 203/216] test(mir): fail on performance contract drift --- ...est_structure_lane_resource_measurement.py | 25 +++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/services/analysis-engine/tests/test_structure_lane_resource_measurement.py b/services/analysis-engine/tests/test_structure_lane_resource_measurement.py index b81f4d539..f166f5796 100644 --- a/services/analysis-engine/tests/test_structure_lane_resource_measurement.py +++ b/services/analysis-engine/tests/test_structure_lane_resource_measurement.py @@ -53,6 +53,31 @@ def test_performance_contract_is_part_of_metric_aware_registration_identity() -> } +@pytest.mark.parametrize( + ("field", "drifted_value"), + ( + ("contract_id", "isolated-single-shot-v2"), + ("measured_trials", 19), + ("lane_order", "baseline_then_candidate"), + ("latency_quantiles", [0.5, 0.9]), + ("quantile_method", "nearest"), + ), +) +def test_performance_measurement_contract_drift_fails_closed( + monkeypatch: pytest.MonkeyPatch, + field: str, + drifted_value: object, +) -> None: + """Reject preregistration metadata that no longer describes runtime measurement semantics.""" + module = _measurement_module() + drifted_contract = dict(module.PERFORMANCE_MEASUREMENT_CONTRACT) + drifted_contract[field] = drifted_value + monkeypatch.setattr(module, "PERFORMANCE_MEASUREMENT_CONTRACT", drifted_contract) + + with pytest.raises(RuntimeError, match="performance measurement contract drift"): + module._validate_performance_measurement_contract() + + def test_paired_measurement_alternates_order_and_uses_linear_quantiles( monkeypatch: pytest.MonkeyPatch, ) -> None: From f08a35392eabbc759760d2f0a5b9fd87241a2e61 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Mon, 21 Sep 2026 21:10:26 +0900 Subject: [PATCH 204/216] fix(mir): bind performance contract to runtime semantics --- .../measure_structure_lane_resources.py | 104 ++++++++++++------ 1 file changed, 68 insertions(+), 36 deletions(-) diff --git a/scripts/research/measure_structure_lane_resources.py b/scripts/research/measure_structure_lane_resources.py index 01e27d75c..9c471df12 100644 --- a/scripts/research/measure_structure_lane_resources.py +++ b/scripts/research/measure_structure_lane_resources.py @@ -43,25 +43,48 @@ ChromaFeature = Literal["cqt", "stft"] _REPOSITORY_ROOT = Path(__file__).resolve().parents[2] +_PERFORMANCE_CONTRACT_ID = "isolated-single-shot-v1" +_SUPPORTED_PLATFORMS = ("darwin", "win32") +_WARMUP_TRIALS = 0 _MEASURED_TRIALS = 20 +_LANE_ORDER = "alternate_baseline_candidate_by_trial_index" +_LATENCY_QUANTILES = (0.5, 0.95) +_QUANTILE_METHOD = "linear" + + +def _expected_performance_measurement_contract() -> dict[str, object]: + """Return the runtime-backed scientific performance contract.""" + return { + "contract_id": _PERFORMANCE_CONTRACT_ID, + "supported_platforms": list(_SUPPORTED_PLATFORMS), + "warmup_trials": _WARMUP_TRIALS, + "measured_trials": _MEASURED_TRIALS, + "trial_process": "fresh_subprocess_per_lane_trial", + "lane_order": _LANE_ORDER, + "timer": "time.perf_counter_ns", + "timer_scope": "repository_structure_segmenter_only", + "worker_startup_in_latency": False, + "input_transfer_in_latency": False, + "latency_quantiles": list(_LATENCY_QUANTILES), + "quantile_method": _QUANTILE_METHOD, + "memory_metric": "process_peak_resident_set_size", + "memory_scope": "entire_worker_process_lifetime_including_pcm_input", + "peak_rss_aggregation": "maximum_across_trials", + } + + +PERFORMANCE_MEASUREMENT_CONTRACT: dict[str, object] = ( + _expected_performance_measurement_contract() +) + + +def _validate_performance_measurement_contract() -> None: + """Fail closed if published preregistration metadata drifts from runtime semantics.""" + if PERFORMANCE_MEASUREMENT_CONTRACT != _expected_performance_measurement_contract(): + raise RuntimeError("performance measurement contract drifted from runtime implementation") -PERFORMANCE_MEASUREMENT_CONTRACT: dict[str, object] = { - "contract_id": "isolated-single-shot-v1", - "supported_platforms": ["darwin", "win32"], - "warmup_trials": 0, - "measured_trials": _MEASURED_TRIALS, - "trial_process": "fresh_subprocess_per_lane_trial", - "lane_order": "alternate_baseline_candidate_by_trial_index", - "timer": "time.perf_counter_ns", - "timer_scope": "repository_structure_segmenter_only", - "worker_startup_in_latency": False, - "input_transfer_in_latency": False, - "latency_quantiles": [0.5, 0.95], - "quantile_method": "linear", - "memory_metric": "process_peak_resident_set_size", - "memory_scope": "entire_worker_process_lifetime_including_pcm_input", - "peak_rss_aggregation": "maximum_across_trials", -} + +_validate_performance_measurement_contract() @dataclass(frozen=True, slots=True) @@ -166,40 +189,48 @@ def _macos_peak_rss_mib() -> float: def _windows_peak_rss_mib() -> float: """Return Windows PeakWorkingSetSize for the current worker process.""" - import ctypes - from ctypes import wintypes + from ctypes import ( + POINTER, + Structure, + WinDLL, + byref, + c_size_t, + get_last_error, + sizeof, + wintypes, + ) - class ProcessMemoryCounters(ctypes.Structure): + class ProcessMemoryCounters(Structure): """Win32 PROCESS_MEMORY_COUNTERS layout required by GetProcessMemoryInfo.""" _fields_ = [ ("cb", wintypes.DWORD), ("PageFaultCount", wintypes.DWORD), - ("PeakWorkingSetSize", ctypes.c_size_t), - ("WorkingSetSize", ctypes.c_size_t), - ("QuotaPeakPagedPoolUsage", ctypes.c_size_t), - ("QuotaPagedPoolUsage", ctypes.c_size_t), - ("QuotaPeakNonPagedPoolUsage", ctypes.c_size_t), - ("QuotaNonPagedPoolUsage", ctypes.c_size_t), - ("PagefileUsage", ctypes.c_size_t), - ("PeakPagefileUsage", ctypes.c_size_t), + ("PeakWorkingSetSize", c_size_t), + ("WorkingSetSize", c_size_t), + ("QuotaPeakPagedPoolUsage", c_size_t), + ("QuotaPagedPoolUsage", c_size_t), + ("QuotaPeakNonPagedPoolUsage", c_size_t), + ("QuotaNonPagedPoolUsage", c_size_t), + ("PagefileUsage", c_size_t), + ("PeakPagefileUsage", c_size_t), ] - kernel32 = ctypes.WinDLL("kernel32", use_last_error=True) - psapi = ctypes.WinDLL("psapi", use_last_error=True) + kernel32 = WinDLL("kernel32", use_last_error=True) + psapi = WinDLL("psapi", use_last_error=True) kernel32.GetCurrentProcess.restype = wintypes.HANDLE psapi.GetProcessMemoryInfo.argtypes = [ wintypes.HANDLE, - ctypes.POINTER(ProcessMemoryCounters), + POINTER(ProcessMemoryCounters), wintypes.DWORD, ] psapi.GetProcessMemoryInfo.restype = wintypes.BOOL counters = ProcessMemoryCounters() - counters.cb = ctypes.sizeof(counters) + counters.cb = sizeof(counters) handle = kernel32.GetCurrentProcess() - if not psapi.GetProcessMemoryInfo(handle, ctypes.byref(counters), counters.cb): - error_code = ctypes.get_last_error() + if not psapi.GetProcessMemoryInfo(handle, byref(counters), counters.cb): + error_code = get_last_error() raise OSError(error_code, "GetProcessMemoryInfo failed") peak_bytes = int(counters.PeakWorkingSetSize) if peak_bytes <= 0: @@ -338,7 +369,7 @@ def _summarize_trials(trials: Sequence[IsolatedLaneTrial]) -> LaneResourceSummar ) if not np.all(np.isfinite(latencies)) or np.any(latencies <= 0.0): raise ValueError("lane evidence contains invalid latency") - p50, p95 = np.quantile(latencies, [0.5, 0.95], method="linear") + p50, p95 = np.quantile(latencies, _LATENCY_QUANTILES, method=_QUANTILE_METHOD) peak_rss_mib = max(trial.peak_rss_mib for trial in trials) if not math.isfinite(peak_rss_mib) or peak_rss_mib <= 0.0: raise ValueError("lane evidence contains invalid peak RSS") @@ -356,6 +387,7 @@ def measure_paired_repository_lane_resources( duration_seconds: object, ) -> PairedLaneResourceEvidence: """Measure paired CQT/STFT performance on the exact immutable admitted PCM.""" + _validate_performance_measurement_contract() pcm = _pcm_view(decoded_pcm) sample_rate = _positive_sample_rate(sample_rate_hz) exact_duration = _duration( @@ -380,7 +412,7 @@ def measure_paired_repository_lane_resources( else: candidate_trials.append(observation) return PairedLaneResourceEvidence( - contract_id="isolated-single-shot-v1", + contract_id=_PERFORMANCE_CONTRACT_ID, baseline=_summarize_trials(baseline_trials), candidate=_summarize_trials(candidate_trials), ) From 53866fe86af791f20daa56cc3eb618a3af02a00f Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Mon, 21 Sep 2026 21:11:32 +0900 Subject: [PATCH 205/216] docs(mir): trace performance-contract drift repair --- .../structure-performance-measurement-contract.md | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/docs/traceability/mir/structure-performance-measurement-contract.md b/docs/traceability/mir/structure-performance-measurement-contract.md index 4214bdc83..a96dd987f 100644 --- a/docs/traceability/mir/structure-performance-measurement-contract.md +++ b/docs/traceability/mir/structure-performance-measurement-contract.md @@ -11,6 +11,8 @@ The structure noninferiority result schema already carries per-track `p50_latenc That is not sufficient for a production decision between `chroma_cqt` and `chroma_stft`. Performance evidence must be fixed before candidate results are observed, must consume the same admitted decoded PCM identity as the quality comparison, and must avoid a cache-warmed benchmark path that is unlike a user's first analysis of a song. +A later review found a second reproducibility gap in that producer. `PERFORMANCE_MEASUREMENT_CONTRACT` participated in registration identity, but runtime measurement did not validate that the published mapping still matched the implementation constants used for trial count, contract identity, lane order, quantiles, and quantile method. A future edit could therefore change the public preregistration mapping without making scientific execution fail closed. The same review also found mixed `ctypes` import styles in the Windows RSS boundary; that was repaired without changing the Win32 API or measurement semantics. + ## Decision `isolated-single-shot-v1` is the only preregistered structure performance contract in schema v1. Its complete semantics are owned by `scripts/research/measure_structure_lane_resources.py` and copied into `STRUCTURE_METRIC_CONTRACT["performance_measurement"]`, so they participate in the metric-aware registration SHA-256. @@ -31,6 +33,8 @@ The contract is: Twenty observations are a preregistered engineering sampling plan, not a claim that twenty observations estimate an asymptotic tail distribution. The production decision still uses the registered paired track-level bootstrap across the rights-cleared corpus; the within-track repetitions create the frozen p50/p95 performance summary supplied to that higher-level procedure. +The public `PERFORMANCE_MEASUREMENT_CONTRACT` is now built from the runtime constants that control contract identity, supported platforms, warm-up count, measured-trial count, lane-order identity, latency quantiles, and quantile method. Scientific execution calls `_validate_performance_measurement_contract()` before paired measurement. Any mutation or source drift that makes the published mapping differ from those runtime-backed semantics fails before evidence is produced. Existing behavior tests continue to verify the actual alternating lane sequence and numerical quantile result, so the mapping comparison is not treated as a substitute for executable behavior evidence. + ## Timer boundary Python documents `perf_counter()` as a performance counter with the highest available resolution for short durations and states that it is monotonic in CPython; `perf_counter_ns()` returns the same clock as integer nanoseconds and avoids float precision loss. The worker therefore records `perf_counter_ns()` immediately before and after the one repository-owned structure-segmentation call. @@ -47,7 +51,7 @@ This does not claim that operating-system file cache, CPU frequency, thermal sta On macOS, the worker reads `getrusage(RUSAGE_SELF).ru_maxrss`. Apple's `getrusage(2)` documentation defines `ru_maxrss` as maximum resident set size and documents the value in kilobytes, so the implementation converts KiB to MiB by dividing by 1024. -On Windows, the worker calls `GetProcessMemoryInfo` and uses `PROCESS_MEMORY_COUNTERS.PeakWorkingSetSize`. Microsoft documents the working set as physical memory mapped into the process address space and exposes both current and peak working-set size. The implementation converts bytes to MiB. +On Windows, the worker calls `GetProcessMemoryInfo` and uses `PROCESS_MEMORY_COUNTERS.PeakWorkingSetSize`. Microsoft documents the working set as physical memory mapped into the process address space and exposes both current and peak working-set size. The implementation converts bytes to MiB. The Windows-only function now uses a single `ctypes` import style while preserving `Structure`, `wintypes`, `WinDLL`, pointer setup, `GetProcessMemoryInfo`, and `get_last_error` behavior. Linux is intentionally not admitted for `isolated-single-shot-v1`. Python's `resource` API is Unix-specific and `ru_maxrss` unit conventions are platform-dependent; adding Linux would require a separately reviewed platform contract rather than silently treating all `getrusage` values as equivalent. @@ -69,6 +73,9 @@ The scientific receipt records the contract identity plus baseline/candidate p50 - Contract fixture `a05c345a23910e0e0c48f327ded8ea1a8d09bb9a` cross-checks registration metadata against the executable performance owner. - Hygiene repair `6d49d7a54672eda000d94682c2040915f6b4657e` removed an unused typing import before hosted lint evidence. - Focused-test alignment `11c59ade85b1866e550edec2ea26cb185670c9ab` prevents the recognized-metric unit boundary from accidentally launching platform performance workers; the dedicated performance tests own that contract. +- Hosted CI on exact `e99ccecc20a72d9531d77be5b854ae6090500aff` reached `ci / build-and-test` and failed `check:python-docstrings` on seven #1228-owned test functions. Commits `8bf2fe6f31e0e65b55014e47c2ea12b032e57e4c` and `22da683509cf2e47c427b7ff3797149a7f67430f` repair those exact D103 findings without changing scientific behavior. The predecessor failure is retained only as causal evidence. +- RED `2265cc612166e642968e1c6780e50abc2f311748` adds focused regressions that mutate contract identity, measured-trial count, lane-order identity, latency quantiles, and quantile method and require scientific execution to fail closed on drift. Repair followed before a terminal hosted verdict, so no hosted RED is claimed for the test-only head. +- GREEN `f08a35392eabbc759760d2f0a5b9fd87241a2e61` makes the published contract runtime-backed, validates it before paired measurement, reuses the contract ID and quantile constants in execution, and removes the mixed Windows `ctypes` import style without changing the Win32 measurement API. ## Rejected alternatives @@ -84,17 +91,19 @@ The scientific receipt records the contract identity plus baseline/candidate p50 **Use fewer measured trials when a track is expensive.** Rejected because data-dependent or operator-dependent trial reduction would change the estimator after corpus identity is known. An execution that cannot complete the preregistered count is a failed track under the existing no-exclusion policy. +**Keep the public performance mapping as registration-only metadata.** Rejected because registration identity is not enough if runtime execution can proceed after that mapping drifts away from the constants and behavior used to produce evidence. The producer now fails before paired measurement when the public mapping no longer equals the runtime-backed contract. + ## Security Notes - Input remains local-only and path-free after corpus admission. - Subprocess invocation uses `sys.executable`, the repository-owned worker path, exact numeric arguments, `shell=False`, and stdin PCM bytes; no generic execution surface is introduced. -- Any non-zero exit, worker stderr, malformed JSON, unexpected receipt field, unsupported platform, invalid timer, or invalid RSS value fails the scientific measurement. +- Any non-zero exit, worker stderr, malformed JSON, unexpected receipt field, unsupported platform, invalid timer, invalid RSS value, or performance-contract drift fails the scientific measurement. - Raw licensed audio and workstation paths are not emitted in durable performance evidence. - This contract adds no telemetry or network-dependent runtime path. ## Claim boundary and remaining work -This contract closes the missing canonical producer for per-track p50/p95 latency and peak RSS. It does not establish that STFT is faster, that either lane meets a product latency target, or that twenty within-track repetitions make p95 a universal tail-latency estimate. +This contract closes the missing canonical producer for per-track p50/p95 latency and peak RSS and now makes drift between the public preregistration mapping and its runtime-backed constants an executable failure. It does not establish that STFT is faster, that either lane meets a product latency target, or that twenty within-track repetitions make p95 a universal tail-latency estimate. Before rights-cleared real-audio execution, reviewers still must freeze the concrete corpus membership and representativeness, numeric quality noninferiority margins, maximum candidate latency ratio, exact bootstrap resample count and seed, exact host/runtime profile, and claim boundary. Candidate results must not be inspected before those choices are frozen. From d01649192eb7869e8476706f8ed150c2ce0d8fa0 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Tue, 22 Sep 2026 06:09:31 +0900 Subject: [PATCH 206/216] fix(mir): apply Ruff import order to structure evidence tests --- .../tests/test_structure_corpus_consumer_integrity_policy.py | 1 - 1 file changed, 1 deletion(-) diff --git a/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py b/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py index ab151e974..7807768d3 100644 --- a/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py +++ b/services/analysis-engine/tests/test_structure_corpus_consumer_integrity_policy.py @@ -8,7 +8,6 @@ from pathlib import Path import pytest - from test_structure_corpus_admission import ( _admission, _digest, From 2b92ce8a2f2e1b98471339df1749abb6a14a046e Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Tue, 22 Sep 2026 06:09:43 +0900 Subject: [PATCH 207/216] fix(mir): apply Ruff import order to structure evidence tests --- .../tests/test_structure_corpus_manifest_loader_policy.py | 1 - 1 file changed, 1 deletion(-) diff --git a/services/analysis-engine/tests/test_structure_corpus_manifest_loader_policy.py b/services/analysis-engine/tests/test_structure_corpus_manifest_loader_policy.py index 7a2435be9..3df319456 100644 --- a/services/analysis-engine/tests/test_structure_corpus_manifest_loader_policy.py +++ b/services/analysis-engine/tests/test_structure_corpus_manifest_loader_policy.py @@ -5,7 +5,6 @@ from pathlib import Path import pytest - from test_structure_corpus_admission import _admission From e37ff615bc292be072acff1c6bf196e8c8c9f742 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Tue, 22 Sep 2026 06:09:55 +0900 Subject: [PATCH 208/216] fix(mir): apply Ruff import order to structure evidence tests --- .../tests/test_structure_functional_label_contract_policy.py | 1 - 1 file changed, 1 deletion(-) diff --git a/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py b/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py index 67c6a6d35..4974769dc 100644 --- a/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py +++ b/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py @@ -5,7 +5,6 @@ import copy import pytest - from test_structure_noninferiority_policy import _metrics, _registration, _validator From dd32c8a00d644a03aac1dce557f8a0a5e66e02ba Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Tue, 22 Sep 2026 06:10:14 +0900 Subject: [PATCH 209/216] fix(mir): apply Ruff import order to structure evidence tests --- .../tests/test_structure_metric_preregistration_contract.py | 1 - 1 file changed, 1 deletion(-) diff --git a/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py b/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py index f8cfc6084..8ab742567 100644 --- a/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py +++ b/services/analysis-engine/tests/test_structure_metric_preregistration_contract.py @@ -9,7 +9,6 @@ import pytest from conftest import load_module - from test_structure_noninferiority_policy import _registration, _result _EXPECTED_METRIC_CONTRACT = { From 7d9ee2bae5cb26b71840be0d92444d4479bb756b Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Tue, 22 Sep 2026 06:10:38 +0900 Subject: [PATCH 210/216] fix(mir): apply Ruff import order to structure evidence tests --- .../tests/test_structure_noninferiority_aggregation.py | 1 - 1 file changed, 1 deletion(-) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py b/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py index a448b3ac8..22f744651 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_aggregation.py @@ -7,7 +7,6 @@ import pytest from conftest import load_module - from test_structure_noninferiority_policy import _registration, _result From 212c195a569a044fd6d941b6da0f604a7d9ebdb4 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Tue, 22 Sep 2026 06:10:57 +0900 Subject: [PATCH 211/216] fix(mir): apply Ruff import order to structure evidence tests --- .../test_structure_noninferiority_envelope_schema_policy.py | 1 - 1 file changed, 1 deletion(-) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py index 5f791fe17..318627b38 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_envelope_schema_policy.py @@ -3,7 +3,6 @@ from __future__ import annotations import pytest - from test_structure_noninferiority_policy import ( _corpus, _metrics, From eae63a9edb03ef7c1744c3fba3b9fc1b93da27e8 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Tue, 22 Sep 2026 06:11:09 +0900 Subject: [PATCH 212/216] fix(mir): apply Ruff import order to structure evidence tests --- .../tests/test_structure_noninferiority_failure_policy.py | 1 - 1 file changed, 1 deletion(-) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py index 4bace0c68..49ee928ab 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_failure_policy.py @@ -1,7 +1,6 @@ """Regression tests for failed-track scientific acceptance policy.""" import pytest - from test_structure_noninferiority_policy import _registration, _result, _validator From 275cadacf672391951e410c1e3c9292e75bc4c99 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Tue, 22 Sep 2026 06:11:23 +0900 Subject: [PATCH 213/216] fix(mir): apply Ruff import order to structure evidence tests --- .../tests/test_structure_noninferiority_result_schema_policy.py | 1 - 1 file changed, 1 deletion(-) diff --git a/services/analysis-engine/tests/test_structure_noninferiority_result_schema_policy.py b/services/analysis-engine/tests/test_structure_noninferiority_result_schema_policy.py index 2a9d60492..784fdebe7 100644 --- a/services/analysis-engine/tests/test_structure_noninferiority_result_schema_policy.py +++ b/services/analysis-engine/tests/test_structure_noninferiority_result_schema_policy.py @@ -3,7 +3,6 @@ from __future__ import annotations import pytest - from test_structure_noninferiority_policy import _registration, _result, _validator From 6cdea53db6502e994e26adc6665b5be19df4d602 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Tue, 22 Sep 2026 14:23:29 +0900 Subject: [PATCH 214/216] fix(mir): remove residual Ruff import ambiguity --- .../test_structure_functional_label_contract_policy.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py b/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py index 4974769dc..8c147d44e 100644 --- a/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py +++ b/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py @@ -5,9 +5,13 @@ import copy import pytest -from test_structure_noninferiority_policy import _metrics, _registration, _validator +import test_structure_noninferiority_policy as noninferiority_policy +_metrics = noninferiority_policy._metrics +_registration = noninferiority_policy._registration +_validator = noninferiority_policy._validator + _MIREX_2025_ACC_IMPLEMENTATION = ( "ismir-mirex/mirex-evaluation@" "b9fa0b0b32e2145af31f35830f78fc9d09a4301b:" From 823cd35ce22935e4e072aa3995e7cb85063fbde0 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 23 Sep 2026 06:35:45 +0900 Subject: [PATCH 215/216] fix(research): sort functional-label policy imports --- .../tests/test_structure_functional_label_contract_policy.py | 1 + 1 file changed, 1 insertion(+) diff --git a/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py b/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py index 8c147d44e..a184275ae 100644 --- a/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py +++ b/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py @@ -5,6 +5,7 @@ import copy import pytest + import test_structure_noninferiority_policy as noninferiority_policy From 7eb2b9d1f08254d7ce7300a6f9a5b72ea06b3977 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 23 Sep 2026 21:06:45 +0900 Subject: [PATCH 216/216] repair(mir): apply hosted Ruff import grouping --- .../tests/test_structure_functional_label_contract_policy.py | 1 - 1 file changed, 1 deletion(-) diff --git a/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py b/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py index a184275ae..8c147d44e 100644 --- a/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py +++ b/services/analysis-engine/tests/test_structure_functional_label_contract_policy.py @@ -5,7 +5,6 @@ import copy import pytest - import test_structure_noninferiority_policy as noninferiority_policy