diff --git a/.github/ISSUE_TEMPLATE/benchmark-question.md b/.github/ISSUE_TEMPLATE/benchmark-question.md new file mode 100644 index 0000000..92241ae --- /dev/null +++ b/.github/ISSUE_TEMPLATE/benchmark-question.md @@ -0,0 +1,28 @@ +--- +name: Benchmark question +description: Ask about a sanitized benchmark result or protocol +labels: [] +assignees: [] +--- + +> Public issue: do not paste credentials, private URLs/IPs, raw production logs, cookies, headers, customer data, or infrastructure details. + +## Benchmark goal + +Describe the operator metric you are trying to improve. + +## Sanitized setup + +Describe the baseline and candidate policy at a high level without naming private systems, endpoints, or accounts. + +## Metrics + +Paste only aggregate `proxybench` output or synthetic fixture data. + +## Question + +What interpretation or benchmark-design issue do you want help with? + +## Authorization / responsible use + +Confirm the workload is authorized and respects applicable target policies, rate limits, and law. diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml new file mode 100644 index 0000000..75d5682 --- /dev/null +++ b/.github/workflows/test.yml @@ -0,0 +1,52 @@ +name: tests + +on: + push: + pull_request: + +permissions: + contents: read + +jobs: + test: + runs-on: ubuntu-latest + strategy: + fail-fast: false + matrix: + python-version: ["3.10", "3.11", "3.12"] + + steps: + - name: Check out repository + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + + - name: Set up Python + uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0 + with: + python-version: ${{ matrix.python-version }} + + - name: Public repository hygiene check + run: python scripts/public_hygiene_check.py + + - name: Install package from source + run: python -m pip install --disable-pip-version-check . + + - name: Compile package + run: python -m compileall -q src + + - name: Run tests against installed package + run: python -m unittest discover -s tests -v + + - name: Installed CLI smoke tests + shell: bash + run: | + proxybench summarize examples/baseline.jsonl \ + | python -c 'import json,sys; d=json.load(sys.stdin); assert d["requests"] == 20; assert d["usable_results"] == 13' + + python -m proxybench compare examples/baseline.jsonl examples/candidate.jsonl \ + --min-requests 20 \ + --min-success-uplift-pp 10 \ + --max-rpu-regression-pct 0 \ + --max-cost-regression-pct 0 \ + | python -c 'import json,sys; d=json.load(sys.stdin); assert d["gate"]["verdict"] == "PASS"' diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..38feb8a --- /dev/null +++ b/.gitignore @@ -0,0 +1,35 @@ +# Python +__pycache__/ +*.py[cod] +*.egg-info/ +build/ +dist/ +.venv/ +venv/ +.coverage +.pytest_cache/ +.mypy_cache/ +.ruff_cache/ + +# Local secrets and credentials +.env +.env.* +!.env.example +*.pem +*.key +*.p12 +*.pfx +credentials* +secrets* + +# Potentially sensitive captures/data +*.har +*.pcap +*.pcapng +*.sqlite +*.sqlite3 +*.db +*.log +private/ +local-data/ +raw-data/ diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..d39984e --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 PN Labs + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/README.md b/README.md index 6cd7453..0627f01 100644 --- a/README.md +++ b/README.md @@ -1 +1,137 @@ -# proxybench \ No newline at end of file +# proxybench + +**Measure usable results, not proxy count.** + +`proxybench` is a local, deterministic benchmark utility for proxy and web-retrieval workloads. It compares sanitized request outcomes using metrics that matter to operators: usable success rate, requests per usable result, rotations per usable result, latency, and cost per usable result. + +Built by **PN Labs**. + +## Why this exists + +Adding more proxies or rotating more often does not guarantee better retrieval outcomes. A benchmark should answer a narrower question: + +```text +baseline policy + vs +candidate policy + ↓ +usable success rate +requests / usable result +rotations / usable result +latency distribution +cost / usable result +``` + +`proxybench` deliberately does **not** perform crawling, make network requests, accept proxy credentials, or select providers. It measures evidence you already collected from an authorized workload. + +## Relationship to proxy-outcome + +[`proxy-outcome`](https://github.com/pnlabs-dev/proxy-outcome) answers: + +> What observation do we actually have, and how strong is the proxy-layer evidence? + +`proxybench` answers: + +> Did policy A or policy B produce better usable outcomes and efficiency? + +Classification and benchmarking stay separate. + +## Install + +```bash +python -m pip install . +``` + +Runtime dependencies: **none**. + +## Input format + +Input is JSON Lines (`.jsonl`). Every line is one sanitized retrieval event. + +Allowed fields only: + +```json +{"usable": true, "latency_ms": 420, "cost_units": 0.0021, "rotated": false, "outcome": "SUCCESS"} +``` + +- `usable` — required boolean. Whether the result was usable for the workload. +- `latency_ms` — optional non-negative number. +- `cost_units` — optional non-negative number in any consistent cost unit. +- `rotated` — optional boolean. +- `outcome` — optional uppercase categorical token such as `HTTP_RATE_LIMIT`. + +Unknown fields are rejected. URLs, IP addresses, proxy identifiers, provider credentials, cookies, headers, payloads, customer identifiers, and other production context are neither required nor part of the schema. + +## Summarize one arm + +```bash +proxybench summarize examples/baseline.jsonl +``` + +Output includes: + +- request count; +- usable result count; +- usable success rate + descriptive Wilson interval; +- requests per usable result; +- rotation coverage and rotations per usable result; +- latency coverage, p50, and p95; +- cost coverage and cost per usable result; +- categorical outcome counts. + +## Compare A/B arms + +```bash +proxybench compare examples/baseline.jsonl examples/candidate.jsonl +``` + +The comparison reports directional deltas without pretending that request events are necessarily independent or causal. + +Optional operator-defined gates: + +```bash +proxybench compare examples/baseline.jsonl examples/candidate.jsonl \ + --min-requests 20 \ + --min-success-uplift-pp 2 \ + --max-rpu-regression-pct 5 \ + --max-cost-regression-pct 5 +``` + +Gate verdicts are `PASS`, `FAIL`, `INCONCLUSIVE`, or `NO_GATES_CONFIGURED`. + +## Design principles + +- **Usable-result first** — HTTP success alone is not the buyer KPI. +- **Data minimization** — no URLs, IPs, credentials, raw headers, or production payloads are required. +- **Local only** — no network I/O or telemetry. +- **Evidence before claims** — descriptive intervals are not presented as causal proof. +- **Explicit coverage** — cost/latency/rotation metrics report how much of the input actually contained that field. +- **Zero runtime dependencies** — easy to embed in CI and benchmark harnesses. + +## Public / commercial boundary + +This repository contains the transparent measurement baseline only. It does not contain PN Labs' private routing/scoring implementation, provider selection logic, target × egress learning, promotion/demotion intelligence, private benchmark datasets, production infrastructure, or cost-optimization control plane. + +See [`docs/PUBLIC-BOUNDARY.md`](docs/PUBLIC-BOUNDARY.md). + +## Validation + +```bash +python -m pip install . +python -m unittest discover -s tests -v +python scripts/public_hygiene_check.py +``` + +CI installs the package before testing, smoke-tests the installed CLI, and runs a non-echoing vendor-neutral public repository hygiene gate. + +## Security + +See [`SECURITY.md`](SECURITY.md). Do not put production logs, credentials, private endpoint information, personal/customer data, or private infrastructure details into public benchmark fixtures or issues. + +## Responsible use + +Use `proxybench` only with workloads you are authorized to run. Respect applicable target policies, rate limits, robots directives where relevant, and law. + +## License + +MIT. diff --git a/SECURITY.md b/SECURITY.md new file mode 100644 index 0000000..cf68e45 --- /dev/null +++ b/SECURITY.md @@ -0,0 +1,49 @@ +# Security policy + +`proxybench` is intentionally local-only. It performs no network I/O, telemetry, crawling, provider login, proxy authentication, or credential storage. + +## Public input boundary + +The benchmark event schema is deliberately small. It accepts only: + +- `usable` +- `latency_ms` +- `cost_units` +- `rotated` +- `outcome` + +Unknown fields are rejected. This is a data-minimization boundary: target URLs, proxy endpoints, IP addresses, provider names, headers, cookies, credentials, payloads, customer identifiers, and infrastructure details are not required. + +## Never publish sensitive material + +Do not place any of the following in issues, pull requests, examples, fixtures, screenshots, benchmark artifacts, or CI logs: + +- API keys, access tokens, passwords, or private keys; +- proxy credentials or credential-bearing proxy URLs; +- session cookies, authorization headers, or authenticated request dumps; +- private endpoint URLs, IP addresses, internal hostnames, or infrastructure topology; +- production HAR/PCAP captures or raw production logs; +- customer data, personal data, or private datasets; +- non-public commercial implementation details. + +Use synthetic fixtures and aggregate counts instead. + +## Error behavior + +Input-validation errors do not intentionally echo raw JSON lines, unsupported values, or local input paths. This reduces the chance that CI output becomes a secondary disclosure path. + +## Statistical safety + +The Wilson interval reported for usable success rate is descriptive. `proxybench` does not claim that request events are independent, randomized, or causal. Operator-defined gates are policy checks, not scientific proof. + +## Repository hygiene gate + +CI runs `scripts/public_hygiene_check.py`, which searches public text files for common secret/token shapes, credential-bearing URLs, IPv4/IPv6 literals, internal-hostname shapes, sensitive filenames, and email addresses. + +Detected values are never printed; only the rule class and file path are reported. + +This is defense-in-depth, not a replacement for review or dedicated secret scanning. + +## Reporting security issues + +Do not open a public issue containing sensitive reproduction material. Reduce the problem to a synthetic reproducer or sanitized aggregate description before sharing. diff --git a/docs/METRICS.md b/docs/METRICS.md new file mode 100644 index 0000000..35d36b7 --- /dev/null +++ b/docs/METRICS.md @@ -0,0 +1,81 @@ +# Metrics + +`proxybench` focuses on operator metrics tied to usable outcomes rather than raw proxy inventory. + +## Usable success rate + +```text +usable results / requests +``` + +The event producer decides what `usable=true` means for the workload. That definition should be fixed before comparing arms. + +A descriptive 95% Wilson interval is reported for the proportion. It is not a causal claim and does not establish request independence. + +## Requests per usable result + +```text +requests / usable results +``` + +Lower is generally more efficient. If an arm produces zero usable results, this metric is `null` rather than infinity. + +## Rotations per usable result + +```text +rotations / usable results +``` + +This metric is emitted only when `rotated` coverage is 100%. Partial coverage is reported, but no complete-arm ratio is invented. + +## Cost per usable result + +```text +total cost units / usable results +``` + +`cost_units` can represent currency, provider credits, bandwidth-equivalent units, or another consistent operator-defined unit. + +The ratio is emitted only when cost coverage is 100%. Do not compare arms that use different cost units. + +## Latency + +`proxybench` reports p50 and p95 over supplied non-negative `latency_ms` values using linear interpolation between ordered samples. + +Latency coverage is always reported so partial instrumentation remains visible. + +## Outcome counts + +The optional `outcome` field is an uppercase categorical token. It is intended for sanitized categories such as `SUCCESS`, `HTTP_RATE_LIMIT`, or `PROXY_PATH_FAILURE`—not target names, URLs, provider endpoints, or private identifiers. + +## A/B deltas + +Candidate deltas are reported relative to baseline: + +- usable success uplift in percentage points; +- usable success relative change in percent; +- requests-per-usable-result percent change; +- rotations-per-usable-result percent change; +- cost-per-usable-result percent change; +- p95 latency percent change. + +A negative change is favorable for ratios where lower is better. + +## Gate semantics + +Gates are operator-defined acceptance criteria. For example: + +```text +min success uplift >= 2 percentage points +requests / usable result regression <= 5% +cost / usable result regression <= 5% +``` + +Possible verdicts: + +- `PASS` — all configured gates are evaluable and pass; +- `FAIL` — at least one configured gate fails and none is missing; +- `INCONCLUSIVE` — minimum sample count is not met or a required metric lacks full coverage; +- `NO_GATES_CONFIGURED` — comparison is descriptive only. + +These verdicts are operational policy results, not hypothesis-test or causal conclusions. diff --git a/docs/PROTOCOL.md b/docs/PROTOCOL.md new file mode 100644 index 0000000..aefcca5 --- /dev/null +++ b/docs/PROTOCOL.md @@ -0,0 +1,65 @@ +# Benchmark protocol + +Use this protocol when comparing a baseline routing/rotation policy against a candidate policy. + +## 1. Freeze the definition of usable + +Define `usable=true` before collecting either arm. Examples might include a parsed record, a validated page state, or another workload-specific success criterion. + +Do not change that definition between arms. + +## 2. Keep workload conditions comparable + +Where practical, compare arms over similar target classes, request mixes, time windows, and instrumentation. Record meaningful differences outside the public fixture if they are sensitive. + +`proxybench` does not make a causal claim when conditions differ. + +## 3. Sanitize before export + +Export only the public event schema: + +```text +usable +latency_ms +cost_units +rotated +outcome +``` + +Do not export URLs, IPs, credentials, cookies, headers, payloads, customer identifiers, provider account data, or infrastructure details. + +## 4. Preserve missingness + +If latency, rotation, or cost was not measured for an event, omit the field or use `null`. + +Do not silently fill missing values with zero or `false`. Coverage is part of the benchmark result. + +## 5. Compare operator KPIs + +Prioritize: + +1. usable success rate; +2. requests per usable result; +3. cost per usable result; +4. rotations per usable result; +5. p95 latency. + +Proxy count by itself is not an outcome metric. + +## 6. Set gates before looking at the result + +When possible, choose acceptance thresholds before running the candidate arm. This reduces post-hoc goal shifting. + +Example: + +```text +minimum usable-success uplift: +2 percentage points +maximum requests/usable regression: +5% +maximum cost/usable regression: +5% +``` + +## 7. Treat statistical output conservatively + +The Wilson interval is descriptive. Requests from the same session, target, account, subnet, or time window may be correlated. + +For production decisions, repeat the benchmark over multiple independent runs or periods where feasible and preserve rollback criteria. diff --git a/docs/PUBLIC-BOUNDARY.md b/docs/PUBLIC-BOUNDARY.md new file mode 100644 index 0000000..8dae499 --- /dev/null +++ b/docs/PUBLIC-BOUNDARY.md @@ -0,0 +1,34 @@ +# Public boundary + +`proxybench` is intentionally a transparent measurement baseline, not a source dump of PN Labs commercial systems. + +## Included publicly + +- sanitized benchmark event schema; +- deterministic summary metrics; +- descriptive Wilson interval for usable success rate; +- operator-defined A/B gate engine; +- JSON CLI; +- synthetic example fixtures; +- tests, documentation, and CI; +- public security/hygiene controls. + +## Intentionally excluded + +This repository does not contain: + +- live proxy routing or traffic orchestration; +- provider selection logic or provider-private configuration; +- target × egress historical learning; +- private scoring weights or optimization heuristics; +- promotion/demotion intelligence; +- automatic provider purchasing or account logic; +- production endpoints, IP addresses, credentials, tokens, cookies, or private hostnames; +- infrastructure topology, deployment secrets, operational runbooks, or incident data; +- non-public benchmark datasets, customer data, or private design-partner material. + +## Contribution rule + +Contributions should improve the public measurement baseline without requiring disclosure of private infrastructure, production traffic, third-party secrets, or commercial decision logic. + +If a change needs sensitive evidence, reduce it to a synthetic reproducer or sanitized aggregate result before opening a public issue or pull request. diff --git a/examples/baseline.jsonl b/examples/baseline.jsonl new file mode 100644 index 0000000..52fab1f --- /dev/null +++ b/examples/baseline.jsonl @@ -0,0 +1,20 @@ +{"usable":true,"latency_ms":410,"cost_units":0.01,"rotated":false,"outcome":"SUCCESS"} +{"usable":false,"latency_ms":520,"cost_units":0.01,"rotated":true,"outcome":"HTTP_RATE_LIMIT"} +{"usable":true,"latency_ms":390,"cost_units":0.01,"rotated":false,"outcome":"SUCCESS"} +{"usable":false,"latency_ms":610,"cost_units":0.01,"rotated":true,"outcome":"HTTP_ACCESS_DENIED"} +{"usable":true,"latency_ms":430,"cost_units":0.01,"rotated":false,"outcome":"SUCCESS"} +{"usable":true,"latency_ms":455,"cost_units":0.01,"rotated":true,"outcome":"SUCCESS"} +{"usable":false,"latency_ms":720,"cost_units":0.01,"rotated":true,"outcome":"HTTP_5XX"} +{"usable":true,"latency_ms":405,"cost_units":0.01,"rotated":false,"outcome":"SUCCESS"} +{"usable":true,"latency_ms":440,"cost_units":0.01,"rotated":false,"outcome":"SUCCESS"} +{"usable":false,"latency_ms":580,"cost_units":0.01,"rotated":true,"outcome":"HTTP_RATE_LIMIT"} +{"usable":true,"latency_ms":395,"cost_units":0.01,"rotated":false,"outcome":"SUCCESS"} +{"usable":true,"latency_ms":465,"cost_units":0.01,"rotated":true,"outcome":"SUCCESS"} +{"usable":false,"latency_ms":640,"cost_units":0.01,"rotated":true,"outcome":"HTTP_REDIRECT"} +{"usable":true,"latency_ms":420,"cost_units":0.01,"rotated":false,"outcome":"SUCCESS"} +{"usable":false,"latency_ms":690,"cost_units":0.01,"rotated":true,"outcome":"HTTP_ACCESS_DENIED"} +{"usable":true,"latency_ms":450,"cost_units":0.01,"rotated":false,"outcome":"SUCCESS"} +{"usable":true,"latency_ms":400,"cost_units":0.01,"rotated":false,"outcome":"SUCCESS"} +{"usable":false,"latency_ms":560,"cost_units":0.01,"rotated":true,"outcome":"HTTP_RATE_LIMIT"} +{"usable":true,"latency_ms":435,"cost_units":0.01,"rotated":false,"outcome":"SUCCESS"} +{"usable":true,"latency_ms":445,"cost_units":0.01,"rotated":false,"outcome":"SUCCESS"} diff --git a/examples/candidate.jsonl b/examples/candidate.jsonl new file mode 100644 index 0000000..fd223ad --- /dev/null +++ b/examples/candidate.jsonl @@ -0,0 +1,20 @@ +{"usable":true,"latency_ms":360,"cost_units":0.009,"rotated":false,"outcome":"SUCCESS"} +{"usable":true,"latency_ms":390,"cost_units":0.009,"rotated":false,"outcome":"SUCCESS"} +{"usable":true,"latency_ms":350,"cost_units":0.009,"rotated":false,"outcome":"SUCCESS"} +{"usable":false,"latency_ms":500,"cost_units":0.009,"rotated":true,"outcome":"HTTP_ACCESS_DENIED"} +{"usable":true,"latency_ms":370,"cost_units":0.009,"rotated":false,"outcome":"SUCCESS"} +{"usable":true,"latency_ms":405,"cost_units":0.009,"rotated":false,"outcome":"SUCCESS"} +{"usable":false,"latency_ms":540,"cost_units":0.009,"rotated":true,"outcome":"HTTP_5XX"} +{"usable":true,"latency_ms":365,"cost_units":0.009,"rotated":false,"outcome":"SUCCESS"} +{"usable":true,"latency_ms":380,"cost_units":0.009,"rotated":false,"outcome":"SUCCESS"} +{"usable":false,"latency_ms":470,"cost_units":0.009,"rotated":true,"outcome":"HTTP_RATE_LIMIT"} +{"usable":true,"latency_ms":355,"cost_units":0.009,"rotated":false,"outcome":"SUCCESS"} +{"usable":true,"latency_ms":395,"cost_units":0.009,"rotated":false,"outcome":"SUCCESS"} +{"usable":false,"latency_ms":515,"cost_units":0.009,"rotated":true,"outcome":"HTTP_REDIRECT"} +{"usable":true,"latency_ms":375,"cost_units":0.009,"rotated":false,"outcome":"SUCCESS"} +{"usable":true,"latency_ms":410,"cost_units":0.009,"rotated":false,"outcome":"SUCCESS"} +{"usable":true,"latency_ms":385,"cost_units":0.009,"rotated":false,"outcome":"SUCCESS"} +{"usable":true,"latency_ms":345,"cost_units":0.009,"rotated":false,"outcome":"SUCCESS"} +{"usable":false,"latency_ms":495,"cost_units":0.009,"rotated":true,"outcome":"HTTP_RATE_LIMIT"} +{"usable":true,"latency_ms":400,"cost_units":0.009,"rotated":false,"outcome":"SUCCESS"} +{"usable":true,"latency_ms":365,"cost_units":0.009,"rotated":false,"outcome":"SUCCESS"} diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..1923756 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,22 @@ +[build-system] +requires = ["setuptools>=69"] +build-backend = "setuptools.build_meta" + +[project] +name = "proxybench" +version = "0.1.0" +description = "Evidence-first benchmarking for proxy and web-retrieval workloads" +readme = "README.md" +requires-python = ">=3.10" +license = {text = "MIT"} +authors = [{name = "PN Labs"}] +dependencies = [] + +[project.scripts] +proxybench = "proxybench.cli:main" + +[tool.setuptools] +package-dir = {"" = "src"} + +[tool.setuptools.packages.find] +where = ["src"] diff --git a/scripts/public_hygiene_check.py b/scripts/public_hygiene_check.py new file mode 100644 index 0000000..abe0198 --- /dev/null +++ b/scripts/public_hygiene_check.py @@ -0,0 +1,98 @@ +from __future__ import annotations + +import pathlib +import re +import sys + + +ROOT = pathlib.Path(__file__).resolve().parents[1] +TEXT_SUFFIXES = {".md", ".py", ".toml", ".yml", ".yaml", ".txt", ".json", ".jsonl"} + +# The checker reports only the rule and file path. It never prints a matched +# value, which avoids turning CI logs into a secondary disclosure path. +# Rules intentionally use vendor-neutral labels and context-aware secret shapes. +RULES: tuple[tuple[str, re.Pattern[str]], ...] = ( + ("private-key material", re.compile("-----BEGIN " + r"(?:RSA |EC |OPENSSH )?PRIVATE KEY-----")), + ("cloud access-key shaped token", re.compile(r"\b[A-Z]{4}[0-9A-Z]{16}\b")), + ( + "credential assignment shaped value", + re.compile( + r"(?i)\b(?:api[_-]?key|access[_-]?token|auth[_-]?token|secret|password)" + r"\s*[:=]\s*[\"'][A-Za-z0-9_./+=-]{16,}[\"']" + ), + ), + ( + "bearer-token shaped value", + re.compile(r"(?i)\bBearer\s+[A-Za-z0-9_./+=-]{20,}"), + ), + ("API key shaped value", re.compile(r"\bsk-(?:[A-Za-z0-9_-]{3,}-)?[A-Za-z0-9_-]{20,}\b")), + ( + "credential-bearing proxy/HTTP URL", + re.compile(r"(?:https?|socks5?)://[^\s/:@]+:[^\s/@]+@", re.IGNORECASE), + ), + ( + "IPv4 literal", + re.compile(r"(? list[pathlib.Path]: + files: list[pathlib.Path] = [] + for path in ROOT.rglob("*"): + if not path.is_file(): + continue + if ".git" in path.parts: + continue + if path.suffix.lower() in TEXT_SUFFIXES or path.name == "LICENSE": + files.append(path) + return files + + +def main() -> int: + failures: list[tuple[str, str]] = [] + + for path in iter_text_files(): + rel = path.relative_to(ROOT).as_posix() + if path.name in BANNED_FILENAMES: + failures.append((rel, "banned sensitive filename")) + continue + + text = path.read_text(encoding="utf-8", errors="replace") + for label, pattern in RULES: + if pattern.search(text): + failures.append((rel, label)) + + if failures: + print("public hygiene check: FAIL") + for rel, label in failures: + print(f"- {rel}: {label}") + print("Matched values are intentionally suppressed.") + return 1 + + print("public hygiene check: PASS") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/src/proxybench/__init__.py b/src/proxybench/__init__.py new file mode 100644 index 0000000..18f6291 --- /dev/null +++ b/src/proxybench/__init__.py @@ -0,0 +1,13 @@ +from .core import compare_summaries, parse_event, summarize, summarize_jsonl, wilson_interval +from .models import Event, InputError, Summary + +__all__ = [ + "Event", + "InputError", + "Summary", + "compare_summaries", + "parse_event", + "summarize", + "summarize_jsonl", + "wilson_interval", +] diff --git a/src/proxybench/__main__.py b/src/proxybench/__main__.py new file mode 100644 index 0000000..2f05ddc --- /dev/null +++ b/src/proxybench/__main__.py @@ -0,0 +1,5 @@ +from .cli import main + + +if __name__ == "__main__": + main() diff --git a/src/proxybench/cli.py b/src/proxybench/cli.py new file mode 100644 index 0000000..ac296e2 --- /dev/null +++ b/src/proxybench/cli.py @@ -0,0 +1,80 @@ +from __future__ import annotations + +import argparse +import json +import sys + +from .core import compare_summaries, summarize_jsonl +from .models import InputError + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser( + prog="proxybench", + description="Benchmark sanitized proxy/web-retrieval outcomes without network I/O.", + ) + sub = parser.add_subparsers(dest="command", required=True) + + summarize_cmd = sub.add_parser("summarize", help="summarize one JSONL benchmark arm") + summarize_cmd.add_argument("input") + + compare_cmd = sub.add_parser("compare", help="compare baseline and candidate JSONL arms") + compare_cmd.add_argument("baseline") + compare_cmd.add_argument("candidate") + compare_cmd.add_argument("--min-requests", type=int, default=1) + compare_cmd.add_argument("--min-success-uplift-pp", type=float) + compare_cmd.add_argument("--max-rpu-regression-pct", type=float) + compare_cmd.add_argument("--max-cost-regression-pct", type=float) + compare_cmd.add_argument("--max-p95-latency-regression-pct", type=float) + + return parser + + +def _validate_thresholds(args: argparse.Namespace) -> None: + if getattr(args, "min_requests", 1) < 1: + raise InputError("min_requests must be >= 1") + + for name in ( + "max_rpu_regression_pct", + "max_cost_regression_pct", + "max_p95_latency_regression_pct", + ): + value = getattr(args, name, None) + if value is not None and value < 0: + raise InputError(f"{name} must be >= 0") + + +def run(argv: list[str] | None = None) -> int: + args = build_parser().parse_args(argv) + + try: + _validate_thresholds(args) + + if args.command == "summarize": + payload = summarize_jsonl(args.input).to_dict() + else: + baseline = summarize_jsonl(args.baseline) + candidate = summarize_jsonl(args.candidate) + payload = compare_summaries( + baseline, + candidate, + min_requests=args.min_requests, + min_success_uplift_pp=args.min_success_uplift_pp, + max_rpu_regression_pct=args.max_rpu_regression_pct, + max_cost_regression_pct=args.max_cost_regression_pct, + max_p95_latency_regression_pct=args.max_p95_latency_regression_pct, + ) + except (InputError, OSError): + print("proxybench: input validation failed", file=sys.stderr) + return 2 + + print(json.dumps(payload, indent=2, sort_keys=True)) + return 0 + + +def main() -> None: + raise SystemExit(run()) + + +if __name__ == "__main__": + main() diff --git a/src/proxybench/core.py b/src/proxybench/core.py new file mode 100644 index 0000000..c6c20fa --- /dev/null +++ b/src/proxybench/core.py @@ -0,0 +1,329 @@ +from __future__ import annotations + +import json +import math +import pathlib +import re +from collections import Counter +from collections.abc import Iterable, Mapping + +from .models import Event, InputError, Summary + + +_ALLOWED_FIELDS = {"usable", "latency_ms", "cost_units", "rotated", "outcome"} +_OUTCOME_RE = re.compile(r"^[A-Z][A-Z0-9_]{0,63}$") +_Z95 = 1.959963984540054 + + +def _round(value: float | None, digits: int = 6) -> float | None: + if value is None: + return None + return round(value, digits) + + +def _finite_nonnegative(value: object, *, line_number: int, field: str) -> float: + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise InputError(f"line {line_number}: {field} must be a non-negative number") + number = float(value) + if not math.isfinite(number) or number < 0: + raise InputError(f"line {line_number}: {field} must be a finite non-negative number") + return number + + +def parse_event(raw: object, *, line_number: int) -> Event: + """Parse one sanitized benchmark event without echoing raw input values.""" + if not isinstance(raw, Mapping): + raise InputError(f"line {line_number}: event must be a JSON object") + + if set(raw) - _ALLOWED_FIELDS: + raise InputError(f"line {line_number}: event contains unsupported fields") + + if "usable" not in raw or not isinstance(raw["usable"], bool): + raise InputError(f"line {line_number}: usable must be a boolean") + + latency_ms = None + if "latency_ms" in raw and raw["latency_ms"] is not None: + latency_ms = _finite_nonnegative( + raw["latency_ms"], line_number=line_number, field="latency_ms" + ) + + cost_units = None + if "cost_units" in raw and raw["cost_units"] is not None: + cost_units = _finite_nonnegative( + raw["cost_units"], line_number=line_number, field="cost_units" + ) + + rotated = None + if "rotated" in raw and raw["rotated"] is not None: + if not isinstance(raw["rotated"], bool): + raise InputError(f"line {line_number}: rotated must be a boolean") + rotated = raw["rotated"] + + outcome = None + if "outcome" in raw and raw["outcome"] is not None: + if not isinstance(raw["outcome"], str) or not _OUTCOME_RE.fullmatch(raw["outcome"]): + raise InputError( + f"line {line_number}: outcome must be an uppercase categorical token" + ) + outcome = raw["outcome"] + + return Event( + usable=raw["usable"], + latency_ms=latency_ms, + cost_units=cost_units, + rotated=rotated, + outcome=outcome, + ) + + +def iter_jsonl(path: str | pathlib.Path) -> Iterable[Event]: + """Yield validated events from JSONL while suppressing raw-line echo in errors.""" + with pathlib.Path(path).open("r", encoding="utf-8") as handle: + for line_number, line in enumerate(handle, start=1): + if not line.strip(): + continue + try: + raw = json.loads(line) + except json.JSONDecodeError as exc: + raise InputError(f"line {line_number}: invalid JSON") from exc + yield parse_event(raw, line_number=line_number) + + +def wilson_interval(successes: int, total: int, *, z: float = _Z95) -> tuple[float, float] | None: + """Return a descriptive Wilson score interval for a binomial proportion.""" + if total <= 0: + return None + p = successes / total + denominator = 1 + (z * z / total) + center = (p + z * z / (2 * total)) / denominator + margin = ( + z + * math.sqrt((p * (1 - p) + z * z / (4 * total)) / total) + / denominator + ) + return (_round(max(0.0, center - margin)), _round(min(1.0, center + margin))) + + +def _quantile(values: list[float], q: float) -> float | None: + if not values: + return None + ordered = sorted(values) + if len(ordered) == 1: + return ordered[0] + position = (len(ordered) - 1) * q + lower = math.floor(position) + upper = math.ceil(position) + if lower == upper: + return ordered[lower] + fraction = position - lower + return ordered[lower] + (ordered[upper] - ordered[lower]) * fraction + + +def summarize(events: Iterable[Event]) -> Summary: + requests = 0 + usable = 0 + rotations = 0 + rotation_seen = 0 + latencies: list[float] = [] + total_cost = 0.0 + cost_seen = 0 + outcome_seen = 0 + outcomes: Counter[str] = Counter() + + for event in events: + requests += 1 + usable += int(event.usable) + + if event.rotated is not None: + rotation_seen += 1 + rotations += int(event.rotated) + + if event.latency_ms is not None: + latencies.append(event.latency_ms) + + if event.cost_units is not None: + cost_seen += 1 + total_cost += event.cost_units + + if event.outcome is not None: + outcome_seen += 1 + outcomes[event.outcome] += 1 + + if requests == 0: + raise InputError("input contains no benchmark events") + + success_rate = usable / requests + rpu = requests / usable if usable else None + + full_rotation_coverage = rotation_seen == requests + rotations_per_usable = ( + rotations / usable if usable and full_rotation_coverage else None + ) + + full_cost_coverage = cost_seen == requests + cost_per_usable = ( + total_cost / usable if usable and full_cost_coverage else None + ) + + return Summary( + requests=requests, + usable_results=usable, + usable_success_rate=_round(success_rate) or 0.0, + usable_success_rate_wilson95=wilson_interval(usable, requests), + requests_per_usable_result=_round(rpu), + rotation_coverage=_round(rotation_seen / requests) or 0.0, + rotations=rotations, + rotations_per_usable_result=_round(rotations_per_usable), + latency_coverage=_round(len(latencies) / requests) or 0.0, + latency_ms_p50=_round(_quantile(latencies, 0.50)), + latency_ms_p95=_round(_quantile(latencies, 0.95)), + cost_coverage=_round(cost_seen / requests) or 0.0, + total_cost_units=_round(total_cost) if cost_seen else None, + cost_units_per_usable_result=_round(cost_per_usable), + outcome_coverage=_round(outcome_seen / requests) or 0.0, + outcome_counts=dict(sorted(outcomes.items())), + ) + + +def summarize_jsonl(path: str | pathlib.Path) -> Summary: + return summarize(iter_jsonl(path)) + + +def _pct_change(baseline: float | None, candidate: float | None) -> float | None: + if baseline is None or candidate is None or baseline == 0: + return None + return _round(((candidate / baseline) - 1) * 100) + + +def compare_summaries( + baseline: Summary, + candidate: Summary, + *, + min_requests: int = 1, + min_success_uplift_pp: float | None = None, + max_rpu_regression_pct: float | None = None, + max_cost_regression_pct: float | None = None, + max_p95_latency_regression_pct: float | None = None, +) -> dict[str, object]: + """Compare two summaries and optionally evaluate operator-defined gates.""" + if min_requests < 1: + raise ValueError("min_requests must be >= 1") + + success_delta_pp = _round( + (candidate.usable_success_rate - baseline.usable_success_rate) * 100 + ) + success_relative_pct = _pct_change( + baseline.usable_success_rate, candidate.usable_success_rate + ) + rpu_change_pct = _pct_change( + baseline.requests_per_usable_result, + candidate.requests_per_usable_result, + ) + cost_change_pct = _pct_change( + baseline.cost_units_per_usable_result, + candidate.cost_units_per_usable_result, + ) + rotation_change_pct = _pct_change( + baseline.rotations_per_usable_result, + candidate.rotations_per_usable_result, + ) + p95_latency_change_pct = _pct_change( + baseline.latency_ms_p95, + candidate.latency_ms_p95, + ) + + configured = any( + value is not None + for value in ( + min_success_uplift_pp, + max_rpu_regression_pct, + max_cost_regression_pct, + max_p95_latency_regression_pct, + ) + ) + + checks: dict[str, dict[str, object]] = {} + verdict = "NO_GATES_CONFIGURED" + + if configured: + if baseline.requests < min_requests or candidate.requests < min_requests: + verdict = "INCONCLUSIVE" + else: + inconclusive = False + failed = False + + def add_check(name: str, actual: float | None, threshold: float, *, mode: str) -> None: + nonlocal inconclusive, failed + if actual is None: + checks[name] = { + "actual": None, + "threshold": threshold, + "result": "INCONCLUSIVE", + } + inconclusive = True + return + passed = actual >= threshold if mode == "min" else actual <= threshold + checks[name] = { + "actual": actual, + "threshold": threshold, + "result": "PASS" if passed else "FAIL", + } + failed = failed or not passed + + if min_success_uplift_pp is not None: + add_check( + "min_success_uplift_pp", + success_delta_pp, + min_success_uplift_pp, + mode="min", + ) + if max_rpu_regression_pct is not None: + add_check( + "max_rpu_regression_pct", + rpu_change_pct, + max_rpu_regression_pct, + mode="max", + ) + if max_cost_regression_pct is not None: + add_check( + "max_cost_regression_pct", + cost_change_pct, + max_cost_regression_pct, + mode="max", + ) + if max_p95_latency_regression_pct is not None: + add_check( + "max_p95_latency_regression_pct", + p95_latency_change_pct, + max_p95_latency_regression_pct, + mode="max", + ) + + if inconclusive: + verdict = "INCONCLUSIVE" + elif failed: + verdict = "FAIL" + else: + verdict = "PASS" + + return { + "baseline": baseline.to_dict(), + "candidate": candidate.to_dict(), + "delta": { + "usable_success_uplift_pp": success_delta_pp, + "usable_success_relative_pct": success_relative_pct, + "requests_per_usable_result_change_pct": rpu_change_pct, + "rotations_per_usable_result_change_pct": rotation_change_pct, + "cost_per_usable_result_change_pct": cost_change_pct, + "latency_p95_change_pct": p95_latency_change_pct, + }, + "gate": { + "min_requests_per_arm": min_requests, + "verdict": verdict, + "checks": checks, + }, + "interpretation_note": ( + "Metrics and Wilson intervals are descriptive; request events may be correlated " + "and this output is not a causal or independence claim." + ), + } diff --git a/src/proxybench/models.py b/src/proxybench/models.py new file mode 100644 index 0000000..ff620f6 --- /dev/null +++ b/src/proxybench/models.py @@ -0,0 +1,45 @@ +from __future__ import annotations + +from dataclasses import asdict, dataclass +from typing import Any + + +class InputError(ValueError): + """Raised when sanitized benchmark input does not satisfy the public schema.""" + + +@dataclass(frozen=True) +class Event: + usable: bool + latency_ms: float | None = None + cost_units: float | None = None + rotated: bool | None = None + outcome: str | None = None + + +@dataclass(frozen=True) +class Summary: + requests: int + usable_results: int + usable_success_rate: float + usable_success_rate_wilson95: tuple[float, float] | None + requests_per_usable_result: float | None + rotation_coverage: float + rotations: int + rotations_per_usable_result: float | None + latency_coverage: float + latency_ms_p50: float | None + latency_ms_p95: float | None + cost_coverage: float + total_cost_units: float | None + cost_units_per_usable_result: float | None + outcome_coverage: float + outcome_counts: dict[str, int] + + def to_dict(self) -> dict[str, Any]: + data = asdict(self) + interval = self.usable_success_rate_wilson95 + data["usable_success_rate_wilson95"] = ( + list(interval) if interval is not None else None + ) + return data diff --git a/tests/test_cli.py b/tests/test_cli.py new file mode 100644 index 0000000..47d03a1 --- /dev/null +++ b/tests/test_cli.py @@ -0,0 +1,78 @@ +import contextlib +import io +import json +import pathlib +import tempfile +import unittest + +from proxybench.cli import run + + +class CliTests(unittest.TestCase): + def test_summarize_outputs_json(self): + with tempfile.TemporaryDirectory() as tmp: + path = pathlib.Path(tmp) / "events.jsonl" + path.write_text('{"usable": true}\n{"usable": false}\n', encoding="utf-8") + stdout = io.StringIO() + with contextlib.redirect_stdout(stdout): + code = run(["summarize", str(path)]) + self.assertEqual(code, 0) + payload = json.loads(stdout.getvalue()) + self.assertEqual(payload["requests"], 2) + self.assertEqual(payload["usable_results"], 1) + + def test_compare_outputs_gate_verdict(self): + with tempfile.TemporaryDirectory() as tmp: + baseline = pathlib.Path(tmp) / "baseline.jsonl" + candidate = pathlib.Path(tmp) / "candidate.jsonl" + baseline.write_text( + '{"usable": true}\n{"usable": false}\n{"usable": false}\n{"usable": false}\n', + encoding="utf-8", + ) + candidate.write_text( + '{"usable": true}\n{"usable": true}\n{"usable": false}\n{"usable": false}\n', + encoding="utf-8", + ) + stdout = io.StringIO() + with contextlib.redirect_stdout(stdout): + code = run( + [ + "compare", + str(baseline), + str(candidate), + "--min-requests", + "4", + "--min-success-uplift-pp", + "20", + ] + ) + self.assertEqual(code, 0) + payload = json.loads(stdout.getvalue()) + self.assertEqual(payload["gate"]["verdict"], "PASS") + + def test_validation_error_does_not_echo_input(self): + marker = "DO-NOT-ECHO-SECRET-CONTEXT" + with tempfile.TemporaryDirectory() as tmp: + path = pathlib.Path(tmp) / "events.jsonl" + path.write_text( + '{"usable": true, "private_context": "' + marker + '"}\n', + encoding="utf-8", + ) + stderr = io.StringIO() + with contextlib.redirect_stderr(stderr): + code = run(["summarize", str(path)]) + self.assertEqual(code, 2) + self.assertNotIn(marker, stderr.getvalue()) + self.assertEqual(stderr.getvalue().strip(), "proxybench: input validation failed") + + def test_missing_file_does_not_echo_path(self): + sensitive_path = "/private/customer-secret/events.jsonl" + stderr = io.StringIO() + with contextlib.redirect_stderr(stderr): + code = run(["summarize", sensitive_path]) + self.assertEqual(code, 2) + self.assertNotIn(sensitive_path, stderr.getvalue()) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_core.py b/tests/test_core.py new file mode 100644 index 0000000..24a2881 --- /dev/null +++ b/tests/test_core.py @@ -0,0 +1,133 @@ +import io +import json +import math +import pathlib +import tempfile +import unittest + +from proxybench import Event, InputError, compare_summaries, parse_event, summarize, summarize_jsonl, wilson_interval + + +class CoreTests(unittest.TestCase): + def test_parse_minimal_event(self): + event = parse_event({"usable": True}, line_number=1) + self.assertEqual(event, Event(usable=True)) + + def test_unknown_fields_are_rejected(self): + with self.assertRaises(InputError): + parse_event({"usable": True, "url": "https://example.invalid"}, line_number=1) + + def test_private_context_fields_are_not_part_of_schema(self): + for field in ("proxy_ip", "provider", "cookie", "authorization", "target"): + with self.subTest(field=field): + with self.assertRaises(InputError): + parse_event({"usable": True, field: "redacted"}, line_number=1) + + def test_non_finite_values_are_rejected(self): + for value in (math.inf, -math.inf, math.nan): + with self.subTest(value=value): + with self.assertRaises(InputError): + parse_event({"usable": True, "latency_ms": value}, line_number=1) + + def test_negative_cost_is_rejected(self): + with self.assertRaises(InputError): + parse_event({"usable": True, "cost_units": -1}, line_number=1) + + def test_outcome_must_be_categorical_token(self): + with self.assertRaises(InputError): + parse_event({"usable": True, "outcome": "https://private.invalid/x"}, line_number=1) + + def test_summary_metrics(self): + summary = summarize( + [ + Event(True, latency_ms=100, cost_units=1, rotated=False, outcome="SUCCESS"), + Event(False, latency_ms=200, cost_units=1, rotated=True, outcome="HTTP_RATE_LIMIT"), + Event(True, latency_ms=300, cost_units=1, rotated=False, outcome="SUCCESS"), + Event(False, latency_ms=400, cost_units=1, rotated=True, outcome="HTTP_5XX"), + ] + ) + self.assertEqual(summary.requests, 4) + self.assertEqual(summary.usable_results, 2) + self.assertEqual(summary.usable_success_rate, 0.5) + self.assertEqual(summary.requests_per_usable_result, 2.0) + self.assertEqual(summary.rotations, 2) + self.assertEqual(summary.rotations_per_usable_result, 1.0) + self.assertEqual(summary.latency_ms_p50, 250.0) + self.assertEqual(summary.latency_ms_p95, 385.0) + self.assertEqual(summary.total_cost_units, 4.0) + self.assertEqual(summary.cost_units_per_usable_result, 2.0) + self.assertEqual(summary.outcome_counts["SUCCESS"], 2) + + def test_partial_cost_coverage_does_not_invent_cost_per_success(self): + summary = summarize([Event(True, cost_units=1), Event(False)]) + self.assertEqual(summary.cost_coverage, 0.5) + self.assertIsNone(summary.cost_units_per_usable_result) + + def test_partial_rotation_coverage_does_not_invent_rotation_rate(self): + summary = summarize([Event(True, rotated=True), Event(True)]) + self.assertEqual(summary.rotation_coverage, 0.5) + self.assertIsNone(summary.rotations_per_usable_result) + + def test_empty_input_is_rejected(self): + with self.assertRaises(InputError): + summarize([]) + + def test_wilson_interval_is_bounded(self): + interval = wilson_interval(5, 10) + self.assertIsNotNone(interval) + low, high = interval + self.assertGreaterEqual(low, 0) + self.assertLessEqual(high, 1) + self.assertLess(low, 0.5) + self.assertGreater(high, 0.5) + + def test_compare_directional_metrics_and_pass_gate(self): + baseline = summarize([Event(True), Event(False), Event(False), Event(False)]) + candidate = summarize([Event(True), Event(True), Event(False), Event(False)]) + result = compare_summaries( + baseline, + candidate, + min_requests=4, + min_success_uplift_pp=20, + max_rpu_regression_pct=0, + ) + self.assertEqual(result["delta"]["usable_success_uplift_pp"], 25.0) + self.assertEqual(result["gate"]["verdict"], "PASS") + + def test_compare_missing_cost_is_inconclusive_when_cost_gate_enabled(self): + baseline = summarize([Event(True), Event(False)]) + candidate = summarize([Event(True), Event(True)]) + result = compare_summaries( + baseline, + candidate, + max_cost_regression_pct=5, + ) + self.assertEqual(result["gate"]["verdict"], "INCONCLUSIVE") + + def test_compare_without_thresholds_does_not_claim_pass(self): + baseline = summarize([Event(True)]) + candidate = summarize([Event(True)]) + result = compare_summaries(baseline, candidate) + self.assertEqual(result["gate"]["verdict"], "NO_GATES_CONFIGURED") + + def test_jsonl_error_does_not_include_raw_line(self): + marker = "DO-NOT-ECHO-THIS-SENSITIVE-VALUE" + with tempfile.TemporaryDirectory() as tmp: + path = pathlib.Path(tmp) / "events.jsonl" + path.write_text('{"usable": true, "x": "' + marker + '"}\n', encoding="utf-8") + with self.assertRaises(InputError) as caught: + summarize_jsonl(path) + self.assertNotIn(marker, str(caught.exception)) + + def test_invalid_json_error_does_not_include_raw_line(self): + marker = "DO-NOT-ECHO-INVALID-LINE" + with tempfile.TemporaryDirectory() as tmp: + path = pathlib.Path(tmp) / "events.jsonl" + path.write_text("{" + marker + "\n", encoding="utf-8") + with self.assertRaises(InputError) as caught: + summarize_jsonl(path) + self.assertNotIn(marker, str(caught.exception)) + + +if __name__ == "__main__": + unittest.main()