diff --git a/eng/pipelines/pr-validation-pipeline.yml b/eng/pipelines/pr-validation-pipeline.yml index 6e5038d89..85e3e2ae5 100644 --- a/eng/pipelines/pr-validation-pipeline.yml +++ b/eng/pipelines/pr-validation-pipeline.yml @@ -468,9 +468,7 @@ jobs: - job: PytestOnMacOS displayName: 'macOS x86_64' - # Reserve 60 minutes outside the 100-minute benchmark step for Colima/SQL - # setup, pytest, fixture restore and artifact publication. - timeoutInMinutes: 160 + timeoutInMinutes: 90 pool: vmImage: 'macos-latest' @@ -568,11 +566,7 @@ jobs: pip install -r requirements.txt echo "Building pybind bindings (.so) (overlapped with container setup)..." - PROFILER_BUILD=0 - if [ "$(Build.Reason)" = "PullRequest" ]; then - PROFILER_BUILD=1 - fi - ( cd mssql_python/pybind && ENABLE_PROFILING="$PROFILER_BUILD" ./build.sh ) + ( cd mssql_python/pybind && ./build.sh ) echo "Waiting for container setup (Colima + SQL Server) to finish..." SQL_STATUS=0 @@ -585,10 +579,6 @@ jobs: env: DB_PASSWORD: $(DB_PASSWORD) - - script: python -m eng.profiler_benchmarks.controller --check-build on - displayName: 'Verify native configuration and recording OFF before pytest' - condition: and(succeeded(), eq(variables['Build.Reason'], 'PullRequest')) - - template: steps/install-mssql-py-core.yml parameters: platform: unix @@ -637,59 +627,9 @@ jobs: testResultsFiles: '**/test-results.xml' testRunTitle: 'Publish pytest results on macOS $(sqlVersion)' - # Download and restore AdventureWorks2022 database for benchmarking - - script: | - echo "Downloading AdventureWorks2022.bak..." - curl -sSL -o /tmp/AdventureWorks2022.bak \ - https://github.com/Microsoft/sql-server-samples/releases/download/adventureworks/AdventureWorks2022.bak - - echo "Copying backup into SQL Server container..." - docker cp /tmp/AdventureWorks2022.bak sqlserver:/tmp/AdventureWorks2022.bak - - echo "Restoring AdventureWorks2022 database..." - docker exec sqlserver /opt/mssql-tools18/bin/sqlcmd \ - -S localhost -U SA -P "$DB_PASSWORD" -C \ - -Q "RESTORE DATABASE AdventureWorks2022 FROM DISK = '/tmp/AdventureWorks2022.bak' WITH MOVE 'AdventureWorks2022' TO '/var/opt/mssql/data/AdventureWorks2022.mdf', MOVE 'AdventureWorks2022_log' TO '/var/opt/mssql/data/AdventureWorks2022_log.ldf', REPLACE" - - if [ $? -eq 0 ]; then - echo "AdventureWorks2022 database restored successfully" - else - echo "Failed to restore AdventureWorks2022 database" - fi - - rm -f /tmp/AdventureWorks2022.bak - docker exec sqlserver rm -f /tmp/AdventureWorks2022.bak || true - displayName: 'Download and restore AdventureWorks2022 database on macOS' - condition: and(succeeded(), or(eq(variables['sqlVersion'], 'SQL2022'), eq(variables['sqlVersion'], 'SQL2025'))) - continueOnError: true - env: - DB_PASSWORD: $(DB_PASSWORD) - - # Run performance benchmarks on macOS - - script: | - set -euo pipefail - echo "Restoring build dependencies for isolated profiling builds..." - brew tap microsoft/mssql-release https://github.com/Microsoft/homebrew-mssql-release - # Newer Homebrew refuses to load formulae from third-party taps unless the tap is trusted - brew trust microsoft/mssql-release || echo "brew trust failed; attempting install anyway" - HOMEBREW_ACCEPT_EULA=Y brew install msodbcsql18 - python -m eng.profiler_benchmarks.controller --reuse-candidate --leg "macOS-$(sqlVersion)" --output profiler-results - displayName: 'Compare profiling builds on macOS $(sqlVersion)' - condition: and(succeeded(), eq(variables['Build.Reason'], 'PullRequest'), or(eq(variables['sqlVersion'], 'SQL2022'), eq(variables['sqlVersion'], 'SQL2025'))) - timeoutInMinutes: 100 - continueOnError: true - env: - SYSTEM_PULLREQUEST_SOURCECOMMITID: $(System.PullRequest.SourceCommitId) - DB_CONNECTION_STRING: 'Server=tcp:127.0.0.1,1433;Database=AdventureWorks2022;Uid=SA;Pwd=$(DB_PASSWORD);TrustServerCertificate=yes' - - # Both revisions and samples travel together; no historical baseline lookup. - - task: PublishPipelineArtifact@1 - inputs: - targetPath: profiler-results - artifact: 'profiler-macOS-$(sqlVersion)' - displayName: 'Publish paired profiler measurements' - condition: and(succeededOrFailed(), eq(variables['Build.Reason'], 'PullRequest'), or(eq(variables['sqlVersion'], 'SQL2022'), eq(variables['sqlVersion'], 'SQL2025'))) - continueOnError: true + # Routine PR profiling excludes hosted macOS because Colima-backed control + # runs produced regressions with no product-code change. Functional macOS + # coverage remains above; Unix performance signal comes from Ubuntu. - script: | python3 eng/scripts/setup_sql_container.py --cleanup --colima \ @@ -713,11 +653,13 @@ jobs: distroName: 'Ubuntu' sqlServerImage: 'mcr.microsoft.com/mssql/server:2022-latest' useAzureSQL: 'false' + profilerLeg: 'Linux-SQL2022' Ubuntu_SQL2025: dockerImage: 'ubuntu:24.04' distroName: 'Ubuntu-SQL2025' sqlServerImage: 'mcr.microsoft.com/mssql/server:2025-latest' useAzureSQL: 'false' + profilerLeg: 'Linux-SQL2025' ${{ if ne(variables['AZURE_CONNECTION_STRING'], '') }}: Ubuntu_AzureSQL: dockerImage: 'ubuntu:24.04' @@ -832,7 +774,8 @@ jobs: - script: | # Build pybind bindings in the container PROFILER_BUILD=0 - if [ "$(Build.Reason)" = "PullRequest" ] && [ "$(distroName)" = "Ubuntu" ]; then + if [ "$(Build.Reason)" = "PullRequest" ] && + [[ "$(distroName)" =~ ^Ubuntu(-SQL2025)?$ ]]; then PROFILER_BUILD=1 fi docker exec -e ENABLE_PROFILING="$PROFILER_BUILD" test-container-$(distroName) bash -c " @@ -912,8 +855,8 @@ jobs: DB_PASSWORD: $(DB_PASSWORD) - script: | - # Download and restore AdventureWorks2022 database for benchmarking on Ubuntu only - if [ "$(distroName)" = "Ubuntu" ] && [ "$(useAzureSQL)" = "false" ]; then + # Download and restore AdventureWorks2022 for both Unix profiler legs. + if [[ "$(distroName)" =~ ^Ubuntu(-SQL2025)?$ ]] && [ "$(useAzureSQL)" = "false" ]; then echo "Downloading AdventureWorks2022.bak..." wget -q https://github.com/Microsoft/sql-server-samples/releases/download/adventureworks/AdventureWorks2022.bak -O /tmp/AdventureWorks2022.bak @@ -939,14 +882,16 @@ jobs: docker exec sqlserver-$(distroName) rm -f /tmp/AdventureWorks2022.bak || true fi displayName: 'Download and restore AdventureWorks2022 database in $(distroName)' - condition: and(succeeded(), eq(variables['distroName'], 'Ubuntu'), eq(variables['useAzureSQL'], 'false')) + condition: and(succeeded(), eq(variables['useAzureSQL'], 'false'), or(eq(variables['distroName'], 'Ubuntu'), eq(variables['distroName'], 'Ubuntu-SQL2025'))) continueOnError: true env: DB_PASSWORD: $(DB_PASSWORD) + # Unix profiling uses Ubuntu because it covers the shared Unix hot paths while + # producing stable paired measurements on hosted agents. - script: | - # Run performance benchmarks on Ubuntu with SQL Server 2022 only - if [ "$(distroName)" = "Ubuntu" ] && [ "$(useAzureSQL)" = "false" ]; then + # Run Unix performance benchmarks on Ubuntu with SQL Server 2022 and 2025. + if [[ "$(distroName)" =~ ^Ubuntu(-SQL2025)?$ ]] && [ "$(useAzureSQL)" = "false" ]; then export SQLSERVER_IP=$(docker inspect sqlserver-$(distroName) --format='{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}') echo "Running performance benchmarks on Ubuntu with SQL Server IP: $SQLSERVER_IP" @@ -967,13 +912,13 @@ jobs: ACCEPT_EULA=Y apt-get install -y --no-install-recommends git unixodbc unixodbc-dev libodbc2 libodbcinst2 odbcinst msodbcsql18 git config --global --add safe.directory /workspace odbcinst -q -d -n "ODBC Driver 18 for SQL Server" - python -m eng.profiler_benchmarks.controller --reuse-candidate --leg Linux-SQL2022 --output profiler-results + python -m eng.profiler_benchmarks.controller --reuse-candidate --leg "$(profilerLeg)" --output profiler-results ' else echo "Skipping performance benchmarks on $(distroName) (only runs on Ubuntu with local SQL Server)" fi - displayName: 'Compare profiling builds in $(distroName) container' - condition: and(succeeded(), eq(variables['Build.Reason'], 'PullRequest'), eq(variables['distroName'], 'Ubuntu'), eq(variables['useAzureSQL'], 'false')) + displayName: 'Compare profiling builds on Unix in $(distroName) container' + condition: and(succeeded(), eq(variables['Build.Reason'], 'PullRequest'), eq(variables['useAzureSQL'], 'false'), or(eq(variables['distroName'], 'Ubuntu'), eq(variables['distroName'], 'Ubuntu-SQL2025'))) continueOnError: true timeoutInMinutes: 100 env: @@ -984,9 +929,9 @@ jobs: - task: PublishPipelineArtifact@1 inputs: targetPath: profiler-results - artifact: profiler-Linux-SQL2022 + artifact: profiler-$(profilerLeg) displayName: 'Publish paired profiler measurements' - condition: and(succeededOrFailed(), eq(variables['Build.Reason'], 'PullRequest'), eq(variables['distroName'], 'Ubuntu'), eq(variables['useAzureSQL'], 'false')) + condition: and(succeededOrFailed(), eq(variables['Build.Reason'], 'PullRequest'), eq(variables['useAzureSQL'], 'false'), or(eq(variables['distroName'], 'Ubuntu'), eq(variables['distroName'], 'Ubuntu-SQL2025'))) continueOnError: true - script: | diff --git a/eng/profiler_benchmarks/README.md b/eng/profiler_benchmarks/README.md index 490802036..8578d6904 100644 --- a/eng/profiler_benchmarks/README.md +++ b/eng/profiler_benchmarks/README.md @@ -27,11 +27,14 @@ Partial results never produce a verdict. ## Publication -Five environments publish raw samples: Windows and macOS on SQL Server 2022/2025, -and Ubuntu on SQL Server 2022. The privileged publisher runs trusted base code, -authenticates benchmark producers, validates bounded artifacts, and ignores stale -heads. A failed aggregate build can still publish when its authenticated artifacts -validate. Missing, malformed, canceled, incomplete, or invalid data remains unavailable. +Four environments publish raw samples: Windows and Unix on SQL Server 2022/2025. +Unix measurements run on Ubuntu. Routine macOS profiling is intentionally excluded: +hosted macOS plus Colima produced false regressions on a documentation-only control +PR, while macOS remains covered by functional CI. The privileged publisher runs +trusted base code, authenticates benchmark producers, validates bounded artifacts, +and ignores stale heads. A failed aggregate build can still publish when its +authenticated artifacts validate. Missing, malformed, canceled, incomplete, or +invalid data remains unavailable. The publisher waits up to 220 minutes inside a 230-minute workflow. The first main comparison after introduction may be incomplete because its parent lacks this diff --git a/eng/profiler_benchmarks/report.py b/eng/profiler_benchmarks/report.py index b5768861e..7d7aa2175 100644 --- a/eng/profiler_benchmarks/report.py +++ b/eng/profiler_benchmarks/report.py @@ -14,7 +14,9 @@ import zipfile import zlib -LEGS = ("Windows-SQL2022", "Windows-SQL2025", "macOS-SQL2022", "macOS-SQL2025", "Linux-SQL2022") +# Hosted macOS plus Colima produced false regressions on a documentation-only +# control PR. Routine reports use stable Ubuntu measurements as the Unix signal. +LEGS = ("Windows-SQL2022", "Windows-SQL2025", "Linux-SQL2022", "Linux-SQL2025") TASK_NAMES = { "connect": "Connection opening", "select": "SELECT queries", @@ -208,7 +210,7 @@ def _validate(report, build_id=None, head=None, source=None, base=None, suite=No for value in env.values(): text(value) expected_os, sql = report["leg"].split("-") - if env["os"] != {"macOS": "Darwin"}.get(expected_os, expected_os): + if env["os"] != expected_os: raise ValueError("Artifact platform does not match its leg") if not env["sql_version"].startswith({"SQL2022": "16.", "SQL2025": "17."}[sql]): raise ValueError("Artifact SQL version does not match its leg") @@ -391,6 +393,7 @@ def escape(value): def environment_name(leg): operating_system, sql = leg.split("-") + operating_system = {"Linux": "Unix"}.get(operating_system, operating_system) return f"{operating_system} / SQL Server {sql.removeprefix('SQL')}" diff --git a/tests/test_036_profiler_ci.py b/tests/test_036_profiler_ci.py index c5223d106..1d8c04b47 100644 --- a/tests/test_036_profiler_ci.py +++ b/tests/test_036_profiler_ci.py @@ -126,8 +126,8 @@ def test_consistent_slowdown_is_advisory_regression(report): ) body = reporting.render([report], "c" * 40, 42) assert "20 consistent slowdown signals" in body - assert "| Linux / SQL Server 2022 | Connection opening |" in body - assert "| macOS / SQL Server 2022 | No result available" in body + assert "| Unix / SQL Server 2022 | Connection opening |" in body + assert "| Windows / SQL Server 2022 | No result available" in body assert body.index("consistent slowdown signals") < body.index( "Build, commits and measurement details" ) @@ -212,9 +212,7 @@ def set_leg(report, leg): operating_system, sql = leg.split("-") for pair in report["pairs"]: for sample in pair.values(): - sample["environment"]["os"] = {"macOS": "Darwin"}.get( - operating_system, operating_system - ) + sample["environment"]["os"] = operating_system sample["environment"]["sql_version"] = "16.0" if sql == "SQL2022" else "17.0" return report @@ -260,7 +258,7 @@ def test_render_bounds_schema_valid_diagnostics(report): reporting.validate(item) body = reporting.render(reports, "c" * 40, 42) assert len(body) <= 60000 - assert "100 diagnostic rows are available in the raw ADO artifacts" in body + assert "80 diagnostic rows are available in the raw ADO artifacts" in body assert "All database tasks and timings" in body assert "Build, commits and measurement details" in body @@ -307,15 +305,15 @@ def test_impact_summary_handles_single_inconsistent_and_complete_clean_results(r pair["candidate"]["scenarios"]["fetchone"]["wall_ms"] *= scale noisy = reporting.render([clean], "c" * 40, 42) assert ( - "**Row-by-row fetching was slower on Linux / SQL Server 2022, " + "**Row-by-row fetching was slower on Unix / SQL Server 2022, " "but the repeated comparisons were inconsistent.**" ) in noisy assert "Inconsistent slowdowns to review:" in noisy complete = [set_leg(clear_slowdowns(copy.deepcopy(report)), leg) for leg in reporting.LEGS] clean_body = reporting.render(complete, "c" * 40, 42) - assert "**No consistent slowdowns detected across all 5 environments.**" in clean_body - assert "**Coverage:** 5 of 5 environments completed." in clean_body + assert "**No consistent slowdowns detected across all 4 environments.**" in clean_body + assert "**Coverage:** 4 of 4 environments completed." in clean_body def test_impact_summary_handles_single_regression_partial_and_no_results(report): @@ -324,7 +322,7 @@ def test_impact_summary_handles_single_regression_partial_and_no_results(report) pair["candidate"]["scenarios"]["fetchone"]["wall_ms"] *= 1.3 body = reporting.render([single], "c" * 40, 42) assert ( - "**This PR consistently slows row-by-row fetching on Linux / SQL Server 2022 " "by 30.0%.**" + "**This PR consistently slows row-by-row fetching on Unix / SQL Server 2022 " "by 30.0%.**" ) in body assert "Affected phases and call counts" in body assert "All database tasks and timings" in body @@ -338,7 +336,7 @@ def test_impact_summary_handles_single_regression_partial_and_no_results(report) ["Windows-SQL2022 (missing)"], ) assert "No consistent slowdowns in the 1 completed environment." in partial - assert "No result is available for 4 environments." in partial + assert "No result is available for 3 environments." in partial assert "| Windows / SQL Server 2022 | No result available (missing) |" in partial assert "pending" not in partial.lower() @@ -843,7 +841,7 @@ def measure(path, output, scenarios, timeout): def test_ci_deadlines_include_setup_queueing_and_publication(): pipeline = (ROOT / "eng/pipelines/pr-validation-pipeline.yml").read_text(encoding="utf-8") - for job in ("pytestonwindows", "PytestOnMacOS", "PytestOnLinux"): + for job in ("pytestonwindows", "PytestOnLinux"): section = pipeline.split(f"- job: {job}\n", 1)[1].split("\n- job:", 1)[0] job_minutes = int(re.search(r"^ timeoutInMinutes: (\d+)$", section, re.M)[1]) benchmark_step = section.split( @@ -861,9 +859,9 @@ def test_ci_deadlines_include_setup_queueing_and_publication(): ) -def test_linux_profiler_step_does_not_put_database_password_on_command_line(): +def test_unix_profiler_step_does_not_put_database_password_on_command_line(): pipeline = (ROOT / "eng/pipelines/pr-validation-pipeline.yml").read_text(encoding="utf-8") - benchmark = pipeline.split("# Run performance benchmarks on Ubuntu", 1)[1] + benchmark = pipeline.split("# Run Unix performance benchmarks on Ubuntu", 1)[1] benchmark = benchmark.split("displayName: 'Compare profiling builds", 1)[0] assert "-e DB_PASSWORD \\" in benchmark assert "Pwd=$(DB_PASSWORD)" not in benchmark @@ -1042,14 +1040,14 @@ def corrupt_deflate(raw): if corrupt in ("base", "provenance"): assert "Build provenance validation failed" in posted[1] return - assert "| macOS / SQL Server 2022 | No result available" in posted[1] + assert "| Windows / SQL Server 2025 | No result available" in posted[1] if corrupt in ("suite", "source"): assert "workload version differs from trusted base" in posted[1] assert "consistent slowdown signals" not in posted[1] elif corrupt in ("zip", "timeout", "scenarios", "recursion", "deflate"): assert "### Windows / SQL Server 2022" in posted[1] assert reporting.escape("Linux-SQL2022 (invalid artifact)") in posted[1] - assert "| Linux / SQL Server 2022 | No result available (invalid artifact) |" in posted[1] + assert "| Unix / SQL Server 2022 | No result available (invalid artifact) |" in posted[1] assert posted[1].count("20 consistent slowdown signals") == 1 else: assert "### Windows / SQL Server 2022" in posted[1] @@ -1232,13 +1230,13 @@ def test_report_leg_must_match_measured_environment(report, environment): def test_ci_reuses_profiling_builds_without_changing_release_defaults(): pipeline = (ROOT / "eng/pipelines/pr-validation-pipeline.yml").read_text(encoding="utf-8") assert "benchmarks/perf-benchmarking.py" not in pipeline - assert pipeline.count("python -m eng.profiler_benchmarks.controller --reuse-candidate") == 3 + assert pipeline.count("python -m eng.profiler_benchmarks.controller --reuse-candidate") == 2 profiler_conditions = re.findall( r"displayName: '(?:Compare profiling builds[^']*|Publish paired profiler measurements)'\n" r" condition: ([^\n]+)", pipeline, ) - assert len(profiler_conditions) == 6 + assert len(profiler_conditions) == 4 assert all( "eq(variables['Build.Reason'], 'PullRequest')" in condition for condition in profiler_conditions @@ -1259,16 +1257,22 @@ def test_ci_reuses_profiling_builds_without_changing_release_defaults(): "eq(variables['sqlVersion'], 'LocalDB')))" ) in windows macos = pipeline.split("- job: PytestOnMacOS\n", 1)[1].split("\n- job:", 1)[0] - assert 'if [ "$(Build.Reason)" = "PullRequest" ]; then' in macos - assert 'ENABLE_PROFILING="$PROFILER_BUILD" ./build.sh' in macos - assert "condition: and(succeeded(), eq(variables['Build.Reason'], 'PullRequest'))" in macos + assert "timeoutInMinutes: 90" in macos + assert "ENABLE_PROFILING" not in macos + assert "--reuse-candidate" not in macos + assert "AdventureWorks2022" not in macos + assert "profiler-macOS" not in macos + assert "Routine PR profiling excludes hosted macOS" in macos linux = pipeline.split("- job: PytestOnLinux\n", 1)[1].split("\n- job:", 1)[0] - assert ( - 'if [ "$(Build.Reason)" = "PullRequest" ] && [ "$(distroName)" = "Ubuntu" ]; then' in linux - ) + assert 'if [ "$(Build.Reason)" = "PullRequest" ] &&' in linux + assert '[[ "$(distroName)" =~ ^Ubuntu(-SQL2025)?$ ]]' in linux assert 'if [ "$PROFILER_BUILD" = "1" ]; then' in linux assert "python -m eng.profiler_benchmarks.controller --check-build on" in linux - benchmark = linux.split("# Run performance benchmarks on Ubuntu", 1)[1] + assert "profilerLeg: 'Linux-SQL2022'" in linux + assert "profilerLeg: 'Linux-SQL2025'" in linux + assert '--leg "$(profilerLeg)"' in linux + assert "artifact: profiler-$(profilerLeg)" in linux + benchmark = linux.split("# Run Unix performance benchmarks on Ubuntu with", 1)[1] assert "-e BUILD_BUILDID \\" in benchmark assert "BUILD_BUILDID: $(Build.BuildId)" in benchmark assert "git config --global --add safe.directory /workspace" in benchmark