diff --git a/eng/pipelines/pr-validation-pipeline.yml b/eng/pipelines/pr-validation-pipeline.yml
index 6e5038d89..85e3e2ae5 100644
--- a/eng/pipelines/pr-validation-pipeline.yml
+++ b/eng/pipelines/pr-validation-pipeline.yml
@@ -468,9 +468,7 @@ jobs:
- job: PytestOnMacOS
displayName: 'macOS x86_64'
- # Reserve 60 minutes outside the 100-minute benchmark step for Colima/SQL
- # setup, pytest, fixture restore and artifact publication.
- timeoutInMinutes: 160
+ timeoutInMinutes: 90
pool:
vmImage: 'macos-latest'
@@ -568,11 +566,7 @@ jobs:
pip install -r requirements.txt
echo "Building pybind bindings (.so) (overlapped with container setup)..."
- PROFILER_BUILD=0
- if [ "$(Build.Reason)" = "PullRequest" ]; then
- PROFILER_BUILD=1
- fi
- ( cd mssql_python/pybind && ENABLE_PROFILING="$PROFILER_BUILD" ./build.sh )
+ ( cd mssql_python/pybind && ./build.sh )
echo "Waiting for container setup (Colima + SQL Server) to finish..."
SQL_STATUS=0
@@ -585,10 +579,6 @@ jobs:
env:
DB_PASSWORD: $(DB_PASSWORD)
- - script: python -m eng.profiler_benchmarks.controller --check-build on
- displayName: 'Verify native configuration and recording OFF before pytest'
- condition: and(succeeded(), eq(variables['Build.Reason'], 'PullRequest'))
-
- template: steps/install-mssql-py-core.yml
parameters:
platform: unix
@@ -637,59 +627,9 @@ jobs:
testResultsFiles: '**/test-results.xml'
testRunTitle: 'Publish pytest results on macOS $(sqlVersion)'
- # Download and restore AdventureWorks2022 database for benchmarking
- - script: |
- echo "Downloading AdventureWorks2022.bak..."
- curl -sSL -o /tmp/AdventureWorks2022.bak \
- https://github.com/Microsoft/sql-server-samples/releases/download/adventureworks/AdventureWorks2022.bak
-
- echo "Copying backup into SQL Server container..."
- docker cp /tmp/AdventureWorks2022.bak sqlserver:/tmp/AdventureWorks2022.bak
-
- echo "Restoring AdventureWorks2022 database..."
- docker exec sqlserver /opt/mssql-tools18/bin/sqlcmd \
- -S localhost -U SA -P "$DB_PASSWORD" -C \
- -Q "RESTORE DATABASE AdventureWorks2022 FROM DISK = '/tmp/AdventureWorks2022.bak' WITH MOVE 'AdventureWorks2022' TO '/var/opt/mssql/data/AdventureWorks2022.mdf', MOVE 'AdventureWorks2022_log' TO '/var/opt/mssql/data/AdventureWorks2022_log.ldf', REPLACE"
-
- if [ $? -eq 0 ]; then
- echo "AdventureWorks2022 database restored successfully"
- else
- echo "Failed to restore AdventureWorks2022 database"
- fi
-
- rm -f /tmp/AdventureWorks2022.bak
- docker exec sqlserver rm -f /tmp/AdventureWorks2022.bak || true
- displayName: 'Download and restore AdventureWorks2022 database on macOS'
- condition: and(succeeded(), or(eq(variables['sqlVersion'], 'SQL2022'), eq(variables['sqlVersion'], 'SQL2025')))
- continueOnError: true
- env:
- DB_PASSWORD: $(DB_PASSWORD)
-
- # Run performance benchmarks on macOS
- - script: |
- set -euo pipefail
- echo "Restoring build dependencies for isolated profiling builds..."
- brew tap microsoft/mssql-release https://github.com/Microsoft/homebrew-mssql-release
- # Newer Homebrew refuses to load formulae from third-party taps unless the tap is trusted
- brew trust microsoft/mssql-release || echo "brew trust failed; attempting install anyway"
- HOMEBREW_ACCEPT_EULA=Y brew install msodbcsql18
- python -m eng.profiler_benchmarks.controller --reuse-candidate --leg "macOS-$(sqlVersion)" --output profiler-results
- displayName: 'Compare profiling builds on macOS $(sqlVersion)'
- condition: and(succeeded(), eq(variables['Build.Reason'], 'PullRequest'), or(eq(variables['sqlVersion'], 'SQL2022'), eq(variables['sqlVersion'], 'SQL2025')))
- timeoutInMinutes: 100
- continueOnError: true
- env:
- SYSTEM_PULLREQUEST_SOURCECOMMITID: $(System.PullRequest.SourceCommitId)
- DB_CONNECTION_STRING: 'Server=tcp:127.0.0.1,1433;Database=AdventureWorks2022;Uid=SA;Pwd=$(DB_PASSWORD);TrustServerCertificate=yes'
-
- # Both revisions and samples travel together; no historical baseline lookup.
- - task: PublishPipelineArtifact@1
- inputs:
- targetPath: profiler-results
- artifact: 'profiler-macOS-$(sqlVersion)'
- displayName: 'Publish paired profiler measurements'
- condition: and(succeededOrFailed(), eq(variables['Build.Reason'], 'PullRequest'), or(eq(variables['sqlVersion'], 'SQL2022'), eq(variables['sqlVersion'], 'SQL2025')))
- continueOnError: true
+ # Routine PR profiling excludes hosted macOS because Colima-backed control
+ # runs produced regressions with no product-code change. Functional macOS
+ # coverage remains above; Unix performance signal comes from Ubuntu.
- script: |
python3 eng/scripts/setup_sql_container.py --cleanup --colima \
@@ -713,11 +653,13 @@ jobs:
distroName: 'Ubuntu'
sqlServerImage: 'mcr.microsoft.com/mssql/server:2022-latest'
useAzureSQL: 'false'
+ profilerLeg: 'Linux-SQL2022'
Ubuntu_SQL2025:
dockerImage: 'ubuntu:24.04'
distroName: 'Ubuntu-SQL2025'
sqlServerImage: 'mcr.microsoft.com/mssql/server:2025-latest'
useAzureSQL: 'false'
+ profilerLeg: 'Linux-SQL2025'
${{ if ne(variables['AZURE_CONNECTION_STRING'], '') }}:
Ubuntu_AzureSQL:
dockerImage: 'ubuntu:24.04'
@@ -832,7 +774,8 @@ jobs:
- script: |
# Build pybind bindings in the container
PROFILER_BUILD=0
- if [ "$(Build.Reason)" = "PullRequest" ] && [ "$(distroName)" = "Ubuntu" ]; then
+ if [ "$(Build.Reason)" = "PullRequest" ] &&
+ [[ "$(distroName)" =~ ^Ubuntu(-SQL2025)?$ ]]; then
PROFILER_BUILD=1
fi
docker exec -e ENABLE_PROFILING="$PROFILER_BUILD" test-container-$(distroName) bash -c "
@@ -912,8 +855,8 @@ jobs:
DB_PASSWORD: $(DB_PASSWORD)
- script: |
- # Download and restore AdventureWorks2022 database for benchmarking on Ubuntu only
- if [ "$(distroName)" = "Ubuntu" ] && [ "$(useAzureSQL)" = "false" ]; then
+ # Download and restore AdventureWorks2022 for both Unix profiler legs.
+ if [[ "$(distroName)" =~ ^Ubuntu(-SQL2025)?$ ]] && [ "$(useAzureSQL)" = "false" ]; then
echo "Downloading AdventureWorks2022.bak..."
wget -q https://github.com/Microsoft/sql-server-samples/releases/download/adventureworks/AdventureWorks2022.bak -O /tmp/AdventureWorks2022.bak
@@ -939,14 +882,16 @@ jobs:
docker exec sqlserver-$(distroName) rm -f /tmp/AdventureWorks2022.bak || true
fi
displayName: 'Download and restore AdventureWorks2022 database in $(distroName)'
- condition: and(succeeded(), eq(variables['distroName'], 'Ubuntu'), eq(variables['useAzureSQL'], 'false'))
+ condition: and(succeeded(), eq(variables['useAzureSQL'], 'false'), or(eq(variables['distroName'], 'Ubuntu'), eq(variables['distroName'], 'Ubuntu-SQL2025')))
continueOnError: true
env:
DB_PASSWORD: $(DB_PASSWORD)
+ # Unix profiling uses Ubuntu because it covers the shared Unix hot paths while
+ # producing stable paired measurements on hosted agents.
- script: |
- # Run performance benchmarks on Ubuntu with SQL Server 2022 only
- if [ "$(distroName)" = "Ubuntu" ] && [ "$(useAzureSQL)" = "false" ]; then
+ # Run Unix performance benchmarks on Ubuntu with SQL Server 2022 and 2025.
+ if [[ "$(distroName)" =~ ^Ubuntu(-SQL2025)?$ ]] && [ "$(useAzureSQL)" = "false" ]; then
export SQLSERVER_IP=$(docker inspect sqlserver-$(distroName) --format='{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}')
echo "Running performance benchmarks on Ubuntu with SQL Server IP: $SQLSERVER_IP"
@@ -967,13 +912,13 @@ jobs:
ACCEPT_EULA=Y apt-get install -y --no-install-recommends git unixodbc unixodbc-dev libodbc2 libodbcinst2 odbcinst msodbcsql18
git config --global --add safe.directory /workspace
odbcinst -q -d -n "ODBC Driver 18 for SQL Server"
- python -m eng.profiler_benchmarks.controller --reuse-candidate --leg Linux-SQL2022 --output profiler-results
+ python -m eng.profiler_benchmarks.controller --reuse-candidate --leg "$(profilerLeg)" --output profiler-results
'
else
echo "Skipping performance benchmarks on $(distroName) (only runs on Ubuntu with local SQL Server)"
fi
- displayName: 'Compare profiling builds in $(distroName) container'
- condition: and(succeeded(), eq(variables['Build.Reason'], 'PullRequest'), eq(variables['distroName'], 'Ubuntu'), eq(variables['useAzureSQL'], 'false'))
+ displayName: 'Compare profiling builds on Unix in $(distroName) container'
+ condition: and(succeeded(), eq(variables['Build.Reason'], 'PullRequest'), eq(variables['useAzureSQL'], 'false'), or(eq(variables['distroName'], 'Ubuntu'), eq(variables['distroName'], 'Ubuntu-SQL2025')))
continueOnError: true
timeoutInMinutes: 100
env:
@@ -984,9 +929,9 @@ jobs:
- task: PublishPipelineArtifact@1
inputs:
targetPath: profiler-results
- artifact: profiler-Linux-SQL2022
+ artifact: profiler-$(profilerLeg)
displayName: 'Publish paired profiler measurements'
- condition: and(succeededOrFailed(), eq(variables['Build.Reason'], 'PullRequest'), eq(variables['distroName'], 'Ubuntu'), eq(variables['useAzureSQL'], 'false'))
+ condition: and(succeededOrFailed(), eq(variables['Build.Reason'], 'PullRequest'), eq(variables['useAzureSQL'], 'false'), or(eq(variables['distroName'], 'Ubuntu'), eq(variables['distroName'], 'Ubuntu-SQL2025')))
continueOnError: true
- script: |
diff --git a/eng/profiler_benchmarks/README.md b/eng/profiler_benchmarks/README.md
index 490802036..8578d6904 100644
--- a/eng/profiler_benchmarks/README.md
+++ b/eng/profiler_benchmarks/README.md
@@ -27,11 +27,14 @@ Partial results never produce a verdict.
## Publication
-Five environments publish raw samples: Windows and macOS on SQL Server 2022/2025,
-and Ubuntu on SQL Server 2022. The privileged publisher runs trusted base code,
-authenticates benchmark producers, validates bounded artifacts, and ignores stale
-heads. A failed aggregate build can still publish when its authenticated artifacts
-validate. Missing, malformed, canceled, incomplete, or invalid data remains unavailable.
+Four environments publish raw samples: Windows and Unix on SQL Server 2022/2025.
+Unix measurements run on Ubuntu. Routine macOS profiling is intentionally excluded:
+hosted macOS plus Colima produced false regressions on a documentation-only control
+PR, while macOS remains covered by functional CI. The privileged publisher runs
+trusted base code, authenticates benchmark producers, validates bounded artifacts,
+and ignores stale heads. A failed aggregate build can still publish when its
+authenticated artifacts validate. Missing, malformed, canceled, incomplete, or
+invalid data remains unavailable.
The publisher waits up to 220 minutes inside a 230-minute workflow. The first main
comparison after introduction may be incomplete because its parent lacks this
diff --git a/eng/profiler_benchmarks/report.py b/eng/profiler_benchmarks/report.py
index b5768861e..7d7aa2175 100644
--- a/eng/profiler_benchmarks/report.py
+++ b/eng/profiler_benchmarks/report.py
@@ -14,7 +14,9 @@
import zipfile
import zlib
-LEGS = ("Windows-SQL2022", "Windows-SQL2025", "macOS-SQL2022", "macOS-SQL2025", "Linux-SQL2022")
+# Hosted macOS plus Colima produced false regressions on a documentation-only
+# control PR. Routine reports use stable Ubuntu measurements as the Unix signal.
+LEGS = ("Windows-SQL2022", "Windows-SQL2025", "Linux-SQL2022", "Linux-SQL2025")
TASK_NAMES = {
"connect": "Connection opening",
"select": "SELECT queries",
@@ -208,7 +210,7 @@ def _validate(report, build_id=None, head=None, source=None, base=None, suite=No
for value in env.values():
text(value)
expected_os, sql = report["leg"].split("-")
- if env["os"] != {"macOS": "Darwin"}.get(expected_os, expected_os):
+ if env["os"] != expected_os:
raise ValueError("Artifact platform does not match its leg")
if not env["sql_version"].startswith({"SQL2022": "16.", "SQL2025": "17."}[sql]):
raise ValueError("Artifact SQL version does not match its leg")
@@ -391,6 +393,7 @@ def escape(value):
def environment_name(leg):
operating_system, sql = leg.split("-")
+ operating_system = {"Linux": "Unix"}.get(operating_system, operating_system)
return f"{operating_system} / SQL Server {sql.removeprefix('SQL')}"
diff --git a/tests/test_036_profiler_ci.py b/tests/test_036_profiler_ci.py
index c5223d106..1d8c04b47 100644
--- a/tests/test_036_profiler_ci.py
+++ b/tests/test_036_profiler_ci.py
@@ -126,8 +126,8 @@ def test_consistent_slowdown_is_advisory_regression(report):
)
body = reporting.render([report], "c" * 40, 42)
assert "20 consistent slowdown signals" in body
- assert "| Linux / SQL Server 2022 | Connection opening |" in body
- assert "| macOS / SQL Server 2022 | No result available" in body
+ assert "| Unix / SQL Server 2022 | Connection opening |" in body
+ assert "| Windows / SQL Server 2022 | No result available" in body
assert body.index("consistent slowdown signals") < body.index(
"Build, commits and measurement details"
)
@@ -212,9 +212,7 @@ def set_leg(report, leg):
operating_system, sql = leg.split("-")
for pair in report["pairs"]:
for sample in pair.values():
- sample["environment"]["os"] = {"macOS": "Darwin"}.get(
- operating_system, operating_system
- )
+ sample["environment"]["os"] = operating_system
sample["environment"]["sql_version"] = "16.0" if sql == "SQL2022" else "17.0"
return report
@@ -260,7 +258,7 @@ def test_render_bounds_schema_valid_diagnostics(report):
reporting.validate(item)
body = reporting.render(reports, "c" * 40, 42)
assert len(body) <= 60000
- assert "100 diagnostic rows are available in the raw ADO artifacts" in body
+ assert "80 diagnostic rows are available in the raw ADO artifacts" in body
assert "All database tasks and timings" in body
assert "Build, commits and measurement details" in body
@@ -307,15 +305,15 @@ def test_impact_summary_handles_single_inconsistent_and_complete_clean_results(r
pair["candidate"]["scenarios"]["fetchone"]["wall_ms"] *= scale
noisy = reporting.render([clean], "c" * 40, 42)
assert (
- "**Row-by-row fetching was slower on Linux / SQL Server 2022, "
+ "**Row-by-row fetching was slower on Unix / SQL Server 2022, "
"but the repeated comparisons were inconsistent.**"
) in noisy
assert "Inconsistent slowdowns to review:" in noisy
complete = [set_leg(clear_slowdowns(copy.deepcopy(report)), leg) for leg in reporting.LEGS]
clean_body = reporting.render(complete, "c" * 40, 42)
- assert "**No consistent slowdowns detected across all 5 environments.**" in clean_body
- assert "**Coverage:** 5 of 5 environments completed." in clean_body
+ assert "**No consistent slowdowns detected across all 4 environments.**" in clean_body
+ assert "**Coverage:** 4 of 4 environments completed." in clean_body
def test_impact_summary_handles_single_regression_partial_and_no_results(report):
@@ -324,7 +322,7 @@ def test_impact_summary_handles_single_regression_partial_and_no_results(report)
pair["candidate"]["scenarios"]["fetchone"]["wall_ms"] *= 1.3
body = reporting.render([single], "c" * 40, 42)
assert (
- "**This PR consistently slows row-by-row fetching on Linux / SQL Server 2022 " "by 30.0%.**"
+ "**This PR consistently slows row-by-row fetching on Unix / SQL Server 2022 " "by 30.0%.**"
) in body
assert "Affected phases and call counts" in body
assert "All database tasks and timings" in body
@@ -338,7 +336,7 @@ def test_impact_summary_handles_single_regression_partial_and_no_results(report)
["Windows-SQL2022 (missing)"],
)
assert "No consistent slowdowns in the 1 completed environment." in partial
- assert "No result is available for 4 environments." in partial
+ assert "No result is available for 3 environments." in partial
assert "| Windows / SQL Server 2022 | No result available (missing) |" in partial
assert "pending" not in partial.lower()
@@ -843,7 +841,7 @@ def measure(path, output, scenarios, timeout):
def test_ci_deadlines_include_setup_queueing_and_publication():
pipeline = (ROOT / "eng/pipelines/pr-validation-pipeline.yml").read_text(encoding="utf-8")
- for job in ("pytestonwindows", "PytestOnMacOS", "PytestOnLinux"):
+ for job in ("pytestonwindows", "PytestOnLinux"):
section = pipeline.split(f"- job: {job}\n", 1)[1].split("\n- job:", 1)[0]
job_minutes = int(re.search(r"^ timeoutInMinutes: (\d+)$", section, re.M)[1])
benchmark_step = section.split(
@@ -861,9 +859,9 @@ def test_ci_deadlines_include_setup_queueing_and_publication():
)
-def test_linux_profiler_step_does_not_put_database_password_on_command_line():
+def test_unix_profiler_step_does_not_put_database_password_on_command_line():
pipeline = (ROOT / "eng/pipelines/pr-validation-pipeline.yml").read_text(encoding="utf-8")
- benchmark = pipeline.split("# Run performance benchmarks on Ubuntu", 1)[1]
+ benchmark = pipeline.split("# Run Unix performance benchmarks on Ubuntu", 1)[1]
benchmark = benchmark.split("displayName: 'Compare profiling builds", 1)[0]
assert "-e DB_PASSWORD \\" in benchmark
assert "Pwd=$(DB_PASSWORD)" not in benchmark
@@ -1042,14 +1040,14 @@ def corrupt_deflate(raw):
if corrupt in ("base", "provenance"):
assert "Build provenance validation failed" in posted[1]
return
- assert "| macOS / SQL Server 2022 | No result available" in posted[1]
+ assert "| Windows / SQL Server 2025 | No result available" in posted[1]
if corrupt in ("suite", "source"):
assert "workload version differs from trusted base" in posted[1]
assert "consistent slowdown signals" not in posted[1]
elif corrupt in ("zip", "timeout", "scenarios", "recursion", "deflate"):
assert "### Windows / SQL Server 2022" in posted[1]
assert reporting.escape("Linux-SQL2022 (invalid artifact)") in posted[1]
- assert "| Linux / SQL Server 2022 | No result available (invalid artifact) |" in posted[1]
+ assert "| Unix / SQL Server 2022 | No result available (invalid artifact) |" in posted[1]
assert posted[1].count("20 consistent slowdown signals") == 1
else:
assert "### Windows / SQL Server 2022" in posted[1]
@@ -1232,13 +1230,13 @@ def test_report_leg_must_match_measured_environment(report, environment):
def test_ci_reuses_profiling_builds_without_changing_release_defaults():
pipeline = (ROOT / "eng/pipelines/pr-validation-pipeline.yml").read_text(encoding="utf-8")
assert "benchmarks/perf-benchmarking.py" not in pipeline
- assert pipeline.count("python -m eng.profiler_benchmarks.controller --reuse-candidate") == 3
+ assert pipeline.count("python -m eng.profiler_benchmarks.controller --reuse-candidate") == 2
profiler_conditions = re.findall(
r"displayName: '(?:Compare profiling builds[^']*|Publish paired profiler measurements)'\n"
r" condition: ([^\n]+)",
pipeline,
)
- assert len(profiler_conditions) == 6
+ assert len(profiler_conditions) == 4
assert all(
"eq(variables['Build.Reason'], 'PullRequest')" in condition
for condition in profiler_conditions
@@ -1259,16 +1257,22 @@ def test_ci_reuses_profiling_builds_without_changing_release_defaults():
"eq(variables['sqlVersion'], 'LocalDB')))"
) in windows
macos = pipeline.split("- job: PytestOnMacOS\n", 1)[1].split("\n- job:", 1)[0]
- assert 'if [ "$(Build.Reason)" = "PullRequest" ]; then' in macos
- assert 'ENABLE_PROFILING="$PROFILER_BUILD" ./build.sh' in macos
- assert "condition: and(succeeded(), eq(variables['Build.Reason'], 'PullRequest'))" in macos
+ assert "timeoutInMinutes: 90" in macos
+ assert "ENABLE_PROFILING" not in macos
+ assert "--reuse-candidate" not in macos
+ assert "AdventureWorks2022" not in macos
+ assert "profiler-macOS" not in macos
+ assert "Routine PR profiling excludes hosted macOS" in macos
linux = pipeline.split("- job: PytestOnLinux\n", 1)[1].split("\n- job:", 1)[0]
- assert (
- 'if [ "$(Build.Reason)" = "PullRequest" ] && [ "$(distroName)" = "Ubuntu" ]; then' in linux
- )
+ assert 'if [ "$(Build.Reason)" = "PullRequest" ] &&' in linux
+ assert '[[ "$(distroName)" =~ ^Ubuntu(-SQL2025)?$ ]]' in linux
assert 'if [ "$PROFILER_BUILD" = "1" ]; then' in linux
assert "python -m eng.profiler_benchmarks.controller --check-build on" in linux
- benchmark = linux.split("# Run performance benchmarks on Ubuntu", 1)[1]
+ assert "profilerLeg: 'Linux-SQL2022'" in linux
+ assert "profilerLeg: 'Linux-SQL2025'" in linux
+ assert '--leg "$(profilerLeg)"' in linux
+ assert "artifact: profiler-$(profilerLeg)" in linux
+ benchmark = linux.split("# Run Unix performance benchmarks on Ubuntu with", 1)[1]
assert "-e BUILD_BUILDID \\" in benchmark
assert "BUILD_BUILDID: $(Build.BuildId)" in benchmark
assert "git config --global --add safe.directory /workspace" in benchmark