Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
40 changes: 28 additions & 12 deletions .github/workflows/all.yml
Original file line number Diff line number Diff line change
Expand Up @@ -160,7 +160,7 @@ jobs:
CCACHE_IGNOREOPTIONS: "-fno-strict-overflow -fwrapv -W* -arch x86_64 arm64 -dynamic -fno-common -g -I/usr/local/opt/*"
CCACHE_LOGFILE: ${{ github.workspace }}/.ccache_log
CCACHE_DEBUG: "1"
USE_PORTABLE_SIMD: "1" # Use portable SIMD to avoid "Illegal instruction" on different runner CPUs
Comment thread
cjacoby-sptfy marked this conversation as resolved.
USE_MARCH_NATIVE: "0" # Keep cached objects compatible across runner CPUs
DISABLE_LTO: "1" # Speeds up un-cacheable link step which doesn't really increase performance in tests anyways
CC: ${{ matrix.cc }}
CXX: ${{ matrix.cxx }}
Expand Down Expand Up @@ -357,7 +357,7 @@ jobs:
CCACHE_IGNOREOPTIONS: "-fno-strict-overflow -fwrapv -W* -arch x86_64 arm64 -dynamic -fno-common -g -I/usr/local/opt/*"
CCACHE_LOGFILE: ${{ github.workspace }}/.ccache_log
USE_ASAN: "1"
USE_PORTABLE_SIMD: "1" # Use portable SIMD to avoid "Illegal instruction" on different runner CPUs
USE_MARCH_NATIVE: "0" # Keep cached objects compatible across runner CPUs
DISABLE_LTO: "1" # Speeds up un-cacheable link step which doesn't really increase performance in tests anyways
CC: ${{ matrix.cc }}
CXX: ${{ matrix.cxx }}
Expand Down Expand Up @@ -471,7 +471,7 @@ jobs:
CCACHE_NOINODECACHE: 1
CCACHE_IGNOREOPTIONS: "-fno-strict-overflow -fwrapv -W* -arch x86_64 arm64 -dynamic -fno-common -g -I/usr/local/opt/*"
CCACHE_LOGFILE: ${{ github.workspace }}/.ccache_log
USE_PORTABLE_SIMD: "1" # Use portable SIMD to avoid "Illegal instruction" on different runner CPUs
USE_MARCH_NATIVE: "0" # Keep cached objects compatible across runner CPUs
DISABLE_LTO: "1" # Speeds up un-cacheable link step which doesn't really increase performance in tests anyways
CC: ${{ matrix.cc }}
CXX: ${{ matrix.cxx }}
Expand Down Expand Up @@ -726,7 +726,7 @@ jobs:
CCACHE_IGNOREOPTIONS: "-fno-strict-overflow -fwrapv -W* -arch x86_64 arm64 -dynamic -fno-common -g -I/usr/local/opt/*"
CCACHE_LOGFILE: ${{ github.workspace }}/.ccache_log
CCACHE_DEBUG: "1"
USE_PORTABLE_SIMD: "1" # Use portable SIMD to avoid "Illegal instruction" on different runner CPUs
USE_MARCH_NATIVE: "0" # Keep cached objects compatible across runner CPUs
# Use the minimum macOS deployment target supported by our version of PyBind:
MACOSX_DEPLOYMENT_TARGET: "10.14"
# This build caching is only to speed up tests on CI, so we only care about x86_64 (macos-15-intel runners).
Expand Down Expand Up @@ -831,7 +831,7 @@ jobs:
env:
DEBUG: "0"
USE_ASAN: "1"
USE_PORTABLE_SIMD: "1" # Use portable SIMD to avoid "Illegal instruction" on different runner CPUs
USE_MARCH_NATIVE: "0" # Keep cached objects compatible across runner CPUs
CCACHE_DIR: ${{ github.workspace }}/.ccache
CCACHE_BASEDIR: ${{ github.workspace }}
CCACHE_DEBUGDIR: ${{ github.workspace }}/.ccache_debug
Expand Down Expand Up @@ -891,26 +891,26 @@ jobs:
- { os: windows-latest, build: cp314-win_amd64 }
- { os: windows-latest, build: cp315-win_amd64 }
# - { os: windows-latest, build: cp312-win32 }
- { os: "ubuntu-24.04", build: cp310-manylinux_x86_64 }
- { os: "ubuntu-24.04", build: cp310-manylinux_x86_64, python: "3.10" }
- { os: "ubuntu-24.04-arm", build: cp310-manylinux_aarch64 }
- { os: "ubuntu-24.04", build: cp310-musllinux_x86_64 }
- { os: "ubuntu-24.04-arm", build: cp310-musllinux_aarch64 }
- { os: "ubuntu-24.04", build: cp311-manylinux_x86_64 }
- { os: "ubuntu-24.04", build: cp311-manylinux_x86_64, python: "3.11" }
- { os: "ubuntu-24.04-arm", build: cp311-manylinux_aarch64 }
# - { os: 'ubuntu-24.04', build: cp311-musllinux_x86_64 }
# - { os: 'ubuntu-24.04', build: cp311-musllinux_aarch64 }
- { os: "ubuntu-24.04", build: cp312-manylinux_x86_64 }
- { os: "ubuntu-24.04", build: cp312-manylinux_x86_64, python: "3.12" }
- { os: "ubuntu-24.04-arm", build: cp312-manylinux_aarch64 }
- { os: "ubuntu-24.04", build: cp312-musllinux_x86_64 }
- { os: "ubuntu-24.04-arm", build: cp312-musllinux_aarch64 }
- { os: "ubuntu-24.04", build: cp313-manylinux_x86_64 }
- { os: "ubuntu-24.04", build: cp313-manylinux_x86_64, python: "3.13" }
- { os: "ubuntu-24.04-arm", build: cp313-manylinux_aarch64 }
# - { os: 'ubuntu-24.04', build: cp313-musllinux_x86_64 }
# - { os: 'ubuntu-24.04', build: cp313-musllinux_aarch64 }
- { os: "ubuntu-24.04", build: cp314-manylinux_x86_64 }
- { os: "ubuntu-24.04", build: cp314-manylinux_x86_64, python: "3.14" }
- { os: "ubuntu-24.04-arm", build: cp314-manylinux_aarch64 }
- { os: "ubuntu-24.04-arm", build: cp314t-manylinux_aarch64 }
- { os: "ubuntu-24.04", build: cp315-manylinux_x86_64 }
- { os: "ubuntu-24.04", build: cp315-manylinux_x86_64, python: "3.15" }
- { os: "ubuntu-24.04-arm", build: cp315-manylinux_aarch64 }
- { os: "ubuntu-24.04-arm", build: cp315t-manylinux_aarch64 }
name: Build wheel for ${{ matrix.build }}
Expand All @@ -933,9 +933,25 @@ jobs:
CIBW_TEST_COMMAND: "python -c \"import pedalboard; print(pedalboard.__version__)\""
CIBW_TEST_REQUIRES: ""
# on macOS and with Python 3.10: building NumPy from source fails without these options:
CIBW_ENVIRONMENT: NPY_BLAS_ORDER="" NPY_LAPACK_ORDER="" CIBW_BUILD="${{ matrix.build }}"
CIBW_ENVIRONMENT: NPY_BLAS_ORDER="" NPY_LAPACK_ORDER="" CIBW_BUILD="${{ matrix.build }}" USE_MARCH_NATIVE=0
# Use the minimum macOS deployment target supported by our version of PyBind:
MACOSX_DEPLOYMENT_TARGET: "10.14"
- name: Set up the wheel's Python version
if: contains(matrix.build, 'manylinux_x86_64')
uses: actions/setup-python@v5
with:
python-version: ${{ matrix.python }}
allow-prereleases: true
- name: Install the x86_64 CPU emulator
if: contains(matrix.build, 'manylinux_x86_64')
run: |
sudo apt-get update
sudo apt-get install --yes qemu-user
- name: Test wheel on supported x86_64 CPUs
if: contains(matrix.build, 'manylinux_x86_64')
run: |
python -m pip install ./wheelhouse/*.whl
python scripts/test_linux_wheel_cpu_compatibility.py
- name: Ensure wheels have required files
if: runner.os != 'Windows'
run: |
Expand Down
11 changes: 5 additions & 6 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -204,13 +204,12 @@ elseif(UNIX AND NOT APPLE)
if(CMAKE_SYSTEM_PROCESSOR MATCHES "arm|aarch64")
add_compile_definitions(HAVE_NEON=1)
else()
# Use -march=native for local builds to optimize for the current CPU,
# but use a portable baseline for CI builds to avoid "Illegal instruction" errors
# when ccache restores objects built on different runner hardware.
if(DEFINED ENV{USE_PORTABLE_SIMD})
add_compile_options(-mavx)
else()
# Keep distributable builds compatible with AVX-capable x86_64 CPUs.
# Local builds may explicitly target the build machine for extra optimization.
if("$ENV{USE_MARCH_NATIVE}" STREQUAL "1")
add_compile_options(-march=native)
else()
add_compile_options(-mavx)
endif()
add_compile_definitions(HAVE_AVX)
endif()
Expand Down
5 changes: 5 additions & 0 deletions CONTRIBUTING.md
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,11 @@ python3 setup.py build develop

Then, you can `import pedalboard` from Python (or run the tests with `tox`) to test out your local changes.

Linux x86_64 builds target the portable AVX baseline by default. To optimize a local
build for the current machine instead, set `USE_MARCH_NATIVE=1` while building.
The previous `USE_PORTABLE_SIMD` variable is no longer used; builds that set it remain
portable because AVX is now the default.

> If you're on macOS or Linux, you can try to compile a debug build _faster_ by using [Ccache](https://ccache.dev/):
> ## macOS
> ```shell
Expand Down
1 change: 1 addition & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -62,6 +62,7 @@ If you are new to Python, follow [INSTALLATION.md](https://github.com/spotify/pe
- Tested heavily in production use cases at Spotify
- Tested automatically on GitHub with VSTs
- Platform `manylinux` and `musllinux` wheels built for `x86_64` (Intel/AMD) and `aarch64` (ARM/Apple Silicon)
- `x86_64` wheels require AVX support
- Most Linux VSTs require a relatively modern Linux installation (with glibc > 2.27)
- macOS
- Tested manually with VSTs and Audio Units
Expand Down
85 changes: 85 additions & 0 deletions scripts/test_linux_wheel_cpu_compatibility.py
Comment thread
cjacoby-sptfy marked this conversation as resolved.
Original file line number Diff line number Diff line change
@@ -0,0 +1,85 @@
#!/usr/bin/env python3
"""Smoke-test an installed Linux wheel across its x86_64 CPU compatibility boundary."""

import argparse
import signal
import subprocess
import sys
import tempfile


SUPPORTED_CPU_MODELS = ("IvyBridge", "EPYC-Milan")
UNSUPPORTED_CPU_MODELS = ("Nehalem",)
DEFAULT_CPU_MODELS = SUPPORTED_CPU_MODELS + UNSUPPORTED_CPU_MODELS
SMOKE_TEST_TIMEOUT_SECONDS = 60
RUNTIME_PROBE = "print('Python runtime started successfully')"
WHEEL_SMOKE_TEST = (
"import numpy as np; "
"from pedalboard import Gain; "
"output = Gain(gain_db=-6)(np.zeros((1, 1024), dtype=np.float32), 48000); "
"assert output.shape == (1, 1024)"
)


def run_on_cpu(
cpu_model: str, python_code: str, python: str = sys.executable
) -> subprocess.CompletedProcess[str]:
with tempfile.TemporaryDirectory() as working_directory:
return subprocess.run(
[
"qemu-x86_64",
"-cpu",
cpu_model,
python,
"-c",
python_code,
],
capture_output=True,
cwd=working_directory,
text=True,
timeout=SMOKE_TEST_TIMEOUT_SECONDS,
)


def format_failure(result: subprocess.CompletedProcess[str]) -> str:
return f"exit code {result.returncode}\n{result.stdout}{result.stderr}"


def smoke_test_wheel(cpu_model: str, python: str = sys.executable) -> None:
print(f"Testing installed wheel on {cpu_model}...")
if cpu_model in UNSUPPORTED_CPU_MODELS:
runtime_probe = run_on_cpu(cpu_model, RUNTIME_PROBE, python)
if runtime_probe.returncode != 0:
raise RuntimeError(
f"Python runtime failed to start on {cpu_model} ({format_failure(runtime_probe)})"
)

result = run_on_cpu(cpu_model, WHEEL_SMOKE_TEST, python)
if cpu_model in UNSUPPORTED_CPU_MODELS:
expected_returncode = -signal.SIGILL
if result.returncode != expected_returncode:
raise RuntimeError(
f"pedalboard wheel unexpectedly ran on non-AVX CPU {cpu_model}; "
f"expected SIGILL ({expected_returncode}), got {format_failure(result)}"
)
print(f"Installed wheel correctly requires AVX on {cpu_model}.")
return

if result.returncode != 0:
raise RuntimeError(
f"pedalboard wheel failed its smoke test on {cpu_model} ({format_failure(result)})"
)
print(f"Installed wheel passed on {cpu_model}.")


def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("cpu_models", nargs="*", default=DEFAULT_CPU_MODELS)
args = parser.parse_args()

for cpu_model in args.cpu_models:
smoke_test_wheel(cpu_model)


if __name__ == "__main__":
main()
11 changes: 5 additions & 6 deletions setup.py
Original file line number Diff line number Diff line change
Expand Up @@ -175,13 +175,12 @@ def ignore_files_matching(files, *matches):
else:
# And on x86, ignore the ARM-specific SIMD code (and KCVI; not GCC or Clang compatible).
fftw_paths = ignore_files_matching(fftw_paths, "neon")
# Use -march=native for local builds to optimize for the current CPU,
# but use a portable baseline for CI builds to avoid "Illegal instruction" errors
# when ccache restores objects built on different runner hardware.
if os.getenv("USE_PORTABLE_SIMD"):
ALL_CFLAGS.append("-mavx")
else:
# Keep distributable builds compatible with AVX-capable x86_64 CPUs.
# Local builds may explicitly target the build machine for extra optimization.
if os.getenv("USE_MARCH_NATIVE") == "1":
ALL_CFLAGS.append("-march=native")
else:
ALL_CFLAGS.append("-mavx")
# Enable SIMD instructions:
ALL_CFLAGS.extend(
[
Expand Down
93 changes: 93 additions & 0 deletions tests/test_linux_wheel_cpu_compatibility.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,93 @@
import os
from pathlib import Path
import subprocess
import sys

import pytest


pytestmark = pytest.mark.skipif(
sys.platform == "win32", reason="The wheel compatibility harness runs only on Linux"
)


HARNESS = Path(__file__).parents[1] / "scripts" / "test_linux_wheel_cpu_compatibility.py"
NON_AVX_CPU = "Nehalem"


def run_harness(
tmp_path: Path,
emulator_body: str,
cpu_models: tuple[str, ...] = (NON_AVX_CPU,),
) -> tuple[subprocess.CompletedProcess, list[str]]:
invocation_log = tmp_path / "qemu-invocations"
emulator = tmp_path / "qemu-x86_64"
emulator.write_text(
"#!/bin/sh\n"
'cpu_model="$2"\n'
'for argument do code="$argument"; done\n'
f'printf "%s|%s\\n" "$cpu_model" "$code" >> "{invocation_log}"\n'
f"{emulator_body}\n"
)
emulator.chmod(0o755)

env = os.environ.copy()
env["PATH"] = f"{tmp_path}{os.pathsep}{env['PATH']}"
result = subprocess.run(
[sys.executable, HARNESS, *cpu_models],
capture_output=True,
env=env,
text=True,
)
invocations = invocation_log.read_text().splitlines()
return result, invocations


def test_non_avx_cpu_requires_wheel_import_to_terminate_with_sigill(tmp_path: Path) -> None:
result, invocations = run_harness(
tmp_path,
'case "$code" in *pedalboard*) kill -ILL $$ ;; *) exit 0 ;; esac',
)

assert result.returncode == 0, result.stderr
assert len(invocations) == 2
assert "pedalboard" not in invocations[0]
assert "pedalboard" in invocations[1]


def test_default_cpu_models_include_the_non_avx_boundary_check(tmp_path: Path) -> None:
result, invocations = run_harness(
tmp_path,
'case "$cpu_model:$code" in Nehalem:*pedalboard*) kill -ILL $$ ;; *) exit 0 ;; esac',
cpu_models=(),
)

assert result.returncode == 0, result.stderr
assert any(invocation.startswith("IvyBridge|") for invocation in invocations)
assert any(invocation.startswith("EPYC-Milan|") for invocation in invocations)
non_avx_invocations = [
invocation for invocation in invocations if invocation.startswith(f"{NON_AVX_CPU}|")
]
assert len(non_avx_invocations) == 2
assert "pedalboard" not in non_avx_invocations[0]
assert "pedalboard" in non_avx_invocations[1]


def test_non_avx_cpu_rejects_a_broken_emulator_or_python_runtime(tmp_path: Path) -> None:
result, invocations = run_harness(tmp_path, "kill -ILL $$")

assert result.returncode != 0
assert len(invocations) == 1
assert "pedalboard" not in invocations[0]


def test_non_avx_cpu_rejects_a_non_sigill_wheel_failure(tmp_path: Path) -> None:
result, invocations = run_harness(
tmp_path,
'case "$code" in *pedalboard*) exit 1 ;; *) exit 0 ;; esac',
)

assert result.returncode != 0
assert len(invocations) == 2
assert "pedalboard" not in invocations[0]
assert "pedalboard" in invocations[1]
Loading