Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
30 commits
Select commit Hold shift + click to select a range
a5b37ad
Add first version of experiment framework
eliasachermann Jun 11, 2026
b988bc4
Measure casting times separately
eliasachermann Jun 15, 2026
d5d9e4f
Merge branch 'elias/dev' into elias/dev-experiment-framework
eliasachermann Jun 15, 2026
e36aeae
Add l1,l2,linf,snr error metrics
eliasachermann Jun 15, 2026
5ae48c5
Merge branch 'elias/dev' into elias/dev-experiment-framework
eliasachermann Jun 16, 2026
ee8c80c
Merge branch 'main' into elias/dev-experiment-framework
eliasachermann Jun 18, 2026
2b5e27d
Narrow tasklet outputs to destination array dtype
eliasachermann Jun 22, 2026
4368089
Store symbols and scalars in db
eliasachermann Jun 24, 2026
eca8e4d
Show progress
eliasachermann Jun 24, 2026
6576f61
Use GPUEvents for timing on GPUs
eliasachermann Jun 25, 2026
5d52d92
Add support for GPU block size
eliasachermann Jun 25, 2026
f055fb9
Remove timers from change_and_propagate_fp_types
eliasachermann Jun 26, 2026
f0beb17
Insert map-boundary casts for mixed precisions
eliasachermann Jun 29, 2026
c8e975c
Add empty side-effect tasklet to prevent statefusion of casting states
eliasachermann Jun 29, 2026
3e61204
Reuse input buffers across performance reps
eliasachermann Jun 29, 2026
7744c26
Add first version off sensitivity analysis
eliasachermann Jul 6, 2026
056c7ff
Merge branch 'main' into elias/dev-experiment-framework
eliasachermann Jul 7, 2026
3b931e3
Add default value to pertubation config
eliasachermann Jul 7, 2026
6644b55
cleanup
eliasachermann Jul 7, 2026
75da8ef
Use default rules correctly
eliasachermann Jul 7, 2026
fa249fa
Remove unused parameter
eliasachermann Jul 7, 2026
af2aa7b
Throw error when noise is half-specified
eliasachermann Jul 7, 2026
e02361f
Run reference only once and cache it
eliasachermann Jul 7, 2026
821182c
Throw error when timer sample count does not match
eliasachermann Jul 7, 2026
1eaac62
cleanup
eliasachermann Jul 7, 2026
967941d
Add examples on how to use experiment framework
eliasachermann Jul 8, 2026
b7b9ccb
Add scipy and tqdm in CI
eliasachermann Jul 8, 2026
2f8e691
Add corpus for shared kernels
eliasachermann Jul 9, 2026
6ef1ff4
Do not simplify GPU transoframtion to keep host<->device copies in se…
eliasachermann Jul 9, 2026
5a4dfa8
Unique build folder per point.
eliasachermann Jul 9, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/fp-arena-ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -53,7 +53,7 @@ jobs:
# FP-Arena without deps so it uses the DaCe just installed (any main),
# rather than re-pulling pinned main from git.
pip install --no-deps ./fp-arena
pip install pytest
pip install pytest scipy tqdm

- name: Run FP-Arena tests
run: |
Expand Down
4 changes: 4 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -22,3 +22,7 @@ out.sdfg
.idea/
*.swp
.DS_Store


#Experiment results database
*.db
2 changes: 2 additions & 0 deletions corpus/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
# Copyright 2019-2026 ETH Zurich and the FP-Arena authors. All rights reserved.
"""A corpus of benchmark kernels written in the DaCe Python frontend."""
31 changes: 5 additions & 26 deletions examples/heat3d.py → corpus/heat3d.py
Original file line number Diff line number Diff line change
@@ -1,20 +1,18 @@
# Copyright 2019-2026 ETH Zurich and the FP-Arena authors. All rights reserved.
"""Run the heat3d stencil in reduced precision."""
"""
The heat3d stencil kernel (Polybench), a 3D Jacobi-style heat-diffusion step.
"""

import dace as dc
import numpy as np

from fp_arena.transformations.change_and_propagate_fp_types import (
change_and_propagate_fp_types,
)

# Default problem size: an N x N x N grid advanced over TSTEPS time steps.
GRID_N, TSTEPS = 40, 20

N = dc.symbol("N", dtype=dc.int64)


@dc.program
def kernel(TSTEPS: dc.int64, A: dc.float64[N, N, N], B: dc.float64[N, N, N]):
def heat3d_kernel(TSTEPS: dc.int64, A: dc.float64[N, N, N], B: dc.float64[N, N, N]):
for t in range(1, TSTEPS):
B[1:-1, 1:-1, 1:-1] = (
0.125 * (A[2:, 1:-1, 1:-1] - 2.0 * A[1:-1, 1:-1, 1:-1] + A[:-2, 1:-1, 1:-1])
Expand All @@ -32,22 +30,3 @@ def kernel(TSTEPS: dc.int64, A: dc.float64[N, N, N], B: dc.float64[N, N, N]):
* (B[1:-1, 1:-1, 2:] - 2.0 * B[1:-1, 1:-1, 1:-1] + B[1:-1, 1:-1, :-2])
+ B[1:-1, 1:-1, 1:-1]
)


def main():
sdfg = kernel.to_sdfg(simplify=True)

change_and_propagate_fp_types(
sdfg,
{"A": dc.float16, "B": dc.float16},
constant_type=dc.float16,
)

rng = np.random.default_rng(0)
A = rng.uniform(0, 100, (GRID_N,) * 3)
B = rng.uniform(0, 100, (GRID_N,) * 3)
sdfg(TSTEPS=TSTEPS, A=A, B=B, N=GRID_N)


if __name__ == "__main__":
main()
55 changes: 55 additions & 0 deletions examples/experiment_error.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,55 @@
# Copyright 2019-2026 ETH Zurich and the FP-Arena authors. All rights reserved.
"""Measure the heat3d stencil's error in reduced precision against an fp64 reference."""

from scipy import stats

from corpus.heat3d import GRID_N, TSTEPS, heat3d_kernel
from fp_arena.experiment import (
ErrorAnalysisConfig,
ExperimentConfig,
ResultStore,
run_error,
)

N_SAMPLES = 3


def main():
experiment = ExperimentConfig(
name="heat3d",
program=heat3d_kernel.to_sdfg(simplify=True),
symbols={"N": GRID_N},
scalar_args={"TSTEPS": TSTEPS},
target="cpu",
inputs={
"A": stats.uniform(0, 100),
"B": stats.uniform(0, 100),
},
)

store = ResultStore("heat3d_results.db")
results = run_error(
ErrorAnalysisConfig(
experiment,
precisions=[{"A": "fp32", "B": "fp32"}, {"A": "fp16", "B": "fp16"}],
reference="fp64",
n_samples=N_SAMPLES,
),
store=store,
)

for r in results:
label = " ".join(f"{k}={v}" for k, v in r.precision.items())
for arr in sorted(r.errors):
e = r.errors[arr]
print(
f"{label:>16} {arr}: rel_mean {e.rel_mean:.3e} "
f"rel_max {e.rel_max:.3e} linf {e.linf:.3e}"
)

print(f"{len(store.query(experiment='heat3d', kind='error'))} rows in the database")
store.close()


if __name__ == "__main__":
main()
54 changes: 54 additions & 0 deletions examples/experiment_performance.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,54 @@
# Copyright 2019-2026 ETH Zurich and the FP-Arena authors. All rights reserved.
"""Time the heat3d stencil across precision points with the experiment framework."""

import statistics

from scipy import stats

from corpus.heat3d import GRID_N, TSTEPS, heat3d_kernel
from fp_arena.experiment import (
ExperimentConfig,
PerformanceAnalysisConfig,
ResultStore,
run_performance,
)


def main():
experiment = ExperimentConfig(
name="heat3d",
program=heat3d_kernel.to_sdfg(simplify=True),
symbols={"N": GRID_N},
scalar_args={"TSTEPS": TSTEPS},
target="cpu",
inputs={
"A": stats.uniform(0, 100),
"B": stats.uniform(0, 100),
},
)

store = ResultStore("heat3d_results.db")
results = run_performance(
PerformanceAnalysisConfig(
experiment,
precisions=[
{}, # the unmodified fp64 program
{"A": "fp32", "B": "fp32"},
{"A": "fp16", "B": "fp16"},
],
n_warmup=1,
n_reps=5,
),
store=store,
)

for r in results:
label = " ".join(f"{k}={v}" for k, v in r.precision.items()) or "fp64 baseline"
print(
f"{label:>16}: total {statistics.median(r.total_times):8.3f} ms "
f"kernel {statistics.median(r.kernel_times):8.3f} ms"
)


if __name__ == "__main__":
main()
50 changes: 50 additions & 0 deletions examples/experiment_perturbation.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,50 @@
# Copyright 2019-2026 ETH Zurich and the FP-Arena authors. All rights reserved.
"""Measure the heat3d stencil's sensitivity to input noise with the experiment framework."""

from scipy import stats

from corpus.heat3d import GRID_N, TSTEPS, heat3d_kernel
from fp_arena.experiment import (
ExperimentConfig,
Noise,
PerturbationAnalysisConfig,
ResultStore,
run_perturbation,
)


def main():
experiment = ExperimentConfig(
name="heat3d",
program=heat3d_kernel.to_sdfg(simplify=True),
symbols={"N": GRID_N},
scalar_args={"TSTEPS": TSTEPS},
target="cpu",
inputs={
"A": stats.uniform(0, 100),
"B": stats.uniform(0, 100),
},
)

noise = Noise(relative=1e-3, relative_dist=stats.uniform(-1.0, 2.0))
store = ResultStore("heat3d_results.db")
results = run_perturbation(
PerturbationAnalysisConfig(
experiment,
noise={"A": noise, "B": noise},
precisions=[{}], # the unmodified fp64 program
),
store=store,
)

for r in results:
for arr in sorted(r.errors):
e = r.errors[arr]
print(
f"perturbed {r.perturbed} -> {arr}: rel_mean {e.rel_mean:.3e} "
f"rel_max {e.rel_max:.3e}"
)


if __name__ == "__main__":
main()
29 changes: 29 additions & 0 deletions examples/reduced_precision.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
# Copyright 2019-2026 ETH Zurich and the FP-Arena authors. All rights reserved.
"""Run the heat3d stencil in reduced precision."""

import dace as dc
import numpy as np

from corpus.heat3d import GRID_N, TSTEPS, heat3d_kernel
from fp_arena.transformations.change_and_propagate_fp_types import (
change_and_propagate_fp_types,
)


def main():
sdfg = heat3d_kernel.to_sdfg(simplify=True)

change_and_propagate_fp_types(
sdfg,
{"A": dc.float16, "B": dc.float16},
constant_type=dc.float16,
)

rng = np.random.default_rng(0)
A = rng.uniform(0, 100, (GRID_N,) * 3)
B = rng.uniform(0, 100, (GRID_N,) * 3)
sdfg(TSTEPS=TSTEPS, A=A, B=B, N=GRID_N)


if __name__ == "__main__":
main()
41 changes: 41 additions & 0 deletions fp_arena/experiment/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,41 @@
# Copyright 2019-2026 ETH Zurich and the FP-Arena authors. All rights reserved.
"""
FP-Arena experiment framework: precision performance and error analysis.
"""

from fp_arena.experiment.config import (
ErrorAnalysisConfig,
ExperimentConfig,
PerformanceAnalysisConfig,
PerturbationAnalysisConfig,
PrecisionMap,
)
from fp_arena.experiment.results import (
ErrorResult,
ErrorStats,
PerfResult,
PerturbationResult,
)
from fp_arena.experiment.inputs import Noise
from fp_arena.experiment import registry
from fp_arena.experiment.runner import run_error, run_performance, run_perturbation
from fp_arena.experiment.store import ResultStore, StoredResult

__all__ = [
"ExperimentConfig",
"PerformanceAnalysisConfig",
"ErrorAnalysisConfig",
"PerturbationAnalysisConfig",
"PrecisionMap",
"PerfResult",
"ErrorResult",
"ErrorStats",
"PerturbationResult",
"Noise",
"run_performance",
"run_error",
"run_perturbation",
"registry",
"ResultStore",
"StoredResult",
]
87 changes: 87 additions & 0 deletions fp_arena/experiment/config.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,87 @@
# Copyright 2019-2026 ETH Zurich and the FP-Arena authors. All rights reserved.
"""
Experiment config.
"""

from dataclasses import dataclass, field
from typing import Any, Dict, FrozenSet, List, Optional, Union

import dace

from fp_arena.experiment.inputs import DistributionLike, Noise

#: ``{array_name: precision_key}`` -- pins a subset of arrays to a target format.
PrecisionMap = Dict[str, str]


@dataclass
class ExperimentConfig:
"""
The program under test plus everything needed to run it -- shared by all analyses.

:param name: identifier used to group results in the database.
:param program: the ``dace.SDFG`` under test
:param inputs: per-array input distributions for the *read* arrays.
:param promotion_rules: promotion rules for resolving precision conflicts (e.g. ``{frozenset({fp16, fp32}): fp32}``); ``None`` uses ``DEFAULT_PROMOTION_RULES``.
:param symbols: values for the SDFG's free symbols (e.g. ``{"N": 100}``);
:param scalar_args: values for non-array scalar arguments.
:param target: ``"cpu"`` (default) or ``"gpu"``;
:param seed: base RNG seed for inputs and noise, shared across analyses.
:param gpu_block_size: GPU thread-block size ``[x, y, z]`` (x = contiguous dim) set on every GPU_Device map; ``None`` uses DaCe's default.
"""

name: str
program: dace.SDFG
inputs: Dict[str, DistributionLike] = field(default_factory=dict)
promotion_rules: Optional[
Dict[FrozenSet[dace.dtypes.typeclass], dace.dtypes.typeclass]
] = None
symbols: Dict[str, int] = field(default_factory=dict)
scalar_args: Dict[str, Any] = field(default_factory=dict)
target: str = "cpu"
seed: int = 0
gpu_block_size: Optional[List[int]] = None


@dataclass
class PerformanceAnalysisConfig:
"""
Measure wall-clock runtime across precision points.
"""

experiment: ExperimentConfig
precisions: List[PrecisionMap]
noise: Dict[str, Noise] = field(default_factory=dict)
n_warmup: int = 1
n_reps: int = 10


@dataclass
class ErrorAnalysisConfig:
"""
Measure per-array error of each precision point against a high-precision
reference, aggregated over ``n_samples`` input realisations.
``reference`` is a single key (e.g. ``"mpfr128"``, ``"fp64"``) or a per-array ``{name: key}`` map.
"""

experiment: ExperimentConfig
precisions: List[PrecisionMap]
noise: Dict[str, Noise] = field(default_factory=dict)
reference: Union[str, Dict[str, str]] = "fp64"
n_samples: int = 1


@dataclass
class PerturbationAnalysisConfig:
"""
Measure input sensitivity: perturb one input array at a time and compare
each written array against the clean run at the same precision point,
aggregated over ``n_samples`` input realisations.
``noise`` names the inputs to perturb (each analysed separately);
``precisions`` lists the points to analyse at (default: the unmodified program).
"""

experiment: ExperimentConfig
noise: Dict[str, Noise]
precisions: List[PrecisionMap] = field(default_factory=lambda: [{}])
n_samples: int = 1
Loading
Loading