diff --git a/.gitignore b/.gitignore index 20fe29d..cad9aab 100644 --- a/.gitignore +++ b/.gitignore @@ -3,3 +3,6 @@ *.jl.mem /Manifest.toml /docs/build/ +.vscode +.perftest_logs +.perftests \ No newline at end of file diff --git a/Project.toml b/Project.toml index 7a0236c..73702a2 100644 --- a/Project.toml +++ b/Project.toml @@ -4,12 +4,14 @@ authors = ["Dvegrod , Samuel Omlin , a version = "0.2.1" [deps] +BandwidthBenchmark = "68eb07c1-04fd-4e62-9736-d6127c4c03c6" BenchmarkTools = "6e4b80f9-dd63-53aa-95a3-0cdb28fa8baf" Configurations = "5218b696-f38b-4ac9-8b61-a12ec717816d" CountFlops = "1db9610d-79e1-487a-8d40-77f3295c7593" -CpuId = "adafc99b-e345-5852-983c-f28acb93d879" +DataFrames = "a93c6f00-e57d-5684-b7b6-d8193f3e46c0" Dates = "ade2ca70-3891-5945-98fb-dc099432e06a" HTTP = "cd3eb016-35fb-5094-929b-558a96fad6f3" +Hwloc = "0e44f5e4-bd66-52a0-8798-143a42290a1d" JLD2 = "033835bb-8acc-5ee8-8aae-3f567f8a3819" JSON = "682c06a0-de6a-54ab-a142-c8b1cf79cde6" LibGit2 = "76f85450-5226-5b5a-8eaa-529ad045b433" @@ -17,12 +19,12 @@ LinearAlgebra = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e" MLStyle = "d8e11817-5142-5d16-987a-aa16d5891078" MacroTools = "1914dd2f-81c6-5fcd-8719-6d5c9610ff09" Pkg = "44cfe95a-1eb2-52ea-b672-e2afdf69b78f" +PrecompileTools = "aea7be01-6a6a-4083-8856-8a6e6704d82a" Printf = "de0858da-6303-5e67-8744-51eddeeeb8d7" -Revise = "295af30f-e4ad-537b-8983-00126c2a3abe" -STREAMBenchmark = "05e9033e-e298-417a-adae-495536c11ad4" Suppressor = "fd094767-a336-5f1f-9728-57cf17d0bbfb" TOML = "fa267f1f-6049-4f14-aa54-33bafae1ed76" Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40" +ThreadPinningCore = "6f48bc29-05ce-4cc8-baad-4adcba581a18" UnicodePlots = "b8865327-cd53-5732-bb35-84acbb429228" [weakdeps] @@ -32,16 +34,19 @@ MPI = "da04e1cc-30fd-572f-bb4f-1f8673147195" PerfTest_MPIExt = "MPI" [compat] -BenchmarkTools="1" -Configurations="0" -CountFlops="0" -CpuId="0" -HTTP="1" -JLD2="0" -JSON="0" -MLStyle="0" -MacroTools="0" -Revise="3" -STREAMBenchmark="0" -Suppressor="0" -UnicodePlots="3" +julia = "1, 1.11" +BandwidthBenchmark = "0.2.0" +BenchmarkTools = "1" +Configurations = "0" +CountFlops = "0" +DataFrames = "1" +HTTP = "1" +Hwloc = "3.3" +JLD2 = "0" +JSON = "0" +MLStyle = "0" +MacroTools = "0" +PrecompileTools = "1.2" +Suppressor = "0" +ThreadPinningCore = "0.4" +UnicodePlots = "3" diff --git a/docs/make.jl b/docs/make.jl index 9a4eebc..9abca67 100644 --- a/docs/make.jl +++ b/docs/make.jl @@ -38,13 +38,14 @@ makedocs(; warnonly = [:missing_docs], pages = [ "Introduction" => "index.md", - "Usage" => "usage.md", + "Quickstart" => "usage.md", "Macros" => "macros.md", "Examples" => [hide("..." => "examples.md"), "examples/mock2-memorythroughput.md", "examples/mock3-roofline.md", "examples/mock4-recursive.md", ], + "Configuration" => "configuration_params.md", "Internals" => "internals.md", "Limitations" => "limitations.md", "API reference" => "api.md", diff --git a/docs/src/configuration_params.MD b/docs/src/configuration_params.MD new file mode 100644 index 0000000..7427f43 --- /dev/null +++ b/docs/src/configuration_params.MD @@ -0,0 +1,104 @@ +# Configuration Parameters + +This document describes all configuration parameters available. Parameters are organized by section, they are specified in TOML format. + +--- + +## `[general]` + +General settings that control the overall behavior of the package. + +| Parameter | Type | Default | Description | +|---|---|---|---| +| `autoflops` | `Bool` | `true` | Whether the autoflop counter is enabled and accessible during testing. | +| `numas` | `String \| Integer \| Float64` | `"single"` | Amount of NUMAS to use when pinning threads. | +| `threads_per_numa` | `String \| Integer \| Float64` | `"single"` | Amount of threads to pin per NUMA. | +| `save_results` | `Bool` | `true` | Whether to record the results of executing the performance suite. | +| `logs_enabled` | `Bool` | `true` | Enable or disable the log subsystem. If false, this will prevent PerfTest from creating a log folder and log files.| +| `save_folder` | `String` | `".perftests"` | Folder where performance suite results shall be stored. | +| `max_saved_results` | `Int` | `20` | Maximum amount of performance test suite execution results to be saved in the result file for each suite. | +| `plotting` | `Bool` | `true` | Whether to have plots in the test output of methodologies that support it. | +| `verbose` | `Int` | `0` | Whether to output the collected logs in the standard output. | +| `recursive` | `Bool` | `true` | If enabled, whenever a file is included inside a recipe that is being transformed, it will be transformed as well. | +| `safe_formulas` | `Bool` | `false` | Deprecated. | +| `suppress_output` | `Bool` | `true` | Whether to hide the output of the test targets. If false the output of each execution will be shown, which will be probably a long output. | + +--- + +## `[regression]` + +Settings for regression testing. + +| Parameter | Type | Default | Description | +|---|---|---|---| +| `enabled` | `Bool` | `true` | Whether to use regression testing in this suite. | +| `dedicated_reference_file` | `String` | `""` | If not an empty string, a succesful suite execution will be saved into this file, as well as in the (result file see `[general] save_folder`). The last saved execution will be used as reference for regression testing. If empty the package will look at the result file instead, and find the latest sucessful execution as reference. | +| `default_threshold` | `Number` | `1.1` | The default threshold of the `@regression` macro if no threshold is specified. | +| `use_bencher` | `Bool` | `false` | Whether to enable Bencher to upload results to an online CI/CD performance benchmark platform. THIS FEATURE IS EXPERIMENTAL and likely unstable. | + +--- + +## `[roofline]` + +Settings for roofline analysis. + +| Parameter | Type | Default | Description | +|---|---|---|---| +| `enabled` | `Bool` | `true` | Whether to use roofline methodologies in this suite. | +| `default_threshold` | `Number` | `0.5` | Default threshold of the `@roofline` macro if no threshold is specified. | + +--- + +## `[memory_bandwidth]` + +Settings for memory bandwidth analysis. + +| Parameter | Type | Default | Description | +|---|---|---|---| +| `enabled` | `Bool` | `true` | Whether to use effective memory throughput methodology in this suite. | +| `default_threshold` | `Number` | `0.5` | Default thresholds of the `@define_eff_mem_throughput` macro if no threshold is specified. | + +--- + +## `[perfcompare]` + +Settings for performance comparison. + +| Parameter | Type | Default | Description | +|---|---|---|---| +| `enabled` | `Bool` | `true` | Whether to use the `@perfcompare` macro in this suite. | + +--- + +## `[machine_benchmarking]` + +Settings for benchmarking the underlying machine. + +| Parameter | Type | Default | Description | +|---|---|---|---| +| `memory_bandwidth_test_buffer_size` | `Int` | `0` | If 0 a buffer size for bandwith benchmarks is used so its at least 4 times bigger than the biggest cache level of the machine. If non-zero this value is used to set the buffer size instead. | + +--- + +## `[MPI]` + +Settings for MPI-based execution. + +| Parameter | Type | Default | Description | +|---|---|---|---| +| `enabled` | `Bool` | `false` | Whether to enable the MPI aware performance suite generation, this will mainly make sure tests are measured in all ranks but evaluated on the main rank only, and that machine benchmarks take into account all ranks.| +| `mode` | `String` | `"reduce"` | This has no function as of now, but it will be used in future versions of perftest. | + +--- + +## `[bencher]` + +Settings for [Bencher](https://bencher.dev) integration. This feature is experimental, and likely prone to bugs. + +| Parameter | Type | Default | Description | +|---|---|---|---| +| `api_key` | `String` | `""` | The Bencher platform key to use to connect to bencher. | +| `api_url` | `String` | `"https://api.bencher.dev"` | The API url. | +| `project_name` | `String` | `""` | The name of the project where metrics and testbeds shall be posted. | +| `organization` | `String` | `""` | The organization that holds the project. | +| `custom_testbed_name` | `String` | `""` | The customized identification of this machine (a.k.a testbed for Bencher) when uploading posting suite execution results. | diff --git a/docs/src/index.MD b/docs/src/index.MD index f3d1bba..bd67a94 100644 --- a/docs/src/index.MD +++ b/docs/src/index.MD @@ -4,10 +4,11 @@ The package `PerfTest` provides the user with a performance regression unit test ## Dependencies `PerfTest` relies on: - BenchmarkTools + - BandwidthBenchmark - Configurations - CountFlops - - CpuId - Dates + - Hwloc - HTTP - JLD2 - JSON @@ -16,9 +17,9 @@ The package `PerfTest` provides the user with a performance regression unit test - MLStyle - MacroTools - Pkg + - PrecompileTools - Printf - Revise - - STREAMBenchmark - Suppressor - TOML - Test diff --git a/docs/src/usage.MD b/docs/src/usage.MD index a17b5b4..fd35468 100644 --- a/docs/src/usage.MD +++ b/docs/src/usage.MD @@ -1,4 +1,4 @@ -# Usage +# Quickstart `PerfTest` provides a set of macros to instrument ordinary Julia test files with performance tests. The idea is to have the posibility of having a functional and a performance suite all in the same place. @@ -8,7 +8,7 @@ The underlying idea of declaring performance tests can be boiled down the follow 2. Tell PerfTest what is the target to be tested by using the macro @perftest 3. Tell PerfTest how the target shall be tested, which metrics are interesting, which of those metrics values would be considered a failure, this can be declared using the metric and methodology macros (see Macros) -The following dummy example embodies the paradigm of the package: +The following dummy example presents how a recipe file looks like: ```julia using ExampleModule : innerProduct, Test, PerfTest # Importing the target and test libraries @@ -29,14 +29,95 @@ The following things can be appreciated in this example: 2. The target of the perftest is the innerProduct function 3. The performance test methodology is a roofline model, the developer expects innerProduct to perform at least at 50% of the maximum flop performance set by the roofline. The operational intensity is defined on the main block of the macro. :autoflop is a symbol that enables the use of an automatic flop count feature. -## Execution +## How to use, a first PerfTest.jl recipe -To execute the functional test, simply run the file. +Lets assume we are a developer that wants to track performance regressions on the components of a package in development. The files discussed here can be accessed in `examples/example_quickstart`. We have the following module: + +```julia +# module.jl +module MyPackage + +# Add [a[1],a[2],...a[end]] and [b[end], b[end-1],...,b[1]] elementwise +function addReversed(A :: Vector{<: Number}, B:: Vector{<: Number}) :: Vector{<:Number} + return [a + b for (a,b) in zip(A, reverse(B))] +end + +end +``` + +We want to track the performance of this package in order to detect performance regressions. To do so we build the following test recipe: + +```julia +# testfile.jl +using Test,PerfTest + +include("module.jl") + +@perftest_config " +[general] +verbose = 3 +[regression] +dedicated_reference_file='reference.JLD2' +" + +@testset "addReversed tests" begin + N = 10 + # We want the size to be bigger on the performance test + @on_perftest_exec begin + N = 1_000_000 + end + # We set the regression checker, we dont specify a metric therefore the default (median time elapsed) is used + # low_is_bad=false time elapsed metrics are considered worse the bigger they are + # threshold = 1.05 the test will fail if the time is 105% of the reference or greater, in other words: @test time_elapsed < 1.05 * reference + @regression threshold=1.05 low_is_bad=false + + A = [i for i in 1:N] + B = [N-i for i in 1:N] + + result = @perftest MyPackage.addReversed(A, B) + @test sum(result) == N*N +end +``` + +### Running the functional test side of the recipe: + +```julia + include("testfile.jl") +``` + +Or if the module is setup as its own package (not this example): +```julia + using Pkg; Pkg.test() +``` + +### Running the performance test side of the recipe: + +We are doing regression, so in case no reference has been made we will need to execute it at least twice. Once to record the reference, and after that whenever the developer wants to check for a performance regression. + +The first time PerfTest is run on a specific directory, a configuration file will be created with a set of default values. Configuration parameters can be set as well by the `@perftest_config` macro, as seen above. The macro takes the highest priority over any other configuration source. +```julia + # This will record a successful performance suite + using PerfTest; runperftests("testfile.jl") +``` + +In between, the developer may add new changes to the implementation or checks out a different git branch. + +```julia + # This will compare the suite results against the reference, if the tests are sucessful the results become the reference. + using PerfTest; runperftests("testfile.jl") +``` -To execute the performance test, pass the file path to `PerfTest.transform` and evaluate the resulting test suite. The result of transform can be also saved as a file (see `PerfTest.saveExprAsFile`) for later execution. For more information have a look at the [Examples](@ref) and see the [API reference](@ref) for details on the usage of `PerfTest`. +## Main use cases of this package: + +This package supports both performance testing by regression checks and by performance methodologies, e.g roofline models. It is meant to cover the following two situations: + + 1. A developer wants to track potential regressions over the course of package development. + 2. A developer wants to set machine-agnostic performance tests by using methodologies or custom metrics that are normalized by machine parameters, with the purpose of verifying that the package profits from the capabilities of the machines its been executed on. + 2.1. During development. + 2.2. During the whole software lifecycle, users can check if the package is properly set up in their machine. ## Installation diff --git a/examples/example-paper-implicitglobalgrid/EXP_test_halo_thr.jl b/examples/example-paper-implicitglobalgrid/EXP_test_halo_thr.jl index 78e89a8..1596627 100644 --- a/examples/example-paper-implicitglobalgrid/EXP_test_halo_thr.jl +++ b/examples/example-paper-implicitglobalgrid/EXP_test_halo_thr.jl @@ -1,10 +1,7 @@ -# IMPORTANT: To replicate the same results as in the original paper, divide the GB/s values by 3 -# this difference is due to the STREAM benchmark policy, which counts each byte transferred thrice (write-allocate cache) and thus the target does as write_allocate +# IMPORTANT: To replicate the same results as in the original paper, divide the GB/s values by 2 +# this difference is due to the benchmark policy, which counts each byte transferred twice (no write-allocate in copy in PerfTest 1.2.3 in PerfTest 1.2.1 this would be thrice instead) # in our example the policy is to measure throughput as the amount of bytes transferred on an execution which is a different policy # Nevertheless, the percentages remain constant and the test is valid regardless of the used criterion -# NOTE: All tests of this file can be run with any number of processes. -# Nearly all of the functionality can however be verified with one single process -# (thanks to the usage of periodic boundaries in most of the full halo update tests). push!(LOAD_PATH, "../src") using Test @@ -21,7 +18,7 @@ NTHREADS = 16 # 256M Elements on STREAM Benchmark (2GB) @perftest_config " [general] -verbose = true +verbose = 3 autoflops = false [regression] @@ -74,7 +71,7 @@ dz = 1.0 i = 0 # For info in the x3 multiplier see beggining of file @define_eff_memory_throughput ratio=0.9 begin - (nx * ny * 8) * 3 * MPI.Comm_size(MPI.COMM_WORLD) / :median_time + (nx * ny * 8) * 2 * MPI.Comm_size(MPI.COMM_WORLD) / :median_time end @auxiliary_metric name="Time" units="s" begin :median_time diff --git a/examples/example-paper-implicitglobalgrid/Manifest.toml b/examples/example-paper-implicitglobalgrid/Manifest.toml deleted file mode 100644 index bc51c8f..0000000 --- a/examples/example-paper-implicitglobalgrid/Manifest.toml +++ /dev/null @@ -1,1481 +0,0 @@ -# This file is machine-generated - editing it directly is not advised - -julia_version = "1.10.4" -manifest_format = "2.0" -project_hash = "70db121c19a2d7298f11b17ad98951819ce15980" - -[[deps.AMDGPU]] -deps = ["AbstractFFTs", "AcceleratedKernels", "Adapt", "Atomix", "CEnum", "ExprTools", "GPUArrays", "GPUCompiler", "KernelAbstractions", "LLD_jll", "LLVM", "LLVM_jll", "Libdl", "LinearAlgebra", "Pkg", "Preferences", "PrettyTables", "Printf", "ROCmDeviceLibs_jll", "Random", "Random123", "RandomNumbers", "SparseArrays", "SpecialFunctions", "StaticArraysCore", "Statistics", "UnsafeAtomics"] -git-tree-sha1 = "915c5604d60710fe0bc13f87fbf31e3827f016ea" -uuid = "21141c5a-9bdb-4563-92ae-f87d6854732e" -version = "1.2.3" - - [deps.AMDGPU.extensions] - AMDGPUChainRulesCoreExt = "ChainRulesCore" - AMDGPUEnzymeCoreExt = "EnzymeCore" - - [deps.AMDGPU.weakdeps] - ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4" - EnzymeCore = "f151be2c-9106-41f4-ab19-57ee4f262869" - -[[deps.AbstractFFTs]] -deps = ["LinearAlgebra"] -git-tree-sha1 = "d92ad398961a3ed262d8bf04a1a2b8340f915fef" -uuid = "621f4979-c628-5d54-868e-fcf4e3e8185c" -version = "1.5.0" - - [deps.AbstractFFTs.extensions] - AbstractFFTsChainRulesCoreExt = "ChainRulesCore" - AbstractFFTsTestExt = "Test" - - [deps.AbstractFFTs.weakdeps] - ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4" - Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40" - -[[deps.AcceleratedKernels]] -deps = ["ArgCheck", "GPUArrays", "KernelAbstractions", "Markdown", "OhMyThreads", "Polyester"] -git-tree-sha1 = "0af0b85e3e08a1dadeea28ec688898906eab399a" -uuid = "6a4ca0a5-0e36-4168-a932-d9be78d558f1" -version = "0.3.3" - - [deps.AcceleratedKernels.extensions] - AcceleratedKernelsMetalExt = "Metal" - AcceleratedKernelsoneAPIExt = "oneAPI" - - [deps.AcceleratedKernels.weakdeps] - Metal = "dde4c033-4e86-420c-a63e-0dd931031962" - oneAPI = "8f75cd03-7ff8-4ecb-9b8f-daf728133b1b" - -[[deps.Accessors]] -deps = ["CompositionsBase", "ConstructionBase", "Dates", "InverseFunctions", "MacroTools"] -git-tree-sha1 = "856ecd7cebb68e5fc87abecd2326ad59f0f911f3" -uuid = "7d9f7c33-5ae7-4f3b-8dc6-eff91059b697" -version = "0.1.43" - - [deps.Accessors.extensions] - AxisKeysExt = "AxisKeys" - IntervalSetsExt = "IntervalSets" - LinearAlgebraExt = "LinearAlgebra" - StaticArraysExt = "StaticArrays" - StructArraysExt = "StructArrays" - TestExt = "Test" - UnitfulExt = "Unitful" - - [deps.Accessors.weakdeps] - AxisKeys = "94b1ba4f-4ee9-5380-92f1-94cde586c3c5" - IntervalSets = "8197267c-284f-5f27-9208-e0e47529a953" - LinearAlgebra = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e" - StaticArrays = "90137ffa-7385-5640-81b9-e52037218182" - StructArrays = "09ab397b-f2b6-538f-b94a-2f83cf4a842a" - Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40" - Unitful = "1986cc42-f94f-5a68-af5c-568840ba703d" - -[[deps.Adapt]] -deps = ["LinearAlgebra", "Requires"] -git-tree-sha1 = "7e35fca2bdfba44d797c53dfe63a51fabf39bfc0" -uuid = "79e6a3ab-5dfb-504d-930d-738a2a938a0e" -version = "4.4.0" -weakdeps = ["SparseArrays", "StaticArrays"] - - [deps.Adapt.extensions] - AdaptSparseArraysExt = "SparseArrays" - AdaptStaticArraysExt = "StaticArrays" - -[[deps.AliasTables]] -deps = ["PtrArrays", "Random"] -git-tree-sha1 = "9876e1e164b144ca45e9e3198d0b689cadfed9ff" -uuid = "66dad0bd-aa9a-41b7-9441-69ab47430ed8" -version = "1.1.3" - -[[deps.ArgCheck]] -git-tree-sha1 = "f9e9a66c9b7be1ad7372bbd9b062d9230c30c5ce" -uuid = "dce04be8-c92d-5529-be00-80e4d2c0e197" -version = "2.5.0" - -[[deps.ArgTools]] -uuid = "0dad84c5-d112-42e6-8d28-ef12dabb789f" -version = "1.1.1" - -[[deps.ArrayInterface]] -deps = ["Adapt", "LinearAlgebra"] -git-tree-sha1 = "d81ae5489e13bc03567d4fbbb06c546a5e53c857" -uuid = "4fba245c-0d91-5ea0-9b3e-6abc04ee57a9" -version = "7.22.0" - - [deps.ArrayInterface.extensions] - ArrayInterfaceBandedMatricesExt = "BandedMatrices" - ArrayInterfaceBlockBandedMatricesExt = "BlockBandedMatrices" - ArrayInterfaceCUDAExt = "CUDA" - ArrayInterfaceCUDSSExt = ["CUDSS", "CUDA"] - ArrayInterfaceChainRulesCoreExt = "ChainRulesCore" - ArrayInterfaceChainRulesExt = "ChainRules" - ArrayInterfaceGPUArraysCoreExt = "GPUArraysCore" - ArrayInterfaceMetalExt = "Metal" - ArrayInterfaceReverseDiffExt = "ReverseDiff" - ArrayInterfaceSparseArraysExt = "SparseArrays" - ArrayInterfaceStaticArraysCoreExt = "StaticArraysCore" - ArrayInterfaceTrackerExt = "Tracker" - - [deps.ArrayInterface.weakdeps] - BandedMatrices = "aae01518-5342-5314-be14-df237901396f" - BlockBandedMatrices = "ffab5731-97b5-5995-9138-79e8c1846df0" - CUDA = "052768ef-5323-5732-b1bb-66c8b64840ba" - CUDSS = "45b445bb-4962-46a0-9369-b4df9d0f772e" - ChainRules = "082447d4-558c-5d27-93f4-14fc19e9eca2" - ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4" - GPUArraysCore = "46192b85-c4d5-4398-a991-12ede77f4527" - Metal = "dde4c033-4e86-420c-a63e-0dd931031962" - ReverseDiff = "37e2e3b7-166d-5795-8a7a-e32c996b4267" - SparseArrays = "2f01184e-e22b-5df5-ae63-d93ebab69eaf" - StaticArraysCore = "1e83bf80-4336-4d27-bf5d-d5a4f845583c" - Tracker = "9f7883ad-71c0-57eb-9f7f-b5c9e6d3789c" - -[[deps.Artifacts]] -uuid = "56f22d72-fd6d-98f1-02f0-08ddc0907c33" - -[[deps.Atomix]] -deps = ["UnsafeAtomics"] -git-tree-sha1 = "29bb0eb6f578a587a49da16564705968667f5fa8" -uuid = "a9b6321e-bd34-4604-b9c9-b65b8de01458" -version = "1.1.2" - - [deps.Atomix.extensions] - AtomixCUDAExt = "CUDA" - AtomixMetalExt = "Metal" - AtomixOpenCLExt = "OpenCL" - AtomixoneAPIExt = "oneAPI" - - [deps.Atomix.weakdeps] - CUDA = "052768ef-5323-5732-b1bb-66c8b64840ba" - Metal = "dde4c033-4e86-420c-a63e-0dd931031962" - OpenCL = "08131aa3-fb12-5dee-8b74-c09406e224a2" - oneAPI = "8f75cd03-7ff8-4ecb-9b8f-daf728133b1b" - -[[deps.BFloat16s]] -deps = ["LinearAlgebra", "Printf", "Random"] -git-tree-sha1 = "3b642331600250f592719140c60cf12372b82d66" -uuid = "ab4f0b2a-ad5b-11e8-123f-65d77653426b" -version = "0.5.1" - -[[deps.BangBang]] -deps = ["Accessors", "ConstructionBase", "InitialValues", "LinearAlgebra"] -git-tree-sha1 = "7edecc3b90af8373014393e98e8fcfbdf3b52543" -uuid = "198e06fe-97b7-11e9-32a5-e1d131e6ad66" -version = "0.4.7" - - [deps.BangBang.extensions] - BangBangChainRulesCoreExt = "ChainRulesCore" - BangBangDataFramesExt = "DataFrames" - BangBangStaticArraysExt = "StaticArrays" - BangBangStructArraysExt = "StructArrays" - BangBangTablesExt = "Tables" - BangBangTypedTablesExt = "TypedTables" - - [deps.BangBang.weakdeps] - ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4" - DataFrames = "a93c6f00-e57d-5684-b7b6-d8193f3e46c0" - StaticArrays = "90137ffa-7385-5640-81b9-e52037218182" - StructArrays = "09ab397b-f2b6-538f-b94a-2f83cf4a842a" - Tables = "bd369af6-aec1-5ad0-b16a-f7cc5008161c" - TypedTables = "9d95f2ec-7b3d-5a63-8d20-e2491e220bb9" - -[[deps.Base64]] -uuid = "2a0f44e3-6c83-55bd-87e4-b1978d98bd5f" - -[[deps.BenchmarkTools]] -deps = ["Compat", "JSON", "Logging", "Printf", "Profile", "Statistics", "UUIDs"] -git-tree-sha1 = "7fecfb1123b8d0232218e2da0c213004ff15358d" -uuid = "6e4b80f9-dd63-53aa-95a3-0cdb28fa8baf" -version = "1.6.3" - -[[deps.BitFlags]] -git-tree-sha1 = "0691e34b3bb8be9307330f88d1a3c3f25466c24d" -uuid = "d1d4a3ce-64b1-5f1a-9ba4-7e7e69966f35" -version = "0.1.9" - -[[deps.BitTwiddlingConvenienceFunctions]] -deps = ["Static"] -git-tree-sha1 = "f21cfd4950cb9f0587d5067e69405ad2acd27b87" -uuid = "62783981-4cbd-42fc-bca8-16325de8dc4b" -version = "0.1.6" - -[[deps.CEnum]] -git-tree-sha1 = "389ad5c84de1ae7cf0e28e381131c98ea87d54fc" -uuid = "fa961155-64e5-5f13-b03f-caf6b980ea82" -version = "0.5.0" - -[[deps.CPUSummary]] -deps = ["CpuId", "IfElse", "PrecompileTools", "Preferences", "Static"] -git-tree-sha1 = "f3a21d7fc84ba618a779d1ed2fcca2e682865bab" -uuid = "2a0fbf3d-bb9c-48f3-b0a9-814d99fd7ab9" -version = "0.2.7" - -[[deps.CUDA]] -deps = ["AbstractFFTs", "Adapt", "BFloat16s", "CEnum", "CUDA_Compiler_jll", "CUDA_Driver_jll", "CUDA_Runtime_Discovery", "CUDA_Runtime_jll", "Crayons", "DataFrames", "ExprTools", "GPUArrays", "GPUCompiler", "GPUToolbox", "KernelAbstractions", "LLVM", "LLVMLoopInfo", "LazyArtifacts", "Libdl", "LinearAlgebra", "Logging", "NVTX", "Preferences", "PrettyTables", "Printf", "Random", "Random123", "RandomNumbers", "Reexport", "Requires", "SparseArrays", "StaticArrays", "Statistics", "demumble_jll"] -git-tree-sha1 = "224c078c40f7cbd98f393d1722953f3a7f24a3ca" -uuid = "052768ef-5323-5732-b1bb-66c8b64840ba" -version = "5.8.5" - - [deps.CUDA.extensions] - ChainRulesCoreExt = "ChainRulesCore" - EnzymeCoreExt = "EnzymeCore" - SparseMatricesCSRExt = "SparseMatricesCSR" - SpecialFunctionsExt = "SpecialFunctions" - - [deps.CUDA.weakdeps] - ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4" - EnzymeCore = "f151be2c-9106-41f4-ab19-57ee4f262869" - SparseMatricesCSR = "a0a7dd2c-ebf4-11e9-1f05-cf50bc540ca1" - SpecialFunctions = "276daf66-3868-5448-9aa4-cd146d93841b" - -[[deps.CUDA_Compiler_jll]] -deps = ["Artifacts", "CUDA_Driver_jll", "CUDA_Runtime_jll", "JLLWrappers", "LazyArtifacts", "Libdl", "TOML"] -git-tree-sha1 = "68bac021c5d511fe39c2f4306cc9623a6626dbb4" -uuid = "d1e2174e-dfdc-576e-b43e-73b79eb1aca8" -version = "0.2.2+0" - -[[deps.CUDA_Driver_jll]] -deps = ["Artifacts", "JLLWrappers", "Libdl", "Pkg"] -git-tree-sha1 = "23bf4e60006b78544f753880fbcf1aa158a7669c" -uuid = "4ee394cb-3365-5eb0-8335-949819d2adfc" -version = "13.1.0+2" - -[[deps.CUDA_Runtime_Discovery]] -deps = ["Libdl"] -git-tree-sha1 = "f9a521f52d236fe49f1028d69e549e7f2644bb72" -uuid = "1af6417a-86b4-443c-805f-a4643ffb695f" -version = "1.0.0" - -[[deps.CUDA_Runtime_jll]] -deps = ["Artifacts", "CUDA_Driver_jll", "JLLWrappers", "LazyArtifacts", "Libdl", "TOML"] -git-tree-sha1 = "92cd84e2b760e471d647153ea5efc5789fc5e8b2" -uuid = "76a88914-d11a-5bdc-97e0-2f5a05c973a2" -version = "0.19.2+0" - -[[deps.Cassette]] -git-tree-sha1 = "f8764df8d9d2aec2812f009a1ac39e46c33354b8" -uuid = "7057c7e9-c182-5462-911a-8362d720325c" -version = "0.3.14" - -[[deps.CellArrays]] -deps = ["Adapt", "StaticArrays"] -git-tree-sha1 = "db45cc84e9a2ef63e65c1ae206c9d4706197c099" -uuid = "d35fcfd7-7af4-4c67-b1aa-d78070614af4" -version = "0.3.2" - - [deps.CellArrays.weakdeps] - AMDGPU = "21141c5a-9bdb-4563-92ae-f87d6854732e" - CUDA = "052768ef-5323-5732-b1bb-66c8b64840ba" - Metal = "dde4c033-4e86-420c-a63e-0dd931031962" - -[[deps.ChunkCodecCore]] -git-tree-sha1 = "1a3ad7e16a321667698a19e77362b35a1e94c544" -uuid = "0b6fb165-00bc-4d37-ab8b-79f91016dbe1" -version = "1.0.1" - -[[deps.ChunkCodecLibZlib]] -deps = ["ChunkCodecCore", "Zlib_jll"] -git-tree-sha1 = "cee8104904c53d39eb94fd06cbe60cb5acde7177" -uuid = "4c0bbee4-addc-4d73-81a0-b6caacae83c8" -version = "1.0.0" - -[[deps.ChunkCodecLibZstd]] -deps = ["ChunkCodecCore", "Zstd_jll"] -git-tree-sha1 = "34d9873079e4cb3d0c62926a225136824677073f" -uuid = "55437552-ac27-4d47-9aa3-63184e8fd398" -version = "1.0.0" - -[[deps.ChunkSplitters]] -git-tree-sha1 = "63a3903063d035260f0f6eab00f517471c5dc784" -uuid = "ae650224-84b6-46f8-82ea-d812ca08434e" -version = "3.1.2" - -[[deps.CloseOpenIntervals]] -deps = ["Static", "StaticArrayInterface"] -git-tree-sha1 = "05ba0d07cd4fd8b7a39541e31a7b0254704ea581" -uuid = "fb6a15b2-703c-40df-9091-08a04967cfa9" -version = "0.1.13" - -[[deps.CodeTracking]] -deps = ["InteractiveUtils", "UUIDs"] -git-tree-sha1 = "b7231a755812695b8046e8471ddc34c8268cbad5" -uuid = "da1fd8a2-8d9e-5ec2-8556-3022fb5608a2" -version = "3.0.0" - -[[deps.CodecZlib]] -deps = ["TranscodingStreams", "Zlib_jll"] -git-tree-sha1 = "962834c22b66e32aa10f7611c08c8ca4e20749a9" -uuid = "944b1d66-785c-5afd-91f1-9de20f533193" -version = "0.7.8" - -[[deps.ColorSchemes]] -deps = ["ColorTypes", "ColorVectorSpace", "Colors", "FixedPointNumbers", "PrecompileTools", "Random"] -git-tree-sha1 = "b0fd3f56fa442f81e0a47815c92245acfaaa4e34" -uuid = "35d6a980-a343-548e-a6ea-1d62b119f2f4" -version = "3.31.0" - -[[deps.ColorTypes]] -deps = ["FixedPointNumbers", "Random"] -git-tree-sha1 = "67e11ee83a43eb71ddc950302c53bf33f0690dfe" -uuid = "3da002f7-5984-5a60-b8a6-cbb66c0b333f" -version = "0.12.1" - - [deps.ColorTypes.extensions] - StyledStringsExt = "StyledStrings" - - [deps.ColorTypes.weakdeps] - StyledStrings = "f489334b-da3d-4c2e-b8f0-e476e12c162b" - -[[deps.ColorVectorSpace]] -deps = ["ColorTypes", "FixedPointNumbers", "LinearAlgebra", "Requires", "Statistics", "TensorCore"] -git-tree-sha1 = "8b3b6f87ce8f65a2b4f857528fd8d70086cd72b1" -uuid = "c3611d14-8923-5661-9e6a-0046d554d3a4" -version = "0.11.0" -weakdeps = ["SpecialFunctions"] - - [deps.ColorVectorSpace.extensions] - SpecialFunctionsExt = "SpecialFunctions" - -[[deps.Colors]] -deps = ["ColorTypes", "FixedPointNumbers", "Reexport"] -git-tree-sha1 = "37ea44092930b1811e666c3bc38065d7d87fcc74" -uuid = "5ae59095-9a9b-59fe-a467-6f913c188581" -version = "0.13.1" - -[[deps.CommonWorldInvalidations]] -git-tree-sha1 = "ae52d1c52048455e85a387fbee9be553ec2b68d0" -uuid = "f70d9fcc-98c5-4d4a-abd7-e4cdeebd8ca8" -version = "1.0.0" - -[[deps.Compat]] -deps = ["TOML", "UUIDs"] -git-tree-sha1 = "9d8a54ce4b17aa5bdce0ea5c34bc5e7c340d16ad" -uuid = "34da2185-b29b-5c13-b0c7-acf172513d20" -version = "4.18.1" -weakdeps = ["Dates", "LinearAlgebra"] - - [deps.Compat.extensions] - CompatLinearAlgebraExt = "LinearAlgebra" - -[[deps.Compiler]] -git-tree-sha1 = "382d79bfe72a406294faca39ef0c3cef6e6ce1f1" -uuid = "807dbc54-b67e-4c79-8afb-eafe4df6f2e1" -version = "0.1.1" - -[[deps.CompilerSupportLibraries_jll]] -deps = ["Artifacts", "Libdl"] -uuid = "e66e0078-7015-5450-92f7-15fbd957f2ae" -version = "1.1.1+0" - -[[deps.CompositionsBase]] -git-tree-sha1 = "802bb88cd69dfd1509f6670416bd4434015693ad" -uuid = "a33af91c-f02d-484b-be07-31d278c5ca2b" -version = "0.1.2" -weakdeps = ["InverseFunctions"] - - [deps.CompositionsBase.extensions] - CompositionsBaseInverseFunctionsExt = "InverseFunctions" - -[[deps.ConcurrentUtilities]] -deps = ["Serialization", "Sockets"] -git-tree-sha1 = "d9d26935a0bcffc87d2613ce14c527c99fc543fd" -uuid = "f0e56b4a-5159-44fe-b623-3e5288b988bb" -version = "2.5.0" - -[[deps.Configurations]] -deps = ["ExproniconLite", "OrderedCollections", "TOML"] -git-tree-sha1 = "4358750bb58a3caefd5f37a4a0c5bfdbbf075252" -uuid = "5218b696-f38b-4ac9-8b61-a12ec717816d" -version = "0.17.6" - -[[deps.ConstructionBase]] -git-tree-sha1 = "b4b092499347b18a015186eae3042f72267106cb" -uuid = "187b0558-2788-49d3-abe0-74a17ed4e7c9" -version = "1.6.0" - - [deps.ConstructionBase.extensions] - ConstructionBaseIntervalSetsExt = "IntervalSets" - ConstructionBaseLinearAlgebraExt = "LinearAlgebra" - ConstructionBaseStaticArraysExt = "StaticArrays" - - [deps.ConstructionBase.weakdeps] - IntervalSets = "8197267c-284f-5f27-9208-e0e47529a953" - LinearAlgebra = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e" - StaticArrays = "90137ffa-7385-5640-81b9-e52037218182" - -[[deps.Contour]] -git-tree-sha1 = "439e35b0b36e2e5881738abc8857bd92ad6ff9a8" -uuid = "d38c429a-6771-53c6-b99e-75d170b6e991" -version = "0.6.3" - -[[deps.CountFlops]] -deps = ["Cassette"] -git-tree-sha1 = "3d4b4b334a3a78d1339e4ab2878c688233a469aa" -uuid = "1db9610d-79e1-487a-8d40-77f3295c7593" -version = "0.1.0" - -[[deps.CpuId]] -deps = ["Markdown"] -git-tree-sha1 = "fcbb72b032692610bfbdb15018ac16a36cf2e406" -uuid = "adafc99b-e345-5852-983c-f28acb93d879" -version = "0.3.1" - -[[deps.Crayons]] -git-tree-sha1 = "249fe38abf76d48563e2f4556bebd215aa317e15" -uuid = "a8cc5b0e-0ffa-5ad4-8c14-923d3ee1735f" -version = "4.1.1" - -[[deps.DataAPI]] -git-tree-sha1 = "abe83f3a2f1b857aac70ef8b269080af17764bbe" -uuid = "9a962f9c-6df0-11e9-0e5d-c546b8b5ee8a" -version = "1.16.0" - -[[deps.DataFrames]] -deps = ["Compat", "DataAPI", "DataStructures", "Future", "InlineStrings", "InvertedIndices", "IteratorInterfaceExtensions", "LinearAlgebra", "Markdown", "Missings", "PooledArrays", "PrecompileTools", "PrettyTables", "Printf", "Random", "Reexport", "SentinelArrays", "SortingAlgorithms", "Statistics", "TableTraits", "Tables", "Unicode"] -git-tree-sha1 = "d8928e9169ff76c6281f39a659f9bca3a573f24c" -uuid = "a93c6f00-e57d-5684-b7b6-d8193f3e46c0" -version = "1.8.1" - -[[deps.DataStructures]] -deps = ["OrderedCollections"] -git-tree-sha1 = "e357641bb3e0638d353c4b29ea0e40ea644066a6" -uuid = "864edb3b-99cc-5e75-8d2d-829cb0a9cfe8" -version = "0.19.3" - -[[deps.DataValueInterfaces]] -git-tree-sha1 = "bfc1187b79289637fa0ef6d4436ebdfe6905cbd6" -uuid = "e2d170a0-9d28-54be-80f0-106bbe20a464" -version = "1.0.0" - -[[deps.Dates]] -deps = ["Printf"] -uuid = "ade2ca70-3891-5945-98fb-dc099432e06a" - -[[deps.DelimitedFiles]] -deps = ["Mmap"] -git-tree-sha1 = "9e2f36d3c96a820c678f2f1f1782582fcf685bae" -uuid = "8bb1440f-4735-579b-a4ab-409b98df4dab" -version = "1.9.1" - -[[deps.Distributed]] -deps = ["Random", "Serialization", "Sockets"] -uuid = "8ba89e20-285c-5b6f-9357-94700520ee1b" - -[[deps.DocStringExtensions]] -git-tree-sha1 = "7442a5dfe1ebb773c29cc2962a8980f47221d76c" -uuid = "ffbed154-4ef7-542d-bbb7-c09d3a79fcae" -version = "0.9.5" - -[[deps.Downloads]] -deps = ["ArgTools", "FileWatching", "LibCURL", "NetworkOptions"] -uuid = "f43a241f-c20a-4ad4-852c-f6b1247861c6" -version = "1.6.0" - -[[deps.ExceptionUnwrapping]] -deps = ["Test"] -git-tree-sha1 = "d36f682e590a83d63d1c7dbd287573764682d12a" -uuid = "460bff9d-24e4-43bc-9d9f-a8973cb893f4" -version = "0.1.11" - -[[deps.ExprTools]] -git-tree-sha1 = "27415f162e6028e81c72b82ef756bf321213b6ec" -uuid = "e2ba6199-217a-4e67-a87a-7c52f15ade04" -version = "0.1.10" - -[[deps.ExproniconLite]] -git-tree-sha1 = "c13f0b150373771b0fdc1713c97860f8df12e6c2" -uuid = "55351af7-c7e9-48d6-89ff-24e801d99491" -version = "0.10.14" - -[[deps.FileIO]] -deps = ["Pkg", "Requires", "UUIDs"] -git-tree-sha1 = "d60eb76f37d7e5a40cc2e7c36974d864b82dc802" -uuid = "5789e2e9-d7fb-5bc7-8068-2c6fae9b9549" -version = "1.17.1" -weakdeps = ["HTTP"] - - [deps.FileIO.extensions] - HTTPExt = "HTTP" - -[[deps.FileWatching]] -uuid = "7b1f6079-737a-58dc-b8bc-7a2ca5c1b5ee" - -[[deps.FixedPointNumbers]] -deps = ["Statistics"] -git-tree-sha1 = "05882d6995ae5c12bb5f36dd2ed3f61c98cbb172" -uuid = "53c48c17-4a7d-5ca2-90c5-79b7896eea93" -version = "0.8.5" - -[[deps.Future]] -deps = ["Random"] -uuid = "9fa8497b-333b-5362-9e8d-4d0656e87820" - -[[deps.GPUArrays]] -deps = ["Adapt", "GPUArraysCore", "KernelAbstractions", "LLVM", "LinearAlgebra", "Printf", "Random", "Reexport", "ScopedValues", "Serialization", "SparseArrays", "Statistics"] -git-tree-sha1 = "4afe2c81e0160a7190c24361defc8c28f3616bac" -uuid = "0c68f7d7-f131-5f86-a1c3-88cf8149b2d7" -version = "11.3.4" -weakdeps = ["JLD2"] - - [deps.GPUArrays.extensions] - JLD2Ext = "JLD2" - -[[deps.GPUArraysCore]] -deps = ["Adapt"] -git-tree-sha1 = "83cf05ab16a73219e5f6bd1bdfa9848fa24ac627" -uuid = "46192b85-c4d5-4398-a991-12ede77f4527" -version = "0.2.0" - -[[deps.GPUCompiler]] -deps = ["ExprTools", "InteractiveUtils", "LLVM", "Libdl", "Logging", "PrecompileTools", "Preferences", "Scratch", "Serialization", "TOML", "Tracy", "UUIDs"] -git-tree-sha1 = "966946d226e8b676ca6409454718accb18c34c54" -uuid = "61eb1bfa-7361-4325-ad38-22787b887f55" -version = "1.8.2" - -[[deps.GPUToolbox]] -deps = ["LLVM"] -git-tree-sha1 = "5bfe837129bf49e2e049b4f1517546055cc16a93" -uuid = "096a3bc2-3ced-46d0-87f4-dd12716f4bfc" -version = "0.3.0" - -[[deps.HTTP]] -deps = ["Base64", "CodecZlib", "ConcurrentUtilities", "Dates", "ExceptionUnwrapping", "Logging", "LoggingExtras", "MbedTLS", "NetworkOptions", "OpenSSL", "PrecompileTools", "Random", "SimpleBufferStream", "Sockets", "URIs", "UUIDs"] -git-tree-sha1 = "5e6fe50ae7f23d171f44e311c2960294aaa0beb5" -uuid = "cd3eb016-35fb-5094-929b-558a96fad6f3" -version = "1.10.19" - -[[deps.HashArrayMappedTries]] -git-tree-sha1 = "2eaa69a7cab70a52b9687c8bf950a5a93ec895ae" -uuid = "076d061b-32b6-4027-95e0-9a2c6f6d7e74" -version = "0.2.0" - -[[deps.HostCPUFeatures]] -deps = ["BitTwiddlingConvenienceFunctions", "IfElse", "Libdl", "Preferences", "Static"] -git-tree-sha1 = "af9ab7d1f70739a47f03be78771ebda38c3c71bf" -uuid = "3e5b6fbb-0976-4d2c-9146-d79de83f2fb0" -version = "0.1.18" - -[[deps.Hwloc]] -deps = ["CEnum", "Hwloc_jll", "Printf"] -git-tree-sha1 = "6a3d80f31ff87bc94ab22a7b8ec2f263f9a6a583" -uuid = "0e44f5e4-bd66-52a0-8798-143a42290a1d" -version = "3.3.0" - - [deps.Hwloc.extensions] - HwlocTrees = "AbstractTrees" - - [deps.Hwloc.weakdeps] - AbstractTrees = "1520ce14-60c1-5f80-bbc7-55ef81b5835c" - -[[deps.Hwloc_jll]] -deps = ["Artifacts", "JLLWrappers", "Libdl", "XML2_jll", "Xorg_libpciaccess_jll"] -git-tree-sha1 = "3d468106a05408f9f7b6f161d9e7715159af247b" -uuid = "e33a78d0-f292-5ffc-b300-72abe9b543c8" -version = "2.12.2+0" - -[[deps.IfElse]] -git-tree-sha1 = "debdd00ffef04665ccbb3e150747a77560e8fad1" -uuid = "615f187c-cbe4-4ef1-ba3b-2fcf58d6d173" -version = "0.1.1" - -[[deps.ImplicitGlobalGrid]] -deps = ["CellArrays", "MPI"] -git-tree-sha1 = "92ce4ec7993527451a56c5e963b80a6eb2c134ae" -uuid = "4d7a3746-15be-11ea-1130-334b0c4f5fa0" -version = "0.16.2" -weakdeps = ["AMDGPU", "CUDA", "Polyester", "StaticArrays"] - - [deps.ImplicitGlobalGrid.extensions] - ImplicitGlobalGrid_AMDGPUExt = "AMDGPU" - ImplicitGlobalGrid_CUDAExt = "CUDA" - ImplicitGlobalGrid_PolyesterExt = "Polyester" - -[[deps.InitialValues]] -git-tree-sha1 = "4da0f88e9a39111c2fa3add390ab15f3a44f3ca3" -uuid = "22cec73e-a1b8-11e9-2c92-598750a2cf9c" -version = "0.3.1" - -[[deps.InlineStrings]] -git-tree-sha1 = "8f3d257792a522b4601c24a577954b0a8cd7334d" -uuid = "842dd82b-1e85-43dc-bf29-5d0ee9dffc48" -version = "1.4.5" - - [deps.InlineStrings.extensions] - ArrowTypesExt = "ArrowTypes" - ParsersExt = "Parsers" - - [deps.InlineStrings.weakdeps] - ArrowTypes = "31f734f8-188a-4ce0-8406-c8a06bd891cd" - Parsers = "69de0a69-1ddd-5017-9359-2bf0b02dc9f0" - -[[deps.InteractiveUtils]] -deps = ["Markdown"] -uuid = "b77e0a4c-d291-57a0-90e8-8db25a27a240" - -[[deps.InverseFunctions]] -git-tree-sha1 = "a779299d77cd080bf77b97535acecd73e1c5e5cb" -uuid = "3587e190-3f89-42d0-90ee-14403ec27112" -version = "0.1.17" -weakdeps = ["Dates", "Test"] - - [deps.InverseFunctions.extensions] - InverseFunctionsDatesExt = "Dates" - InverseFunctionsTestExt = "Test" - -[[deps.InvertedIndices]] -git-tree-sha1 = "6da3c4316095de0f5ee2ebd875df8721e7e0bdbe" -uuid = "41ab1584-1d38-5bbf-9106-f11c6c58b48f" -version = "1.3.1" - -[[deps.IrrationalConstants]] -git-tree-sha1 = "b2d91fe939cae05960e760110b328288867b5758" -uuid = "92d709cd-6900-40b7-9082-c6be49f344b6" -version = "0.2.6" - -[[deps.IteratorInterfaceExtensions]] -git-tree-sha1 = "a3f24677c21f5bbe9d2a714f95dcd58337fb2856" -uuid = "82899510-4779-5014-852e-03e436cf321d" -version = "1.0.0" - -[[deps.JLD2]] -deps = ["ChunkCodecLibZlib", "ChunkCodecLibZstd", "FileIO", "MacroTools", "Mmap", "OrderedCollections", "PrecompileTools", "ScopedValues"] -git-tree-sha1 = "8f8ff711442d1f4cfc0d86133e7ee03d62ec9b98" -uuid = "033835bb-8acc-5ee8-8aae-3f567f8a3819" -version = "0.6.3" -weakdeps = ["UnPack"] - - [deps.JLD2.extensions] - UnPackExt = "UnPack" - -[[deps.JLLWrappers]] -deps = ["Artifacts", "Preferences"] -git-tree-sha1 = "0533e564aae234aff59ab625543145446d8b6ec2" -uuid = "692b3bcd-3c85-4b1f-b108-f13ce0eb3210" -version = "1.7.1" - -[[deps.JSON]] -deps = ["Dates", "Logging", "Parsers", "PrecompileTools", "StructUtils", "UUIDs", "Unicode"] -git-tree-sha1 = "b3ad4a0255688dcb895a52fafbaae3023b588a90" -uuid = "682c06a0-de6a-54ab-a142-c8b1cf79cde6" -version = "1.4.0" - - [deps.JSON.extensions] - JSONArrowExt = ["ArrowTypes"] - - [deps.JSON.weakdeps] - ArrowTypes = "31f734f8-188a-4ce0-8406-c8a06bd891cd" - -[[deps.JuliaInterpreter]] -deps = ["CodeTracking", "InteractiveUtils", "Random", "UUIDs"] -git-tree-sha1 = "80580012d4ed5a3e8b18c7cd86cebe4b816d17a6" -uuid = "aa1ae85d-cabe-5617-a682-6adf51b2e16a" -version = "0.10.9" - -[[deps.JuliaNVTXCallbacks_jll]] -deps = ["Artifacts", "JLLWrappers", "Libdl", "Pkg"] -git-tree-sha1 = "af433a10f3942e882d3c671aacb203e006a5808f" -uuid = "9c1d0b0a-7046-5b2e-a33f-ea22f176ac7e" -version = "0.2.1+0" - -[[deps.KernelAbstractions]] -deps = ["Adapt", "Atomix", "InteractiveUtils", "MacroTools", "PrecompileTools", "Requires", "StaticArrays", "UUIDs"] -git-tree-sha1 = "b5a371fcd1d989d844a4354127365611ae1e305f" -uuid = "63c18a36-062a-441e-b654-da1e3ab1ce7c" -version = "0.9.39" - - [deps.KernelAbstractions.extensions] - EnzymeExt = "EnzymeCore" - LinearAlgebraExt = "LinearAlgebra" - SparseArraysExt = "SparseArrays" - - [deps.KernelAbstractions.weakdeps] - EnzymeCore = "f151be2c-9106-41f4-ab19-57ee4f262869" - LinearAlgebra = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e" - SparseArrays = "2f01184e-e22b-5df5-ae63-d93ebab69eaf" - -[[deps.LLD_jll]] -deps = ["Artifacts", "Libdl", "Zlib_jll", "libLLVM_jll"] -uuid = "d55e3150-da41-5e91-b323-ecfd1eec6109" -version = "15.0.7+10" - -[[deps.LLVM]] -deps = ["CEnum", "LLVMExtra_jll", "Libdl", "Preferences", "Printf", "Unicode"] -git-tree-sha1 = "ce8614210409eaa54ed5968f4b50aa96da7ae543" -uuid = "929cbde3-209d-540e-8aea-75f648917ca0" -version = "9.4.4" -weakdeps = ["BFloat16s"] - - [deps.LLVM.extensions] - BFloat16sExt = "BFloat16s" - -[[deps.LLVMExtra_jll]] -deps = ["Artifacts", "JLLWrappers", "LazyArtifacts", "Libdl", "TOML"] -git-tree-sha1 = "8e76807afb59ebb833e9b131ebf1a8c006510f33" -uuid = "dad2f222-ce93-54a1-a47d-0025e8a3acab" -version = "0.0.38+0" - -[[deps.LLVMLoopInfo]] -git-tree-sha1 = "2e5c102cfc41f48ae4740c7eca7743cc7e7b75ea" -uuid = "8b046642-f1f6-4319-8d3c-209ddc03c586" -version = "1.0.0" - -[[deps.LLVM_jll]] -deps = ["Artifacts", "JLLWrappers", "Libdl", "TOML", "Zlib_jll", "libLLVM_jll"] -git-tree-sha1 = "806ba59dcf2edf25188193ce687024c7762105c6" -uuid = "86de99a1-58d6-5da7-8064-bd56ce2e322c" -version = "15.0.7+10" - -[[deps.LaTeXStrings]] -git-tree-sha1 = "dda21b8cbd6a6c40d9d02a73230f9d70fed6918c" -uuid = "b964fa9f-0449-5b57-a5c2-d3ea65f4040f" -version = "1.4.0" - -[[deps.LayoutPointers]] -deps = ["ArrayInterface", "LinearAlgebra", "ManualMemory", "SIMDTypes", "Static", "StaticArrayInterface"] -git-tree-sha1 = "a9eaadb366f5493a5654e843864c13d8b107548c" -uuid = "10f19ff3-798f-405d-979b-55457f8fc047" -version = "0.1.17" - -[[deps.LazyArtifacts]] -deps = ["Artifacts", "Pkg"] -uuid = "4af54fe1-eca0-43a8-85a7-787d91b784e3" - -[[deps.LibCURL]] -deps = ["LibCURL_jll", "MozillaCACerts_jll"] -uuid = "b27032c2-a3e7-50c8-80cd-2d36dbcbfd21" -version = "0.6.4" - -[[deps.LibCURL_jll]] -deps = ["Artifacts", "LibSSH2_jll", "Libdl", "MbedTLS_jll", "Zlib_jll", "nghttp2_jll"] -uuid = "deac9b47-8bc7-5906-a0fe-35ac56dc84c0" -version = "8.4.0+0" - -[[deps.LibGit2]] -deps = ["Base64", "LibGit2_jll", "NetworkOptions", "Printf", "SHA"] -uuid = "76f85450-5226-5b5a-8eaa-529ad045b433" - -[[deps.LibGit2_jll]] -deps = ["Artifacts", "LibSSH2_jll", "Libdl", "MbedTLS_jll"] -uuid = "e37daf67-58a4-590a-8e99-b0245dd2ffc5" -version = "1.6.4+0" - -[[deps.LibSSH2_jll]] -deps = ["Artifacts", "Libdl", "MbedTLS_jll"] -uuid = "29816b5a-b9ab-546f-933c-edad1886dfa8" -version = "1.11.0+1" - -[[deps.LibTracyClient_jll]] -deps = ["Artifacts", "JLLWrappers", "Libdl"] -git-tree-sha1 = "d4e20500d210247322901841d4eafc7a0c52642d" -uuid = "ad6e5548-8b26-5c9f-8ef3-ef0ad883f3a5" -version = "0.13.1+0" - -[[deps.Libdl]] -uuid = "8f399da3-3557-5675-b5ff-fb832c97cbdb" - -[[deps.Libiconv_jll]] -deps = ["Artifacts", "JLLWrappers", "Libdl"] -git-tree-sha1 = "be484f5c92fad0bd8acfef35fe017900b0b73809" -uuid = "94ce4f54-9a6c-5748-9c1c-f9c7231a4531" -version = "1.18.0+0" - -[[deps.LinearAlgebra]] -deps = ["Libdl", "OpenBLAS_jll", "libblastrampoline_jll"] -uuid = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e" - -[[deps.LogExpFunctions]] -deps = ["DocStringExtensions", "IrrationalConstants", "LinearAlgebra"] -git-tree-sha1 = "13ca9e2586b89836fd20cccf56e57e2b9ae7f38f" -uuid = "2ab3a3ac-af41-5b50-aa03-7779005ae688" -version = "0.3.29" - - [deps.LogExpFunctions.extensions] - LogExpFunctionsChainRulesCoreExt = "ChainRulesCore" - LogExpFunctionsChangesOfVariablesExt = "ChangesOfVariables" - LogExpFunctionsInverseFunctionsExt = "InverseFunctions" - - [deps.LogExpFunctions.weakdeps] - ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4" - ChangesOfVariables = "9e997f8a-9a97-42d5-a9f1-ce6bfc15e2c0" - InverseFunctions = "3587e190-3f89-42d0-90ee-14403ec27112" - -[[deps.Logging]] -uuid = "56ddb016-857b-54e1-b83d-db4d58db5568" - -[[deps.LoggingExtras]] -deps = ["Dates", "Logging"] -git-tree-sha1 = "f00544d95982ea270145636c181ceda21c4e2575" -uuid = "e6f89c97-d47a-5376-807f-9c37f3926c36" -version = "1.2.0" - -[[deps.LoopVectorization]] -deps = ["ArrayInterface", "CPUSummary", "CloseOpenIntervals", "DocStringExtensions", "HostCPUFeatures", "IfElse", "LayoutPointers", "LinearAlgebra", "OffsetArrays", "PolyesterWeave", "PrecompileTools", "SIMDTypes", "SLEEFPirates", "Static", "StaticArrayInterface", "ThreadingUtilities", "UnPack", "VectorizationBase"] -git-tree-sha1 = "a9fc7883eb9b5f04f46efb9a540833d1fad974b3" -uuid = "bdcacae8-1622-11e9-2a5c-532679323890" -version = "0.12.173" - - [deps.LoopVectorization.extensions] - ForwardDiffExt = ["ChainRulesCore", "ForwardDiff"] - ForwardDiffNNlibExt = ["ForwardDiff", "NNlib"] - SpecialFunctionsExt = "SpecialFunctions" - - [deps.LoopVectorization.weakdeps] - ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4" - ForwardDiff = "f6369f11-7733-5829-9624-2563aa707210" - NNlib = "872c559c-99b0-510c-b3b7-b6c96a88d5cd" - SpecialFunctions = "276daf66-3868-5448-9aa4-cd146d93841b" - -[[deps.LoweredCodeUtils]] -deps = ["CodeTracking", "Compiler", "JuliaInterpreter"] -git-tree-sha1 = "65ae3db6ab0e5b1b5f217043c558d9d1d33cc88d" -uuid = "6f1432cf-f94c-5a45-995e-cdbf5db27b0b" -version = "3.5.0" - -[[deps.MLStyle]] -git-tree-sha1 = "bc38dff0548128765760c79eb7388a4b37fae2c8" -uuid = "d8e11817-5142-5d16-987a-aa16d5891078" -version = "0.4.17" - -[[deps.MPI]] -deps = ["Distributed", "DocStringExtensions", "Libdl", "MPICH_jll", "MPIPreferences", "MPItrampoline_jll", "MicrosoftMPI_jll", "OpenMPI_jll", "PkgVersion", "PrecompileTools", "Requires", "Serialization", "Sockets"] -git-tree-sha1 = "a61ecf714d71064b766d481ef43c094d4c6e3c52" -uuid = "da04e1cc-30fd-572f-bb4f-1f8673147195" -version = "0.20.23" -weakdeps = ["AMDGPU", "CUDA"] - - [deps.MPI.extensions] - AMDGPUExt = "AMDGPU" - CUDAExt = "CUDA" - -[[deps.MPICH_jll]] -deps = ["Artifacts", "CompilerSupportLibraries_jll", "Hwloc_jll", "JLLWrappers", "LazyArtifacts", "Libdl", "MPIPreferences", "TOML"] -git-tree-sha1 = "9341048b9f723f2ae2a72a5269ac2f15f80534dc" -uuid = "7cb0a576-ebde-5e09-9194-50597f1243b4" -version = "4.3.2+0" - -[[deps.MPIPreferences]] -deps = ["Libdl", "Preferences"] -git-tree-sha1 = "c105fe467859e7f6e9a852cb15cb4301126fac07" -uuid = "3da0fdf6-3ccc-4f1b-acd9-58baa6c99267" -version = "0.1.11" - -[[deps.MPItrampoline_jll]] -deps = ["Artifacts", "CompilerSupportLibraries_jll", "JLLWrappers", "LazyArtifacts", "Libdl", "MPIPreferences", "TOML"] -git-tree-sha1 = "e214f2a20bdd64c04cd3e4ff62d3c9be7e969a59" -uuid = "f1f71cc9-e9ae-5b93-9b94-4fe0e1ad3748" -version = "5.5.4+0" - -[[deps.MacroTools]] -git-tree-sha1 = "1e0228a030642014fe5cfe68c2c0a818f9e3f522" -uuid = "1914dd2f-81c6-5fcd-8719-6d5c9610ff09" -version = "0.5.16" - -[[deps.ManualMemory]] -git-tree-sha1 = "bcaef4fc7a0cfe2cba636d84cda54b5e4e4ca3cd" -uuid = "d125e4d3-2237-4719-b19c-fa641b8a4667" -version = "0.1.8" - -[[deps.MarchingCubes]] -deps = ["PrecompileTools", "StaticArrays"] -git-tree-sha1 = "0e893025924b6becbae4109f8020ac0e12674b01" -uuid = "299715c1-40a9-479a-aaf9-4a633d36f717" -version = "0.1.11" - -[[deps.Markdown]] -deps = ["Base64"] -uuid = "d6f4376e-aef5-505a-96c1-9c027394607a" - -[[deps.MbedTLS]] -deps = ["Dates", "MbedTLS_jll", "MozillaCACerts_jll", "NetworkOptions", "Random", "Sockets"] -git-tree-sha1 = "c067a280ddc25f196b5e7df3877c6b226d390aaf" -uuid = "739be429-bea8-5141-9913-cc70e7f3736d" -version = "1.1.9" - -[[deps.MbedTLS_jll]] -deps = ["Artifacts", "Libdl"] -uuid = "c8ffd9c3-330d-5841-b78e-0817d7145fa1" -version = "2.28.2+1" - -[[deps.MicrosoftMPI_jll]] -deps = ["Artifacts", "JLLWrappers", "Libdl", "Pkg"] -git-tree-sha1 = "bc95bf4149bf535c09602e3acdf950d9b4376227" -uuid = "9237b28f-5490-5468-be7b-bb81f5f5e6cf" -version = "10.1.4+3" - -[[deps.Missings]] -deps = ["DataAPI"] -git-tree-sha1 = "ec4f7fbeab05d7747bdf98eb74d130a2a2ed298d" -uuid = "e1d29d7a-bbdc-5cf2-9ac0-f12de2c33e28" -version = "1.2.0" - -[[deps.Mmap]] -uuid = "a63ad114-7e13-5084-954f-fe012c677804" - -[[deps.MozillaCACerts_jll]] -uuid = "14a3606d-f60d-562e-9121-12d972cd8159" -version = "2023.1.10" - -[[deps.NVTX]] -deps = ["JuliaNVTXCallbacks_jll", "Libdl", "NVTX_jll"] -git-tree-sha1 = "a9083c3e469e63cca454d1fc3b19472d9d92c14a" -uuid = "5da4648a-3479-48b8-97b9-01cb529c0a1f" -version = "1.0.3" -weakdeps = ["Colors"] - - [deps.NVTX.extensions] - NVTXColorsExt = "Colors" - -[[deps.NVTX_jll]] -deps = ["Artifacts", "JLLWrappers", "Libdl"] -git-tree-sha1 = "af2232f69447494514c25742ba1503ec7e9877fe" -uuid = "e98f9f5b-d649-5603-91fd-7774390e6439" -version = "3.2.2+0" - -[[deps.NaNMath]] -deps = ["OpenLibm_jll"] -git-tree-sha1 = "9b8215b1ee9e78a293f99797cd31375471b2bcae" -uuid = "77ba4419-2d1f-58cd-9bb1-8ffee604a2e3" -version = "1.1.3" - -[[deps.NetworkOptions]] -uuid = "ca575930-c2e3-43a9-ace4-1e988b2c1908" -version = "1.2.0" - -[[deps.OffsetArrays]] -git-tree-sha1 = "117432e406b5c023f665fa73dc26e79ec3630151" -uuid = "6fe1bfb0-de20-5000-8ca7-80f57d26f881" -version = "1.17.0" -weakdeps = ["Adapt"] - - [deps.OffsetArrays.extensions] - OffsetArraysAdaptExt = "Adapt" - -[[deps.OhMyThreads]] -deps = ["BangBang", "ChunkSplitters", "ScopedValues", "StableTasks", "TaskLocalValues"] -git-tree-sha1 = "b3c63491156b66f60c2cd181ba05bc0b327e5c7d" -uuid = "67456a42-1dca-4109-a031-0a68de7e3ad5" -version = "0.8.5" -weakdeps = ["Markdown"] - - [deps.OhMyThreads.extensions] - MarkdownExt = "Markdown" - -[[deps.OpenBLAS_jll]] -deps = ["Artifacts", "CompilerSupportLibraries_jll", "Libdl"] -uuid = "4536629a-c528-5b80-bd46-f80d51c5b363" -version = "0.3.23+4" - -[[deps.OpenLibm_jll]] -deps = ["Artifacts", "Libdl"] -uuid = "05823500-19ac-5b8b-9628-191a04bc5112" -version = "0.8.1+2" - -[[deps.OpenMPI_jll]] -deps = ["Artifacts", "CompilerSupportLibraries_jll", "Hwloc_jll", "JLLWrappers", "LazyArtifacts", "Libdl", "MPIPreferences", "TOML", "Zlib_jll"] -git-tree-sha1 = "ab6596a9d8236041dcd59b5b69316f28a8753592" -uuid = "fe0851c0-eecd-5654-98d4-656369965a5c" -version = "5.0.9+0" - -[[deps.OpenSSL]] -deps = ["BitFlags", "Dates", "MozillaCACerts_jll", "NetworkOptions", "OpenSSL_jll", "Sockets"] -git-tree-sha1 = "1d1aaa7d449b58415f97d2839c318b70ffb525a0" -uuid = "4d8831e6-92b7-49fb-bdf8-b643e874388c" -version = "1.6.1" - -[[deps.OpenSSL_jll]] -deps = ["Artifacts", "JLLWrappers", "Libdl"] -git-tree-sha1 = "c9cbeda6aceffc52d8a0017e71db27c7a7c0beaf" -uuid = "458c3c95-2e84-50aa-8efc-19380b2a3a95" -version = "3.5.5+0" - -[[deps.OpenSpecFun_jll]] -deps = ["Artifacts", "CompilerSupportLibraries_jll", "JLLWrappers", "Libdl"] -git-tree-sha1 = "1346c9208249809840c91b26703912dff463d335" -uuid = "efe28fd5-8261-553b-a9e1-b2916fc3738e" -version = "0.5.6+0" - -[[deps.OrderedCollections]] -git-tree-sha1 = "05868e21324cede2207c6f0f466b4bfef6d5e7ee" -uuid = "bac558e1-5e72-5ebc-8fee-abe8a469f55d" -version = "1.8.1" - -[[deps.Parsers]] -deps = ["Dates", "PrecompileTools", "UUIDs"] -git-tree-sha1 = "7d2f8f21da5db6a806faf7b9b292296da42b2810" -uuid = "69de0a69-1ddd-5017-9359-2bf0b02dc9f0" -version = "2.8.3" - -[[deps.PerfTest]] -deps = ["BenchmarkTools", "Configurations", "CountFlops", "CpuId", "Dates", "HTTP", "JLD2", "JSON", "LibGit2", "LinearAlgebra", "MLStyle", "MacroTools", "Pkg", "Printf", "Revise", "STREAMBenchmark", "Suppressor", "TOML", "Test", "UnicodePlots"] -path = "../.." -uuid = "1dca261b-fc56-4a8c-a7e2-9798d8a75978" -version = "0.1.5" -weakdeps = ["MPI"] - - [deps.PerfTest.extensions] - PerfTest_MPIExt = "MPI" - -[[deps.Pkg]] -deps = ["Artifacts", "Dates", "Downloads", "FileWatching", "LibGit2", "Libdl", "Logging", "Markdown", "Printf", "REPL", "Random", "SHA", "Serialization", "TOML", "Tar", "UUIDs", "p7zip_jll"] -uuid = "44cfe95a-1eb2-52ea-b672-e2afdf69b78f" -version = "1.10.0" - -[[deps.PkgVersion]] -deps = ["Pkg"] -git-tree-sha1 = "f9501cc0430a26bc3d156ae1b5b0c1b47af4d6da" -uuid = "eebad327-c553-4316-9ea0-9fa01ccd7688" -version = "0.3.3" - -[[deps.Polyester]] -deps = ["ArrayInterface", "BitTwiddlingConvenienceFunctions", "CPUSummary", "IfElse", "ManualMemory", "PolyesterWeave", "Static", "StaticArrayInterface", "StrideArraysCore", "ThreadingUtilities"] -git-tree-sha1 = "6f7cd22a802094d239824c57d94c8e2d0f7cfc7d" -uuid = "f517fe37-dbe3-4b94-8317-1923a5111588" -version = "0.7.18" - -[[deps.PolyesterWeave]] -deps = ["BitTwiddlingConvenienceFunctions", "CPUSummary", "IfElse", "Static", "ThreadingUtilities"] -git-tree-sha1 = "645bed98cd47f72f67316fd42fc47dee771aefcd" -uuid = "1d0040c9-8b98-4ee7-8388-3f51789ca0ad" -version = "0.2.2" - -[[deps.PooledArrays]] -deps = ["DataAPI", "Future"] -git-tree-sha1 = "36d8b4b899628fb92c2749eb488d884a926614d3" -uuid = "2dfb63ee-cc39-5dd5-95bd-886bf059d720" -version = "1.4.3" - -[[deps.PrecompileTools]] -deps = ["Preferences"] -git-tree-sha1 = "5aa36f7049a63a1528fe8f7c3f2113413ffd4e1f" -uuid = "aea7be01-6a6a-4083-8856-8a6e6704d82a" -version = "1.2.1" - -[[deps.Preferences]] -deps = ["TOML"] -git-tree-sha1 = "522f093a29b31a93e34eaea17ba055d850edea28" -uuid = "21216c6a-2e73-6563-6e65-726566657250" -version = "1.5.1" - -[[deps.PrettyTables]] -deps = ["Crayons", "LaTeXStrings", "Markdown", "PrecompileTools", "Printf", "Reexport", "StringManipulation", "Tables"] -git-tree-sha1 = "1101cd475833706e4d0e7b122218257178f48f34" -uuid = "08abe8d2-0d0c-5749-adfa-8a2ac140af0d" -version = "2.4.0" - -[[deps.Printf]] -deps = ["Unicode"] -uuid = "de0858da-6303-5e67-8744-51eddeeeb8d7" - -[[deps.Profile]] -deps = ["Printf"] -uuid = "9abbd945-dff8-562f-b5e8-e1ebf5ef1b79" - -[[deps.PtrArrays]] -git-tree-sha1 = "1d36ef11a9aaf1e8b74dacc6a731dd1de8fd493d" -uuid = "43287f4e-b6f4-7ad1-bb20-aadabca52c3d" -version = "1.3.0" - -[[deps.REPL]] -deps = ["InteractiveUtils", "Markdown", "Sockets", "Unicode"] -uuid = "3fa0cd96-eef1-5676-8a61-b3b8758bbffb" - -[[deps.ROCmDeviceLibs_jll]] -deps = ["Artifacts", "JLLWrappers", "Libdl", "Zlib_jll"] -git-tree-sha1 = "9c5b123e62df15d3512ecff91cfce00e36387151" -uuid = "873c0968-716b-5aa7-bb8d-d1e2e2aeff2d" -version = "5.6.1+1" - -[[deps.Random]] -deps = ["SHA"] -uuid = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c" - -[[deps.Random123]] -deps = ["Random", "RandomNumbers"] -git-tree-sha1 = "dbe5fd0b334694e905cb9fda73cd8554333c46e2" -uuid = "74087812-796a-5b5d-8853-05524746bad3" -version = "1.7.1" - -[[deps.RandomNumbers]] -deps = ["Random"] -git-tree-sha1 = "c6ec94d2aaba1ab2ff983052cf6a606ca5985902" -uuid = "e6cf234a-135c-5ec9-84dd-332b85af5143" -version = "1.6.0" - -[[deps.Reexport]] -git-tree-sha1 = "45e428421666073eab6f2da5c9d310d99bb12f9b" -uuid = "189a3867-3050-52da-a836-e630ba90ab69" -version = "1.2.2" - -[[deps.Requires]] -deps = ["UUIDs"] -git-tree-sha1 = "62389eeff14780bfe55195b7204c0d8738436d64" -uuid = "ae029012-a4dd-5104-9daa-d747884805df" -version = "1.3.1" - -[[deps.Revise]] -deps = ["CodeTracking", "FileWatching", "InteractiveUtils", "JuliaInterpreter", "LibGit2", "LoweredCodeUtils", "OrderedCollections", "Preferences", "REPL", "UUIDs"] -git-tree-sha1 = "14d1bfb0a30317edc77e11094607ace3c800f193" -uuid = "295af30f-e4ad-537b-8983-00126c2a3abe" -version = "3.13.2" -weakdeps = ["Distributed"] - - [deps.Revise.extensions] - DistributedExt = "Distributed" - -[[deps.SHA]] -uuid = "ea8e919c-243c-51af-8825-aaa63cd721ce" -version = "0.7.0" - -[[deps.SIMDTypes]] -git-tree-sha1 = "330289636fb8107c5f32088d2741e9fd7a061a5c" -uuid = "94e857df-77ce-4151-89e5-788b33177be4" -version = "0.1.0" - -[[deps.SLEEFPirates]] -deps = ["IfElse", "Static", "VectorizationBase"] -git-tree-sha1 = "456f610ca2fbd1c14f5fcf31c6bfadc55e7d66e0" -uuid = "476501e8-09a2-5ece-8869-fb82de89a1fa" -version = "0.6.43" - -[[deps.STREAMBenchmark]] -deps = ["BenchmarkTools", "Downloads", "LoopVectorization", "Statistics"] -git-tree-sha1 = "72a850a4cde8d1d730b08d7f115890a6c5e3a5fe" -uuid = "05e9033e-e298-417a-adae-495536c11ad4" -version = "0.4.5" - -[[deps.SciMLPublic]] -git-tree-sha1 = "0ba076dbdce87ba230fff48ca9bca62e1f345c9b" -uuid = "431bcebd-1456-4ced-9d72-93c2757fff0b" -version = "1.0.1" - -[[deps.ScopedValues]] -deps = ["HashArrayMappedTries", "Logging"] -git-tree-sha1 = "c3b2323466378a2ba15bea4b2f73b081e022f473" -uuid = "7e506255-f358-4e82-b7e4-beb19740aa63" -version = "1.5.0" - -[[deps.Scratch]] -deps = ["Dates"] -git-tree-sha1 = "9b81b8393e50b7d4e6d0a9f14e192294d3b7c109" -uuid = "6c6a2e73-6563-6170-7368-637461726353" -version = "1.3.0" - -[[deps.SentinelArrays]] -deps = ["Dates", "Random"] -git-tree-sha1 = "ebe7e59b37c400f694f52b58c93d26201387da70" -uuid = "91c51154-3ec4-41a3-a24f-3f23e20d615c" -version = "1.4.9" - -[[deps.Serialization]] -uuid = "9e88b42a-f829-5b0c-bbe9-9e923198166b" - -[[deps.SimpleBufferStream]] -git-tree-sha1 = "f305871d2f381d21527c770d4788c06c097c9bc1" -uuid = "777ac1f9-54b0-4bf8-805c-2214025038e7" -version = "1.2.0" - -[[deps.Sockets]] -uuid = "6462fe0b-24de-5631-8697-dd941f90decc" - -[[deps.SortingAlgorithms]] -deps = ["DataStructures"] -git-tree-sha1 = "64d974c2e6fdf07f8155b5b2ca2ffa9069b608d9" -uuid = "a2af1166-a08f-5f64-846c-94a0d3cef48c" -version = "1.2.2" - -[[deps.SparseArrays]] -deps = ["Libdl", "LinearAlgebra", "Random", "Serialization", "SuiteSparse_jll"] -uuid = "2f01184e-e22b-5df5-ae63-d93ebab69eaf" -version = "1.10.0" - -[[deps.SpecialFunctions]] -deps = ["IrrationalConstants", "LogExpFunctions", "OpenLibm_jll", "OpenSpecFun_jll"] -git-tree-sha1 = "f2685b435df2613e25fc10ad8c26dddb8640f547" -uuid = "276daf66-3868-5448-9aa4-cd146d93841b" -version = "2.6.1" - - [deps.SpecialFunctions.extensions] - SpecialFunctionsChainRulesCoreExt = "ChainRulesCore" - - [deps.SpecialFunctions.weakdeps] - ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4" - -[[deps.StableTasks]] -git-tree-sha1 = "c4f6610f85cb965bee5bfafa64cbeeda55a4e0b2" -uuid = "91464d47-22a1-43fe-8b7f-2d57ee82463f" -version = "0.1.7" - -[[deps.Static]] -deps = ["CommonWorldInvalidations", "IfElse", "PrecompileTools", "SciMLPublic"] -git-tree-sha1 = "49440414711eddc7227724ae6e570c7d5559a086" -uuid = "aedffcd0-7271-4cad-89d0-dc628f76c6d3" -version = "1.3.1" - -[[deps.StaticArrayInterface]] -deps = ["ArrayInterface", "Compat", "IfElse", "LinearAlgebra", "PrecompileTools", "Static"] -git-tree-sha1 = "96381d50f1ce85f2663584c8e886a6ca97e60554" -uuid = "0d7ed370-da01-4f52-bd93-41d350b8b718" -version = "1.8.0" -weakdeps = ["OffsetArrays", "StaticArrays"] - - [deps.StaticArrayInterface.extensions] - StaticArrayInterfaceOffsetArraysExt = "OffsetArrays" - StaticArrayInterfaceStaticArraysExt = "StaticArrays" - -[[deps.StaticArrays]] -deps = ["LinearAlgebra", "PrecompileTools", "Random", "StaticArraysCore"] -git-tree-sha1 = "eee1b9ad8b29ef0d936e3ec9838c7ec089620308" -uuid = "90137ffa-7385-5640-81b9-e52037218182" -version = "1.9.16" - - [deps.StaticArrays.extensions] - StaticArraysChainRulesCoreExt = "ChainRulesCore" - StaticArraysStatisticsExt = "Statistics" - - [deps.StaticArrays.weakdeps] - ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4" - Statistics = "10745b16-79ce-11e8-11f9-7d13ad32a3b2" - -[[deps.StaticArraysCore]] -git-tree-sha1 = "6ab403037779dae8c514bad259f32a447262455a" -uuid = "1e83bf80-4336-4d27-bf5d-d5a4f845583c" -version = "1.4.4" - -[[deps.Statistics]] -deps = ["LinearAlgebra", "SparseArrays"] -uuid = "10745b16-79ce-11e8-11f9-7d13ad32a3b2" -version = "1.10.0" - -[[deps.StatsAPI]] -deps = ["LinearAlgebra"] -git-tree-sha1 = "178ed29fd5b2a2cfc3bd31c13375ae925623ff36" -uuid = "82ae8749-77ed-4fe6-ae5f-f523153014b0" -version = "1.8.0" - -[[deps.StatsBase]] -deps = ["AliasTables", "DataAPI", "DataStructures", "IrrationalConstants", "LinearAlgebra", "LogExpFunctions", "Missings", "Printf", "Random", "SortingAlgorithms", "SparseArrays", "Statistics", "StatsAPI"] -git-tree-sha1 = "aceda6f4e598d331548e04cc6b2124a6148138e3" -uuid = "2913bbd2-ae8a-5f71-8c99-4fb6c76f3a91" -version = "0.34.10" - -[[deps.StrideArraysCore]] -deps = ["ArrayInterface", "CloseOpenIntervals", "IfElse", "LayoutPointers", "LinearAlgebra", "ManualMemory", "SIMDTypes", "Static", "StaticArrayInterface", "ThreadingUtilities"] -git-tree-sha1 = "83151ba8065a73f53ca2ae98bc7274d817aa30f2" -uuid = "7792a7ef-975c-4747-a70f-980b88e8d1da" -version = "0.5.8" - -[[deps.StringManipulation]] -deps = ["PrecompileTools"] -git-tree-sha1 = "a3c1536470bf8c5e02096ad4853606d7c8f62721" -uuid = "892a3eda-7b42-436c-8928-eab12a02cf0e" -version = "0.4.2" - -[[deps.StructUtils]] -deps = ["Dates", "UUIDs"] -git-tree-sha1 = "9297459be9e338e546f5c4bedb59b3b5674da7f1" -uuid = "ec057cc2-7a8d-4b58-b3b3-92acb9f63b42" -version = "2.6.2" - - [deps.StructUtils.extensions] - StructUtilsMeasurementsExt = ["Measurements"] - StructUtilsTablesExt = ["Tables"] - - [deps.StructUtils.weakdeps] - Measurements = "eff96d63-e80a-5855-80a2-b1b0885c5ab7" - Tables = "bd369af6-aec1-5ad0-b16a-f7cc5008161c" - -[[deps.SuiteSparse_jll]] -deps = ["Artifacts", "Libdl", "libblastrampoline_jll"] -uuid = "bea87d4a-7f5b-5778-9afe-8cc45184846c" -version = "7.2.1+1" - -[[deps.Suppressor]] -deps = ["Logging"] -git-tree-sha1 = "6dbb5b635c5437c68c28c2ac9e39b87138f37c0a" -uuid = "fd094767-a336-5f1f-9728-57cf17d0bbfb" -version = "0.2.8" - -[[deps.SysInfo]] -deps = ["Dates", "DelimitedFiles", "Hwloc", "PrecompileTools", "Random", "Serialization"] -git-tree-sha1 = "7aaebfbf5b3a39268f4a0caaa43e878e1138d25c" -uuid = "90a7ee08-a23f-48b9-9006-0e0e2a9e4608" -version = "0.3.0" - -[[deps.TOML]] -deps = ["Dates"] -uuid = "fa267f1f-6049-4f14-aa54-33bafae1ed76" -version = "1.0.3" - -[[deps.TableTraits]] -deps = ["IteratorInterfaceExtensions"] -git-tree-sha1 = "c06b2f539df1c6efa794486abfb6ed2022561a39" -uuid = "3783bdb8-4a98-5b6b-af9a-565f29a5fe9c" -version = "1.0.1" - -[[deps.Tables]] -deps = ["DataAPI", "DataValueInterfaces", "IteratorInterfaceExtensions", "OrderedCollections", "TableTraits"] -git-tree-sha1 = "f2c1efbc8f3a609aadf318094f8fc5204bdaf344" -uuid = "bd369af6-aec1-5ad0-b16a-f7cc5008161c" -version = "1.12.1" - -[[deps.Tar]] -deps = ["ArgTools", "SHA"] -uuid = "a4e569a6-e804-4fa4-b0f3-eef7a1d5b13e" -version = "1.10.0" - -[[deps.TaskLocalValues]] -git-tree-sha1 = "67e469338d9ce74fc578f7db1736a74d93a49eb8" -uuid = "ed4db957-447d-4319-bfb6-7fa9ae7ecf34" -version = "0.1.3" - -[[deps.TensorCore]] -deps = ["LinearAlgebra"] -git-tree-sha1 = "1feb45f88d133a655e001435632f019a9a1bcdb6" -uuid = "62fd8b95-f654-4bbd-a8a5-9c27f68ccd50" -version = "0.1.1" - -[[deps.Test]] -deps = ["InteractiveUtils", "Logging", "Random", "Serialization"] -uuid = "8dfed614-e22c-5e08-85e1-65c5234f0b40" - -[[deps.ThreadPinning]] -deps = ["DelimitedFiles", "Libdl", "LinearAlgebra", "PrecompileTools", "Preferences", "Random", "StableTasks", "SysInfo", "ThreadPinningCore"] -git-tree-sha1 = "d47dbc7862f69ce1973fff227237275ff4a10781" -uuid = "811555cd-349b-4f26-b7bc-1f208b848042" -version = "1.0.2" -weakdeps = ["Distributed", "MPI"] - - [deps.ThreadPinning.extensions] - DistributedExt = "Distributed" - MPIExt = "MPI" - -[[deps.ThreadPinningCore]] -deps = ["LinearAlgebra", "PrecompileTools", "StableTasks"] -git-tree-sha1 = "bb3c6f3b5600fbff028c43348365681b34d06499" -uuid = "6f48bc29-05ce-4cc8-baad-4adcba581a18" -version = "0.4.5" - -[[deps.ThreadingUtilities]] -deps = ["ManualMemory"] -git-tree-sha1 = "d969183d3d244b6c33796b5ed01ab97328f2db85" -uuid = "8290d209-cae3-49c0-8002-c8c24d57dab5" -version = "0.5.5" - -[[deps.Tracy]] -deps = ["ExprTools", "LibTracyClient_jll", "Libdl"] -git-tree-sha1 = "73e3ff50fd3990874c59fef0f35d10644a1487bc" -uuid = "e689c965-62c8-4b79-b2c5-8359227902fd" -version = "0.1.6" - - [deps.Tracy.extensions] - TracyProfilerExt = "TracyProfiler_jll" - - [deps.Tracy.weakdeps] - TracyProfiler_jll = "0c351ed6-8a68-550e-8b79-de6f926da83c" - -[[deps.TranscodingStreams]] -git-tree-sha1 = "0c45878dcfdcfa8480052b6ab162cdd138781742" -uuid = "3bb67fe8-82b1-5028-8e26-92a6c54297fa" -version = "0.11.3" - -[[deps.URIs]] -git-tree-sha1 = "bef26fb046d031353ef97a82e3fdb6afe7f21b1a" -uuid = "5c2747f8-b7ea-4ff2-ba2e-563bfd36b1d4" -version = "1.6.1" - -[[deps.UUIDs]] -deps = ["Random", "SHA"] -uuid = "cf7118a7-6976-5b1a-9a39-7adc72f591a4" - -[[deps.UnPack]] -git-tree-sha1 = "387c1f73762231e86e0c9c5443ce3b4a0a9a0c2b" -uuid = "3a884ed6-31ef-47d7-9d2a-63182c4928ed" -version = "1.0.2" - -[[deps.Unicode]] -uuid = "4ec0a83e-493e-50e2-b9ac-8f72acf5a8f5" - -[[deps.UnicodePlots]] -deps = ["ColorSchemes", "ColorTypes", "Contour", "Crayons", "Dates", "LinearAlgebra", "MarchingCubes", "NaNMath", "PrecompileTools", "Printf", "SparseArrays", "StaticArrays", "StatsBase"] -git-tree-sha1 = "0087c82cf98f2c2bb7df350c02b0b1fc6ae087d6" -uuid = "b8865327-cd53-5732-bb35-84acbb429228" -version = "3.8.1" - - [deps.UnicodePlots.extensions] - FreeTypeExt = ["FileIO", "FreeType"] - ImageInTerminalExt = "ImageInTerminal" - IntervalSetsExt = "IntervalSets" - TermExt = "Term" - UnitfulExt = "Unitful" - - [deps.UnicodePlots.weakdeps] - FileIO = "5789e2e9-d7fb-5bc7-8068-2c6fae9b9549" - FreeType = "b38be410-82b0-50bf-ab77-7b57e271db43" - ImageInTerminal = "d8c32880-2388-543b-8c61-d9f865259254" - IntervalSets = "8197267c-284f-5f27-9208-e0e47529a953" - Term = "22787eb5-b846-44ae-b979-8e399b8463ab" - Unitful = "1986cc42-f94f-5a68-af5c-568840ba703d" - -[[deps.UnsafeAtomics]] -git-tree-sha1 = "b13c4edda90890e5b04ba24e20a310fbe6f249ff" -uuid = "013be700-e6cd-48c3-b4a1-df204f14c38f" -version = "0.3.0" -weakdeps = ["LLVM"] - - [deps.UnsafeAtomics.extensions] - UnsafeAtomicsLLVM = ["LLVM"] - -[[deps.VectorizationBase]] -deps = ["ArrayInterface", "CPUSummary", "HostCPUFeatures", "IfElse", "LayoutPointers", "Libdl", "LinearAlgebra", "SIMDTypes", "Static", "StaticArrayInterface"] -git-tree-sha1 = "d1d9a935a26c475ebffd54e9c7ad11627c43ea85" -uuid = "3d5dd08c-fd9d-11e8-17fa-ed2836048c2f" -version = "0.21.72" - -[[deps.XML2_jll]] -deps = ["Artifacts", "JLLWrappers", "Libdl", "Libiconv_jll", "Zlib_jll"] -git-tree-sha1 = "80d3930c6347cfce7ccf96bd3bafdf079d9c0390" -uuid = "02c8fc9c-b97f-50b9-bbe4-9be30ff0a78a" -version = "2.13.9+0" - -[[deps.Xorg_libpciaccess_jll]] -deps = ["Artifacts", "JLLWrappers", "Libdl", "Zlib_jll"] -git-tree-sha1 = "4909eb8f1cbf6bd4b1c30dd18b2ead9019ef2fad" -uuid = "a65dc6b1-eb27-53a1-bb3e-dea574b5389e" -version = "0.18.1+0" - -[[deps.Zlib_jll]] -deps = ["Libdl"] -uuid = "83775a58-1f1d-513f-b197-d71354ab007a" -version = "1.2.13+1" - -[[deps.Zstd_jll]] -deps = ["Artifacts", "JLLWrappers", "Libdl"] -git-tree-sha1 = "446b23e73536f84e8037f5dce465e92275f6a308" -uuid = "3161d3a3-bdf6-5164-811a-617609db77b4" -version = "1.5.7+1" - -[[deps.demumble_jll]] -deps = ["Artifacts", "JLLWrappers", "Libdl"] -git-tree-sha1 = "6498e3581023f8e530f34760d18f75a69e3a4ea8" -uuid = "1e29f10c-031c-5a83-9565-69cddfc27673" -version = "1.3.0+0" - -[[deps.libLLVM_jll]] -deps = ["Artifacts", "Libdl"] -uuid = "8f36deef-c2a5-5394-99ed-8e07531fb29a" -version = "15.0.7+10" - -[[deps.libblastrampoline_jll]] -deps = ["Artifacts", "Libdl"] -uuid = "8e850b90-86db-534c-a0d3-1478176c7d93" -version = "5.8.0+1" - -[[deps.nghttp2_jll]] -deps = ["Artifacts", "Libdl"] -uuid = "8e850ede-7688-5339-a07c-302acd2aaf8d" -version = "1.52.0+1" - -[[deps.p7zip_jll]] -deps = ["Artifacts", "Libdl"] -uuid = "3f19e933-33d8-53b3-aaab-bd5110c3b7a0" -version = "17.4.0+2" diff --git a/examples/example-paper-implicitglobalgrid/perftest_config.toml b/examples/example-paper-implicitglobalgrid/perftest_config.toml index 8d716ee..06d97b2 100644 --- a/examples/example-paper-implicitglobalgrid/perftest_config.toml +++ b/examples/example-paper-implicitglobalgrid/perftest_config.toml @@ -6,22 +6,24 @@ mode = "reduce" enabled = false [general] -recursive = true save_results = true -autoflops = false +max_saved_results = 20 suppress_output = true -plotting = true save_folder = ".perftests" -max_saved_results = 20 -verbose = true +threads_per_numa = "single" +recursive = true +verbose = 0 +autoflops = true +plotting = true +numas = "single" logs_enabled = true safe_formulas = false [regression] -enabled = false -default_threshold = 0.9 +enabled = true +dedicated_reference_file = "" +default_threshold = 1.1 use_bencher = false -custom_file = "" [roofline] enabled = true diff --git a/examples/example-paper-implicitglobalgrid/transform.jl b/examples/example-paper-implicitglobalgrid/transform.jl index 89f4db5..bd7c421 100644 --- a/examples/example-paper-implicitglobalgrid/transform.jl +++ b/examples/example-paper-implicitglobalgrid/transform.jl @@ -3,8 +3,7 @@ using Pkg Pkg.update() Pkg.instantiate() -using Revise,PerfTest -#PerfTest.toggleMPI() +using PerfTest @info "Transforming expression" expr = PerfTest.transform(ARGS[1]) diff --git a/examples/example-paper-pardiso/output.jl b/examples/example-paper-pardiso/output.jl deleted file mode 100644 index a01278e..0000000 --- a/examples/example-paper-pardiso/output.jl +++ /dev/null @@ -1,332 +0,0 @@ -module __PERFTEST__ -using Test -using PerfTest -using Pkg -using Pardiso -using Random -using SparseArrays -using LinearAlgebra -using MatrixMarket -ps = PardisoSolver() -set_msglvl!(ps, 1) -function build_sparse_matrix(nx, ny, nz, hx2, hy2, hz2) - data = Float64[] - row_indices = Int[] - col_indices = Int[] - for j = 0:ny - 1 - for i = 0:nx - 1 - for k = 0:nz - 1 - row = k * nx * ny + j * nx + i + 1 - if k > 0 - col = (k - 1) * nx * ny + j * nx + i + 1 - push!(data, -hz2) - push!(row_indices, row) - push!(col_indices, col) - end - if i > 0 - col = k * nx * ny + j * nx + (i - 1) + 1 - push!(data, -hx2) - push!(row_indices, row) - push!(col_indices, col) - end - if j > 0 - col = k * nx * ny + (j - 1) * nx + i + 1 - push!(data, -hy2) - push!(row_indices, row) - push!(col_indices, col) - end - push!(data, 2 * (hx2 + hy2 + hz2)) - push!(row_indices, row) - push!(col_indices, row) - if k < nz - 1 - col = (k + 1) * nx * ny + j * nx + i + 1 - push!(data, -hz2) - push!(row_indices, row) - push!(col_indices, col) - end - if i < nx - 1 - col = k * nx * ny + j * nx + (i + 1) + 1 - push!(data, -hx2) - push!(row_indices, row) - push!(col_indices, col) - end - if j < ny - 1 - col = k * nx * ny + (j + 1) * nx + i + 1 - push!(data, -hy2) - push!(row_indices, row) - push!(col_indices, col) - end - end - end - end - sparse(row_indices, col_indices, data) -end -using Test, Dates -using PerfTest: DepthRecord, Metric_Test, Methodology_Result, StrOrSym, Metric_Result, magnitudeAdjust, MPISetup, newMetricResult, buildPrimitiveMetrics!, measureCPUPeakFlops!, measureMemBandwidth!, addLog, @PRFTBenchmark, PRFTBenchmarkGroup, @PRFTCapture_out, @PRFTCount_ops, PRFTflop, @PRFTSuppress, Test_Result, by_index, regression, Suite_Execution_Result, savePrimitives, main_rank, GlobalSuiteData, @perftestset, PerfTestSet, extractTestResults, saveMethodologyData, Configuration -if main_rank() - path = "./.perftests/pardisotest.jl_PERFORMANCE.JLD2" - nofile = true - if isfile(path) - nofile = false - datafile = PerfTest.openDataFile(path) - else - datafile = PerfTest.Perftest_Datafile_Root(PerfTest.Suite_Execution_Result[]) - PerfTest.p_yellow("[!]") - println("Regression: No previous performance reference for this configuration has been found, measuring performance without evaluation.") - end - regression_path = (Configuration.CONFIG["regression"])["custom_file"] - if regression_path != "" && isfile(regression_path) - regression_file = PerfTest.openDataFile(regression_path) - else - if regression_path != "" - @error "Regression data file $(regression_file) could not be opened or found" - else - regression_file = datafile - end - end - _PRFT_GLOBALS = GlobalSuiteData(datafile, path, "pardisotest.jl") - MPISetup(PerfTest.NormalMode, _PRFT_GLOBALS) - if length(regression_file.results) > 0 - _PRFT_GLOBALS.old = (regression_file.results[end]).perftests - else - _PRFT_GLOBALS.old = nothing - end -end -let - size = try - CpuId.cachesize() - catch - addLog("machine", "[MACHINE] CpuId failed, using default cache size") - [1024 * 1024 * 16] - end - global _PRFT_GLOBALS.builtins[:MEM_CACHE_SIZES] = size - addLog("machine", "[MACHINE] Memory buffer size for benchmarking = $((size ./ 1024) ./ 1024) [MB]") - measureCPUPeakFlops!(PerfTest.NormalMode, _PRFT_GLOBALS) - measureMemBandwidth!(PerfTest.NormalMode, _PRFT_GLOBALS) -end -TS = @perftestset(PerfTestSet, "Pardiso GFLOP tests", begin - local ts = Test.get_testset() - ts.old_test_results = _PRFT_GLOBALS.old - nothing - TS = @perftestset(PerfTestSet, "Laplace for different N", for N = [i for i = [40, 42, 45, 47, 50, 52, 55, 57, 60]] - local ts = Test.get_testset() - ts.iterator = N - nothing - A = build_sparse_matrix(N, N, N, (N - 1) ^ 3, (N - 1) ^ 3, (N - 1) ^ 3) - x = zeros(Float64, N ^ 3) - b = zeros(Float64, N ^ 3) - nothing - set_phase!(ps, 11) - pardiso(ps, x, A, b) - set_phase!(ps, 22) - ts.benchmarks["Test 1"] = @PRFTBenchmark(samples = 5, ($pardiso)($ps, $x, $A, $b)) - test_res = Test_Result("Test 1") - ts.test_results["Test 1"] = test_res - keys = ["Pardiso GFLOP tests", "Laplace for different N"] - append!(keys, ["Test 1"]) - old_test_res = _PRFT_GLOBALS.old - for key = keys - if !(old_test_res isa Nothing) && haskey(old_test_res, key) - old_test_res = old_test_res[key] - else - old_test_res = nothing - break - end - end - test_res.primitives[:autoflop] = PRFTflop(@PRFTCount_ops(($pardiso)($ps, $x, $A, $b))) - test_res.primitives[:printed_output] = @PRFTCapture_out(test_res.primitives[:ret_value] = pardiso(ps, x, A, b)) - buildPrimitiveMetrics!(PerfTest.NormalMode, ts, test_res) - let m = newMetricResult(PerfTest.NormalMode, name = "Operational intensity", units = "Flop/Byte", value = begin - flop = 1.0e9 * PerfTest.grepOutputXGetNumber(test_res.primitives[:printed_output], "Gflop for the numerical factorization:") - mem = sizeof(Float64) * PerfTest.grepOutputXGetNumber(test_res.primitives[:printed_output], "number of nonzeros in L") - flop / mem - end, auxiliary = false) - test_res.metrics[:opInt] = m - end - let m = newMetricResult(PerfTest.NormalMode, name = "Attained Flops", units = "FLOP/s", value = (flop = 1.0e9 * PerfTest.grepOutputXGetNumber(test_res.primitives[:printed_output], "Gflop for the numerical factorization:")) / test_res.primitives[:median_time], auxiliary = false) - test_res.metrics[:attainedFLOPS] = m - end - let m = newMetricResult(PerfTest.NormalMode, name = "OUT", units = "String", value = test_res.primitives[:printed_output], auxiliary = true) - test_res.auxiliar[:OUT] = m - end - let - opint = (test_res.metrics[:opInt]).value - flop_s = (test_res.metrics[:attainedFLOPS]).value - roof = PerfTest.rooflineCalc(_PRFT_GLOBALS.builtins[:CPU_FLOPS_PEAK], _PRFT_GLOBALS.builtins[:MEM_STREAM_COPY]) - result_flop_ratio = newMetricResult(PerfTest.NormalMode, name = "Attained FLOP/S by expected FLOP/S", units = "%", value = (flop_s / roof(opint)) * 100) - methodology_res = Methodology_Result(name = "Roofline Model") - success_flop = result_flop_ratio.value >= 0.4 * 100 - flop_test = Metric_Test(reference = 100, threshold_min_percent = 0.4 * 100, threshold_max_percent = nothing, low_is_bad = true, succeeded = success_flop, custom_plotting = Symbol[], full_print = true) - push!(methodology_res.metrics, result_flop_ratio => flop_test) - methodology_res.custom_elements[:realf] = magnitudeAdjust(test_res.metrics[:attainedFLOPS]) - methodology_res.custom_elements[:opint] = test_res.metrics[:opInt] - aux_mem = newMetricResult(PerfTest.NormalMode, name = "Peak empirical bandwidth", units = "B/s", value = _PRFT_GLOBALS.builtins[:MEM_STREAM_COPY]) - aux_flops = newMetricResult(PerfTest.NormalMode, name = "Peak empirical flops", units = "FLOP/s", value = _PRFT_GLOBALS.builtins[:CPU_FLOPS_PEAK]) - aux_rcorner = newMetricResult(PerfTest.NormalMode, name = "Roofline Corner", units = "Flop/Byte", value = aux_flops.value / aux_mem.value) - methodology_res.custom_elements[:mem_peak] = magnitudeAdjust(aux_mem) - methodology_res.custom_elements[:cpu_peak] = magnitudeAdjust(aux_flops) - methodology_res.custom_elements[:roof_corner] = magnitudeAdjust(aux_rcorner) - methodology_res.custom_elements[:roof_corner_raw] = aux_rcorner - methodology_res.custom_elements[:factor] = 0.4 - methodology_res.custom_elements[:plot] = PerfTest.printFullRoofline - try - PerfTest.@_prftest flop_test.succeeded - saveMethodologyData(test_res.name, methodology_res) - catch - end - end - let - methodology_res = Methodology_Result(name = "Performance Regression Testing") - all_succeeded = true - if haskey(test_res.metrics, :median_time) && (!(old_test_res isa Nothing) && haskey(old_test_res.metrics, :median_time)) - ratio = (test_res.metrics[:median_time]).value / (old_test_res.metrics[:median_time]).value - success = ratio > 0.9 - result = newMetricResult(PerfTest.NormalMode, name = ":median_time Difference", units = "%", value = ratio * 100) - test = Metric_Test(reference = 100.0, threshold_min_percent = 0.9 * 100, threshold_max_percent = nothing, low_is_bad = true, succeeded = success, custom_plotting = Symbol[], full_print = true) - push!(methodology_res.metrics, result => test) - methodology_res.custom_elements[:median_time] = newMetricResult(PerfTest.NormalMode, name = (test_res.metrics[:median_time]).name, units = (test_res.metrics[:median_time]).units, value = (test_res.metrics[:median_time]).value) => test - all_succeeded &= success - elseif !(old_test_res isa Nothing) && (!(haskey(test_res.metrics, :median_time)) && (haskey(test_res.primitives, :median_time) && haskey(old_test_res.primitives, :median_time))) - ratio = test_res.primitives[:median_time] / old_test_res.primitives[:median_time] - success = ratio > 0.9 - result = newMetricResult(PerfTest.NormalMode, name = ":median_time Difference", units = "%", value = ratio * 100) - test = Metric_Test(reference = 100.0, threshold_min_percent = 0.9 * 100, threshold_max_percent = nothing, low_is_bad = true, succeeded = success, custom_plotting = Symbol[], full_print = true) - push!(methodology_res.metrics, result => test) - methodology_res.custom_elements[:median_time] = newMetricResult(PerfTest.NormalMode, name = ":median_time", units = "s", value = test_res.primitives[:median_time]) => test - all_succeeded &= success - end - for (r, test) = methodology_res.metrics - PerfTest.@_prftest test.succeeded - end - saveMethodologyData(test_res.name, methodology_res) - end - PerfTest.printAuxiliaries(test_res.auxiliar, Test.get_testset_depth()) - nothing - nothing - end) - nothing - nothing - TS = @perftestset(PerfTestSet, "Custom matrix", for mat = ["af_0_k101/af_0_k101.mtx", "af_shell3/af_shell3.mtx", "pkustk10/pkustk10.mtx", "pkustk11/pkustk11.mtx", "pkustk12/pkustk12.mtx", "pkustk13/pkustk13.mtx", "pkustk14/pkustk14.mtx"] - local ts = Test.get_testset() - ts.iterator = mat - nothing - A = SparseMatrixCSC{Float64}(mmread(mat)) - x = zeros(Float64, size(A, 1)) - b = zeros(Float64, size(A, 1)) - nothing - set_phase!(ps, 11) - pardiso(ps, x, A, b) - set_phase!(ps, 22) - ts.benchmarks["Test 1"] = @PRFTBenchmark(samples = 20, ($pardiso)($ps, $x, $A, $b)) - test_res = Test_Result("Test 1") - ts.test_results["Test 1"] = test_res - keys = ["Pardiso GFLOP tests", "Custom matrix"] - append!(keys, ["Test 1"]) - old_test_res = _PRFT_GLOBALS.old - for key = keys - if !(old_test_res isa Nothing) && haskey(old_test_res, key) - old_test_res = old_test_res[key] - else - old_test_res = nothing - break - end - end - test_res.primitives[:autoflop] = PRFTflop(@PRFTCount_ops(($pardiso)($ps, $x, $A, $b))) - test_res.primitives[:printed_output] = @PRFTCapture_out(test_res.primitives[:ret_value] = pardiso(ps, x, A, b)) - buildPrimitiveMetrics!(PerfTest.NormalMode, ts, test_res) - let m = newMetricResult(PerfTest.NormalMode, name = "Operational intensity", units = "Flop/Byte", value = begin - flop = 1.0e9 * PerfTest.grepOutputXGetNumber(test_res.primitives[:printed_output], "Gflop for the numerical factorization:") - mem = sizeof(Float64) * PerfTest.grepOutputXGetNumber(test_res.primitives[:printed_output], "number of nonzeros in L") - flop / mem - end, auxiliary = false) - test_res.metrics[:opInt] = m - end - let m = newMetricResult(PerfTest.NormalMode, name = "Attained Flops", units = "FLOP/s", value = (flop = 1.0e9 * PerfTest.grepOutputXGetNumber(test_res.primitives[:printed_output], "Gflop for the numerical factorization:")) / test_res.primitives[:median_time], auxiliary = false) - test_res.metrics[:attainedFLOPS] = m - end - let m = newMetricResult(PerfTest.NormalMode, name = "OUT", units = "String", value = test_res.primitives[:printed_output], auxiliary = true) - test_res.auxiliar[:OUT] = m - end - let - opint = (test_res.metrics[:opInt]).value - flop_s = (test_res.metrics[:attainedFLOPS]).value - roof = PerfTest.rooflineCalc(_PRFT_GLOBALS.builtins[:CPU_FLOPS_PEAK], _PRFT_GLOBALS.builtins[:MEM_STREAM_COPY]) - result_flop_ratio = newMetricResult(PerfTest.NormalMode, name = "Attained FLOP/S by expected FLOP/S", units = "%", value = (flop_s / roof(opint)) * 100) - methodology_res = Methodology_Result(name = "Roofline Model") - success_flop = result_flop_ratio.value >= 0.4 * 100 - flop_test = Metric_Test(reference = 100, threshold_min_percent = 0.4 * 100, threshold_max_percent = nothing, low_is_bad = true, succeeded = success_flop, custom_plotting = Symbol[], full_print = true) - push!(methodology_res.metrics, result_flop_ratio => flop_test) - methodology_res.custom_elements[:realf] = magnitudeAdjust(test_res.metrics[:attainedFLOPS]) - methodology_res.custom_elements[:opint] = test_res.metrics[:opInt] - aux_mem = newMetricResult(PerfTest.NormalMode, name = "Peak empirical bandwidth", units = "B/s", value = _PRFT_GLOBALS.builtins[:MEM_STREAM_COPY]) - aux_flops = newMetricResult(PerfTest.NormalMode, name = "Peak empirical flops", units = "FLOP/s", value = _PRFT_GLOBALS.builtins[:CPU_FLOPS_PEAK]) - aux_rcorner = newMetricResult(PerfTest.NormalMode, name = "Roofline Corner", units = "Flop/Byte", value = aux_flops.value / aux_mem.value) - methodology_res.custom_elements[:mem_peak] = magnitudeAdjust(aux_mem) - methodology_res.custom_elements[:cpu_peak] = magnitudeAdjust(aux_flops) - methodology_res.custom_elements[:roof_corner] = magnitudeAdjust(aux_rcorner) - methodology_res.custom_elements[:roof_corner_raw] = aux_rcorner - methodology_res.custom_elements[:factor] = 0.4 - methodology_res.custom_elements[:plot] = PerfTest.printFullRoofline - try - PerfTest.@_prftest flop_test.succeeded - saveMethodologyData(test_res.name, methodology_res) - catch - end - end - let - methodology_res = Methodology_Result(name = "Performance Regression Testing") - all_succeeded = true - if haskey(test_res.metrics, :median_time) && (!(old_test_res isa Nothing) && haskey(old_test_res.metrics, :median_time)) - ratio = (test_res.metrics[:median_time]).value / (old_test_res.metrics[:median_time]).value - success = ratio > 0.9 - result = newMetricResult(PerfTest.NormalMode, name = ":median_time Difference", units = "%", value = ratio * 100) - test = Metric_Test(reference = 100.0, threshold_min_percent = 0.9 * 100, threshold_max_percent = nothing, low_is_bad = true, succeeded = success, custom_plotting = Symbol[], full_print = true) - push!(methodology_res.metrics, result => test) - methodology_res.custom_elements[:median_time] = newMetricResult(PerfTest.NormalMode, name = (test_res.metrics[:median_time]).name, units = (test_res.metrics[:median_time]).units, value = (test_res.metrics[:median_time]).value) => test - all_succeeded &= success - elseif !(old_test_res isa Nothing) && (!(haskey(test_res.metrics, :median_time)) && (haskey(test_res.primitives, :median_time) && haskey(old_test_res.primitives, :median_time))) - ratio = test_res.primitives[:median_time] / old_test_res.primitives[:median_time] - success = ratio > 0.9 - result = newMetricResult(PerfTest.NormalMode, name = ":median_time Difference", units = "%", value = ratio * 100) - test = Metric_Test(reference = 100.0, threshold_min_percent = 0.9 * 100, threshold_max_percent = nothing, low_is_bad = true, succeeded = success, custom_plotting = Symbol[], full_print = true) - push!(methodology_res.metrics, result => test) - methodology_res.custom_elements[:median_time] = newMetricResult(PerfTest.NormalMode, name = ":median_time", units = "s", value = test_res.primitives[:median_time]) => test - all_succeeded &= success - end - for (r, test) = methodology_res.metrics - PerfTest.@_prftest test.succeeded - end - saveMethodologyData(test_res.name, methodology_res) - end - PerfTest.printAuxiliaries(test_res.auxiliar, Test.get_testset_depth()) - nothing - nothing - end) - nothing - nothing - end) -if main_rank() - let - res_num = length(_PRFT_GLOBALS.datafile.results) - if (excess = 20 - res_num) <= 0 - PerfTest.p_yellow("[ℹ]") - println(" Regression: Exceeded maximum recorded results. The oldest $(-1 * excess + 1) result/s will be removed.") - for i = 1:-1 * excess + 1 - popfirst!(_PRFT_GLOBALS.datafile.results) - end - end - end - testresdict = Dict{String, Union{Dict, Test_Result}}() - testresdict[TS.description] = extractTestResults(TS) - newres = Suite_Execution_Result(timestamp = datetime2unix(now()), benchmarks = TS.benchmarks, perftests = testresdict) - push!(_PRFT_GLOBALS.datafile.results, newres) - if sum((Test.get_test_counts(TS))[2:3]) == 0 - PerfTest.saveDataFile(_PRFT_GLOBALS.datafile_path, _PRFT_GLOBALS.datafile) - end - println("[✓] $(path) Performance tests have been finished") - if (Configuration.CONFIG["regression"])["use_bencher"] - bencher_config = Configuration.CONFIG["bencher"] - PerfTest.BencherREST.exportSuiteToBencher(_PRFT_GLOBALS.datafile, bencher_config) - end -end -end \ No newline at end of file diff --git a/examples/example-paper-pardiso/pardisotest.jl b/examples/example-paper-pardiso/pardisotest.jl index 2ea7489..6960e2c 100644 --- a/examples/example-paper-pardiso/pardisotest.jl +++ b/examples/example-paper-pardiso/pardisotest.jl @@ -103,7 +103,7 @@ end ## We define operational intensity and thus enable a roofline test below # - # Target ratio of 5% of roofline performance + # Target ratio of 40% of roofline performance # Actual flops extracted from standard output # Operational intensity extraced from standard output as well @roofline target_ratio=0.4 actual_flops=begin @@ -147,7 +147,7 @@ end ## We define operational intensity and thus enable a roofline test below # - # Target ratio of 5% of roofline performance + # Target ratio of 40% of roofline performance # Actual flops extracted from standard output # Operational intensity extraced from standard output as well @roofline target_ratio=0.4 actual_flops=begin diff --git a/examples/example-paper-pardiso/pardisotest_perfsuite.jl b/examples/example-paper-pardiso/pardisotest_perfsuite.jl new file mode 100644 index 0000000..bb60d36 --- /dev/null +++ b/examples/example-paper-pardiso/pardisotest_perfsuite.jl @@ -0,0 +1,427 @@ +module __PERFTEST__ +using Test +using PerfTest +using Pkg +using Pardiso +using Random +using SparseArrays +using LinearAlgebra +using MatrixMarket +ps = PardisoSolver() +set_msglvl!(ps, 1) +function build_sparse_matrix(nx, ny, nz, hx2, hy2, hz2) + data = Float64[] + row_indices = Int[] + col_indices = Int[] + for j = 0:ny - 1 + for i = 0:nx - 1 + for k = 0:nz - 1 + row = k * nx * ny + j * nx + i + 1 + if k > 0 + col = (k - 1) * nx * ny + j * nx + i + 1 + push!(data, -hz2) + push!(row_indices, row) + push!(col_indices, col) + end + if i > 0 + col = k * nx * ny + j * nx + (i - 1) + 1 + push!(data, -hx2) + push!(row_indices, row) + push!(col_indices, col) + end + if j > 0 + col = k * nx * ny + (j - 1) * nx + i + 1 + push!(data, -hy2) + push!(row_indices, row) + push!(col_indices, col) + end + push!(data, 2 * (hx2 + hy2 + hz2)) + push!(row_indices, row) + push!(col_indices, row) + if k < nz - 1 + col = (k + 1) * nx * ny + j * nx + i + 1 + push!(data, -hz2) + push!(row_indices, row) + push!(col_indices, col) + end + if i < nx - 1 + col = k * nx * ny + j * nx + (i + 1) + 1 + push!(data, -hx2) + push!(row_indices, row) + push!(col_indices, col) + end + if j < ny - 1 + col = k * nx * ny + (j + 1) * nx + i + 1 + push!(data, -hy2) + push!(row_indices, row) + push!(col_indices, col) + end + end + end + end + sparse(row_indices, col_indices, data) +end +using Test, Dates +nothing +using PerfTest: DepthRecord, Metric_Test, Methodology_Result, StrOrSym, Metric_Result, magnitudeAdjust, MPISetup, newMetricResult, buildPrimitiveMetrics!, measureCPUPeakFlops!, measureMemBandwidth!, addLog, @PRFTBenchmark, PRFTBenchmarkGroup, @PRFTCapture_out, @PRFTCount_ops, PRFTflop, @PRFTSuppress, Test_Result, by_index, regression, Suite_Execution_Result, savePrimitives, main_rank, GlobalSuiteData, @perftestset, PerfTestSet, extractTestResults, saveMethodologyData, Configuration +_t_begin = time() +if main_rank(PerfTest.NormalMode) + path = "./.perftests/pardisotest.jl_PERFORMANCE.JLD2" + nofile = true + if isfile(path) + nofile = false + datafile = PerfTest.openDataFile(path) + else + datafile = PerfTest.Perftest_Datafile_Root(PerfTest.Suite_Execution_Result[]) + end + regression_path = (Configuration.CONFIG["regression"])["dedicated_reference_file"] + if regression_path != "" && isfile(regression_path) + regression_file = PerfTest.openDataFile(regression_path) + else + if regression_path != "" + regression_file = PerfTest.Perftest_Datafile_Root(PerfTest.Suite_Execution_Result[]) + else + regression_file = datafile + regression_path = path + end + end + _PRFT_GLOBALS = GlobalSuiteData(datafile, path, "pardisotest.jl") + if length(regression_file.results) > 0 + for i = length(regression_file.results):-1:1 + if length(retrievePerfTests(regression_path, get = :tests, where_pred = testFailed, execution = i)) == 0 + _PRFT_GLOBALS.old = (regression_file.results[i]).perftests + println("Regression: Previous performance reference for this configuration has been found, regression tests could be performed. Regression is currenctly $(if true + "enabled" +else + "disabled" +end)).") + break + end + end + else + _PRFT_GLOBALS.old = nothing + end +else + _PRFT_GLOBALS = GlobalSuiteData() +end +let + PerfTest.Topology.getMachineTopology!() + _PRFT_GLOBALS.builtins[:MEM_CACHE_SIZES] = PerfTest.Topology.getCacheSizes() + measureCPUPeakFlops!(PerfTest.NormalMode, _PRFT_GLOBALS) + measureMemBandwidth!(PerfTest.NormalMode, _PRFT_GLOBALS) +end +TS = @perftestset(PerfTestSet, "Pardiso GFLOP tests", begin + local ts = Test.get_testset() + ts.old_test_results = _PRFT_GLOBALS.old + nothing + TS = @perftestset(PerfTestSet, "Laplace for different N", for N = [i for i = [40, 42, 45, 47, 50, 52, 55, 57, 60]] + local ts = Test.get_testset() + ts.iterator = N + nothing + A = build_sparse_matrix(N, N, N, (N - 1) ^ 3, (N - 1) ^ 3, (N - 1) ^ 3) + x = zeros(Float64, N ^ 3) + b = zeros(Float64, N ^ 3) + nothing + set_phase!(ps, 11) + pardiso(ps, x, A, b) + set_phase!(ps, 22) + ts.benchmarks["Test 1"] = @PRFTBenchmark(($pardiso)($ps, $x, $A, $b), samples = 5) + test_res = Test_Result("Test 1") + ts.test_results["Test 1"] = test_res + keys = ["Pardiso GFLOP tests", "Laplace for different N"] + append!(keys, ["Test 1"]) + old_test_res = _PRFT_GLOBALS.old + for key = keys + if !(old_test_res isa Nothing) && haskey(old_test_res, key) + old_test_res = old_test_res[key] + else + old_test_res = nothing + break + end + end + test_res.primitives[:autoflop] = PRFTflop(@PRFTCount_ops(($pardiso)($ps, $x, $A, $b))) + test_res.primitives[:printed_output] = @PRFTCapture_out(test_res.primitives[:ret_value] = pardiso(ps, x, A, b)) + buildPrimitiveMetrics!(PerfTest.NormalMode, ts, test_res) + let m = newMetricResult(PerfTest.NormalMode, name = "Operational intensity", units = "Flop/Byte", value = begin + flop = 1.0e9 * PerfTest.grepOutputXGetNumber(test_res.primitives[:printed_output], "Gflop for the numerical factorization:") + mem = sizeof(Float64) * PerfTest.grepOutputXGetNumber(test_res.primitives[:printed_output], "number of nonzeros in L") + flop / mem + end, auxiliary = false) + test_res.metrics[:opInt] = m + end + let m = newMetricResult(PerfTest.NormalMode, name = "Attained Flops", units = "FLOP/s", value = (flop = 1.0e9 * PerfTest.grepOutputXGetNumber(test_res.primitives[:printed_output], "Gflop for the numerical factorization:")) / test_res.primitives[:median_time], auxiliary = false) + test_res.metrics[:attainedFLOPS] = m + end + let m = newMetricResult(PerfTest.NormalMode, name = "OUT", units = "String", value = test_res.primitives[:printed_output], auxiliary = true) + test_res.auxiliar[:OUT] = m + end + if main_rank(PerfTest.NormalMode) + let + opint = (test_res.metrics[:opInt]).value + flop_s = (test_res.metrics[:attainedFLOPS]).value + flop_peak = _PRFT_GLOBALS.builtins[:CPU_FLOPS_PEAK] + mem_peak = _PRFT_GLOBALS.builtins[:MEM_BENCH_SDAXPY] + nothing + roof = PerfTest.rooflineCalc(flop_peak, mem_peak) + result_flop_ratio = newMetricResult(PerfTest.NormalMode, name = "Attained FLOP/S by expected FLOP/S", units = "%", value = (flop_s / roof(opint)) * 100) + methodology_res = Methodology_Result(name = "Roofline Model") + success_flop = result_flop_ratio.value >= 0.4 * 100 + flop_test = Metric_Test(reference = 100, threshold_min_percent = 0.4 * 100, threshold_max_percent = nothing, low_is_bad = true, succeeded = success_flop, custom_plotting = Symbol[], full_print = true) + push!(methodology_res.metrics, result_flop_ratio => flop_test) + methodology_res.custom_elements[:realf] = magnitudeAdjust(test_res.metrics[:attainedFLOPS]) + methodology_res.custom_elements[:opint] = test_res.metrics[:opInt] + aux_mem = newMetricResult(PerfTest.NormalMode, name = "Peak empirical bandwidth", units = "B/s", value = mem_peak) + aux_flops = newMetricResult(PerfTest.NormalMode, name = "Peak empirical flops", units = "FLOP/s", value = flop_peak) + aux_rcorner = newMetricResult(PerfTest.NormalMode, name = "Roofline Corner", units = "Flop/Byte", value = aux_flops.value / aux_mem.value) + methodology_res.custom_elements[:mem_peak] = magnitudeAdjust(aux_mem) + methodology_res.custom_elements[:cpu_peak] = magnitudeAdjust(aux_flops) + methodology_res.custom_elements[:roof_corner] = magnitudeAdjust(aux_rcorner) + methodology_res.custom_elements[:roof_corner_raw] = aux_rcorner + methodology_res.custom_elements[:factor] = 0.4 + methodology_res.custom_elements[:plot] = PerfTest.printFullRoofline + try + PerfTest.@_prftest flop_test.succeeded + saveMethodologyData(test_res.name, methodology_res) + catch e + @error "Roofline test failed with error: $(e)" + end + end + let + methodology_res = Methodology_Result(name = "Performance Regression Testing") + all_succeeded = true + if haskey(test_res.metrics, :median_time) && (!(old_test_res isa Nothing) && haskey(old_test_res.metrics, :median_time)) + ratio = (test_res.metrics[:median_time]).value / (old_test_res.metrics[:median_time]).value + success = ratio < 1.1 + result = newMetricResult(PerfTest.NormalMode, name = ":median_time Difference", units = "%", value = ratio * 100) + test = Metric_Test(reference = 100.0, threshold_min_percent = 1.1 * 100, threshold_max_percent = nothing, low_is_bad = false, succeeded = success, custom_plotting = Symbol[], full_print = true) + push!(methodology_res.metrics, result => test) + methodology_res.custom_elements[:median_time] = newMetricResult(PerfTest.NormalMode, name = (test_res.metrics[:median_time]).name, units = (test_res.metrics[:median_time]).units, value = (test_res.metrics[:median_time]).value) => test + all_succeeded &= success + elseif !(old_test_res isa Nothing) && (!(haskey(test_res.metrics, :median_time)) && (haskey(test_res.primitives, :median_time) && haskey(old_test_res.primitives, :median_time))) + ratio = test_res.primitives[:median_time] / old_test_res.primitives[:median_time] + success = ratio < 1.1 + result = newMetricResult(PerfTest.NormalMode, name = ":median_time Difference", units = "%", value = ratio * 100) + test = Metric_Test(reference = 100.0, threshold_min_percent = 1.1 * 100, threshold_max_percent = nothing, low_is_bad = false, succeeded = success, custom_plotting = Symbol[], full_print = true) + push!(methodology_res.metrics, result => test) + methodology_res.custom_elements[:median_time] = newMetricResult(PerfTest.NormalMode, name = ":median_time", units = "s", value = test_res.primitives[:median_time]) => test + all_succeeded &= success + end + if (Configuration.CONFIG["general"])["verbose"] >= 2 && !(old_test_res isa Nothing) + methodology_res.custom_elements[:reference] = newMetricResult(PerfTest.NormalMode, name = ":median_time Reference value", units = if haskey(old_test_res.metrics, :median_time) + (old_test_res.metrics[:median_time]).units + else + "s" + end, value = if haskey(old_test_res.metrics, :median_time) + (old_test_res.metrics[:median_time]).value + else + if haskey(old_test_res.primitives, :median_time) + old_test_res.primitives[:median_time] + else + NaN + end + end) + end + methodology_res.custom_elements[:median_time] = newMetricResult(PerfTest.NormalMode, name = ":median_time", units = if haskey(test_res.metrics, :median_time) + (test_res.metrics[:median_time]).units + else + "s" + end, value = if haskey(test_res.metrics, :median_time) + (test_res.metrics[:median_time]).value + else + if haskey(test_res.primitives, :median_time) + test_res.primitives[:median_time] + else + NaN + end + end) + for (r, test) = methodology_res.metrics + PerfTest.@_prftest test.succeeded + end + saveMethodologyData(test_res.name, methodology_res) + end + end + nothing + end) + nothing + nothing + TS = @perftestset(PerfTestSet, "Custom matrix", for mat = ["af_0_k101/af_0_k101.mtx", "af_shell3/af_shell3.mtx", "pkustk10/pkustk10.mtx", "pkustk11/pkustk11.mtx", "pkustk12/pkustk12.mtx", "pkustk13/pkustk13.mtx", "pkustk14/pkustk14.mtx"] + local ts = Test.get_testset() + ts.iterator = mat + nothing + A = SparseMatrixCSC{Float64}(mmread(mat)) + x = zeros(Float64, size(A, 1)) + b = zeros(Float64, size(A, 1)) + nothing + set_phase!(ps, 11) + pardiso(ps, x, A, b) + set_phase!(ps, 22) + ts.benchmarks["Test 1"] = @PRFTBenchmark(($pardiso)($ps, $x, $A, $b), samples = 20) + test_res = Test_Result("Test 1") + ts.test_results["Test 1"] = test_res + keys = ["Pardiso GFLOP tests", "Custom matrix"] + append!(keys, ["Test 1"]) + old_test_res = _PRFT_GLOBALS.old + for key = keys + if !(old_test_res isa Nothing) && haskey(old_test_res, key) + old_test_res = old_test_res[key] + else + old_test_res = nothing + break + end + end + test_res.primitives[:autoflop] = PRFTflop(@PRFTCount_ops(($pardiso)($ps, $x, $A, $b))) + test_res.primitives[:printed_output] = @PRFTCapture_out(test_res.primitives[:ret_value] = pardiso(ps, x, A, b)) + buildPrimitiveMetrics!(PerfTest.NormalMode, ts, test_res) + let m = newMetricResult(PerfTest.NormalMode, name = "Operational intensity", units = "Flop/Byte", value = begin + flop = 1.0e9 * PerfTest.grepOutputXGetNumber(test_res.primitives[:printed_output], "Gflop for the numerical factorization:") + mem = sizeof(Float64) * PerfTest.grepOutputXGetNumber(test_res.primitives[:printed_output], "number of nonzeros in L") + flop / mem + end, auxiliary = false) + test_res.metrics[:opInt] = m + end + let m = newMetricResult(PerfTest.NormalMode, name = "Attained Flops", units = "FLOP/s", value = (flop = 1.0e9 * PerfTest.grepOutputXGetNumber(test_res.primitives[:printed_output], "Gflop for the numerical factorization:")) / test_res.primitives[:median_time], auxiliary = false) + test_res.metrics[:attainedFLOPS] = m + end + let m = newMetricResult(PerfTest.NormalMode, name = "OUT", units = "String", value = test_res.primitives[:printed_output], auxiliary = true) + test_res.auxiliar[:OUT] = m + end + if main_rank(PerfTest.NormalMode) + let + opint = (test_res.metrics[:opInt]).value + flop_s = (test_res.metrics[:attainedFLOPS]).value + flop_peak = _PRFT_GLOBALS.builtins[:CPU_FLOPS_PEAK] + mem_peak = _PRFT_GLOBALS.builtins[:MEM_BENCH_SDAXPY] + nothing + roof = PerfTest.rooflineCalc(flop_peak, mem_peak) + result_flop_ratio = newMetricResult(PerfTest.NormalMode, name = "Attained FLOP/S by expected FLOP/S", units = "%", value = (flop_s / roof(opint)) * 100) + methodology_res = Methodology_Result(name = "Roofline Model") + success_flop = result_flop_ratio.value >= 0.4 * 100 + flop_test = Metric_Test(reference = 100, threshold_min_percent = 0.4 * 100, threshold_max_percent = nothing, low_is_bad = true, succeeded = success_flop, custom_plotting = Symbol[], full_print = true) + push!(methodology_res.metrics, result_flop_ratio => flop_test) + methodology_res.custom_elements[:realf] = magnitudeAdjust(test_res.metrics[:attainedFLOPS]) + methodology_res.custom_elements[:opint] = test_res.metrics[:opInt] + aux_mem = newMetricResult(PerfTest.NormalMode, name = "Peak empirical bandwidth", units = "B/s", value = mem_peak) + aux_flops = newMetricResult(PerfTest.NormalMode, name = "Peak empirical flops", units = "FLOP/s", value = flop_peak) + aux_rcorner = newMetricResult(PerfTest.NormalMode, name = "Roofline Corner", units = "Flop/Byte", value = aux_flops.value / aux_mem.value) + methodology_res.custom_elements[:mem_peak] = magnitudeAdjust(aux_mem) + methodology_res.custom_elements[:cpu_peak] = magnitudeAdjust(aux_flops) + methodology_res.custom_elements[:roof_corner] = magnitudeAdjust(aux_rcorner) + methodology_res.custom_elements[:roof_corner_raw] = aux_rcorner + methodology_res.custom_elements[:factor] = 0.4 + methodology_res.custom_elements[:plot] = PerfTest.printFullRoofline + try + PerfTest.@_prftest flop_test.succeeded + saveMethodologyData(test_res.name, methodology_res) + catch e + @error "Roofline test failed with error: $(e)" + end + end + let + methodology_res = Methodology_Result(name = "Performance Regression Testing") + all_succeeded = true + if haskey(test_res.metrics, :median_time) && (!(old_test_res isa Nothing) && haskey(old_test_res.metrics, :median_time)) + ratio = (test_res.metrics[:median_time]).value / (old_test_res.metrics[:median_time]).value + success = ratio < 1.1 + result = newMetricResult(PerfTest.NormalMode, name = ":median_time Difference", units = "%", value = ratio * 100) + test = Metric_Test(reference = 100.0, threshold_min_percent = 1.1 * 100, threshold_max_percent = nothing, low_is_bad = false, succeeded = success, custom_plotting = Symbol[], full_print = true) + push!(methodology_res.metrics, result => test) + methodology_res.custom_elements[:median_time] = newMetricResult(PerfTest.NormalMode, name = (test_res.metrics[:median_time]).name, units = (test_res.metrics[:median_time]).units, value = (test_res.metrics[:median_time]).value) => test + all_succeeded &= success + elseif !(old_test_res isa Nothing) && (!(haskey(test_res.metrics, :median_time)) && (haskey(test_res.primitives, :median_time) && haskey(old_test_res.primitives, :median_time))) + ratio = test_res.primitives[:median_time] / old_test_res.primitives[:median_time] + success = ratio < 1.1 + result = newMetricResult(PerfTest.NormalMode, name = ":median_time Difference", units = "%", value = ratio * 100) + test = Metric_Test(reference = 100.0, threshold_min_percent = 1.1 * 100, threshold_max_percent = nothing, low_is_bad = false, succeeded = success, custom_plotting = Symbol[], full_print = true) + push!(methodology_res.metrics, result => test) + methodology_res.custom_elements[:median_time] = newMetricResult(PerfTest.NormalMode, name = ":median_time", units = "s", value = test_res.primitives[:median_time]) => test + all_succeeded &= success + end + if (Configuration.CONFIG["general"])["verbose"] >= 2 && !(old_test_res isa Nothing) + methodology_res.custom_elements[:reference] = newMetricResult(PerfTest.NormalMode, name = ":median_time Reference value", units = if haskey(old_test_res.metrics, :median_time) + (old_test_res.metrics[:median_time]).units + else + "s" + end, value = if haskey(old_test_res.metrics, :median_time) + (old_test_res.metrics[:median_time]).value + else + if haskey(old_test_res.primitives, :median_time) + old_test_res.primitives[:median_time] + else + NaN + end + end) + end + methodology_res.custom_elements[:median_time] = newMetricResult(PerfTest.NormalMode, name = ":median_time", units = if haskey(test_res.metrics, :median_time) + (test_res.metrics[:median_time]).units + else + "s" + end, value = if haskey(test_res.metrics, :median_time) + (test_res.metrics[:median_time]).value + else + if haskey(test_res.primitives, :median_time) + test_res.primitives[:median_time] + else + NaN + end + end) + for (r, test) = methodology_res.metrics + PerfTest.@_prftest test.succeeded + end + saveMethodologyData(test_res.name, methodology_res) + end + end + nothing + end) + nothing + nothing + end) +if main_rank(PerfTest.NormalMode) + testresdict = Dict{String, Union{Dict, Test_Result}}() + if TS isa Vector + benchmarks = PerfTest.newBenchmarkGroup() + for ts = TS + testresdict[ts.description * "_" * string(ts.iterator)] = extractTestResults(TS) + benchmarks[ts.description * "_" * string(ts.iterator)] = ts.benchmarks + end + newres = Suite_Execution_Result(timestamp = datetime2unix(now()), elapsed = time() - _t_begin, benchmarks = benchmarks, perftests = testresdict) + else + testresdict[TS.description] = extractTestResults(TS) + newres = Suite_Execution_Result(timestamp = datetime2unix(now()), elapsed = time() - _t_begin, benchmarks = TS.benchmarks, perftests = testresdict) + end + if PerfTest.testsSucceeded(TS) && ((Configuration.CONFIG["regression"])["enabled"] && regression_path != _PRFT_GLOBALS.datafile_path) + println("All performance tests have passed. Values will be registered as reference for regression testing.") + push!(regression_file.results, newres) + let + res_num = length(regression_file.results) + if (excess = 20 - res_num) <= 0 + PerfTest.p_yellow("[ℹ]") + println(" Regression: Exceeded maximum recorded results. The oldest $(-1 * excess + 1) result/s will be removed.") + for i = 1:-1 * excess + 1 + popfirst!(regression_file.results) + end + end + end + PerfTest.saveDataFile(regression_path, regression_file) + else + println("Some tests failed or errored.") + end + let + push!(_PRFT_GLOBALS.datafile.results, newres) + res_num = length(_PRFT_GLOBALS.datafile.results) + if (excess = 20 - res_num) <= 0 + PerfTest.p_yellow("[ℹ]") + println(" Results File: Exceeded maximum recorded results. The oldest $(-1 * excess + 1) result/s will be removed.") + for i = 1:-1 * excess + 1 + popfirst!(_PRFT_GLOBALS.datafile.results) + end + end + end + PerfTest.saveDataFile(_PRFT_GLOBALS.datafile_path, _PRFT_GLOBALS.datafile) + println("[✓] $(path) Performance tests have been finished (elapsed $(newres.elapsed) s)") + if (Configuration.CONFIG["regression"])["use_bencher"] + bencher_config = Configuration.CONFIG["bencher"] + PerfTest.BencherREST.exportSuiteToBencher(_PRFT_GLOBALS.datafile, bencher_config) + end + PerfTest.clean(PerfTest.NormalMode) +end +end \ No newline at end of file diff --git a/examples/example-paper-pardiso/perftest_config.toml b/examples/example-paper-pardiso/perftest_config.toml index aef5931..06d97b2 100644 --- a/examples/example-paper-pardiso/perftest_config.toml +++ b/examples/example-paper-pardiso/perftest_config.toml @@ -6,22 +6,24 @@ mode = "reduce" enabled = false [general] -recursive = true save_results = true -autoflops = true +max_saved_results = 20 suppress_output = true -plotting = true save_folder = ".perftests" -max_saved_results = 20 -verbose = false +threads_per_numa = "single" +recursive = true +verbose = 0 +autoflops = true +plotting = true +numas = "single" logs_enabled = true safe_formulas = false [regression] enabled = true -default_threshold = 0.9 +dedicated_reference_file = "" +default_threshold = 1.1 use_bencher = false -custom_file = "" [roofline] enabled = true diff --git a/examples/example-paper-pardiso/Manifest.toml b/examples/example-quickstart/Manifest.toml similarity index 54% rename from examples/example-paper-pardiso/Manifest.toml rename to examples/example-quickstart/Manifest.toml index 4768004..2462c42 100644 --- a/examples/example-paper-pardiso/Manifest.toml +++ b/examples/example-quickstart/Manifest.toml @@ -1,19 +1,8 @@ # This file is machine-generated - editing it directly is not advised -julia_version = "1.10.4" +julia_version = "1.11.9" manifest_format = "2.0" -project_hash = "dd74a5711ef35e368dd96d3730e0efa66f070454" - -[[deps.Adapt]] -deps = ["LinearAlgebra", "Requires"] -git-tree-sha1 = "cd8b948862abee8f3d3e9b73a102a9ca924debb0" -uuid = "79e6a3ab-5dfb-504d-930d-738a2a938a0e" -version = "4.2.0" -weakdeps = ["SparseArrays", "StaticArrays"] - - [deps.Adapt.extensions] - AdaptSparseArraysExt = "SparseArrays" - AdaptStaticArraysExt = "StaticArrays" +project_hash = "e4d45922c5e47551d484742427a71ce141aee1e2" [[deps.AliasTables]] deps = ["PtrArrays", "Random"] @@ -23,68 +12,37 @@ version = "1.1.3" [[deps.ArgTools]] uuid = "0dad84c5-d112-42e6-8d28-ef12dabb789f" -version = "1.1.1" - -[[deps.ArrayInterface]] -deps = ["Adapt", "LinearAlgebra"] -git-tree-sha1 = "017fcb757f8e921fb44ee063a7aafe5f89b86dd1" -uuid = "4fba245c-0d91-5ea0-9b3e-6abc04ee57a9" -version = "7.18.0" - - [deps.ArrayInterface.extensions] - ArrayInterfaceBandedMatricesExt = "BandedMatrices" - ArrayInterfaceBlockBandedMatricesExt = "BlockBandedMatrices" - ArrayInterfaceCUDAExt = "CUDA" - ArrayInterfaceCUDSSExt = "CUDSS" - ArrayInterfaceChainRulesCoreExt = "ChainRulesCore" - ArrayInterfaceChainRulesExt = "ChainRules" - ArrayInterfaceGPUArraysCoreExt = "GPUArraysCore" - ArrayInterfaceReverseDiffExt = "ReverseDiff" - ArrayInterfaceSparseArraysExt = "SparseArrays" - ArrayInterfaceStaticArraysCoreExt = "StaticArraysCore" - ArrayInterfaceTrackerExt = "Tracker" - - [deps.ArrayInterface.weakdeps] - BandedMatrices = "aae01518-5342-5314-be14-df237901396f" - BlockBandedMatrices = "ffab5731-97b5-5995-9138-79e8c1846df0" - CUDA = "052768ef-5323-5732-b1bb-66c8b64840ba" - CUDSS = "45b445bb-4962-46a0-9369-b4df9d0f772e" - ChainRules = "082447d4-558c-5d27-93f4-14fc19e9eca2" - ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4" - GPUArraysCore = "46192b85-c4d5-4398-a991-12ede77f4527" - ReverseDiff = "37e2e3b7-166d-5795-8a7a-e32c996b4267" - SparseArrays = "2f01184e-e22b-5df5-ae63-d93ebab69eaf" - StaticArraysCore = "1e83bf80-4336-4d27-bf5d-d5a4f845583c" - Tracker = "9f7883ad-71c0-57eb-9f7f-b5c9e6d3789c" +version = "1.1.2" [[deps.Artifacts]] uuid = "56f22d72-fd6d-98f1-02f0-08ddc0907c33" +version = "1.11.0" + +[[deps.BandwidthBenchmark]] +deps = ["DataFrames", "PrettyTables", "Printf", "Requires", "Statistics", "ThreadPinning"] +git-tree-sha1 = "6b0c18db2cdb386e332cedac2f9fcaf85ab757d2" +uuid = "68eb07c1-04fd-4e62-9736-d6127c4c03c6" +version = "0.2.0" [[deps.Base64]] uuid = "2a0f44e3-6c83-55bd-87e4-b1978d98bd5f" +version = "1.11.0" [[deps.BenchmarkTools]] -deps = ["Compat", "JSON", "Logging", "Printf", "Profile", "Statistics", "UUIDs"] -git-tree-sha1 = "e38fbc49a620f5d0b660d7f543db1009fe0f8336" +deps = ["Compat", "JSON", "Logging", "PrecompileTools", "Printf", "Profile", "Statistics", "UUIDs"] +git-tree-sha1 = "9670d3febc2b6da60a0ae57846ba74670290653f" uuid = "6e4b80f9-dd63-53aa-95a3-0cdb28fa8baf" -version = "1.6.0" +version = "1.8.0" [[deps.BitFlags]] git-tree-sha1 = "0691e34b3bb8be9307330f88d1a3c3f25466c24d" uuid = "d1d4a3ce-64b1-5f1a-9ba4-7e7e69966f35" version = "0.1.9" -[[deps.BitTwiddlingConvenienceFunctions]] -deps = ["Static"] -git-tree-sha1 = "f21cfd4950cb9f0587d5067e69405ad2acd27b87" -uuid = "62783981-4cbd-42fc-bca8-16325de8dc4b" -version = "0.1.6" - -[[deps.CPUSummary]] -deps = ["CpuId", "IfElse", "PrecompileTools", "Static"] -git-tree-sha1 = "5a97e67919535d6841172016c9530fd69494e5ec" -uuid = "2a0fbf3d-bb9c-48f3-b0a9-814d99fd7ab9" -version = "0.2.6" +[[deps.CEnum]] +git-tree-sha1 = "389ad5c84de1ae7cf0e28e381131c98ea87d54fc" +uuid = "fa961155-64e5-5f13-b03f-caf6b980ea82" +version = "0.5.0" [[deps.Cassette]] git-tree-sha1 = "f8764df8d9d2aec2812f009a1ac39e46c33354b8" @@ -108,18 +66,6 @@ git-tree-sha1 = "34d9873079e4cb3d0c62926a225136824677073f" uuid = "55437552-ac27-4d47-9aa3-63184e8fd398" version = "1.0.0" -[[deps.CloseOpenIntervals]] -deps = ["Static", "StaticArrayInterface"] -git-tree-sha1 = "05ba0d07cd4fd8b7a39541e31a7b0254704ea581" -uuid = "fb6a15b2-703c-40df-9091-08a04967cfa9" -version = "0.1.13" - -[[deps.CodeTracking]] -deps = ["InteractiveUtils", "UUIDs"] -git-tree-sha1 = "b7231a755812695b8046e8471ddc34c8268cbad5" -uuid = "da1fd8a2-8d9e-5ec2-8556-3022fb5608a2" -version = "3.0.0" - [[deps.CodecZlib]] deps = ["TranscodingStreams", "Zlib_jll"] git-tree-sha1 = "962834c22b66e32aa10f7611c08c8ca4e20749a9" @@ -162,26 +108,16 @@ git-tree-sha1 = "37ea44092930b1811e666c3bc38065d7d87fcc74" uuid = "5ae59095-9a9b-59fe-a467-6f913c188581" version = "0.13.1" -[[deps.CommonWorldInvalidations]] -git-tree-sha1 = "ae52d1c52048455e85a387fbee9be553ec2b68d0" -uuid = "f70d9fcc-98c5-4d4a-abd7-e4cdeebd8ca8" -version = "1.0.0" - [[deps.Compat]] deps = ["TOML", "UUIDs"] -git-tree-sha1 = "8ae8d32e09f0dcf42a36b90d4e17f5dd2e4c4215" +git-tree-sha1 = "9d8a54ce4b17aa5bdce0ea5c34bc5e7c340d16ad" uuid = "34da2185-b29b-5c13-b0c7-acf172513d20" -version = "4.16.0" +version = "4.18.1" weakdeps = ["Dates", "LinearAlgebra"] [deps.Compat.extensions] CompatLinearAlgebraExt = "LinearAlgebra" -[[deps.Compiler]] -git-tree-sha1 = "382d79bfe72a406294faca39ef0c3cef6e6ce1f1" -uuid = "807dbc54-b67e-4c79-8afb-eafe4df6f2e1" -version = "0.1.1" - [[deps.CompilerSupportLibraries_jll]] deps = ["Artifacts", "Libdl"] uuid = "e66e0078-7015-5450-92f7-15fbd957f2ae" @@ -189,9 +125,9 @@ version = "1.1.1+0" [[deps.ConcurrentUtilities]] deps = ["Serialization", "Sockets"] -git-tree-sha1 = "d9d26935a0bcffc87d2613ce14c527c99fc543fd" +git-tree-sha1 = "21d088c496ea22914fe80906eb5bce65755e5ec8" uuid = "f0e56b4a-5159-44fe-b623-3e5288b988bb" -version = "2.5.0" +version = "2.5.1" [[deps.Configurations]] deps = ["ExproniconLite", "OrderedCollections", "TOML"] @@ -210,12 +146,6 @@ git-tree-sha1 = "3d4b4b334a3a78d1339e4ab2878c688233a469aa" uuid = "1db9610d-79e1-487a-8d40-77f3295c7593" version = "0.1.0" -[[deps.CpuId]] -deps = ["Markdown"] -git-tree-sha1 = "fcbb72b032692610bfbdb15018ac16a36cf2e406" -uuid = "adafc99b-e345-5852-983c-f28acb93d879" -version = "0.3.1" - [[deps.Crayons]] git-tree-sha1 = "249fe38abf76d48563e2f4556bebd215aa317e15" uuid = "a8cc5b0e-0ffa-5ad4-8c14-923d3ee1735f" @@ -226,21 +156,38 @@ git-tree-sha1 = "abe83f3a2f1b857aac70ef8b269080af17764bbe" uuid = "9a962f9c-6df0-11e9-0e5d-c546b8b5ee8a" version = "1.16.0" +[[deps.DataFrames]] +deps = ["Compat", "DataAPI", "DataStructures", "Future", "InlineStrings", "InvertedIndices", "IteratorInterfaceExtensions", "LinearAlgebra", "Markdown", "Missings", "PooledArrays", "PrecompileTools", "PrettyTables", "Printf", "Random", "Reexport", "SentinelArrays", "SortingAlgorithms", "Statistics", "TableTraits", "Tables", "Unicode"] +git-tree-sha1 = "5fab31e2e01e70ad66e3e24c968c264d1cf166d6" +uuid = "a93c6f00-e57d-5684-b7b6-d8193f3e46c0" +version = "1.8.2" + [[deps.DataStructures]] deps = ["OrderedCollections"] -git-tree-sha1 = "e357641bb3e0638d353c4b29ea0e40ea644066a6" +git-tree-sha1 = "e86f4a2805f7f19bec5129bc9150c38208e5dc23" uuid = "864edb3b-99cc-5e75-8d2d-829cb0a9cfe8" -version = "0.19.3" +version = "0.19.4" + +[[deps.DataValueInterfaces]] +git-tree-sha1 = "bfc1187b79289637fa0ef6d4436ebdfe6905cbd6" +uuid = "e2d170a0-9d28-54be-80f0-106bbe20a464" +version = "1.0.0" [[deps.Dates]] deps = ["Printf"] uuid = "ade2ca70-3891-5945-98fb-dc099432e06a" +version = "1.11.0" + +[[deps.DelimitedFiles]] +deps = ["Mmap"] +git-tree-sha1 = "9e2f36d3c96a820c678f2f1f1782582fcf685bae" +uuid = "8bb1440f-4735-579b-a4ab-409b98df4dab" +version = "1.9.1" [[deps.DocStringExtensions]] -deps = ["LibGit2"] -git-tree-sha1 = "2fb1e02f2b635d0845df5d7c167fec4dd739b00d" +git-tree-sha1 = "7442a5dfe1ebb773c29cc2962a8980f47221d76c" uuid = "ffbed154-4ef7-542d-bbb7-c09d3a79fcae" -version = "0.9.3" +version = "0.9.5" [[deps.Downloads]] deps = ["ArgTools", "FileWatching", "LibCURL", "NetworkOptions"] @@ -260,9 +207,9 @@ version = "0.10.14" [[deps.FileIO]] deps = ["Pkg", "Requires", "UUIDs"] -git-tree-sha1 = "d60eb76f37d7e5a40cc2e7c36974d864b82dc802" +git-tree-sha1 = "8e9c059d6857607253e837730dbf780b6b151acd" uuid = "5789e2e9-d7fb-5bc7-8068-2c6fae9b9549" -version = "1.17.1" +version = "1.19.0" weakdeps = ["HTTP"] [deps.FileIO.extensions] @@ -270,6 +217,7 @@ weakdeps = ["HTTP"] [[deps.FileWatching]] uuid = "7b1f6079-737a-58dc-b8bc-7a2ca5c1b5ee" +version = "1.11.0" [[deps.FixedPointNumbers]] deps = ["Statistics"] @@ -277,82 +225,90 @@ git-tree-sha1 = "05882d6995ae5c12bb5f36dd2ed3f61c98cbb172" uuid = "53c48c17-4a7d-5ca2-90c5-79b7896eea93" version = "0.8.5" -[[deps.HDF5]] -deps = ["Compat", "HDF5_jll", "Libdl", "MPIPreferences", "Mmap", "Preferences", "Printf", "Random", "Requires", "UUIDs"] -git-tree-sha1 = "e856eef26cf5bf2b0f95f8f4fc37553c72c8641c" -uuid = "f67ccb44-e63f-5c2f-98bd-6dc0ccc4ba2f" -version = "0.17.2" - - [deps.HDF5.extensions] - MPIExt = "MPI" - - [deps.HDF5.weakdeps] - MPI = "da04e1cc-30fd-572f-bb4f-1f8673147195" - -[[deps.HDF5_jll]] -deps = ["Artifacts", "CompilerSupportLibraries_jll", "JLLWrappers", "LazyArtifacts", "LibCURL_jll", "Libdl", "MPICH_jll", "MPIPreferences", "MPItrampoline_jll", "MicrosoftMPI_jll", "OpenMPI_jll", "OpenSSL_jll", "TOML", "Zlib_jll", "libaec_jll"] -git-tree-sha1 = "e94f84da9af7ce9c6be049e9067e511e17ff89ec" -uuid = "0234f1f7-429e-5d53-9886-15a909be8d59" -version = "1.14.6+0" +[[deps.Future]] +deps = ["Random"] +uuid = "9fa8497b-333b-5362-9e8d-4d0656e87820" +version = "1.11.0" [[deps.HTTP]] deps = ["Base64", "CodecZlib", "ConcurrentUtilities", "Dates", "ExceptionUnwrapping", "Logging", "LoggingExtras", "MbedTLS", "NetworkOptions", "OpenSSL", "PrecompileTools", "Random", "SimpleBufferStream", "Sockets", "URIs", "UUIDs"] -git-tree-sha1 = "5e6fe50ae7f23d171f44e311c2960294aaa0beb5" +git-tree-sha1 = "51059d23c8bb67911a2e6fd5130229113735fc7e" uuid = "cd3eb016-35fb-5094-929b-558a96fad6f3" -version = "1.10.19" +version = "1.11.0" [[deps.HashArrayMappedTries]] git-tree-sha1 = "2eaa69a7cab70a52b9687c8bf950a5a93ec895ae" uuid = "076d061b-32b6-4027-95e0-9a2c6f6d7e74" version = "0.2.0" -[[deps.HostCPUFeatures]] -deps = ["BitTwiddlingConvenienceFunctions", "IfElse", "Libdl", "Static"] -git-tree-sha1 = "8e070b599339d622e9a081d17230d74a5c473293" -uuid = "3e5b6fbb-0976-4d2c-9146-d79de83f2fb0" -version = "0.1.17" +[[deps.Hwloc]] +deps = ["CEnum", "Hwloc_jll", "Printf"] +git-tree-sha1 = "6a3d80f31ff87bc94ab22a7b8ec2f263f9a6a583" +uuid = "0e44f5e4-bd66-52a0-8798-143a42290a1d" +version = "3.3.0" + + [deps.Hwloc.extensions] + HwlocTrees = "AbstractTrees" + + [deps.Hwloc.weakdeps] + AbstractTrees = "1520ce14-60c1-5f80-bbc7-55ef81b5835c" [[deps.Hwloc_jll]] -deps = ["Artifacts", "JLLWrappers", "Libdl"] -git-tree-sha1 = "f93a9ce66cd89c9ba7a4695a47fd93b4c6bc59fa" +deps = ["Artifacts", "JLLWrappers", "Libdl", "XML2_jll", "Xorg_libpciaccess_jll"] +git-tree-sha1 = "baaaebd42ed9ee1bd9173cfd56910e55a8622ee1" uuid = "e33a78d0-f292-5ffc-b300-72abe9b543c8" -version = "2.12.0+0" +version = "2.13.0+1" -[[deps.IfElse]] -git-tree-sha1 = "debdd00ffef04665ccbb3e150747a77560e8fad1" -uuid = "615f187c-cbe4-4ef1-ba3b-2fcf58d6d173" -version = "0.1.1" +[[deps.InlineStrings]] +git-tree-sha1 = "8f3d257792a522b4601c24a577954b0a8cd7334d" +uuid = "842dd82b-1e85-43dc-bf29-5d0ee9dffc48" +version = "1.4.5" -[[deps.IntelOpenMP_jll]] -deps = ["Artifacts", "JLLWrappers", "LazyArtifacts", "Libdl"] -git-tree-sha1 = "0f14a5456bdc6b9731a5682f439a672750a09e48" -uuid = "1d5cc7b8-4909-519e-a0f8-d0f5ad9712d0" -version = "2025.0.4+0" + [deps.InlineStrings.extensions] + ArrowTypesExt = "ArrowTypes" + ParsersExt = "Parsers" + + [deps.InlineStrings.weakdeps] + ArrowTypes = "31f734f8-188a-4ce0-8406-c8a06bd891cd" + Parsers = "69de0a69-1ddd-5017-9359-2bf0b02dc9f0" [[deps.InteractiveUtils]] deps = ["Markdown"] uuid = "b77e0a4c-d291-57a0-90e8-8db25a27a240" +version = "1.11.0" + +[[deps.InvertedIndices]] +git-tree-sha1 = "6da3c4316095de0f5ee2ebd875df8721e7e0bdbe" +uuid = "41ab1584-1d38-5bbf-9106-f11c6c58b48f" +version = "1.3.1" [[deps.IrrationalConstants]] git-tree-sha1 = "b2d91fe939cae05960e760110b328288867b5758" uuid = "92d709cd-6900-40b7-9082-c6be49f344b6" version = "0.2.6" +[[deps.IteratorInterfaceExtensions]] +git-tree-sha1 = "a3f24677c21f5bbe9d2a714f95dcd58337fb2856" +uuid = "82899510-4779-5014-852e-03e436cf321d" +version = "1.0.0" + [[deps.JLD2]] deps = ["ChunkCodecLibZlib", "ChunkCodecLibZstd", "FileIO", "MacroTools", "Mmap", "OrderedCollections", "PrecompileTools", "ScopedValues"] -git-tree-sha1 = "8f8ff711442d1f4cfc0d86133e7ee03d62ec9b98" +git-tree-sha1 = "941f87a0ae1b14d1ac2fa57245425b23a9d7a516" uuid = "033835bb-8acc-5ee8-8aae-3f567f8a3819" -version = "0.6.3" -weakdeps = ["UnPack"] +version = "0.6.4" [deps.JLD2.extensions] UnPackExt = "UnPack" + [deps.JLD2.weakdeps] + UnPack = "3a884ed6-31ef-47d7-9d2a-63182c4928ed" + [[deps.JLLWrappers]] deps = ["Artifacts", "Preferences"] -git-tree-sha1 = "a007feb38b422fbdab534406aeca1b86823cb4d6" +git-tree-sha1 = "7204148362dafe5fe6a273f855b8ccbe4df8173e" uuid = "692b3bcd-3c85-4b1f-b108-f13ce0eb3210" -version = "1.7.0" +version = "1.8.0" [[deps.JSON]] deps = ["Dates", "Mmap", "Parsers", "Unicode"] @@ -360,21 +316,10 @@ git-tree-sha1 = "31e996f0a15c7b280ba9f76636b3ff9e2ae58c9a" uuid = "682c06a0-de6a-54ab-a142-c8b1cf79cde6" version = "0.21.4" -[[deps.JuliaInterpreter]] -deps = ["CodeTracking", "InteractiveUtils", "Random", "UUIDs"] -git-tree-sha1 = "80580012d4ed5a3e8b18c7cd86cebe4b816d17a6" -uuid = "aa1ae85d-cabe-5617-a682-6adf51b2e16a" -version = "0.10.9" - -[[deps.LayoutPointers]] -deps = ["ArrayInterface", "LinearAlgebra", "ManualMemory", "SIMDTypes", "Static", "StaticArrayInterface"] -git-tree-sha1 = "a9eaadb366f5493a5654e843864c13d8b107548c" -uuid = "10f19ff3-798f-405d-979b-55457f8fc047" -version = "0.1.17" - -[[deps.LazyArtifacts]] -deps = ["Artifacts", "Pkg"] -uuid = "4af54fe1-eca0-43a8-85a7-787d91b784e3" +[[deps.LaTeXStrings]] +git-tree-sha1 = "dda21b8cbd6a6c40d9d02a73230f9d70fed6918c" +uuid = "b964fa9f-0449-5b57-a5c2-d3ea65f4040f" +version = "1.4.0" [[deps.LibCURL]] deps = ["LibCURL_jll", "MozillaCACerts_jll"] @@ -384,16 +329,17 @@ version = "0.6.4" [[deps.LibCURL_jll]] deps = ["Artifacts", "LibSSH2_jll", "Libdl", "MbedTLS_jll", "Zlib_jll", "nghttp2_jll"] uuid = "deac9b47-8bc7-5906-a0fe-35ac56dc84c0" -version = "8.4.0+0" +version = "8.6.0+0" [[deps.LibGit2]] deps = ["Base64", "LibGit2_jll", "NetworkOptions", "Printf", "SHA"] uuid = "76f85450-5226-5b5a-8eaa-529ad045b433" +version = "1.11.0" [[deps.LibGit2_jll]] deps = ["Artifacts", "LibSSH2_jll", "Libdl", "MbedTLS_jll"] uuid = "e37daf67-58a4-590a-8e99-b0245dd2ffc5" -version = "1.6.4+0" +version = "1.7.2+0" [[deps.LibSSH2_jll]] deps = ["Artifacts", "Libdl", "MbedTLS_jll"] @@ -402,10 +348,18 @@ version = "1.11.0+1" [[deps.Libdl]] uuid = "8f399da3-3557-5675-b5ff-fb832c97cbdb" +version = "1.11.0" + +[[deps.Libiconv_jll]] +deps = ["Artifacts", "JLLWrappers", "Libdl"] +git-tree-sha1 = "be484f5c92fad0bd8acfef35fe017900b0b73809" +uuid = "94ce4f54-9a6c-5748-9c1c-f9c7231a4531" +version = "1.18.0+0" [[deps.LinearAlgebra]] deps = ["Libdl", "OpenBLAS_jll", "libblastrampoline_jll"] uuid = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e" +version = "1.11.0" [[deps.LogExpFunctions]] deps = ["DocStringExtensions", "IrrationalConstants", "LinearAlgebra"] @@ -425,6 +379,7 @@ version = "0.3.29" [[deps.Logging]] uuid = "56ddb016-857b-54e1-b83d-db4d58db5568" +version = "1.11.0" [[deps.LoggingExtras]] deps = ["Dates", "Logging"] @@ -432,71 +387,15 @@ git-tree-sha1 = "f00544d95982ea270145636c181ceda21c4e2575" uuid = "e6f89c97-d47a-5376-807f-9c37f3926c36" version = "1.2.0" -[[deps.LoopVectorization]] -deps = ["ArrayInterface", "CPUSummary", "CloseOpenIntervals", "DocStringExtensions", "HostCPUFeatures", "IfElse", "LayoutPointers", "LinearAlgebra", "OffsetArrays", "PolyesterWeave", "PrecompileTools", "SIMDTypes", "SLEEFPirates", "Static", "StaticArrayInterface", "ThreadingUtilities", "UnPack", "VectorizationBase"] -git-tree-sha1 = "8084c25a250e00ae427a379a5b607e7aed96a2dd" -uuid = "bdcacae8-1622-11e9-2a5c-532679323890" -version = "0.12.171" - - [deps.LoopVectorization.extensions] - ForwardDiffExt = ["ChainRulesCore", "ForwardDiff"] - SpecialFunctionsExt = "SpecialFunctions" - - [deps.LoopVectorization.weakdeps] - ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4" - ForwardDiff = "f6369f11-7733-5829-9624-2563aa707210" - SpecialFunctions = "276daf66-3868-5448-9aa4-cd146d93841b" - -[[deps.LoweredCodeUtils]] -deps = ["CodeTracking", "Compiler", "JuliaInterpreter"] -git-tree-sha1 = "65ae3db6ab0e5b1b5f217043c558d9d1d33cc88d" -uuid = "6f1432cf-f94c-5a45-995e-cdbf5db27b0b" -version = "3.5.0" - -[[deps.MKL]] -deps = ["Artifacts", "Libdl", "LinearAlgebra", "Logging", "MKL_jll"] -git-tree-sha1 = "39fd578c295085e77aedec501149a63f9ab0c0b6" -uuid = "33e6dc65-8f57-5167-99aa-e5a354878fb2" -version = "0.8.0" - -[[deps.MKL_jll]] -deps = ["Artifacts", "IntelOpenMP_jll", "JLLWrappers", "LazyArtifacts", "Libdl", "oneTBB_jll"] -git-tree-sha1 = "5de60bc6cb3899cd318d80d627560fae2e2d99ae" -uuid = "856f044c-d86e-5d09-b602-aeab76dc8ba7" -version = "2025.0.1+1" - [[deps.MLStyle]] git-tree-sha1 = "bc38dff0548128765760c79eb7388a4b37fae2c8" uuid = "d8e11817-5142-5d16-987a-aa16d5891078" version = "0.4.17" -[[deps.MPICH_jll]] -deps = ["Artifacts", "CompilerSupportLibraries_jll", "Hwloc_jll", "JLLWrappers", "LazyArtifacts", "Libdl", "MPIPreferences", "TOML"] -git-tree-sha1 = "3aa3210044138a1749dbd350a9ba8680869eb503" -uuid = "7cb0a576-ebde-5e09-9194-50597f1243b4" -version = "4.3.0+1" - -[[deps.MPIPreferences]] -deps = ["Libdl", "Preferences"] -git-tree-sha1 = "c105fe467859e7f6e9a852cb15cb4301126fac07" -uuid = "3da0fdf6-3ccc-4f1b-acd9-58baa6c99267" -version = "0.1.11" - -[[deps.MPItrampoline_jll]] -deps = ["Artifacts", "CompilerSupportLibraries_jll", "JLLWrappers", "LazyArtifacts", "Libdl", "MPIPreferences", "TOML"] -git-tree-sha1 = "ff91ca13c7c472cef700f301c8d752bc2aaff1a8" -uuid = "f1f71cc9-e9ae-5b93-9b94-4fe0e1ad3748" -version = "5.5.3+0" - [[deps.MacroTools]] -git-tree-sha1 = "72aebe0b5051e5143a079a4685a46da330a40472" +git-tree-sha1 = "1e0228a030642014fe5cfe68c2c0a818f9e3f522" uuid = "1914dd2f-81c6-5fcd-8719-6d5c9610ff09" -version = "0.5.15" - -[[deps.ManualMemory]] -git-tree-sha1 = "bcaef4fc7a0cfe2cba636d84cda54b5e4e4ca3cd" -uuid = "d125e4d3-2237-4719-b19c-fa641b8a4667" -version = "0.1.8" +version = "0.5.16" [[deps.MarchingCubes]] deps = ["PrecompileTools", "StaticArrays"] @@ -507,29 +406,18 @@ version = "0.1.11" [[deps.Markdown]] deps = ["Base64"] uuid = "d6f4376e-aef5-505a-96c1-9c027394607a" - -[[deps.MatrixMarket]] -deps = ["CodecZlib", "LinearAlgebra", "SparseArrays"] -git-tree-sha1 = "da6fde5ea219bbe414f9d6f878ea9ab5d3476e64" -uuid = "4d4711f2-db25-561a-b6b3-d35e7d4047d3" -version = "0.5.2" +version = "1.11.0" [[deps.MbedTLS]] deps = ["Dates", "MbedTLS_jll", "MozillaCACerts_jll", "NetworkOptions", "Random", "Sockets"] -git-tree-sha1 = "c067a280ddc25f196b5e7df3877c6b226d390aaf" +git-tree-sha1 = "8785729fa736197687541f7053f6d8ab7fc44f92" uuid = "739be429-bea8-5141-9913-cc70e7f3736d" -version = "1.1.9" +version = "1.1.10" [[deps.MbedTLS_jll]] deps = ["Artifacts", "Libdl"] uuid = "c8ffd9c3-330d-5841-b78e-0817d7145fa1" -version = "2.28.2+1" - -[[deps.MicrosoftMPI_jll]] -deps = ["Artifacts", "JLLWrappers", "Libdl", "Pkg"] -git-tree-sha1 = "bc95bf4149bf535c09602e3acdf950d9b4376227" -uuid = "9237b28f-5490-5468-be7b-bb81f5f5e6cf" -version = "10.1.4+3" +version = "2.28.6+0" [[deps.Missings]] deps = ["DataAPI"] @@ -539,10 +427,11 @@ version = "1.2.0" [[deps.Mmap]] uuid = "a63ad114-7e13-5084-954f-fe012c677804" +version = "1.11.0" [[deps.MozillaCACerts_jll]] uuid = "14a3606d-f60d-562e-9121-12d972cd8159" -version = "2023.1.10" +version = "2023.12.12" [[deps.NaNMath]] deps = ["OpenLibm_jll"] @@ -554,30 +443,15 @@ version = "1.1.3" uuid = "ca575930-c2e3-43a9-ace4-1e988b2c1908" version = "1.2.0" -[[deps.OffsetArrays]] -git-tree-sha1 = "5e1897147d1ff8d98883cda2be2187dcf57d8f0c" -uuid = "6fe1bfb0-de20-5000-8ca7-80f57d26f881" -version = "1.15.0" -weakdeps = ["Adapt"] - - [deps.OffsetArrays.extensions] - OffsetArraysAdaptExt = "Adapt" - [[deps.OpenBLAS_jll]] deps = ["Artifacts", "CompilerSupportLibraries_jll", "Libdl"] uuid = "4536629a-c528-5b80-bd46-f80d51c5b363" -version = "0.3.23+4" +version = "0.3.27+1" [[deps.OpenLibm_jll]] deps = ["Artifacts", "Libdl"] uuid = "05823500-19ac-5b8b-9628-191a04bc5112" -version = "0.8.1+2" - -[[deps.OpenMPI_jll]] -deps = ["Artifacts", "CompilerSupportLibraries_jll", "Hwloc_jll", "JLLWrappers", "LazyArtifacts", "Libdl", "MPIPreferences", "TOML", "Zlib_jll"] -git-tree-sha1 = "da913f03f17b449951e0461da960229d4a3d1a8c" -uuid = "fe0851c0-eecd-5654-98d4-656369965a5c" -version = "5.0.7+1" +version = "0.8.5+0" [[deps.OpenSSL]] deps = ["BitFlags", "Dates", "MozillaCACerts_jll", "NetworkOptions", "OpenSSL_jll", "Sockets"] @@ -587,34 +461,26 @@ version = "1.6.1" [[deps.OpenSSL_jll]] deps = ["Artifacts", "JLLWrappers", "Libdl"] -git-tree-sha1 = "a9697f1d06cc3eb3fb3ad49cc67f2cfabaac31ea" +git-tree-sha1 = "2ac022577e5eac7da040de17776d51bb770cd895" uuid = "458c3c95-2e84-50aa-8efc-19380b2a3a95" -version = "3.0.16+0" +version = "3.5.6+0" [[deps.OrderedCollections]] git-tree-sha1 = "05868e21324cede2207c6f0f466b4bfef6d5e7ee" uuid = "bac558e1-5e72-5ebc-8fee-abe8a469f55d" version = "1.8.1" -[[deps.Pardiso]] -deps = ["Libdl", "LinearAlgebra", "MKL_jll", "SparseArrays"] -git-tree-sha1 = "726d408637d989f48c5291f8ad865c4c2e402fa8" -uuid = "46dd5b70-b6fb-5a00-ae2d-e8fea33afaf2" -version = "1.0.0" - [[deps.Parsers]] deps = ["Dates", "PrecompileTools", "UUIDs"] -git-tree-sha1 = "8489905bcdbcfac64d1daa51ca07c0d8f0283821" +git-tree-sha1 = "5d5e0a78e971354b1c7bff0655d11fdc1b0e12c8" uuid = "69de0a69-1ddd-5017-9359-2bf0b02dc9f0" -version = "2.8.1" +version = "2.8.4" [[deps.PerfTest]] -deps = ["BenchmarkTools", "Configurations", "CountFlops", "CpuId", "Dates", "HTTP", "JLD2", "JSON", "LibGit2", "LinearAlgebra", "MLStyle", "MacroTools", "Pkg", "Printf", "Revise", "STREAMBenchmark", "Suppressor", "TOML", "Test", "UnicodePlots"] -git-tree-sha1 = "d250fed091b1968b10af24d8315f7f6dabc5b6e4" -repo-rev = "dev" -repo-url = "/Users/dvegrod/.julia/dev/PerfTest" +deps = ["BandwidthBenchmark", "BenchmarkTools", "Configurations", "CountFlops", "DataFrames", "Dates", "HTTP", "Hwloc", "JLD2", "JSON", "LibGit2", "LinearAlgebra", "MLStyle", "MacroTools", "Pkg", "PrecompileTools", "Printf", "Suppressor", "TOML", "Test", "ThreadPinningCore", "UnicodePlots"] +path = "../.." uuid = "1dca261b-fc56-4a8c-a7e2-9798d8a75978" -version = "0.1.0" +version = "0.2.1" [deps.PerfTest.extensions] PerfTest_MPIExt = "MPI" @@ -623,15 +489,21 @@ version = "0.1.0" MPI = "da04e1cc-30fd-572f-bb4f-1f8673147195" [[deps.Pkg]] -deps = ["Artifacts", "Dates", "Downloads", "FileWatching", "LibGit2", "Libdl", "Logging", "Markdown", "Printf", "REPL", "Random", "SHA", "Serialization", "TOML", "Tar", "UUIDs", "p7zip_jll"] +deps = ["Artifacts", "Dates", "Downloads", "FileWatching", "LibGit2", "Libdl", "Logging", "Markdown", "Printf", "Random", "SHA", "TOML", "Tar", "UUIDs", "p7zip_jll"] uuid = "44cfe95a-1eb2-52ea-b672-e2afdf69b78f" -version = "1.10.0" +version = "1.11.0" + + [deps.Pkg.extensions] + REPLExt = "REPL" -[[deps.PolyesterWeave]] -deps = ["BitTwiddlingConvenienceFunctions", "CPUSummary", "IfElse", "Static", "ThreadingUtilities"] -git-tree-sha1 = "645bed98cd47f72f67316fd42fc47dee771aefcd" -uuid = "1d0040c9-8b98-4ee7-8388-3f51789ca0ad" -version = "0.2.2" + [deps.Pkg.weakdeps] + REPL = "3fa0cd96-eef1-5676-8a61-b3b8758bbffb" + +[[deps.PooledArrays]] +deps = ["DataAPI", "Future"] +git-tree-sha1 = "36d8b4b899628fb92c2749eb488d884a926614d3" +uuid = "2dfb63ee-cc39-5dd5-95bd-886bf059d720" +version = "1.4.3" [[deps.PrecompileTools]] deps = ["Preferences"] @@ -641,30 +513,34 @@ version = "1.2.1" [[deps.Preferences]] deps = ["TOML"] -git-tree-sha1 = "9306f6085165d270f7e3db02af26a400d580f5c6" +git-tree-sha1 = "8b770b60760d4451834fe79dd483e318eee709c4" uuid = "21216c6a-2e73-6563-6e65-726566657250" -version = "1.4.3" +version = "1.5.2" + +[[deps.PrettyTables]] +deps = ["Crayons", "LaTeXStrings", "Markdown", "PrecompileTools", "Printf", "Reexport", "StringManipulation", "Tables"] +git-tree-sha1 = "1101cd475833706e4d0e7b122218257178f48f34" +uuid = "08abe8d2-0d0c-5749-adfa-8a2ac140af0d" +version = "2.4.0" [[deps.Printf]] deps = ["Unicode"] uuid = "de0858da-6303-5e67-8744-51eddeeeb8d7" +version = "1.11.0" [[deps.Profile]] -deps = ["Printf"] uuid = "9abbd945-dff8-562f-b5e8-e1ebf5ef1b79" +version = "1.11.0" [[deps.PtrArrays]] -git-tree-sha1 = "1d36ef11a9aaf1e8b74dacc6a731dd1de8fd493d" +git-tree-sha1 = "4fbbafbc6251b883f4d2705356f3641f3652a7fe" uuid = "43287f4e-b6f4-7ad1-bb20-aadabca52c3d" -version = "1.3.0" - -[[deps.REPL]] -deps = ["InteractiveUtils", "Markdown", "Sockets", "Unicode"] -uuid = "3fa0cd96-eef1-5676-8a61-b3b8758bbffb" +version = "1.4.0" [[deps.Random]] deps = ["SHA"] uuid = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c" +version = "1.11.0" [[deps.Reexport]] git-tree-sha1 = "45e428421666073eab6f2da5c9d310d99bb12f9b" @@ -677,47 +553,25 @@ git-tree-sha1 = "62389eeff14780bfe55195b7204c0d8738436d64" uuid = "ae029012-a4dd-5104-9daa-d747884805df" version = "1.3.1" -[[deps.Revise]] -deps = ["CodeTracking", "FileWatching", "JuliaInterpreter", "LibGit2", "LoweredCodeUtils", "OrderedCollections", "REPL", "Requires", "UUIDs", "Unicode"] -git-tree-sha1 = "ff0bd131abc4ebd9b66d2033144bed6d011d5074" -uuid = "295af30f-e4ad-537b-8983-00126c2a3abe" -version = "3.12.3" - - [deps.Revise.extensions] - DistributedExt = "Distributed" - - [deps.Revise.weakdeps] - Distributed = "8ba89e20-285c-5b6f-9357-94700520ee1b" - [[deps.SHA]] uuid = "ea8e919c-243c-51af-8825-aaa63cd721ce" version = "0.7.0" -[[deps.SIMDTypes]] -git-tree-sha1 = "330289636fb8107c5f32088d2741e9fd7a061a5c" -uuid = "94e857df-77ce-4151-89e5-788b33177be4" -version = "0.1.0" - -[[deps.SLEEFPirates]] -deps = ["IfElse", "Static", "VectorizationBase"] -git-tree-sha1 = "456f610ca2fbd1c14f5fcf31c6bfadc55e7d66e0" -uuid = "476501e8-09a2-5ece-8869-fb82de89a1fa" -version = "0.6.43" - -[[deps.STREAMBenchmark]] -deps = ["BenchmarkTools", "Downloads", "LoopVectorization", "Statistics"] -git-tree-sha1 = "72a850a4cde8d1d730b08d7f115890a6c5e3a5fe" -uuid = "05e9033e-e298-417a-adae-495536c11ad4" -version = "0.4.5" - [[deps.ScopedValues]] deps = ["HashArrayMappedTries", "Logging"] -git-tree-sha1 = "c3b2323466378a2ba15bea4b2f73b081e022f473" +git-tree-sha1 = "67a144433c4ce877ee6d1ada69a124d6b1ecf7be" uuid = "7e506255-f358-4e82-b7e4-beb19740aa63" -version = "1.5.0" +version = "1.6.2" + +[[deps.SentinelArrays]] +deps = ["Dates", "Random"] +git-tree-sha1 = "ebe7e59b37c400f694f52b58c93d26201387da70" +uuid = "91c51154-3ec4-41a3-a24f-3f23e20d615c" +version = "1.4.9" [[deps.Serialization]] uuid = "9e88b42a-f829-5b0c-bbe9-9e923198166b" +version = "1.11.0" [[deps.SimpleBufferStream]] git-tree-sha1 = "f305871d2f381d21527c770d4788c06c097c9bc1" @@ -726,6 +580,7 @@ version = "1.2.0" [[deps.Sockets]] uuid = "6462fe0b-24de-5631-8697-dd941f90decc" +version = "1.11.0" [[deps.SortingAlgorithms]] deps = ["DataStructures"] @@ -736,30 +591,18 @@ version = "1.2.2" [[deps.SparseArrays]] deps = ["Libdl", "LinearAlgebra", "Random", "Serialization", "SuiteSparse_jll"] uuid = "2f01184e-e22b-5df5-ae63-d93ebab69eaf" -version = "1.10.0" - -[[deps.Static]] -deps = ["CommonWorldInvalidations", "IfElse", "PrecompileTools"] -git-tree-sha1 = "87d51a3ee9a4b0d2fe054bdd3fc2436258db2603" -uuid = "aedffcd0-7271-4cad-89d0-dc628f76c6d3" -version = "1.1.1" - -[[deps.StaticArrayInterface]] -deps = ["ArrayInterface", "Compat", "IfElse", "LinearAlgebra", "PrecompileTools", "Static"] -git-tree-sha1 = "96381d50f1ce85f2663584c8e886a6ca97e60554" -uuid = "0d7ed370-da01-4f52-bd93-41d350b8b718" -version = "1.8.0" -weakdeps = ["OffsetArrays", "StaticArrays"] +version = "1.11.0" - [deps.StaticArrayInterface.extensions] - StaticArrayInterfaceOffsetArraysExt = "OffsetArrays" - StaticArrayInterfaceStaticArraysExt = "StaticArrays" +[[deps.StableTasks]] +git-tree-sha1 = "c4f6610f85cb965bee5bfafa64cbeeda55a4e0b2" +uuid = "91464d47-22a1-43fe-8b7f-2d57ee82463f" +version = "0.1.7" [[deps.StaticArrays]] deps = ["LinearAlgebra", "PrecompileTools", "Random", "StaticArraysCore"] -git-tree-sha1 = "eee1b9ad8b29ef0d936e3ec9838c7ec089620308" +git-tree-sha1 = "246a8bb2e6667f832eea063c3a56aef96429a3db" uuid = "90137ffa-7385-5640-81b9-e52037218182" -version = "1.9.16" +version = "1.9.18" [deps.StaticArrays.extensions] StaticArraysChainRulesCoreExt = "ChainRulesCore" @@ -775,9 +618,14 @@ uuid = "1e83bf80-4336-4d27-bf5d-d5a4f845583c" version = "1.4.4" [[deps.Statistics]] -deps = ["LinearAlgebra", "SparseArrays"] +deps = ["LinearAlgebra"] +git-tree-sha1 = "ae3bb1eb3bba077cd276bc5cfc337cc65c3075c0" uuid = "10745b16-79ce-11e8-11f9-7d13ad32a3b2" -version = "1.10.0" +version = "1.11.1" +weakdeps = ["SparseArrays"] + + [deps.Statistics.extensions] + SparseArraysExt = ["SparseArrays"] [[deps.StatsAPI]] deps = ["LinearAlgebra"] @@ -791,10 +639,16 @@ git-tree-sha1 = "aceda6f4e598d331548e04cc6b2124a6148138e3" uuid = "2913bbd2-ae8a-5f71-8c99-4fb6c76f3a91" version = "0.34.10" +[[deps.StringManipulation]] +deps = ["PrecompileTools"] +git-tree-sha1 = "d05693d339e37d6ab134c5ab53c29fce5ee5d7d5" +uuid = "892a3eda-7b42-436c-8928-eab12a02cf0e" +version = "0.4.4" + [[deps.SuiteSparse_jll]] deps = ["Artifacts", "Libdl", "libblastrampoline_jll"] uuid = "bea87d4a-7f5b-5778-9afe-8cc45184846c" -version = "7.2.1+1" +version = "7.7.0+0" [[deps.Suppressor]] deps = ["Logging"] @@ -807,6 +661,18 @@ deps = ["Dates"] uuid = "fa267f1f-6049-4f14-aa54-33bafae1ed76" version = "1.0.3" +[[deps.TableTraits]] +deps = ["IteratorInterfaceExtensions"] +git-tree-sha1 = "c06b2f539df1c6efa794486abfb6ed2022561a39" +uuid = "3783bdb8-4a98-5b6b-af9a-565f29a5fe9c" +version = "1.0.1" + +[[deps.Tables]] +deps = ["DataAPI", "DataValueInterfaces", "IteratorInterfaceExtensions", "OrderedCollections", "TableTraits"] +git-tree-sha1 = "f2c1efbc8f3a609aadf318094f8fc5204bdaf344" +uuid = "bd369af6-aec1-5ad0-b16a-f7cc5008161c" +version = "1.12.1" + [[deps.Tar]] deps = ["ArgTools", "SHA"] uuid = "a4e569a6-e804-4fa4-b0f3-eef7a1d5b13e" @@ -821,12 +687,19 @@ version = "0.1.1" [[deps.Test]] deps = ["InteractiveUtils", "Logging", "Random", "Serialization"] uuid = "8dfed614-e22c-5e08-85e1-65c5234f0b40" - -[[deps.ThreadingUtilities]] -deps = ["ManualMemory"] -git-tree-sha1 = "eda08f7e9818eb53661b3deb74e3159460dfbc27" -uuid = "8290d209-cae3-49c0-8002-c8c24d57dab5" -version = "0.5.2" +version = "1.11.0" + +[[deps.ThreadPinning]] +deps = ["DelimitedFiles", "DocStringExtensions", "Libdl", "LinearAlgebra", "PrecompileTools", "Preferences", "Random", "StableTasks"] +git-tree-sha1 = "333748c6fa62868fa039f00ba670d619776a6752" +uuid = "811555cd-349b-4f26-b7bc-1f208b848042" +version = "0.7.22" + +[[deps.ThreadPinningCore]] +deps = ["LinearAlgebra", "PrecompileTools", "StableTasks"] +git-tree-sha1 = "bb3c6f3b5600fbff028c43348365681b34d06499" +uuid = "6f48bc29-05ce-4cc8-baad-4adcba581a18" +version = "0.4.5" [[deps.TranscodingStreams]] git-tree-sha1 = "0c45878dcfdcfa8480052b6ab162cdd138781742" @@ -841,20 +714,17 @@ version = "1.6.1" [[deps.UUIDs]] deps = ["Random", "SHA"] uuid = "cf7118a7-6976-5b1a-9a39-7adc72f591a4" - -[[deps.UnPack]] -git-tree-sha1 = "387c1f73762231e86e0c9c5443ce3b4a0a9a0c2b" -uuid = "3a884ed6-31ef-47d7-9d2a-63182c4928ed" -version = "1.0.2" +version = "1.11.0" [[deps.Unicode]] uuid = "4ec0a83e-493e-50e2-b9ac-8f72acf5a8f5" +version = "1.11.0" [[deps.UnicodePlots]] deps = ["ColorSchemes", "ColorTypes", "Contour", "Crayons", "Dates", "LinearAlgebra", "MarchingCubes", "NaNMath", "PrecompileTools", "Printf", "SparseArrays", "StaticArrays", "StatsBase"] -git-tree-sha1 = "0087c82cf98f2c2bb7df350c02b0b1fc6ae087d6" +git-tree-sha1 = "32c6b62c4a45c8be876c4916cffabf9b89c05ddd" uuid = "b8865327-cd53-5732-bb35-84acbb429228" -version = "3.8.1" +version = "3.8.3" [deps.UnicodePlots.extensions] FreeTypeExt = ["FileIO", "FreeType"] @@ -871,11 +741,17 @@ version = "3.8.1" Term = "22787eb5-b846-44ae-b979-8e399b8463ab" Unitful = "1986cc42-f94f-5a68-af5c-568840ba703d" -[[deps.VectorizationBase]] -deps = ["ArrayInterface", "CPUSummary", "HostCPUFeatures", "IfElse", "LayoutPointers", "Libdl", "LinearAlgebra", "SIMDTypes", "Static", "StaticArrayInterface"] -git-tree-sha1 = "4ab62a49f1d8d9548a1c8d1a75e5f55cf196f64e" -uuid = "3d5dd08c-fd9d-11e8-17fa-ed2836048c2f" -version = "0.21.71" +[[deps.XML2_jll]] +deps = ["Artifacts", "JLLWrappers", "Libdl", "Libiconv_jll", "Zlib_jll"] +git-tree-sha1 = "80d3930c6347cfce7ccf96bd3bafdf079d9c0390" +uuid = "02c8fc9c-b97f-50b9-bbe4-9be30ff0a78a" +version = "2.13.9+0" + +[[deps.Xorg_libpciaccess_jll]] +deps = ["Artifacts", "JLLWrappers", "Libdl", "Zlib_jll"] +git-tree-sha1 = "58972370b81423fc546c56a60ed1a009450177c3" +uuid = "a65dc6b1-eb27-53a1-bb3e-dea574b5389e" +version = "0.19.0+0" [[deps.Zlib_jll]] deps = ["Libdl"] @@ -888,27 +764,15 @@ git-tree-sha1 = "446b23e73536f84e8037f5dce465e92275f6a308" uuid = "3161d3a3-bdf6-5164-811a-617609db77b4" version = "1.5.7+1" -[[deps.libaec_jll]] -deps = ["Artifacts", "JLLWrappers", "Libdl"] -git-tree-sha1 = "f5733a5a9047722470b95a81e1b172383971105c" -uuid = "477f73a3-ac25-53e9-8cc3-50b2fa2566f0" -version = "1.1.3+0" - [[deps.libblastrampoline_jll]] deps = ["Artifacts", "Libdl"] uuid = "8e850b90-86db-534c-a0d3-1478176c7d93" -version = "5.8.0+1" +version = "5.11.0+0" [[deps.nghttp2_jll]] deps = ["Artifacts", "Libdl"] uuid = "8e850ede-7688-5339-a07c-302acd2aaf8d" -version = "1.52.0+1" - -[[deps.oneTBB_jll]] -deps = ["Artifacts", "JLLWrappers", "LazyArtifacts", "Libdl"] -git-tree-sha1 = "d5a767a3bb77135a99e433afe0eb14cd7f6914c3" -uuid = "1317d2d5-d96f-522e-a858-c73665f53c3e" -version = "2022.0.0+0" +version = "1.59.0+0" [[deps.p7zip_jll]] deps = ["Artifacts", "Libdl"] diff --git a/examples/example-quickstart/Project.toml b/examples/example-quickstart/Project.toml new file mode 100644 index 0000000..d336dab --- /dev/null +++ b/examples/example-quickstart/Project.toml @@ -0,0 +1,2 @@ +[deps] +PerfTest = "1dca261b-fc56-4a8c-a7e2-9798d8a75978" diff --git a/examples/example-quickstart/module.jl b/examples/example-quickstart/module.jl new file mode 100644 index 0000000..6df17f9 --- /dev/null +++ b/examples/example-quickstart/module.jl @@ -0,0 +1,8 @@ +module MyPackage + +# Add [a[1],a[2],...a[end]] and [b[end], b[end-1],...,b[1]] elementwise +function addReversed(A :: Vector{<: Number}, B:: Vector{<: Number}) :: Vector{<:Number} + return [a + b for (a,b) in zip(A, reverse(B))] +end + +end \ No newline at end of file diff --git a/examples/example-quickstart/testfile.jl b/examples/example-quickstart/testfile.jl new file mode 100644 index 0000000..74b392b --- /dev/null +++ b/examples/example-quickstart/testfile.jl @@ -0,0 +1,28 @@ +using Test,PerfTest + +include("module.jl") + +@perftest_config " +[general] +verbose = 3 +[regression] +dedicated_reference_file='reference.JLD2' +" + +@testset "addReversed tests" begin + N = 10 + # We want the size to be bigger on the performance test + @on_perftest_exec begin + N = 1_000_000 + end + # We set the regression checker, we dont specify a metric therefore the default (median time elapsed) is used + # low_is_bad=false time elapsed metrics are considered worse the bigger they are + # threshold = 1.05 the test will fail if the time is 105% of the reference or greater, in other words: @test time_elapsed < 1.05 * reference + @regression threshold=1.05 low_is_bad=false + + A = [i for i in 1:N] + B = [N-i for i in 1:N] + + result = @perftest MyPackage.addReversed(A, B) + @test sum(result) == N*N +end \ No newline at end of file diff --git a/ext/PerfTest_MPIExt.jl b/ext/PerfTest_MPIExt.jl index 55a5e3b..48d23b2 100644 --- a/ext/PerfTest_MPIExt.jl +++ b/ext/PerfTest_MPIExt.jl @@ -9,7 +9,6 @@ module PerfTest_MPIExt using MPI using PerfTest using LinearAlgebra -using STREAMBenchmark using BenchmarkTools using Base.Threads @@ -22,261 +21,419 @@ function PerfTest.MPISetup(::Type{PerfTest.MPIMode})::Nothing if !mpi_initialized MPI.Init() end - PerfTest.toggleMPI() global mpi_rank = MPI.Comm_rank(MPI.COMM_WORLD) global mpi_size = MPI.Comm_size(MPI.COMM_WORLD) global mpi_initialized = true - + PerfTest.addLog("machine", "[MPI] PerfTest MPI extension enabled - Rank $mpi_rank of $mpi_size") return nothing end # Override main_rank and ranks for MPI mode -function PerfTest.main_rank(mode :: Type{PerfTest.MPIMode})::Bool +function PerfTest.main_rank(mode::Type{PerfTest.MPIMode})::Bool return mpi_rank == 0 end -function PerfTest.mpi_rank(mode :: Type{PerfTest.MPIMode}) :: Int +function PerfTest.mpi_rank(mode::Type{PerfTest.MPIMode})::Int return mpi_rank end -function PerfTest.ranks(mode :: Type{PerfTest.MPIMode})::Int +function PerfTest.ranks(mode::Type{PerfTest.MPIMode})::Int return mpi_size end # MPI Communication utilities +const MAIN_RANK = 0 """ -Gather results from all ranks to the main rank (rank 0) + gather_to_main(value::N; comm=MPI.COMM_WORLD, root=MAIN_RANK) where {N} + +Gather a scalar `value` from every rank into a `Dict{Int64, N}` on the main rank, +keyed by the source rank. Non-main ranks receive an empty `Dict{Int64, N}`. + +`N` must be a type natively supported by MPI (e.g. a `Number` / `isbits` type). """ -function MPICommunicateResults!(results::Number, rank, size) - if rank != 0 - MPI.Send(results, MPI.COMM_WORLD; dest=0) - return Dict{Int64,Number}(rank => results) +function gatherToMain(value::N; comm::MPI.Comm=MPI.COMM_WORLD, + root::Integer=MAIN_RANK) where {N} + rank = MPI.Comm_rank(comm) + nproc = MPI.Comm_size(comm) + + # Pack the local scalar in a length-1 buffer so MPI.Gather can move it. + sendbuf = N[value] + + if rank == root + recvbuf = Vector{N}(undef, nproc) + MPI.Gather!(sendbuf, MPI.UBuffer(recvbuf, 1), comm; root=root) + return Dict{Int64,N}(Int64(r) => recvbuf[r+1] for r in 0:(nproc-1)) else - gathered = Dict{Int64,Number}() - gathered[0] = results - for i in 1:(size-1) - gathered[i] = MPI.Recv(typeof(results), MPI.COMM_WORLD; source=i) - end - return gathered + MPI.Gather!(sendbuf, nothing, comm; root=root) + return Dict{Int64,N}() end end -function MPICommunicateResults!(results::Dict{Int64,Number}, rank, size) - if rank != 0 - MPI.Send(results[rank], MPI.COMM_WORLD; dest=0) +""" + gather_to_main(value::Vector{N}; comm=MPI.COMM_WORLD, root=MAIN_RANK) where {N} + +Vector dispatch: each rank contributes a `Vector{N}` (lengths may differ between +ranks). On the main rank, returns a `Dict{Int64, Vector{N}}` keyed by source rank. +Other ranks receive an empty dict. +""" +function gatherToMain(value::Vector{N}; comm::MPI.Comm=MPI.COMM_WORLD, + root::Integer=MAIN_RANK) where {N} + rank = MPI.Comm_rank(comm) + nproc = MPI.Comm_size(comm) + + # First, communicate per-rank lengths so the root can build a VBuffer. + local_len = Cint(length(value)) + counts = if rank == root + Vector{Cint}(undef, nproc) else - for i in 1:(size-1) - results[i] = MPI.Recv(Number, MPI.COMM_WORLD; source=i) + nothing + end + MPI.Gather!(Cint[local_len], + rank == root ? MPI.UBuffer(counts, 1) : nothing, + comm; root=root) + + if rank == root + total = sum(counts) + recvbuf = Vector{N}(undef, total) + MPI.Gatherv!(value, MPI.VBuffer(recvbuf, counts), comm; root=root) + + out = Dict{Int64,Vector{N}}() + offset = 0 + for r in 0:(nproc-1) + n = Int(counts[r+1]) + out[Int64(r)] = recvbuf[(offset+1):(offset+n)] + offset += n end + return out + else + MPI.Gatherv!(value, nothing, comm; root=root) + return Dict{Int64,Vector{N}}() end - return results end -# Reduction operations -op_sum(acc, new, _...) = acc + new -op_avg(acc, new, num) = acc + new / num -op_max(acc, new, _...) = max(acc, new) -op_min(acc, new, _...) = min(acc, new) - """ -Share values across all ranks and reduce them on the main rank + reduce_to_main(op, value; comm=MPI.COMM_WORLD, root=MAIN_RANK) + +Gather every rank's `value` onto the main rank using [`gather_to_main`](@ref) and +then apply the reduction `op` to the collected values. + +- On the **main rank** returns the reduced value (a scalar for the scalar + dispatch, a `Vector{N}` for the vector dispatch — whatever `op` produces). +- On **other ranks** returns the original local `value` unchanged. + +`op` is any callable accepting a single iterable, e.g. `sum`, `minimum`, +`maximum`, `prod`, or `vs -> reduce(+, vs)`. """ -function MPIShareAndReduce!(value::Number, reduction_op::Function, rank, size)::Number - gathered = MPICommunicateResults!(value, rank, size) - - if rank == 0 - acc = 0.0 - for i in 0:(size-1) - acc = reduction_op(acc, gathered[i], size) +function reduceToMain(value::N, op::Function; comm::MPI.Comm=MPI.COMM_WORLD, + root::Integer=MAIN_RANK) where {N} + gathered = gatherToMain(value; comm=comm, root=root) + if MPI.Comm_rank(comm) == root + ordered = [gathered[Int64(r)] for r in 0:(MPI.Comm_size(comm)-1)] + return op(ordered) + else + return value + end +end + +function reduceToMain(value::Vector{N}, op::Function; comm::MPI.Comm=MPI.COMM_WORLD, + root::Integer=MAIN_RANK) where {N} + gathered = gatherToMain(value; comm=comm, root=root) + if MPI.Comm_rank(comm) == root + ordered = [gathered[Int64(r)] for r in 0:(MPI.Comm_size(comm)-1)] + reduced = [] + for i in 1:length(value) + push!(reduced, op(v[i] for v in ordered)) end - return acc + return reduced + else + return value end - return value end -# MPI-synchronized STREAM benchmark kernels -# Synchronize the local kernels at the end to ensure coordinated memory transfer measurement -function copy_kernel_mpi(C, A; kwargs...) - STREAMBenchmark.copy_nthreads(C, A; kwargs...) - MPI.Barrier(MPI.COMM_WORLD) +""" + Share values of a dictionary from the main rank to all ranks. Value must be a isbits type. +""" +function bcastFromMain(val::N; comm::MPI.Comm=MPI.COMM_WORLD, + root::Integer=MAIN_RANK) where {N} + nbuf = N[val] + MPI.Bcast!(nbuf, comm; root=root) + return nbuf[1] end -function add_kernel_mpi(C, A, B; kwargs...) - STREAMBenchmark.add_nthreads(C, A, B; kwargs...) - MPI.Barrier(MPI.COMM_WORLD) +function bcastFromMain(vals::Vector{N}; comm::MPI.Comm=MPI.COMM_WORLD, + root::Integer=MAIN_RANK) where {N} + if MPI.Comm_rank(comm) == root + nbuf = vals + else + nbuf = Vector{N}(undef, length(vals)) + end + MPI.Bcast!(nbuf, comm; root=root) + return nbuf end +# Reduction operations +op_sum(vals) = sum(vals) +op_avg(vals) = sum(vals) / length(vals) +op_max(vals) = maximum(vals) +op_min(vals) = minimum(vals) + + +# MPI-synchronized bandwidth benchmark kernels +# Synchronize the local kernels at the end to ensure coordinated memory transfer measurement +using DataFrames +using BandwidthBenchmark +init_kernel = BandwidthBenchmark.init_kernel +copy_kernel = BandwidthBenchmark.copy_kernel +update_kernel = BandwidthBenchmark.update_kernel +triad_kernel = BandwidthBenchmark.triad_kernel +daxpy_kernel = BandwidthBenchmark.daxpy_kernel +striad_kernel = BandwidthBenchmark.striad_kernel +sdaxpy_kernel = BandwidthBenchmark.sdaxpy_kernel +allocate = BandwidthBenchmark.allocate + + + + """ -Run STREAM benchmark kernels with MPI synchronization +Run STREAM benchmark kernels with MPI synchronization, shamelessly borrowed from the original STREAMBenchmark.bwbench function """ -function _run_kernels_mpi(copy, add; - verbose=false, - N, - evals_per_sample=10, - write_allocate=true, - nthreads=Threads.nthreads(), - init=:parallel) - α = write_allocate ? 24 : 16 - β = write_allocate ? 32 : 24 - - f = t -> N * α / t - g = t -> N * β / t - - thread_indices = STREAMBenchmark._threadidcs(N, nthreads) - - # Initialize memory - if init == :parallel - A = Vector{Float64}(undef, N) - B = Vector{Float64}(undef, N) - C = Vector{Float64}(undef, N) - - # Fill in parallel (important for NUMA mapping / first-touch policy) - @threads :static for tid in 1:nthreads - @inbounds for i in thread_indices[tid] - A[i] = 0.0 - B[i] = 0.0 - C[i] = 0.0 - end +function _run_kernels_mpi(; + N::Integer=120_000_000, + niter::Integer=10, + verbose::Bool=false, + nthreads::Integer=Threads.nthreads(), + alignment::Integer=64, + write_allocate::Bool=false, +) + + # check arguments + 1 ≤ N || throw(ArgumentError("N must be ≥ 1.")) + 1 ≤ niter || throw(ArgumentError("niter must be ≥ 1.")) + ispow2(alignment) || throw(ArgumentError("alignment $alignment is not a power of 2")) + alignment ≥ sizeof(Float64) || + throw(ArgumentError("alignment $alignment is not a multiple of $(sizeof(Float64))")) + 1 ≤ nthreads ≤ Threads.nthreads() || throw( + ArgumentError( + "nthreads $nthreads must ≥ 1 and ≤ $(Threads.nthreads()). If you want more threads, start Julia with a higher number of threads.", + ), + ) + + # compute thread_indices + # use only first `nthreads` threads, i.e. @threads for tid in 1:nthreads + manual splitting + Nperthread = floor(Int, N / nthreads) + rest = rem(N, nthreads) + thread_indices = collect(Iterators.partition(1:N, Nperthread)) + if rest != 0 + # last thread compensates for the nonzero remainder + thread_indices[end-1] = thread_indices[end-1].start:thread_indices[end].stop + end + + + # allocate data + a = allocate(Float64, N, alignment) + b = allocate(Float64, N, alignment) + c = allocate(Float64, N, alignment) + d = allocate(Float64, N, alignment) + scalar = 3.0 + + # initialize data in parallel (important for NUMA / first-touch policy) + @threads :static for tid in 1:nthreads + @inbounds for i in thread_indices[tid] + a[i] = 2.0 + b[i] = 2.0 + c[i] = 0.5 + d[i] = 1.0 end - else - A = zeros(N) - B = zeros(N) - C = zeros(N) end - # COPY benchmark with MPI barrier setup - t_copy = @belapsed $copy($C, $A; nthreads=$nthreads, thread_indices=$thread_indices) setup = begin - MPI.Barrier(MPI.COMM_WORLD) - end samples = 10 evals = evals_per_sample - bw_copy = f(t_copy) - + # print information if verbose - println("╟─ Rank $(MPI.Comm_rank(MPI.COMM_WORLD)) COPY: ", round(bw_copy; digits=1), " B/s") + nthreads > 1 && println("Threading enabled, using $nthreads (of $(Threads.nthreads())) Julia threads") + alloc = 4.0 * sizeof(Float64) * N * 1.0e-06 + println("Total allocated datasize: $(alloc) MB") end - # ADD benchmark with MPI barrier setup - t_add = @belapsed $add($C, $A, $B; nthreads=$nthreads, thread_indices=$thread_indices) setup = begin - MPI.Barrier(MPI.COMM_WORLD) - end samples = 10 evals = evals_per_sample - bw_add = g(t_add) - - if verbose - println("╟─ Rank $(MPI.Comm_rank(MPI.COMM_WORLD)) ADD: ", round(bw_add; digits=1), " B/s") + # perform measurement + times = zeros(7, niter) + for k in 1:niter + times[1, k] = @elapsed (init_kernel(b, scalar; nthreads=nthreads, thread_indices=thread_indices); MPI.Barrier(MPI.COMM_WORLD)) + times[2, k] = @elapsed (copy_kernel(c, a; nthreads=nthreads, thread_indices=thread_indices); MPI.Barrier(MPI.COMM_WORLD)) + times[3, k] = @elapsed (update_kernel(a, scalar; nthreads=nthreads, thread_indices=thread_indices); MPI.Barrier(MPI.COMM_WORLD)) + times[4, k] = @elapsed (triad_kernel(a, b, c, scalar; nthreads=nthreads, thread_indices=thread_indices); MPI.Barrier(MPI.COMM_WORLD)) + times[5, k] = @elapsed (daxpy_kernel(a, b, scalar; nthreads=nthreads, thread_indices=thread_indices); MPI.Barrier(MPI.COMM_WORLD)) + times[6, k] = @elapsed (striad_kernel(a, b, c, d; nthreads=nthreads, thread_indices=thread_indices); MPI.Barrier(MPI.COMM_WORLD)) + times[7, k] = @elapsed (sdaxpy_kernel(a, b, c; nthreads=nthreads, thread_indices=thread_indices); MPI.Barrier(MPI.COMM_WORLD)) + end + + # analysis / table output of results + results = DataFrame(; + Function=String[], + var"Rate (MB/s)"=Float64[], + var"Rate (MFlop/s)"=Float64[], + var"Avg time"=Float64[], + var"Min time"=Float64[], + var"Max time"=Float64[], + ) + for j in 1:7 + # ignore the first run because of compilation + mintime = @views minimum(times[j, 2:end]) + maxtime = @views maximum(times[j, 2:end]) + avgtime = @views mean(times[j, 2:end]) + if write_allocate + bytes = BandwidthBenchmark.BENCHMARKS[j].words * BandwidthBenchmark.BENCHMARKS[j].write_alloc_factor * sizeof(Float64) * N + else + bytes = BandwidthBenchmark.BENCHMARKS[j].words * sizeof(Float64) * N + end + flops = BandwidthBenchmark.BENCHMARKS[j].flops * N + data_rate = 1.0e-06 * bytes / mintime + flop_rate = 1.0e-06 * flops / mintime + push!( + results, [BandwidthBenchmark.BENCHMARKS[j].label, data_rate, flop_rate, avgtime, mintime, maxtime] + ) end + verbose && pretty_table(results) - return (bw_copy, bw_add) + # validation + BandwidthBenchmark.validate(a, b, c, d, N, niter) + + return results end # Override core measurement functions for MPI mode -function PerfTest.getMachineInfo(::Type{PerfTest.MPIMode}, _PRFT_GLOBALS::PerfTest.GlobalSuiteData)::Expr - if Configuration.CONFIG["machine_benchmarking"]["memory_bandwidth_test_buffer_size"] == 0 - return quote - size = try - CpuId.cachesize() - catch - addLog("machine", "[MACHINE/MPI] CpuId failed, using default cache size") - [1024 * 1024 * 16] - end - _PRFT_GLOBALS.builtins[:MEM_CACHE_SIZES] = size - - addLog("machine", "[MACHINE/MPI] Memory buffer size for benchmarking = $(size ./ 1024 ./ 1024) [MB]") - end - else - return quote - _PRFT_GLOBALS.builtins[:MEM_CACHE_SIZES] = [$(Configuration.CONFIG["machine_benchmarking"]["memory_bandwidth_test_buffer_size"])] - - addLog("machine", "[MACHINE/MPI] Set by config, benchmark buffer size = $(_PRFT_GLOBALS.builtins[:MEM_CACHE_SIZES] ./ 1024 ./ 1024) [MB]") +function PerfTest.machineBenchmarks(mode::Type{PerfTest.MPIMode}, ctx::PerfTest.Context)::Expr + quote + # Block to create a separated scope + let + PerfTest.Topology.getMachineTopology!() + # First element L3, second element L2, third element L1 + _PRFT_GLOBALS.builtins[:MEM_CACHE_SIZES] = PerfTest.Topology.getCacheSizes() + $(ctx._global.uses_benchmarks == Set{Symbol}() ? quote end : quote + measureCPUPeakFlops!($mode, _PRFT_GLOBALS) + measureMemBandwidth!($mode, _PRFT_GLOBALS) + end) end end end function PerfTest.measureCPUPeakFlops!(::Type{PerfTest.MPIMode}, _PRFT_GLOBALS::PerfTest.GlobalSuiteData) LinearAlgebra.BLAS.set_num_threads(Threads.nthreads()) - + # Synchronize all ranks before measurement MPI.Barrier(MPI.COMM_WORLD) - + rank = mpi_rank size = mpi_size # Local peak flops measurement local_peakflops = LinearAlgebra.peakflops(; parallel=true) PerfTest.addLog("machine", "[MACHINE/MPI] (Rank $rank) max flops $local_peakflops") - + # Sum peak flops across all ranks - _PRFT_GLOBALS.builtins[:CPU_FLOPS_PEAK] = MPIShareAndReduce!(local_peakflops, op_sum, rank, size) - - if PerfTest.main_rank(PerfTest.MPIMode) + _PRFT_GLOBALS.builtins[:CPU_FLOPS_PEAK] = reduceToMain(local_peakflops, op_sum) + + if PerfTest.main_rank(PerfTest.MPIMode) PerfTest.addLog("machine", "[MACHINE/MPI] CPU max attainable flops (sum of $size ranks) = $(_PRFT_GLOBALS.builtins[:CPU_FLOPS_PEAK]) [FLOP/S]") end end function PerfTest.measureMemBandwidth!(::Type{PerfTest.MPIMode}, _PRFT_GLOBALS::PerfTest.GlobalSuiteData) - N = _PRFT_GLOBALS.builtins[:MEM_CACHE_SIZES][end] + def = PerfTest.Configuration.CONFIG["machine_benchmarking"]["memory_bandwidth_test_buffer_size"] + if def == 0 + N = _PRFT_GLOBALS.builtins[:MEM_CACHE_SIZES][1] * 4 + else + N = PerfTest.Configuration.CONFIG["machine_benchmarking"]["memory_bandwidth_test_buffer_size"] + end rank = mpi_rank size = mpi_size - - # Run MPI-synchronized STREAM benchmarks - bench_data = _run_kernels_mpi(copy_kernel_mpi, add_kernel_mpi; N=N, verbose=false) - - # Each rank's bandwidth, multiplied by number of ranks for aggregate - local_copy_bw = bench_data[1] - local_add_bw = bench_data[2] - - PerfTest.addLog("machine", "[MACHINE/MPI] (Rank $rank) max stream $bench_data") - # Sum bandwidth across all ranks - total_copy_bw = MPIShareAndReduce!(local_copy_bw, op_sum, rank, size) - total_add_bw = MPIShareAndReduce!(local_add_bw, op_sum, rank, size) - - _PRFT_GLOBALS.builtins[:MEM_STREAM] = (total_copy_bw, total_add_bw) - _PRFT_GLOBALS.builtins[:MEM_STREAM_COPY] = total_copy_bw - _PRFT_GLOBALS.builtins[:MEM_STREAM_ADD] = total_add_bw - if PerfTest.main_rank(PerfTest.MPIMode) - PerfTest.addLog("machine", "[MACHINE/MPI] CPU max attainable bandwidth (sum of $size ranks) = $(_PRFT_GLOBALS.builtins[:MEM_STREAM]) [Byte/s]") + # Run MPI-synchronized bandwidth benchmarks + bench_data = _run_kernels_mpi(N=N, verbose=false) + peakbandwidth_local = bench_data[!, 2] * 10^6 # Convert from MB/s to Byte/s + + peakbandwidth = reduceToMain(peakbandwidth_local, op_sum) + # in Bytes/sec + # "Init" + # "Copy" + # "Update" + # "Triad" + # "Daxpy" + # "STriad" + # "SDaxpy" + if PerfTest.main_rank(PerfTest.MPIMode) + _PRFT_GLOBALS.builtins[:MEM_BENCH] = Float64.(peakbandwidth) + _PRFT_GLOBALS.builtins[:MEM_BENCH_INIT] = peakbandwidth[1] + _PRFT_GLOBALS.builtins[:MEM_BENCH_COPY] = peakbandwidth[2] + _PRFT_GLOBALS.builtins[:MEM_BENCH_UPDATE] = peakbandwidth[3] + _PRFT_GLOBALS.builtins[:MEM_BENCH_TRIAD] = peakbandwidth[4] + _PRFT_GLOBALS.builtins[:MEM_BENCH_DAXPY] = peakbandwidth[5] + _PRFT_GLOBALS.builtins[:MEM_BENCH_STRIAD] = peakbandwidth[6] + _PRFT_GLOBALS.builtins[:MEM_BENCH_SDAXPY] = peakbandwidth[7] + + # COMPAT symbols for benchmarks: + _PRFT_GLOBALS.builtins[:MEM_STREAM_COPY] = peakbandwidth[2] + _PRFT_GLOBALS.builtins[:MEM_STREAM_ADD] = peakbandwidth[4] + _PRFT_GLOBALS.builtins[:MEM_STREAM_STRIAD] = peakbandwidth[6] + + PerfTest.addLog("machine", "[MACHINE/MPI] CPU max attainable bandwidth (sum of $size ranks) = $(_PRFT_GLOBALS.builtins[:MEM_BENCH]) [Byte/s]") + else + # Non-main ranks set the benchmark values to zero to avoid confusion + _PRFT_GLOBALS.builtins[:MEM_BENCH] = zeros(Float64, 7) + _PRFT_GLOBALS.builtins[:MEM_BENCH_INIT] = 0 + _PRFT_GLOBALS.builtins[:MEM_BENCH_COPY] = 0 + _PRFT_GLOBALS.builtins[:MEM_BENCH_UPDATE] = 0 + _PRFT_GLOBALS.builtins[:MEM_BENCH_TRIAD] = 0 + _PRFT_GLOBALS.builtins[:MEM_BENCH_DAXPY] = 0 + _PRFT_GLOBALS.builtins[:MEM_BENCH_STRIAD] = 0 + _PRFT_GLOBALS.builtins[:MEM_BENCH_SDAXPY] = 0 + + _PRFT_GLOBALS.builtins[:MEM_STREAM_COPY] = 0 + _PRFT_GLOBALS.builtins[:MEM_STREAM_ADD] = 0 + _PRFT_GLOBALS.builtins[:MEM_STREAM_STRIAD] = 0 + end + for (k, v) in _PRFT_GLOBALS.builtins + if typeof(v) in [Number, Vector{Number}, Char, Bool] + _PRFT_GLOBALS.builtins[k] = bcastFromMain(v; comm=MPI.COMM_WORLD) + end end end # Override metric creation for MPI mode -function PerfTest.newMetricResult(::Type{PerfTest.MPIMode}; - name, - units, - value, - auxiliary=false, - magnitude_prefix="", - magnitude_mult=1, - reduct="Sum") +function PerfTest.newMetricResult(::Type{PerfTest.MPIMode}; + name, + units, + value, + auxiliary=false, + magnitude_prefix="", + magnitude_mult=1, + reduct="Sum") mpi_info = PerfTest.MPI_MetricInfo(MPI.Comm_size(MPI.COMM_WORLD), reduct) return PerfTest.Metric_Result(name, units, value, auxiliary, magnitude_prefix, magnitude_mult, mpi_info) end # Build primitive metrics with MPI reduction - -function PerfTest.buildPrimitiveMetrics!(::Type{PerfTest.MPIMode}, - ts::PerfTest.PerfTestSet, - test_result::PerfTest.Test_Result) +function PerfTest.buildPrimitiveMetrics!(::Type{PerfTest.MPIMode}, ts::PerfTest.PerfTestSet, test_result::PerfTest.Test_Result) rank = mpi_rank size = mpi_size - + # MEDIAN TIME - use MAX across ranks (slowest determines overall time) local_median_time = median(ts.benchmarks[test_result.name]).time / 1e9 - test_result.primitives[:median_time] = MPIShareAndReduce!(local_median_time, op_max, rank, size) - + test_result.primitives[:median_time] = reduceToMain(local_median_time, op_max) + # MIN TIME - use MIN across ranks local_min_time = minimum(ts.benchmarks[test_result.name]).time / 1e9 - test_result.primitives[:min_time] = MPIShareAndReduce!(local_min_time, op_min, rank, size) - + test_result.primitives[:min_time] = reduceToMain(local_min_time, op_min) + # Iterator count - should be equal among all ranks test_result.primitives[:iterator] = ts.iterator - test_result.primitives[:autoflop] = MPIShareAndReduce!(test_result.primitives[:autoflop], op_sum, rank, size) + test_result.primitives[:autoflop] = reduceToMain(test_result.primitives[:autoflop], op_sum) end -function PerfTest.clean(mode :: Type{PerfTest.MPIMode}) - PerfTest.toggleMPI() +# Nothing to do here (YET) +function PerfTest.clean(mode::Type{PerfTest.MPIMode}) end end # module PerfTest_MPIExt \ No newline at end of file diff --git a/src/PerfTest.jl b/src/PerfTest.jl index 9c166fa..f032719 100644 --- a/src/PerfTest.jl +++ b/src/PerfTest.jl @@ -1,7 +1,7 @@ module PerfTest export @perftest, @on_perftest_exec, @on_perftest_ignore, @perftest_config, @export_vars, @define_benchmark, - @define_eff_memory_throughput, @define_metric, @roofline, @define_test_metric, magnitudeAdjust, @perfcompare, @perfcmp + @define_eff_memory_throughput, @def_eff_mem, @define_metric, @roofline, @define_test_metric, magnitudeAdjust, @perfcompare, @perfcmp, runperftests, @regression using Test using MacroTools @@ -10,18 +10,17 @@ using Configurations using Printf using BenchmarkTools -using STREAMBenchmark using LinearAlgebra -using CpuId +using Hwloc var"@capture" = MacroTools.var"@capture" -abstract type NormalMode end -struct MPIMode <: NormalMode end +abstract type Mode end +struct MPIMode <: Mode end +struct NormalMode <: Mode end mode = NormalMode - ### PARSING TIME # Data structures used in the parse and transform procedures @@ -54,22 +53,24 @@ include("transform/methodologies/roofline.jl") include("transform/prefix.jl") include("transform/suffix.jl") - -# TODO: Separate Macro definitions -include("execution/macros/perftest.jl") -include("execution/macros/perfcompare.jl") -include("execution/macros/roofline.jl") -include("execution/macros/exec_ignore.jl") -include("execution/macros/customs.jl") -include("execution/macros/configuration.jl") +# EXECUTION PART include("execution/structs.jl") include("execution/testset.jl") - # Machine features extraction +include("execution/machine_topology.jl") include("execution/machine_benchmarking.jl") +# Separate Macro definitions +include("execution/macros/perftest.jl") +include("execution/macros/perfcompare.jl") +include("execution/macros/roofline.jl") +include("execution/macros/regression.jl") +include("execution/macros/exec_ignore.jl") +include("execution/macros/customs.jl") +include("execution/macros/configuration.jl") +include("execution/macros/topology.jl") # Rules of the ruleset include("transform/parsing/hierarchy_transform_test_region.jl") @@ -85,6 +86,7 @@ include("transform/auxiliar.jl") # Functions used by the generated suites include("execution/printing.jl") include("execution/data_handling.jl") +include("execution/retrieve.jl") include("execution/units.jl") include("execution/misc.jl") @@ -92,24 +94,24 @@ include("execution/misc.jl") include("bencher/BencherREST.jl") # Base active rules -rules = ASTRule[testset_macro_rule, - test_macro_rule, - test_throws_macro_rule, - test_logs_macro_rule, - inferred_macro_rule, - test_deprecated_macro_rule, - test_warn_macro_rule, - test_nowarn_macro_rule, +first_pass_rules = ASTRule[testset_macro_rule, + test_macro_rule, + test_throws_macro_rule, + test_logs_macro_rule, + inferred_macro_rule, + test_deprecated_macro_rule, + test_warn_macro_rule, + test_nowarn_macro_rule, test_broken_macro_rule, test_skip_macro_rule, perftest_macro_rule, back_macro_rule, - prefix_macro_rule, - suffix_macro_rule, config_macro_rule, + threads_macro_rule, on_perftest_exec_rule, on_perftest_ignore_rule, define_memory_throughput_rule, + regression_macro_rule, define_metric_rule, define_benchmark_rule, export_vars_rule, @@ -118,6 +120,11 @@ rules = ASTRule[testset_macro_rule, recursive_rule ] +second_pass_rules = ASTRule[ + prefix_macro_rule, + suffix_macro_rule, +] + # Transform routines in target_transform.jl perftest_expression_ruleset = [ @@ -127,7 +134,7 @@ perftest_expression_ruleset = [ perftest_dot_interpolation_rule, ] -function parseTarget(expr :: Expr, context::Context)::Expr +function parseTarget(expr::Expr, context::Context)::Expr return MacroTools.prewalk(ruleSet(context, perftest_expression_ruleset), expr) end @@ -141,7 +148,7 @@ WARNING: the rule set will apply the FIRST rule that matches with the expression - `context` the context structure of the tree run, it will be ocassinally used by some rules on the set. - `rules` the collection of rules that will belong to the resulting set. """ -function ruleSet(context::Context, rules :: Vector{ASTRule}) +function ruleSet(context::Context, rules::Vector{ASTRule}) function _ruleSet(x) for rule in rules if rule.match(x) @@ -166,12 +173,14 @@ This method gets a input julia expression, and a context register and executes a """ function _treeRun(input_expr::Expr, context::Context, args...) - return MacroTools.prewalk(ruleSet(context, rules), input_expr) + first_pass = MacroTools.prewalk(ruleSet(context, first_pass_rules), input_expr) + second_pass = MacroTools.prewalk(ruleSet(context, second_pass_rules), first_pass) + return second_pass end ctx = nothing -function setupContext(path :: AbstractString) +function setupContext(path::AbstractString) global ctx = Context(GlobalContext(path, VecErrorCollection(), formula_symbols)) ctx._global.original_file_path = path @@ -184,28 +193,47 @@ The function will return a Julia expression with the resulting performance testi - `path` the path of the script to be transformed. """ -function treeRun(path::AbstractString) +function treeRun(path::AbstractString; config=nothing) # Set log directory setLogFolder() # Clear logs #clearLogs() # Load configuration - if !init_dummy_flag - config = Configuration.load_config() - else + if init_dummy_flag config = Configuration.load_dummy_config() + else + if config === nothing + config = Configuration.load_config() + else + _config = Configuration.load_config() + config = Configuration.merge_configs(_config, config) + Configuration.load_config(config) + end end - if config["general"]["verbose"] + if config["general"]["verbose"] >= 1 verboseOutput() end + if config["MPI"]["enabled"] == true + if isdefined(Main, :MPI) + global mode = MPIMode + addLog("general", "[MPI] MPI enabled in configuration and MPI package found, switching to MPI aware generation") + else + @warn "[MPI] MPI enabled in configuration but MPI package not found, defaulting to non-MPI mode" + end + else + global mode = NormalMode + end + # Load original input_expr = loadFileAsExpr(path) setupContext(path) + Topology.setupLog(addLog, x -> throwParseError!(x, ctx)) + # Run through AST and build new expressions full = _treeRun(input_expr, ctx) @@ -214,24 +242,96 @@ function treeRun(path::AbstractString) # Mount inside a module environment module_full = Expr(:toplevel, - Expr(:module, true, :__PERFTEST__, - Expr(:block, full.args...))) + Expr(:module, true, :__PERFTEST__, + Expr(:block, full.args...))) - if num_errors(ctx._global.errors) > 0 - printErrors(ctx._global.errors) - return quote @warn "Parsing failed" end + if num_errors(ctx) > 0 + printErrors(ctx) + return quote + @warn "Parsing failed" + end end - if config["general"]["verbose"] + if config["general"]["verbose"] >= 2 saveLogFolder() end return MacroTools.prettify(module_full) end + +""" + runperftests(file; ...) + + # Description + Simplified function to access the perftest transformation and subsequent execution of performance test suites. + It takes a recipe script, transforms it into a performance testing suite and executes it. The resulting suite is also saved in a file for later usage. + + # Arguments + `file ::AbstractString`: the path of the recipe script to be transformed + + # Keyword arguments + `execute::Bool = true` : whether the resulting suite should be executed right after generation, by default true. If false, the resulting suite will only be saved in a file with the name of the input with and added "_perfsuite.jl" suffix. + `verbose::Int = 0` : level of verbosity, from 0 to 3 higher is more verbose + 0 : minimal output, only warnings and errors + 1 : general information about the transformation and execution process + 2 : detailed information about the transformation and execution process, logs are saved in a folder + 3 : debug level, very detailed information about the transformation and execution process + `clean ::Bool = false` : whether to leave the config file, the output test suite and the test results or delete everything after the suite has been executed, !!! including previously done results and logs CAREFUL !!!. + `config ::Dict{String,Any} = {}` : other configuration parameters to override the configuration file. Configuration priority: config macro > this argument > configuration file. See configuration for more info. + + + # (!) Do not mistake this method for the macro with the same name, which is used to set test targets inside the recipe script. + + # Example of a config parameter value: + `{"regression" : {"enabled" : true}, "general" : {"recursive" : false}}` + + + See the macro reference for more details about the recipe script format and the possible configurations. +""" +function runperftests(file::AbstractString; execute::Bool=true, verbose::Int=0, clean::Bool=false, config::Union{Dict,Nothing}=nothing) + # Load config file + Configuration.load_config() + # Override with config argument + if config isa Dict + config = Configuration.merge_configs(Configuration.CONFIG, config) + else + config = Configuration.CONFIG + end + # Override with parameters + configp = Dict{String,Any}() + configp["general"] = Dict{String,Any}() + if config["general"]["verbose"] != verbose + configp["general"]["verbose"] = verbose + end + config = Configuration.merge_configs(config, configp) + + expr = treeRun(file, config=config) + name = replace(file, r"\.jl$" => "_perfsuite.jl") + saveExprAsFile(expr, name) + if num_errors(ctx) == 0 + addLog("general", "[SUCCESS] Performance testing suite generated and saved in $name\n") + if mode == MPIMode + addLog("general", "[MPI] Performance testing suite generated in MPI mode, make sure to execute it with an MPI launcher to properly run the tests across all ranks") + end + if execute && mode == NormalMode + addLog("general", "[PERFTEST] Executing performance testing suite $name") + Main.include(name) + end + end + if clean + rm(name) + rm("./perftest_config.toml") + rm("./$(Configuration.CONFIG["general"]["save_folder"])", recursive=true) + if ispath("./.perftest_logs") + rm("./perftest_logs", recursive=true) + end + end +end + """ - In order for the suite to be MPI aware, this function has to be called. Calling it again will disable this feature. + [!] DEPRECATED, do not use, use the Perftest configuration attributes instead. """ function toggleMPI() if mode == NormalMode @@ -245,17 +345,20 @@ transform = treeRun MPItransform(path) = (toggleMPI(); transform(path); toggleMPI()) -init_dummy_flag :: Bool = false +init_dummy_flag::Bool = false -function __init__() - # Precompile the transformation - # This is a quite rudimentary (but effective) solution, a cleaner version is to be expected in the future - try - global init_dummy_flag = true - x = PerfTest.transform(joinpath(dirname(pathof(PerfTest)), "transform/dummy.jl")) - global init_dummy_flag = false - catch - @warn "Precompilation of transform could not be done during initialization. It will be performed during the next function call." +import PrecompileTools +PrecompileTools.@compile_workload begin + try + redirect_stdout(Base.DevNull()) do + global init_dummy_flag = true + x = PerfTest.transform(joinpath(dirname(pathof(PerfTest)), "transform/dummy.jl")) + end + catch err + finally + if ispath("./$(Configuration.PRECOMPILATION_CONFIG["general"]["save_folder"])") + rm("./$(Configuration.PRECOMPILATION_CONFIG["general"]["save_folder"])", recursive=true) + end global init_dummy_flag = false end end diff --git a/src/execution/data_handling.jl b/src/execution/data_handling.jl index 3dd62ff..895c81e 100644 --- a/src/execution/data_handling.jl +++ b/src/execution/data_handling.jl @@ -13,6 +13,11 @@ end This method is used to save historical data of a performance test suite to a save file located in `path`. """ function saveDataFile(path :: AbstractString, contents:: Perftest_Datafile_Root) + # If path has unexisting parent directories, create them + parent_dir = dirname(path) + if !isdir(parent_dir) + mkpath(parent_dir) + end return jldsave(path; contents) end @@ -190,3 +195,4 @@ function get_metric_results(datafile::Perftest_Datafile_Root, metric_name::Abstr end return metric_results end + diff --git a/src/execution/machine_benchmarking.jl b/src/execution/machine_benchmarking.jl index 531f160..4cd4028 100644 --- a/src/execution/machine_benchmarking.jl +++ b/src/execution/machine_benchmarking.jl @@ -1,28 +1,5 @@ # Memory and CPU benchmarks used by different methodologies - -function getMachineInfo()::Expr - if Configuration.CONFIG["machine_benchmarking"]["memory_bandwidth_test_buffer_size"] == 0 - return quote - size = try - CpuId.cachesize() - catch - addLog("machine", "[MACHINE] CpuId failed, using default cache size") - [1024 * 1024 * 16] - end - global _PRFT_GLOBALS.builtins[:MEM_CACHE_SIZES] = size - - addLog("machine", "[MACHINE] Memory buffer size for benchmarking = $(size ./ 1024 ./ 1024) [MB]") - end - else - return quote - global _PRFT_GLOBALS.builtins[:MEM_CACHE_SIZES] = [$(Configuration.CONFIG["machine_benchmarking"]["memory_bandwidth_test_buffer_size"])] - - addLog("machine", "[MACHINE] Set by config, benchmark buffer size = $(_PRFT_GLOBALS.builtins[:MEM_CACHE_SIZES][1] ./ 1024 ./ 1024) [MB]") - end - end -end - function measureCPUPeakFlops!(::Type{<:NormalMode}, _PRFT_GLOBALS::GlobalSuiteData) LinearAlgebra.BLAS.set_num_threads(Threads.nthreads()) # In Flop/s @@ -31,80 +8,55 @@ function measureCPUPeakFlops!(::Type{<:NormalMode}, _PRFT_GLOBALS::GlobalSuiteDa end using Base.Threads +using BandwidthBenchmark -copy_kernel(C,A;kwargs...) = STREAMBenchmark.copy_nthreads(C,A;kwargs...) -add_kernel(C,A,B;kwargs...) = STREAMBenchmark.add_nthreads(C,A,B;kwargs...) - -# This function is heavily based on the respective from STREAMBenchmark -function _run_kernels(copy, add; - verbose = true, - N, - evals_per_sample = 10, - write_allocate = true, - nthreads = Threads.nthreads(), - init = :parallel) - α = write_allocate ? 24 : 16 - β = write_allocate ? 32 : 24 - - f = t -> N * α / t - g = t -> N * β / t - - # N / nthreads if necessary - thread_indices = STREAMBenchmark._threadidcs(N, nthreads) - - # initialize memory - if init == :parallel - A = Vector{Float64}(undef, N) - B = Vector{Float64}(undef, N) - C = Vector{Float64}(undef, N) - s = rand() - - # fill in parallel (important for NUMA mapping / first-touch policy) - @threads :static for tid in 1:nthreads - @inbounds for i in thread_indices[tid] - A[i] = 0.0 - B[i] = 0.0 - C[i] = 0.0 - end - end +function measureMemBandwidth!(::Type{<:NormalMode}, _PRFT_GLOBALS::GlobalSuiteData) + def = Configuration.CONFIG["machine_benchmarking"]["memory_bandwidth_test_buffer_size"] + if def == 0 + N = _PRFT_GLOBALS.builtins[:MEM_CACHE_SIZES][1] * 4 else - A = zeros(N) - B = zeros(N) - C = zeros(N) - s = rand() + N = Configuration.CONFIG["machine_benchmarking"]["memory_bandwidth_test_buffer_size"] end - - # COPY - t_copy = @belapsed $copy($C, $A; nthreads = $nthreads, thread_indices = $thread_indices) samples=10 evals=evals_per_sample - bw_copy = f(t_copy) - - # ADD - t_add = @belapsed $add($C, $A, $B; nthreads = $nthreads, - thread_indices=$thread_indices) samples = 10 evals = evals_per_sample - bw_add = g(t_add) - - return (bw_copy,bw_add) -end - - -function measureMemBandwidth!(::Type{<:NormalMode}, _PRFT_GLOBALS::GlobalSuiteData) - bench_data = _run_kernels(copy_kernel, add_kernel; N=div(_PRFT_GLOBALS.builtins[:MEM_CACHE_SIZES][end], 2)) + bench_data = BandwidthBenchmark.bwbench(N = N, verbose=false) + peakbandwidth = bench_data[!,2] * 10^6 # Convert from MB/s to Byte/s # in Bytes/sec - peakbandwidth = bench_data - _PRFT_GLOBALS.builtins[:MEM_STREAM] = peakbandwidth - _PRFT_GLOBALS.builtins[:MEM_STREAM_COPY] = peakbandwidth[1] - _PRFT_GLOBALS.builtins[:MEM_STREAM_ADD] = peakbandwidth[2] - addLog("machine", "[MACHINE] CPU max attainable bandwidth = $(_PRFT_GLOBALS.builtins[:MEM_STREAM]) [Byte/s]") + # "Init" + # "Copy" + # "Update" + # "Triad" + # "Daxpy" + # "STriad" + # "SDaxpy" + _PRFT_GLOBALS.builtins[:MEM_BENCH] = peakbandwidth + _PRFT_GLOBALS.builtins[:MEM_BENCH_INIT] = peakbandwidth[1] + _PRFT_GLOBALS.builtins[:MEM_BENCH_COPY] = peakbandwidth[2] + _PRFT_GLOBALS.builtins[:MEM_BENCH_UPDATE] = peakbandwidth[3] + _PRFT_GLOBALS.builtins[:MEM_BENCH_TRIAD] = peakbandwidth[4] + _PRFT_GLOBALS.builtins[:MEM_BENCH_DAXPY] = peakbandwidth[5] + _PRFT_GLOBALS.builtins[:MEM_BENCH_STRIAD] = peakbandwidth[6] + _PRFT_GLOBALS.builtins[:MEM_BENCH_SDAXPY] = peakbandwidth[7] + + # COMPAT symbols for benchmarks: + _PRFT_GLOBALS.builtins[:MEM_STREAM_COPY] = peakbandwidth[2] + _PRFT_GLOBALS.builtins[:MEM_STREAM_ADD] = peakbandwidth[4] + _PRFT_GLOBALS.builtins[:MEM_STREAM_STRIAD] = peakbandwidth[6] + + addLog("machine", "[MACHINE] CPU max attainable bandwidth = $(_PRFT_GLOBALS.builtins[:MEM_BENCH]) [Byte/s]") end -function machineBenchmarks(mode ::Type{<:NormalMode})::Expr +function machineBenchmarks(mode ::Type{<:NormalMode}, ctx :: Context)::Expr quote # Block to create a separated scope let - $(getMachineInfo()) - measureCPUPeakFlops!($mode, _PRFT_GLOBALS) - measureMemBandwidth!($mode, _PRFT_GLOBALS) + PerfTest.Topology.getMachineTopology!() + # First element L3, second element L2, third element L1 + _PRFT_GLOBALS.builtins[:MEM_CACHE_SIZES] = PerfTest.Topology.getCacheSizes() + $(ctx._global.uses_benchmarks == Set{Symbol}() ? quote + end : quote + measureCPUPeakFlops!($mode, _PRFT_GLOBALS) + measureMemBandwidth!($mode, _PRFT_GLOBALS) + end) end end end diff --git a/src/execution/machine_topology.jl b/src/execution/machine_topology.jl new file mode 100644 index 0000000..f31806b --- /dev/null +++ b/src/execution/machine_topology.jl @@ -0,0 +1,487 @@ +module Topology + +using Hwloc +using ThreadPinningCore + +addLog = x -> @info "[$(string(x))]" +throwParseError! = x -> @error "Invalid thread arrangement: $x" +struct Index + indexing::Vector{Integer} + symbols::Vector{Symbol} +end + +""" + For a process, the specification on how many threads and how many numa domains to use for the test suite. + (#numas, #threads_per_numa) + Example: + (1, 1) or :single -> One thread on one numa node + (1, 2) -> Two threads on two numa nodes + (1, 1//1) or :numa -> All threads on one numa node + (1, 1//2) -> Half of the threads available for one numa node + (2, 1) -> One thread on each of two numa nodes + (1//1,1//1) or :all -> All posible threads on all numa nodes + # For later: + :gpu_single -> One thread in the same numa domain as a GPU device + :gpu_each -> One thread per GPU device, on the same domains as their respective devices +""" +ArrangementSpec = Union{Symbol,Tuple{Union{Integer,Rational{Int64}},Union{Integer,Rational{Int64}}}} + +cpu_model = Nothing +hwloc_efficiency_mask = 0x0 +hwloc_topology = nothing + +pu_dict = nothing +devices = nothing +numas = nothing + +# Internal methods + +function getChildren(root::Hwloc.Object, indexing::Vector{Integer}) + @assert length(indexing) > 0 "On PerfTest.Topology, empty index to getChildren" + c = root + for i in indexing + c = c.children[i] + end + return c +end + +function matchIdxs(idx1, idx2)::Bool + l = length(idx1) + if length(idx2) != l + return false + end + for i in 1:l + if idx1[i] != idx2[i] + return false + end + end + return true +end + +function isAParentOfB(A, B)::Bool + return matchIdxs(A, B[1:length(A)]) +end + +function findParentPackage(elem::Index)::Index + idx = 0 + for (i, symbol) in enumerate(elem.symbols) + if symbol == :Package || symbol == :Group + idx = i + break + end + end + return Index(copy(elem.indexing[1:idx]), copy(elem.symbols[1:idx])) +end + +function recursiveTopRetrieve!(obj::Hwloc.Object, pu_dict, devices, numas, indexing, symbols) + # Parse NUMA nodes + for (i, mem) in enumerate(obj.memory_children) + if mem.type_ == :NUMANode + push!(indexing, i) + push!(symbols, :NUMANode) + numas[length(numas)+1] = Topology.Index(indexing, symbols) + pop!(indexing) + pop!(symbols) + end + end + # Parse machine hierarchy + for (i, child) in enumerate(obj.children) + push!(indexing, i) + push!(symbols, child.type_) + if child.type_ == :PU + pu_dict[child.os_index] = Topology.Index(indexing, symbols) + end + recursiveTopRetrieve!(child, pu_dict, devices, numas, indexing, symbols) + pop!(indexing) + pop!(symbols) + end + # Parse devices + for (i, io) in enumerate(obj.io_children) + push!(indexing, i) + push!(symbols, io.type_) + devices[length(devices)+1] = Topology.Index(indexing, symbols) + recursiveTopRetrieve!(io, pu_dict, devices, numas, indexing, symbols) + pop!(indexing) + pop!(symbols) + end +end + +# External methods + +""" + In case of heterogeneous levels, returns the biggest of each level +""" +function getCacheSizes()::Vector{Int64} + sizes = [] + for i in [:L3Cache, :L2Cache, :L1Cache] + if (size = getMaxCacheSize(i)) > 0 + push!(sizes, size) + end + end + return Int64.(sizes) +end + +""" + Biggest cache on the topology in bytes + + If the level is not present, 0 will be returned. +""" +function getMaxCacheSize(level::Symbol)::Integer + + current_max = 0 + for (_, x) in pu_dict + indexing, symbols = x.indexing, x.symbols + for (i, symbol) in enumerate(symbols) + if symbol == level + c = getChildren(hwloc_topology, indexing[1:i]) + current_max = current_max < c.attr.size ? c.attr.size : current_max + end + end + end + return current_max +end + +function getNUMADomainCores()::Vector{Vector{Integer}} + cores = [[] for i in 1:length(numas)] + for (keycore, x) in pu_dict + indexing_core = x.indexing + for (keynuma, y) in numas + indexing_numa = y.indexing + if isAParentOfB(indexing_numa[1:end-1], indexing_core) + push!(cores[keynuma], keycore) + end + end + end + return cores +end + +function getDeviceAffineCores(device_id)::Vector{Integer} + cores = [] + parent = findParentPackage(devices[device_id]) + for (key, x) in pu_dict + if isAParentOfB(parent.indexing, x.indexing) + push!(cores, key) + end + end + return cores +end + +function getEfficiencyCores()::Vector{Integer} + eff = [] + for core_id in keys(pu_dict) + if ((1 << Integer(core_id)) & hwloc_efficiency_mask) > 0 + push!(eff, core_id) + end + end + return eff +end + +setupLog(logfunc::Function, errorfunc::Function) = begin + Topology.addLog = logfunc + Topology.throwParseError! = errorfunc +end + +""" +Obtain some basic parameters regarding the machine topology like NUMA arrangement, cache sizes, network interfaces. The info is obtained from Hwloc.jl + Modifies topology submodule. +""" +function getMachineTopology!() + + Topology.cpu_model = Sys.CPU_NAME + Topology.hwloc_topology = gettopology() + cpukind = Hwloc.get_cpukind_info() + for i in cpukind + if i.efficiency_rank == 0 + Topology.hwloc_efficiency_mask = i.masks[end] + break + end + end + + local indexing = [] + local symbols = [] + local pu_dict = Dict{Integer,Topology.Index}() + local devices = Dict{Integer,Topology.Index}() + local numas = Dict{Integer,Topology.Index}() + + recursiveTopRetrieve!(Topology.hwloc_topology, pu_dict, devices, numas, indexing, symbols) + + Topology.pu_dict = pu_dict + Topology.numas = numas + Topology.devices = devices + + addLog("machine", "[MACHINE] $(cpu_model), $(length(numas)) NUMA domains with $([t for t in threadsPerNuma()]) threads per domain.") + addLog("machine", "[MACHINE] Cache sizes (biggest per level): $(getCacheSizes() ./ 1024 ./ 1024) MBytes.") + addLog("machine", "[THREADS] Julia process is running with $(Threads.nthreads()) threads.") + + return pu_dict, devices, numas +end + +function numberOfNUMAS()::Integer + return length(numas) +end + +function threadsPerNuma()::Vector{Integer} + return [length(numa) for numa in getNUMADomainCores()] +end + +# Autovectorized +function literallizeArrangement(arrgmts::Vector{N})::Vector{N} where {N<:Tuple{<:Number,<:Number}} + return literallizeArrangement.(arrgmts) +end + +# Manual pinning +function literallizeArrangement(arrgmts::Vector{N})::Vector{N} where {N<:Integer} + if Topology.hwloc_topology isa Nothing + Topology.getMachineTopology!() + end + threadids = Base.Flatten(Topology.getNUMADomainCores()) + if !all([in(pin, threadids) for pin in arrgmts]) + error("Invalid manual thread pin, thread id not found in machine, (machine threads [$(min(threadids))-$(max(threadids))])") + end + return arrgmts +end + +""" + Prepares a thread arrangement specification for execution by converting it to a literal form `(numas, threads_per_numa)`. + And verifies that the arrangement is satisfiable on the current machine and Julia process. +""" +function literallizeArrangement(arrgmt::ArrangementSpec)::Tuple{Integer,Integer} + # Ensure topology is loaded + if Topology.hwloc_topology isa Nothing + Topology.getMachineTopology!() + end + reason = "" + + total_numas = Topology.numberOfNUMAS() + threads_per_numa = Topology.threadsPerNuma() # Vector{Integer}, one entry per NUMA + + if arrgmt isa Symbol + arrgmt = if arrgmt == :single + (1, 1) + elseif arrgmt == :numa + (1, 1 // 1) + elseif arrgmt == :all + (1 // 1, 1 // 1) + else + error("Invalid thread arrangement: symbol `$arrgmt` is unrecognised. " * + "Must be one of [:single, :numa, :all] or a tuple — see ArrangementSpec docs.") + end + end + + if length(arrgmt) != 2 + error("Invalid thread arrangement: $arrgmt has $(length(arrgmt)) element(s); " * + "expected exactly 2 (#numas, #threads_per_numa).") + end + + numa_spec, thread_spec = arrgmt + + literal_numas::Integer = if numa_spec isa Rational + if !(0 < numa_spec <= 1) + error("Invalid thread arrangement: NUMA ratio $numa_spec is out of range; " * + "must satisfy 0 < ratio ≤ 1.") + end + # e.g. 1//1 → all NUMAs, 1//2 → half of them (at least 1) + max(1, floor(Integer, numa_spec * total_numas)) + else + if numa_spec <= 0 + error("Invalid thread arrangement: if #numa is and integer, it shall be positive, got $numa_spec.") + end + if numa_spec > total_numas + reason *= ("Invalid thread arrangement: requested $numa_spec NUMA domain(s) but " * + "the host only has $total_numas.") + end + Integer(numa_spec) + end + + # Use the minimum available thread count across the selected NUMAs so the + # arrangement is satisfiable on every chosen domain. + min_threads_in_selected = minimum(threads_per_numa[1:literal_numas]) + + literal_threads::Integer = if thread_spec isa Rational + if !(0 < thread_spec <= 1) + error("Invalid thread arrangement: thread ratio $thread_spec is out of range; " * + "must satisfy 0 < ratio ≤ 1.") + end + max(1, floor(Integer, thread_spec * min_threads_in_selected)) + else + if thread_spec <= 0 + error("Invalid thread arrangement: #threads_per_numa must be a positive integer, " * + "got $thread_spec.") + end + if thread_spec > min_threads_in_selected + reason *= ("Invalid thread arrangement: requested $thread_spec thread(s) per NUMA but " * + "the most constrained of the $literal_numas selected NUMA domain(s) only " * + "has $min_threads_in_selected available thread(s).") + end + Integer(thread_spec) + end + + total_needed = literal_numas * literal_threads + + # Check Julia thread count + if Threads.nthreads() < total_needed + reason *= "Not enough threads on the interpreter." + addLog("machine", "Specified thread arrangement $arrgmt -> literalized to ($literal_numas, $literal_threads) cannot be applied on this Julia process. $reason Ignoring.") + return (0, 0) + elseif Threads.nthreads() > total_needed + reason *= "Too many threads on the interpreter ($(Threads.nthreads()) > $total_needed)." + addLog("machine", "Specified thread arrangement $arrgmt -> literalized to ($literal_numas, $literal_threads) cannot be applied on this Julia process. $reason Ignoring.") + return (0, 0) + elseif reason != "" + addLog("machine", "Specified thread arrangement $arrgmt -> literalized to ($literal_numas, $literal_threads) cannot be applied on this Julia process. $reason Ignoring.") + return (0, 0) + end + return (literal_numas, literal_threads) +end + +# Autovectorized +function validateArrangement(arrgmts::Vector)::Bool + return all(validateArrangement.(arrgmts)) +end + +function validateArrangement(arrgmts::Vector{N})::Bool where {N<:Integer} + return all(0 .< arrgmts) +end + +""" + validateArrangement(arrgmt::ArrangementSpec) -> Bool + +Validates the arrangement syntax. + +Throws an error if the arrangement is badly specified. +""" +function validateArrangement(arrgmt::ArrangementSpec)::Bool + if arrgmt isa Symbol + return if arrgmt == :single + true + elseif arrgmt == :numa + true + elseif arrgmt == :all + true + else + throwParseError!("Invalid thread arrangement: symbol `$arrgmt` is unrecognised. " * + "Must be one of [:single, :numa, :all] or a tuple — see ArrangementSpec docs.") + false + end + end + if length(arrgmt) != 2 + throwParseError!("Invalid thread arrangement: $arrgmt has $(length(arrgmt)) element(s); " * + "expected exactly 2 (#numas, #threads_per_numa).") + return false + end + + numa_spec, thread_spec = arrgmt + if numa_spec isa Rational + if !(0 < numa_spec <= 1) + throwParseError!("Invalid thread arrangement: $arrgmt, NUMA ratio $numa_spec is out of range; " * + "must satisfy 0 < ratio ≤ 1.") + return false + end + else + if numa_spec <= 0 + throwParseError!("Invalid thread arrangement: $arrgmt, if #numa is and integer, it shall be positive, got $numa_spec.") + return false + end + end + if thread_spec isa Rational + if !(0 < thread_spec <= 1) + throwParseError!("Invalid thread arrangement: $arrgmt, thread ratio $thread_spec is out of range; " * + "must satisfy 0 < ratio ≤ 1.") + return false + end + else + if thread_spec <= 0 + throwParseError!("Invalid thread arrangement: $arrgmt, #threads_per_numa must be a positive integer, " * + "got $thread_spec.") + return false + end + end + + return true +end + +function isExecutableArrangement(arrgmts::Vector{N}) :: Union{Vector{N}, Nothing} where {N <: Integer} + return validateArrangement(arrgmts) ? arrgmts : nothing +end + +function isExecutableArrangement(arrgmts::Vector)::Union{ArrangementSpec,Nothing} + for arrgmt in arrgmts + if !((found = isExecutableArrangement(arrgmt)) isa Nothing) + return found + end + end + return nothing +end + +function isExecutableArrangement(arrgmt::ArrangementSpec)::Union{ArrangementSpec,Nothing} + return if arrgmt[1] >= 1 && arrgmt[2] >= 1 + arrgmt + else + nothing + end +end + +""" + enforceThreadArrangement(arrgmt::Tuple{Integer, Integer}) -> Bool + enforceThreadArrangement(arrgmt::Vector{Integer}) -> Bool + +Given a *literal* arrangement `(#numa_domains, #threads_per_numa)`, checks +that the running Julia process has at least `#numa_domains × #threads_per_numa` +threads available. + +In case a vector is passed, the vector is interpreted as a manual arrangement, +with the elements being the cores to allocate. + +- Returns `false` if the interpreter does not have enough threads. +- Returns `true` and pins the Julia threads to the appropriate CPU threads + (in NUMA-domain order, `#threads_per_numa` consecutive cores per domain) + if the arrangement can be satisfied. +""" +function enforceThreadArrangement(arrgmt::Tuple{Integer,Integer})::Bool + n_numas, n_threads = arrgmt + total_needed = n_numas * n_threads + + # Check Julia thread count + if Threads.nthreads() < total_needed + @warn "enforceThreadArrangement: need $total_needed Julia thread(s) " * + "($(n_numas) NUMA × $(n_threads) threads/NUMA) but only " * + "$(Threads.nthreads()) are available." + return false + end + + # For each of the first n_numas NUMA domains take the first n_threads cores. + numa_cores = Topology.getNUMADomainCores() # Vector{Vector{Integer}} + + cpu_ids = Vector{Integer}() + for numa_idx in 1:n_numas + # Sort for determinism + domain_cores = sort(numa_cores[numa_idx]) + append!(cpu_ids, domain_cores[1:n_threads]) + end + + pinthreads(cpu_ids) + addLog("machine", "[MACHINE] Pinning enforced, cpuids pinned = $(cpu_ids)") + return true +end + +function enforceThreadArrangement(manual::Vector{<:Integer})::Bool + if length(manual) > sum(length.(Topology.getNUMADomainCores())) + error("enforceThreadArrangement: more threads have been specified than the ones available in the machine") + end + if length(manual) > Threads.nthreads() + error("enforceThreadArrangement: more threads have been specified than the ones avaiable for the interpreter") + end + pinthreads(manual) +end + +""" + freeThreadArrangement() + +Unpins all Julia threads, removing any CPU affinity constraints previously +set by `enforceThreadArrangement`. +""" +function freeThreadArrangement() + unpinthreads() +end + +end # module end \ No newline at end of file diff --git a/src/execution/macros/customs.jl b/src/execution/macros/customs.jl index cc15b67..21bb531 100644 --- a/src/execution/macros/customs.jl +++ b/src/execution/macros/customs.jl @@ -16,8 +16,9 @@ define_metric_validation = defineMacroParams([ true ) ]) + """ -This macro is used to define a new custom metric. +This macro is used to define a new custom metric. The metric result can be used subsequently by other macros. # Arguments - `name` : the name of the metric for identification purposes. @@ -31,6 +32,7 @@ This macro is used to define a new custom metric. - `:autoflop`: will be substituted by the FLOP count the target. - `:printed_output` : will be substituted by the standard output stream of the target. - `:iterator` : will be substituted by the current iterator value in a loop test set. + - Any symbol representing a custom metric or benchmark (e.g metric with name="example" -> :example) """ macro define_metric(args...) return :( @@ -75,7 +77,7 @@ end # Same parameters auxiliary_metric_validation = define_metric_validation """ - Defines a custom metric for informational purposes that will not be used for testing but will be printed as output. + Defines a custom metric for informational purposes that will not be used for testing but will be printed as output. The metric cannot be used by other macros. # Arguments - `name` : the name of the metric for identification purposes. @@ -89,6 +91,7 @@ auxiliary_metric_validation = define_metric_validation - `:autoflop`: will be substituted by the FLOP count the target. - `:printed_output` : will be substituted by the standard output stream of the target. - `:iterator` : will be substituted by the current iterator value in a loop test set. + - Any symbol representing a custom metric or benchmark (e.g metric with name="example" -> :example) """ macro auxiliary_metric(formula, name, units) return :( @@ -105,8 +108,8 @@ define_eff_memory_throughput_validation = defineMacroParams([ ), MacroParameter( :mem_benchmark, - Symbol, - (x) -> x in [:MEM_STREAM_COPY,:MEM_STREAM_ADD], + metricID(), + (x) -> metricID(x) in [:MEM_STREAM_COPY, :MEM_STREAM_ADD, :MEM_BENCH_STRIAD, :MEM_BENCH_SDAXPY], :MEM_STREAM_COPY, #default ), MacroParameter( @@ -117,13 +120,18 @@ define_eff_memory_throughput_validation = defineMacroParams([ true) ]) """ + +`@define_eff_memory_throughput [kwargs...] begin {formula block} end` + +`@def_mem_thr [kwargs...] begin {formula block} end` + This macro is used to define the memory bandwidth of a target in order to execute the effective memory thorughput methodology. # Arguments - - formula block : an expression that returns a single value, which would be the metric value. The formula can have any julia expression inside and additionally some special symbols are supported. The formula may be evaluated several times, so its applied to every target in every test set or just once, if the formula is defined inside a test set, which makes it only applicable to it. + - formula block : an expression that returns a single value, which would be the metric value. The formula can have any julia expression inside and additionally special symbols listed below are supported. The formula may be evaluated several times, so its applied to every target in every test set or just once, if the formula is defined inside a test set, which makes it only applicable on it and onto children testset. - ratio : the allowed minimum percentage over the maximum attainable that is allowed to pass the test, it can be a number or a Julia expression that evaluates to a number - - mem_benchmark : which STREAM kernel benchmark to use (e.g :MEM_STREAM_COPY for transfer operations :MEM_STREAM_ADD for transfer and computing) - - custom_benchmark : in case of using a custom benchmark, the symbol that identifies the chosen benchmark, (must have been defined before) + - benchmark : which bandwidth kernel benchmark to use (e.g :MEM_STREAM_COPY for transfer operation, :MEM_BENCH_STRIAD (Shoenauer triad) for computing operations), :MEM_BENCH_SDAXPY (Shoenauer Daxpy) useful when not sure if copy on write is happening) + - custom_benchmark : if a previously defined benchmark name is passed, the methodology will use it as the maximum bandwidth reference. # Special symbols: - `:median_time` : will be substituted by the median time the target took to execute in the benchmark. @@ -132,6 +140,7 @@ This macro is used to define the memory bandwidth of a target in order to execut - `:autoflop`: will be substituted by the FLOP count the target. - `:printed_output` : will be substituted by the standard output stream of the target. - `:iterator` : will be substituted by the current iterator value in a loop test set. + - Any symbol representing a custom metric or benchmark (e.g metric with name="example" -> :example) # Example: @@ -147,7 +156,7 @@ macro define_eff_memory_throughput(args...) begin end ) end - +var"@def_eff_mem" = var"@define_eff_memory_throughput" """ diff --git a/src/execution/macros/perfcompare.jl b/src/execution/macros/perfcompare.jl index 9691e49..0695bf0 100644 --- a/src/execution/macros/perfcompare.jl +++ b/src/execution/macros/perfcompare.jl @@ -19,6 +19,7 @@ This macro is used to manually declare performance test conditions. - `:autoflop`: will be substituted by the FLOP count the target. - `:printed_output` : will be substituted by the standard output stream of the target. - `:iterator` : will be substituted by the current iterator value in a loop test set. + - Any symbol representing a custom metric or benchmark (e.g metric with name="example" -> :example) # Example: diff --git a/src/execution/macros/regression.jl b/src/execution/macros/regression.jl index 7ac7d98..bb6cf0a 100644 --- a/src/execution/macros/regression.jl +++ b/src/execution/macros/regression.jl @@ -2,30 +2,35 @@ define_regression_validation = defineMacroParams([ MacroParameter(:low_is_bad, - Bool, + Union{Bool,Vector{Bool}}, true, false), MacroParameter(:threshold, - Float64, - (x) -> 0.0 <= x <= 1.0, + Union{Float64,Vector{Float64}}, + (x) -> x isa Float64 ? 0.0 <= x : all(0.0 <= t for t in x), 0.9, #default false), MacroParameter(:metrics, - Union{Symbol,Vector{Symbol}}, + Union{metricID(),Vector{metricID()}}, always_true, [:median_time], #default false), ]) """ -This macro is used to define the memory bandwidth of a target in order to execute the effective memory thorughput methodology. +This macro is used to define regression tests with a customized configuration. Redefining the macro in the same testset will overwrite the previous configuration. # Arguments - - threshold : the minimum ratio over that is allowed to pass the test (e.g 0.9 means that the test is a success if the new metric is at least 90% of the old) - - enable: track regression in the metrics whose names are passed as argument, it accepts a single string or a vector of strings. Non-existent metrics are ignored. -# Example: + - low_is_bad: whether a lower value of the metric is considered a failure (e.g. for time measurements) or a success (e.g. for throughput measurements). It can be a single boolean that applies to all metrics or a vector of booleans that applies to each metric separately. + - threshold : the minimum ratio over that is allowed to pass the test (e.g 0.9 means that the test is a success if the new metric is at least 90% of the old). It can be a single float that applies to all metrics or a vector of floats that applies to each metric separately. + - metrics: track regression in the metrics whose names are passed as argument, it accepts a single string or a vector of strings. Non-existent metrics are ignored. +# Examples: ```julia - @regression threshold=0.9 enable=:median_time + # Tests will fail if median time is lower than 90% of the reference value + @regression threshold=0.9 low_is_bad=false metrics=:median_time + + # Tests will fail if metric results are higher than 110% of the reference value + @regression threshold=1.1 low_is_bad=true metrics=[:custom_metric1, :custom_metric2] ``` """ macro regression(args...) diff --git a/src/execution/macros/roofline.jl b/src/execution/macros/roofline.jl index 9ae9b4b..15aafc2 100644 --- a/src/execution/macros/roofline.jl +++ b/src/execution/macros/roofline.jl @@ -2,17 +2,17 @@ roofline_validation = defineMacroParams([ MacroParameter( :cpu_peak, - Float64, + Number, greaterThan0, ), MacroParameter( :membw_peak, - Float64, + Number, greaterThan0, ), MacroParameter( :target_opint, - Float64, + Number, greaterThan0, ), MacroParameter( @@ -21,7 +21,7 @@ roofline_validation = defineMacroParams([ ), MacroParameter( :target_ratio, - Float64, + Number, (x) -> 0. < x < 2. ), MacroParameter( @@ -38,9 +38,9 @@ roofline_validation = defineMacroParams([ ), # TODO MacroParameter( :mem_benchmark, - Symbol, - (x) -> x in [:MEM_STREAM_COPY,:MEM_STREAM_ADD], - :MEM_STREAM_COPY, #default + metricID(), + (x) -> metricID(x) in [:MEM_STREAM_COPY,:MEM_STREAM_ADD, :MEM_BENCH_STRIAD, :MEM_BENCH_SDAXPY], + :MEM_BENCH_SDAXPY, #default ), MacroParameter( Symbol(""), @@ -59,6 +59,7 @@ This macro enables roofline modelling, if put just before a target declaration ( - `target_opint` : a desired operational intensity for the target, this will turn operational intensity into a test metric - `actual_flops`: another formula that defines the actual performance of the test. - `target_ratio` : the acceptable ratio between the actual performance and the projected performance from the roofline, this will turn actual performance into a test metric. + - `mem_benchmark`: which benchmark to use for the roofline, the default is the Shoenauer DAXPY (:MEM_BENCH_DAXPY), custom benchmarks are supported. Other benchmarks: [:MEM_STREAM_COPY,:MEM_STREAM_ADD, :MEM_BENCH_STRIAD, :MEM_BENCH_SDAXPY] # Special symbols: - `:median_time` : will be substituted by the median time the target took to execute in the benchmark. diff --git a/src/execution/macros/topology.jl b/src/execution/macros/topology.jl new file mode 100644 index 0000000..c27ae5b --- /dev/null +++ b/src/execution/macros/topology.jl @@ -0,0 +1,30 @@ + + +threads_validation = defineMacroParams([ + MacroParameter(Symbol(""), + ExtendedExpr, + validation_function= x -> Topology.validateArrangement(eval(x)), + mandatory= true) +]) +""" + This macro can be used to specify one or more thread pinning configurations for the performance test suite. Not calling this macro on the recipe leaves the threads unpinned. + + # Arguments + A thread specification or a list of them. A thread specification can be: + (#numas, #threads_per_numa) or a shortcut symbol like :single, :numa or :all + Example: + (1, 1) or :single -> One thread on one numa node + (1, 2) -> Two threads on two numa nodes + (1, 1//1) or :numa -> All threads on one numa node + (1, 1//2) -> Half of the threads available for one numa node + (2, 1) -> One thread on each of two numa nodes + (1//1,1//1) or :all -> All posible threads on all numa nodes + + Alternatively a list of integers can be passed which will be interpreted as a manual pinning. + + [!] As of now, the pinning is not MPI aware, (this will come in PerfTest v1.2.4) in MPI cases use the manual pinning, with the list of integers. + [!] This is a feature in development, more comprehensive features will be built over this later on (1.2.4). +""" +macro perftest_threads(anything) + return :(begin end) +end \ No newline at end of file diff --git a/src/execution/printing.jl b/src/execution/printing.jl index a0c8ed7..c698855 100644 --- a/src/execution/printing.jl +++ b/src/execution/printing.jl @@ -209,8 +209,8 @@ end function printAuxiliaries(metrics :: Dict{Symbol, Metric_Result}, tab :: Int) - println(@lpad(tab) * "Auxiliary results:") - for (_,metric) in metrics + length(metrics) > 0 ? println(@lpad(tab) * "Auxiliary results:") : println() + for (_,metric) in metrics if metric.auxiliary PerfTest.auxiliarMetricPrint(metric, tab) end diff --git a/src/execution/retrieve.jl b/src/execution/retrieve.jl new file mode 100644 index 0000000..40ab6ce --- /dev/null +++ b/src/execution/retrieve.jl @@ -0,0 +1,165 @@ +export retrievePerfTests, inTestSet, hasMetric, hasAuxiliar, hasPrimitive, testNamed, testPassed, testFailed + +""" + retrievePerfTests(datafile_path; get=:tests, where=(x)->true, execution=:latest) + +General-purpose retrieval over a performance-test datafile. + +# Arguments +- `datafile_path::AbstractString`: path to a serialized `Perftest_Datafile_Root`. +- `get::Symbol`: what to return. One of: + - `:tests` → `Vector{Test_Result}` + - `:methodologies` → `Vector{Methodology_Result}` + - `:testsets` → `Vector{String}` (fully-qualified testset paths) +- `where::Function`: predicate applied to each candidate element; + only elements for which it returns `true` are kept. +- `execution::Union{Symbol, Int}`: which execution to consider. +- `pathonly ::Bool` : if true, only return the testset path for each result, instead of the full `Test_Result` or `Methodology_Result`. + +# Examples +```julia +tests = retrievePerfTests("results.dat") +fast = retrievePerfTests("results.dat"; where = t -> hasMetric(t, :median_time)) +passed = retrievePerfTests("results.dat"; where = testPassed) +failed = retrievePerfTests("results.dat"; where = testFailed) +meths = retrievePerfTests("results.dat"; get = :methodologies) + +# All predicates: +- `inTestSet(test, "MySuite/MyTest")` → true if `test` is under a testset named "MySuite/MyTest". +- `hasMetric(test, :median_time)` → true if `test` has a metric +- `hasAuxiliar(test, :memory_usage)` → true if `test` has an auxiliar metric +- `hasPrimitive(test, :min_time)` → true if `test` has a primitive +- `testNamed(test, "MyTest")` → true if `test.name == "MyTest"` +- `testPassed(test)` → true if all methodologies of `test` passed +- `testFailed(test)` → true if any methodology of `test` failed +- `methodologyNamed(m, "MyMethodology")` → true if `m.name == "MyMethodology"` +- `methodologyHasMetric(m, "median_time")` → true if `m` has a metric named "median_time" +- `methodologyPassed(m)` → true if all metrics of `m` succeeded +``` + +""" +function retrievePerfTests(datafile_path::AbstractString; + get::Symbol = :tests, + where_pred::Function = _ -> true, + execution::Union{Symbol, Int} = :latest, + pathonly::Bool = false) + root = openDataFile(datafile_path) + if execution isa Symbol + if execution == :latest + suite = root.results[end] + elseif execution == :all + suite = root.results + return [pathonly ? _retrieve(Val(get), s, where_pred)[2] : _retrieve(Val(get), s, where_pred)[1] for s in suite] + else + error("Invalid execution argument: $execution. Expected :latest, :all, or a index.") + end + else + try + suite = root.results[execution] + catch + throw(ArgumentError("No execution with timestamp $execution found in datafile.")) + end + end + return pathonly ? _retrieve(Val(get), suite, where_pred)[2] : _retrieve(Val(get), suite, where_pred)[1] +end + +# Dispatches +_retrieve(::Val{:tests}, suite, pred) = _collect_tests(suite, pred) +_retrieve(::Val{:methodologies}, suite, pred) = _collect_methodologies(suite, pred) +_retrieve(::Val{:testsets}, suite, pred) = _collect_testsets(suite, pred) +function _retrieve(::Val{S}, _, _) where {S} + throw(ArgumentError("retrievePerfTests: unknown get=:$S. " * "Expected one of :tests, :methodologies, :testsets.")) +end + +""" + getExecutionTimestamps(datafile_path::AbstractString) + + Returns a vector of timestamps corresponding to each execution recorded in the datafile at `datafile_path`. +""" +function getExecutionTimestamps(datafile_path::AbstractString) + root = openDataFile(datafile_path) + return [result.timestamp for result in root.results] +end + + +""" + _walk_perftests(f, perftests, path=String[]) + +Recursively walk a `perftests` dict, calling `f(test, path)` for every +`Test_Result` leaf, where `path` is the vector of testset names leading to it. +""" +function _walk_perftests(f, perftests::AbstractDict, path::Vector{String} = String[]) + for (name, entry) in perftests + if entry isa Test_Result + f(entry, path) + elseif entry isa AbstractDict + _walk_perftests(f, entry, vcat(path, String(name))) + end + end +end + +function _collect_tests(suite::Suite_Execution_Result, pred::Function) + out = Test_Result[] + path = String[] + _walk_perftests(suite.perftests) do test, _path + _apply(pred, test, _path) && (push!(out, test);push!(path, join(_path, " > ") * " > " * test.name)) + end + return out, path +end + +function _collect_methodologies(suite::Suite_Execution_Result, pred::Function) + out = Methodology_Result[] + path = String[] + _walk_perftests(suite.perftests) do test, _path + for m in test.methodology_results + _apply(pred, m, _path) && (push!(out, m);push!(path, join(_path, " > ") * " > " * test.name * " > " * m.name)) + end + end + return out, path +end + +function _collect_testsets(suite::Suite_Execution_Result, pred::Function) + seen = Set{String}() + out = String[] + _walk_perftests(suite.perftests) do _test, path + # Every prefix of the path is a valid testset + for i in 1:length(path) + qualified = join(@view(path[1:i]), "/") + if !(qualified in seen) && pred(qualified) + push!(seen, qualified) + push!(out, qualified) + end + end + end + return out,out +end + + +# PREDICATES + +# Accept 1- or 2-arg predicates transparently +_apply(pred, test, path) = hasmethod(pred, Tuple{Test_Result, Vector{String}}) ? + pred(test, path) : pred(test) +#""" +# inTestSet(test, setname) -> Bool +# +#True if `test` lives under a testset named `setname` (matched against any +#path component, or against the full "a/b/c" path). +#""" +#function inTestSet(test::Test_Result, setname::AbstractString) + # TODO +# error("inTestSet requires path info — use the path-aware variant below.") +#end + +# Test predicates that don't need path info: +hasMetric(t::Test_Result, key::Symbol) = haskey(t.metrics, key) +hasAuxiliar(t::Test_Result, key::Symbol) = haskey(t.auxiliar, key) +hasPrimitive(t::Test_Result, key::Symbol) = haskey(t.primitives, key) +testNamed(t::Test_Result, name::AbstractString) = t.name == name +testPassed(t::Test_Result) = all([methodologyPassed(m) for m in t.methodology_results]) +testFailed(t::Test_Result) = !testPassed(t) + +# Methodology predicates: +methodologyNamed(m::Methodology_Result, name::AbstractString) = m.name == name +methodologyHasMetric(m::Methodology_Result, mname::AbstractString) = any(p -> first(p).name == mname, m.metrics) +methodologyPassed(m::Methodology_Result) = all(p -> last(p).succeeded, m.metrics) \ No newline at end of file diff --git a/src/execution/testset.jl b/src/execution/testset.jl index f7db395..7fbfb76 100644 --- a/src/execution/testset.jl +++ b/src/execution/testset.jl @@ -1,5 +1,4 @@ -# Create a custom Julia testset type (from jl) template using Test using Test: AbstractTestSet, Broken, Error, Fail, Pass, record, finish, get_testset_depth, get_testset, print_test_results, get_test_counts, Result @@ -11,7 +10,7 @@ mutable struct PerfTestSet <: AbstractTestSet ## Extra fields needed for PerfTest suites # autonumeric counter to identify tests - test_count :: Int + test_count::Int # Holds the current iteration iterator::Any @@ -19,10 +18,10 @@ mutable struct PerfTestSet <: AbstractTestSet # Holds BenchmarkTools.jl benchmarks benchmarks::BenchmarkGroup # The following includes test results and snapshots of the metrics at test time - test_results::Dict{String, Test_Result} + test_results::Dict{String,Test_Result} # For regression - old_test_results::Union{Nothing, Dict{String,Union{Dict,Test_Result}}} + old_test_results::Union{Nothing,Dict{String,Union{Dict,Test_Result}}} # To print the whole levels parent_chain::Vector{AbstractString} @@ -35,18 +34,19 @@ mutable struct PerfTestSet <: AbstractTestSet else parents = AbstractString[] end - return new(desc, [], 0, 0, nothing, BenchmarkGroup(), Dict{String, Test_Result}(), nothing, parents) + return new(desc, [], 0, 0, nothing, BenchmarkGroup(), Dict{String,Test_Result}(), nothing, parents) end end -function newBenchmarkGroup() :: BenchmarkGroup +function newBenchmarkGroup()::BenchmarkGroup return BenchmarkGroup() end +mpi_and_not_main() :: Bool = !main_rank(Main.__PERFTEST__._GLOBAL_MODE) # Save test results or child test sets -function Test.record(ts::PerfTestSet, t::Result; extra_data = nothing) - if !main_rank(mode) +function Test.record(ts::PerfTestSet, t::Result; extra_data=nothing) + if mpi_and_not_main() return ts end @@ -60,96 +60,188 @@ function Test.record(ts::PerfTestSet, child::AbstractTestSet) return push!(ts.results, child) end -function Test.finish(ts::PerfTestSet) - if !main_rank(mode) - return ts +if isdefined(Test, :TestCounts) + # Julia 1.11+ + # Recursive function that counts the number of test results of each + # type directly in the testset, and totals across the child testsets + function Test.get_test_counts(ts::PerfTestSet) + passes, fails, errors, broken = 0, 0, 0, 0 + c_passes, c_fails, c_errors, c_broken = 0, 0, 0, 0 + for t in ts.results + isa(t, Pass) && (passes += 1) + isa(t, Fail) && (fails += 1) + isa(t, Error) && (errors += 1) + isa(t, Broken) && (broken += 1) + if isa(t, AbstractTestSet) + tc = Test.get_test_counts(t) + c_passes += tc.passes + tc.cumulative_passes + c_fails += tc.fails + tc.cumulative_fails + c_errors += tc.errors + tc.cumulative_errors + c_broken += tc.broken + tc.cumulative_broken + end + end + # We dont use this but we leave the field for compatibility with other types of TestSet + duration = "" + return Test.TestCounts(true, passes, fails, errors, broken, + c_passes, c_fails, c_errors, c_broken, duration) end - # Print failed tests on the current level (or everything if verbose) - for (test_name,test_result) in ts.test_results - if sum(get_test_counts(ts)[2:3]) > 0 || Configuration.CONFIG["general"]["verbose"] - for parent in ts.parent_chain - print("IN: $(parent) ") - end - print( - "\n" - ) - print(" " ^ get_testset_depth() * "AT: $(ts.description)\n") - for methodology in test_result.methodology_results - printMethodology(methodology, get_testset_depth(), Configuration.CONFIG["general"]["plotting"]) + + function Test.finish(ts::PerfTestSet) + if mpi_and_not_main() + return ts + end + + # Print failed tests on the current level (or everything if verbose) + for (test_name, test_result) in ts.test_results + if let c = Test.get_test_counts(ts) + c.fails + c.errors + end > 0 || Configuration.CONFIG["general"]["verbose"] > 0 + for parent in ts.parent_chain + print("IN: $(parent) ") + end + print( + "\n" + ) + print(" "^get_testset_depth() * "AT: $(ts.description)\n") + for methodology in test_result.methodology_results + printMethodology(methodology, get_testset_depth(), Configuration.CONFIG["general"]["plotting"]) + end + + # Print auxiliary metrics + printAuxiliaries(test_result.auxiliar, get_testset_depth()) + else end + end - # Print auxiliary metrics - printAuxiliaries(test_result.auxiliar, get_testset_depth()) + if get_testset_depth() > 0 + record(get_testset(), ts) + return ts else + # show the results are printed if we are at the top level + print_test_results(ts) end + return ts end - if get_testset_depth() > 0 - record(get_testset(), ts) - return ts - else - # show the results are printed if we are at the top level - print_test_results(ts) + """ + Will print the overall result of the test suite execution + """ + function Test.print_test_results(ts::PerfTestSet) + if mpi_and_not_main() + return ts + end + + testcounts = Test.get_test_counts(ts) + passes, fails, errors, _, cp, cf, ce, _ = testcounts.passes, testcounts.fails, testcounts.errors, testcounts.broken, testcounts.cumulative_passes, testcounts.cumulative_fails, testcounts.cumulative_errors, testcounts.cumulative_broken + print("Aggregate Results: $(passes + cp) PASSED, $(fails + cf) FAILED, $(errors+ce) ERRORS\n") + err = get_test_errors(ts) + for _error in err + println("ERROR:") + println(_error.value) + print(_error.backtrace) + end end - return ts -end +else + # Julia 1.10 and below + function Test.finish(ts::PerfTestSet) + if mpi_and_not_main() + return ts + end -# Recursive function that counts the number of test results of each -# type directly in the testset, and totals across the child testsets -function Test.get_test_counts(ts::PerfTestSet) - passes, fails, errors, broken = 0, 0, 0, 0 - c_passes, c_fails, c_errors, c_broken = 0, 0, 0, 0 - for t in ts.results - isa(t, Pass) && (passes += 1) - isa(t, Fail) && (fails += 1) - isa(t, Error) && (errors += 1) - isa(t, Broken) && (broken += 1) - if isa(t, AbstractTestSet) - np, nf, ne, nb, ncp, ncf, nce, ncb, duration = get_test_counts(t) - c_passes += np + ncp - c_fails += nf + ncf - c_errors += ne + nce - c_broken += nb + ncb + # Print failed tests on the current level (or everything if verbose) + for (test_name, test_result) in ts.test_results + if sum(Test.get_test_counts(ts)[2:3]) > 0 || Configuration.CONFIG["general"]["verbose"] > 0 + for parent in ts.parent_chain + print("IN: $(parent) ") + end + print( + "\n" + ) + print(" "^get_testset_depth() * "AT: $(ts.description)\n") + for methodology in test_result.methodology_results + printMethodology(methodology, get_testset_depth(), Configuration.CONFIG["general"]["plotting"]) + end + + # Print auxiliary metrics + printAuxiliaries(test_result.auxiliar, get_testset_depth()) + else + end + end + + if get_testset_depth() > 0 + record(get_testset(), ts) + return ts + else + # show the results are printed if we are at the top level + print_test_results(ts) end + return ts end - # We dont use this but we leave the field for compatibility with other types of TestSet - duration = "" - return passes, fails, errors, broken, c_passes, c_fails, c_errors, c_broken, duration -end -function Test.get_test_counts(tss::Vector{Any}) - passes, fails, errors, broken = 0, 0, 0, 0 - c_passes, c_fails, c_errors, c_broken = 0, 0, 0, 0 - - for ts in tss + # Recursive function that counts the number of test results of each + # type directly in the testset, and totals across the child testsets + function Test.get_test_counts(ts::PerfTestSet) + passes, fails, errors, broken = 0, 0, 0, 0 + c_passes, c_fails, c_errors, c_broken = 0, 0, 0, 0 for t in ts.results - isa(t, Pass) && (passes += 1) - isa(t, Fail) && (fails += 1) - isa(t, Error) && (errors += 1) + isa(t, Pass) && (passes += 1) + isa(t, Fail) && (fails += 1) + isa(t, Error) && (errors += 1) isa(t, Broken) && (broken += 1) if isa(t, AbstractTestSet) - np, nf, ne, nb, ncp, ncf, nce, ncb, duration = get_test_counts(t) + np, nf, ne, nb, ncp, ncf, nce, ncb, duration = Test.get_test_counts(t) c_passes += np + ncp c_fails += nf + ncf c_errors += ne + nce c_broken += nb + ncb end end + # We dont use this but we leave the field for compatibility with other types of TestSet + duration = "" + return passes, fails, errors, broken, c_passes, c_fails, c_errors, c_broken, duration + end + + + """ + Will print the overall result of the test suite execution + """ + function Test.print_test_results(ts::PerfTestSet) + if mpi_and_not_main() + return ts + end + + passes, fails, errors, _, cp, cf, ce, _ = Test.get_test_counts(ts) + print("Aggregate Results: $(passes + cp) PASSED, $(fails + cf) FAILED, $(errors+ce) ERRORS\n") + err = get_test_errors(ts) + for _error in err + println("ERROR:") + println(_error.value) + print(_error.backtrace) + end end +end - # We dont use this but we leave the field for compatibility with other types of TestSet - duration = "" - return passes, fails, errors, broken, c_passes, c_fails, c_errors, c_broken, duration +function testsSucceeded(ts::PerfTestSet) + for t in ts.results + if isa(t, Fail) || isa(t, Error) + return false + elseif isa(t, AbstractTestSet) + if !testsSucceeded(t) + return false + end + end + end + return true end function get_test_errors(ts::PerfTestSet) errors = Error[] for t in ts.results - isa(t, Error) && (push!(errors, t)) + isa(t, Error) && (push!(errors, t)) if isa(t, AbstractTestSet) append!(errors, get_test_errors(t)) end @@ -157,26 +249,9 @@ function get_test_errors(ts::PerfTestSet) return errors end -""" - Will print the overall result of the test suite execution -""" -function Test.print_test_results(ts::PerfTestSet) - if !main_rank(mode) - return ts - end - - passes, fails, errors, _, cp,cf,ce,_ = get_test_counts(ts) - print("Aggregate Results: $(passes + cp) PASSED, $(fails + cf) FAILED, $(errors+ce) ERRORS\n") - err = get_test_errors(ts) - for _error in err - println("ERROR:") - println(_error.value) - print(_error.backtrace) - end -end -function buildPrimitiveMetrics!(::Type{NormalMode}, ts::PerfTestSet, test_result :: Test_Result) +function buildPrimitiveMetrics!(::Type{NormalMode}, ts::PerfTestSet, test_result::Test_Result) # Get testset test_result.primitives[:median_time] = median(ts.benchmarks[test_result.name]).time / 1e9 test_result.primitives[:min_time] = minimum(ts.benchmarks[test_result.name]).time / 1e9 @@ -185,19 +260,19 @@ end var"@perftestset" = Test.var"@testset" -function extractTestResults(tss :: Vector{Any}) :: Dict{String, Union{Dict, Test_Result}} -# In case a testset with for is present - dict = Dict{String,Union{Dict,Test_Result}}() +function extractTestResults(tss::Vector{Any})::Dict{String,Union{Dict,Test_Result}} + # In case a testset with for is present + dict = Dict{String,Union{Dict,Test_Result}}() for ts in tss - dict[ts.description * "_" * string(ts.iterator)] = extractTestResults(ts) + dict[ts.description*"_"*string(ts.iterator)] = extractTestResults(ts) end return dict end -function extractTestResults(ts :: PerfTestSet) :: Dict{String,Union{Dict,Test_Result}} - dict = Dict{String,Union{Dict,Test_Result}}() +function extractTestResults(ts::PerfTestSet)::Dict{String,Union{Dict,Test_Result}} + dict = Dict{String,Union{Dict,Test_Result}}() for t in ts.results if isa(t, Test.Result) @@ -205,7 +280,7 @@ function extractTestResults(ts :: PerfTestSet) :: Dict{String,Union{Dict,Test_Re dict[t.description] = extractTestResults(t) end end - for (k,v) in ts.test_results + for (k, v) in ts.test_results dict[k] = v end @@ -216,8 +291,8 @@ end """ TODO """ -function saveMethodologyData(testname :: AbstractString, data :: Methodology_Result) - ts = Test.get_testset() - res :: Test_Result = ts.test_results[testname] +function saveMethodologyData(testname::AbstractString, data::Methodology_Result) + ts = Test.get_testset() + res::Test_Result = ts.test_results[testname] push!(res.methodology_results, data) end diff --git a/src/execution/units.jl b/src/execution/units.jl index 326e6d1..e1a3adf 100644 --- a/src/execution/units.jl +++ b/src/execution/units.jl @@ -56,4 +56,17 @@ function magnitudeAdjust(m::Metric_Result)::Metric_Result end end - +function absoluteMagnitude(m::Metric_Result)::Metric_Result + if !(m.value isa Number) + return m + end + return newMetricResult( + mode, + name = m.name, + units= m.units, + value = m.value * m.magnitude_mult, + auxiliary = m.auxiliary, + magnitude_prefix="", + magnitude_mult=1 + ) +end diff --git a/src/logs.jl b/src/logs.jl index 3c6049a..097daca 100644 --- a/src/logs.jl +++ b/src/logs.jl @@ -57,9 +57,9 @@ function addLog(channel::AbstractString, message::AbstractString, configuration if !(configuration["general"]["logs_enabled"]) return end - if !(configuration["general"]["verbose"]) && !isempty(LOGS.stdout_bindings) + if !(configuration["general"]["verbose"] > 0) && !isempty(LOGS.stdout_bindings) empty!(LOGS.stdout_bindings) - elseif configuration["general"]["verbose"] && isempty(LOGS.stdout_bindings) + elseif configuration["general"]["verbose"] > 0 && isempty(LOGS.stdout_bindings) verboseOutput() end diff --git a/src/transform/configuration.jl b/src/transform/configuration.jl index cd86ee8..a62195a 100644 --- a/src/transform/configuration.jl +++ b/src/transform/configuration.jl @@ -1,6 +1,5 @@ module Configuration - using Base: DEFAULT_STABLE using TOML """ @@ -44,11 +43,11 @@ end """ -Recursively merge two dictionaries, with values from override_dict taking precedence. +Recursively merge two configs, with values from override_config taking precedence. """ function merge_configs(base_config::Dict, override_config::Dict) - function _mc(bc:: Dict, oc :: Dict) :: Dict + function _mc(bc::Dict, oc::Dict)::Dict local merged = deepcopy(bc) for (key, value) in oc @@ -107,14 +106,13 @@ Load configuration from a TOML file. Args: - filepath: Path to the TOML file -- schema: Optional schema for validation Returns: - Loaded configuration dictionary or nothing """ -function load_config() :: Union{Dict, Nothing} +function load_config(filepath::AbstractString = "")::Union{Dict,Nothing} try - config = TOML.parsefile(CONFIG_FILE) + config = TOML.parsefile(filepath == "" ? CONFIG_FILE : filepath) if !validate_config(config, CONFIG_SHAPE) @error "Configuration validation failed" @@ -125,34 +123,48 @@ function load_config() :: Union{Dict, Nothing} return CONFIG catch e - @info "Configuration not found, loading default configuration" + @info "Configuration not found or invalid, loading default configuration" save_config(DEFAULT) return DEFAULT end end -function load_dummy_config() :: Union{Dict, Nothing} +function load_config(config :: Dict)::Union{Dict,Nothing} + if !validate_config(config, CONFIG_SHAPE) + @error "Configuration validation failed" + return nothing + end + + global CONFIG = config + + return CONFIG +end + +function load_dummy_config()::Union{Dict,Nothing} global CONFIG = PRECOMPILATION_CONFIG return PRECOMPILATION_CONFIG end CONFIG_FILE = "perftest_config.toml" + CONFIG_SHAPE = Dict( "general" => Dict( "autoflops" => Bool, + "numas" => Union{String,Integer,Float64}, + "threads_per_numa" => Union{String,Integer,Float64}, "save_results" => Bool, "logs_enabled" => Bool, "save_folder" => String, "max_saved_results" => Int, "plotting" => Bool, - "verbose" => Bool, + "verbose" => Int, "recursive" => Bool, "safe_formulas" => Bool, "suppress_output" => Bool), "regression" => Dict( "enabled" => Bool, - "custom_file" => String, + "dedicated_reference_file" => String, "default_threshold" => Number, "use_bencher" => Bool, ), @@ -185,20 +197,22 @@ CONFIG_SHAPE = Dict( DEFAULT = Dict( "general" => Dict( - "autoflops" => true, + "autoflops" => false, + "numas" => "single", + "threads_per_numa" => "single", "save_results" => true, "logs_enabled" => true, "save_folder" => ".perftests", "max_saved_results" => 20, "plotting" => true, - "verbose" => false, + "verbose" => 0, "recursive" => true, "suppress_output" => true, "safe_formulas" => false, ), "regression" => Dict( "enabled" => true, - "custom_file" => "", + "dedicated_reference_file" => "", "default_threshold" => 1.1, "use_bencher" => false, ), @@ -232,6 +246,8 @@ DEFAULT = Dict( PRECOMPILATION_CONFIG = Dict( "general" => Dict( "autoflops" => false, + "numas" => 1, + "threads_per_numa" => 1, "save_results" => false, "logs_enabled" => false, "save_folder" => ".thisshouldnotexist", @@ -244,7 +260,7 @@ PRECOMPILATION_CONFIG = Dict( ), "regression" => Dict( "enabled" => false, - "custom_file" => "", + "dedicated_reference_file" => "", "default_threshold" => 0.9, "use_bencher" => false, ), @@ -283,7 +299,7 @@ end # module using TOML -function parseConfigurationMacro(_ :: ExtendedExpr, ctx :: Context, info :: Dict) :: Expr +function parseConfigurationMacro(_::ExtendedExpr, ctx::Context, info::Dict)::Expr # Parse, and merge string = info[Symbol("")] @@ -295,11 +311,11 @@ function parseConfigurationMacro(_ :: ExtendedExpr, ctx :: Context, info :: Dict Configuration.CONFIG = Configuration.merge_configs(Configuration.CONFIG, parsed) if Configuration.CONFIG["general"]["verbose"] != old["general"]["verbose"] - addLog("general", "Verbosity level changed from $(old["general"]["verbose"] ? "HIGH" : "LOW" ) to $(old["general"]["verbose"] ? "LOW" : "HIGH")") + addLog("general", "Verbosity level changed from $(old["general"]["verbose"]) to $(Configuration.CONFIG["general"]["verbose"])") end io = IOBuffer() - TOML.print(nothing,io, Configuration.CONFIG) + TOML.print(nothing, io, Configuration.CONFIG) serialized_config = String(take!(io)) return quote @@ -309,11 +325,43 @@ function parseConfigurationMacro(_ :: ExtendedExpr, ctx :: Context, info :: Dict end end +function parseThreadsMacro(_::ExtendedExpr, ctx::Context, info::Dict)::Expr + try + specs = info[Symbol("")] + + return quote + let + specs = $specs + arrgmt_l = PerfTest.Topology.literallizeArrangement($specs) + arrgmt_e = PerfTest.Topology.isExecutableArrangement(arrgmt_l) + if arrgmt_e === nothing + @warn "Ignoring pinning." + else + try + PerfTest.Topology.enforceThreadArrangement(arrgmt_e) + catch e + if Sys.iswindows() || Sys.isapple() + @warn "Thread pinning is not supported on this OS. Ignoring specified arrangement." + else + @warn "Failed to enforce thread pinning: $e, ignoring specified arrangement." + end + end + end + end + end + + catch + return quote + "INVALID THREAD ARRANGEMENT" + end + end +end + """ Used on a generated test suite to import the configuration set during generation """ -function _perftest_config(config_string :: String) +function _perftest_config(config_string::String) # Parse, and merge parsed = TOML.parse(config_string) diff --git a/src/transform/datastruct.jl b/src/transform/datastruct.jl index 0649fa7..0695437 100644 --- a/src/transform/datastruct.jl +++ b/src/transform/datastruct.jl @@ -15,6 +15,9 @@ struct ASTRule ASTRule(match, macro_params :: Dict, transformation) = new(match, validateMacro(macro_params), transformation) end +metricID() = Union{Symbol, String} +metricID(id :: metricID()) = id isa String ? Symbol(id) : id + struct MacroParameter name :: Symbol type :: Type @@ -138,11 +141,13 @@ mutable struct GlobalContext errors::ErrorCollection valid_symbols::Set{Symbol} custom_benchmarks::Set{Symbol} + # This set is used to track which benchmarks are being used in the recipe, so the machine benchmarking can be done only if needed + uses_benchmarks::Set{Symbol} # Measure suite time elapsed :: Float64 - GlobalContext(path, errors, valid) = new(path, false, errors, valid, Set{Symbol}()) - GlobalContext(path, errors, valid, _) = new(path, true, errors, valid, Set{Symbol}()) + GlobalContext(path, errors, valid) = new(path, false, errors, valid, Set{Symbol}(), Set{Symbol}(), 0.0) + GlobalContext(path, errors, valid, _) = new(path, true, errors, valid, Set{Symbol}(), Set{Symbol}(), 0.0) end """ diff --git a/src/transform/dummy.jl b/src/transform/dummy.jl index b462267..f323b85 100644 --- a/src/transform/dummy.jl +++ b/src/transform/dummy.jl @@ -1,4 +1,4 @@ -# This file is use as a simple way to precomile the transformation function +# This file is use as a simple way to precompile the transformation function # Their contents have no meaning neither affect the other files using Test diff --git a/src/transform/methodologies/manual.jl b/src/transform/methodologies/manual.jl index 0c03e07..5326caa 100644 --- a/src/transform/methodologies/manual.jl +++ b/src/transform/methodologies/manual.jl @@ -15,6 +15,8 @@ function onPerfcmpDefinition(expr :: ExtendedExpr, ctx :: Context, info) override = false, params = params, )) + + addLog("metrics", "[METHODOLOGY] Defined Performance Assertion on $([i.set_name for i in ctx._local.depth_record]) with expression $(info[Symbol("")])") end diff --git a/src/transform/methodologies/mem_bandwidth.jl b/src/transform/methodologies/mem_bandwidth.jl index c946d99..0d081a2 100644 --- a/src/transform/methodologies/mem_bandwidth.jl +++ b/src/transform/methodologies/mem_bandwidth.jl @@ -15,16 +15,19 @@ function onMemoryThroughputDefinition(formula::ExtendedExpr, ctx::Context, info) else info[:ratio] = Configuration.CONFIG["memory_bandwidth"]["default_threshold"] end + info[:benchmark] = metricID(info[:mem_benchmark]) # Check if the user wants to use a custom benchmark if haskey(info, :custom_benchmark) info[:custom] = true # Check TODO if custom b. has been defined - if !( info[:custom_benchmark] in ctx._global.custom_benchmarks) + if !(info[:custom_benchmark] in ctx._global.custom_benchmarks) throwParseError!("Undefined custom benchmark $(info[:custom_benchmark])", ctx) end + info[:benchmark] = info[:custom_benchmark] else info[:custom] = false end + push!(ctx._global.uses_benchmarks, info[:benchmark]) # Adds to the inner scope the request to build the methodology and its needed metric push!(ctx._local.custom_metrics[end], CustomMetric( name="Effective memory throughput", @@ -38,6 +41,8 @@ function onMemoryThroughputDefinition(formula::ExtendedExpr, ctx::Context, info) override=true, params=info, )) + + addLog("metrics", "[METHODOLOGY] Defined Effective Memory Throughput on $([i.set_name for i in ctx._local.depth_record]) with benchmark $(info[:benchmark]) and threshold $(info[:ratio])") return quote end end @@ -45,7 +50,7 @@ end """ Returns an expression used to evaluate the effective memory throughput over a test target """ -function buildMemTRPTMethodology(context :: Context)::Expr +function buildMemTRPTMethodology(context::Context)::Expr info = captureMethodologyInfo(:effMemTP, context._local.enabled_methodologies) @@ -54,21 +59,19 @@ function buildMemTRPTMethodology(context :: Context)::Expr else info = info.params return quote - let - reference_benchmark = $(!info[:custom] - ? :(_PRFT_GLOBALS.builtins[$(QuoteNode(info[:mem_benchmark]))]) - : :(_PRFT_GLOBALS.custom_benchmarks[$(QuoteNode(info[:custom_benchmark]))].value)) + let + reference_benchmark = $(SBMID(info[:benchmark])) value = test_res.metrics[:effMemTP].value / reference_benchmark success = value >= $(info[:ratio]) result = newMetricResult($mode, - name="Effective Throughput Ratio", - units="%", - value=value*100) + name="Effective Throughput Ratio", + units="%", + value=value * 100) test = Metric_Test( reference=100, - threshold_min_percent=$(info[:ratio])*100, - threshold_max_percent=1.0*100, + threshold_min_percent=$(info[:ratio]) * 100, + threshold_max_percent=1.0 * 100, low_is_bad=true, succeeded=success, custom_plotting=Symbol[], @@ -80,7 +83,7 @@ function buildMemTRPTMethodology(context :: Context)::Expr $mode, name="Attained Bandwidth", units="B/s", - value=test_res.metrics[:effMemTP].value + value=test_res.metrics[:effMemTP].value ) aux_ref_value = newMetricResult( $mode, diff --git a/src/transform/methodologies/regression.jl b/src/transform/methodologies/regression.jl index 9c68923..1e5eff1 100644 --- a/src/transform/methodologies/regression.jl +++ b/src/transform/methodologies/regression.jl @@ -2,7 +2,7 @@ """ Called when a regression macro is detected, sets up the regression methodology - + """ function onRegressionDefinition(_::ExtendedExpr, ctx::Context, info) if !(Configuration.CONFIG["regression"]["enabled"]) @@ -21,24 +21,48 @@ function onRegressionDefinition(_::ExtendedExpr, ctx::Context, info) else info[:low_is_bad] = false end + # vvvv Verify that parameters make sense among each other vvvv + if info[:metrics] isa metricID() + info[:metrics] = [metricID(info[:metrics])] + info[:low_is_bad] = info[:low_is_bad] isa Vector ? throwParseError!("Several low_is_bad values provided for low_is_bad for a single metric", ctx) : [info[:low_is_bad]] + info[:threshold] = info[:threshold] isa Vector ? throwParseError!("Several threshold values provided for threshold for a single metric", ctx) : [info[:threshold]] + else + if info[:low_is_bad] isa Bool + info[:low_is_bad] = fill(info[:low_is_bad], length(info[:metrics])) + elseif length(info[:low_is_bad]) != length(info[:metrics]) + throwParseError!("Length of low_is_bad should be either 1 to apply to all, or the same as the length of metrics", ctx) + end + if info[:threshold] isa Float64 + info[:threshold] = fill(info[:threshold], length(info[:metrics])) + elseif length(info[:threshold]) != length(info[:metrics]) + throwParseError!("Length of threshold should be either 1 to apply to all, or the same as the length of metrics", ctx) + end + info[:metrics] = [metricID(m) for m in info[:metrics]] + end + # ^^^^ Verify that parameters make sense among each other ^^^^ + for metric in info[:metrics] + if !(metric in formula_symbols) && !isCustomMetricDefined(ctx, metric) + throwParseError!("Metric $metric is not available in the current context.", ctx) + end + end push!(ctx._local.enabled_methodologies[end], MethodologyParameters( id=:regression, name="Metric regression tracking", - override=true, + override=false, params=info, )) - addLog("metrics", "[METHODOLOGY] Defined REGRESSION on $([i.set_name for i in ctx._local.depth_record]) TH: $(info[:threshold]) LOW_IS_BAD: $(info[:low_is_bad])") + addLog("metrics", "[METHODOLOGY] Defined REGRESSION on $([i.set_name for i in ctx._local.depth_record]) METRICS: $(info[:metrics]) TH: $(info[:threshold]) LOW_IS_BAD: $(info[:low_is_bad])") end """ Executes regression for one single metric """ -function regression(metric :: Symbol, info) +function regression(metric :: Symbol, threshold::Float64, low_is_bad::Bool) metric = QuoteNode(metric) - success_expr = :($(info[:low_is_bad] ? :(>) : :(<))(ratio ,$(info[:threshold]))) + success_expr = :($(low_is_bad ? :(>) : :(<))(ratio ,$(threshold))) return quote if haskey(test_res.metrics, $metric) && !(old_test_res isa Nothing) && haskey(old_test_res.metrics, $metric) ratio = test_res.metrics[$metric].value / old_test_res.metrics[$metric].value @@ -51,9 +75,9 @@ function regression(metric :: Symbol, info) ) test = Metric_Test( reference=100.0, - threshold_min_percent=$(info[:threshold]) * 100, + threshold_min_percent=$(threshold) * 100, threshold_max_percent=nothing, - low_is_bad=$(info[:low_is_bad]), + low_is_bad=$(low_is_bad), succeeded=success, custom_plotting=Symbol[], full_print=true @@ -81,9 +105,9 @@ function regression(metric :: Symbol, info) ) test = Metric_Test( reference=100.0, - threshold_min_percent=$(info[:threshold]) * 100, + threshold_min_percent=$(threshold) * 100, threshold_max_percent=nothing, - low_is_bad=$(info[:low_is_bad]), + low_is_bad=$(low_is_bad), succeeded=success, custom_plotting=Symbol[], full_print=true @@ -101,7 +125,20 @@ function regression(metric :: Symbol, info) all_succeeded &= success end - + if Configuration.CONFIG["general"]["verbose"] >= 2 && !(old_test_res isa Nothing) + methodology_res.custom_elements[:reference] = newMetricResult( + $mode, + name=$("$metric Reference value"), + units=haskey(old_test_res.metrics, $metric) ? old_test_res.metrics[$metric].units : "s", + value=haskey(old_test_res.metrics, $metric) ? old_test_res.metrics[$metric].value : (haskey(old_test_res.primitives, $metric) ? old_test_res.primitives[$metric] : NaN) + ) + end + methodology_res.custom_elements[$metric] = (newMetricResult( + $mode, + name=$("$metric"), + units=haskey(test_res.metrics, $metric) ? test_res.metrics[$metric].units : "s", + value=haskey(test_res.metrics, $metric) ? test_res.metrics[$metric].value : (haskey(test_res.primitives, $metric) ? test_res.primitives[$metric] : NaN) + )) end end @@ -125,11 +162,16 @@ function buildRegression(context::Context)::Expr all_succeeded = true $( if info[:metrics] isa Symbol - regression(info[:metrics], info) + regression(info[:metrics], info[:threshold], info[:low_is_bad]) else - for i in info[:metrics] - regression(i, info) - end + q = quote end; + for (i, metric) in enumerate(info[:metrics]) + q = quote + $q; + $(regression(metric, info[:threshold][i], info[:low_is_bad][i])) + end + end; + q end ) diff --git a/src/transform/methodologies/roofline.jl b/src/transform/methodologies/roofline.jl index 82a0f64..dd11d9e 100644 --- a/src/transform/methodologies/roofline.jl +++ b/src/transform/methodologies/roofline.jl @@ -3,7 +3,7 @@ using MacroTools: blockunify using Configurations: ExproniconLite using UnicodePlots -function rooflineCalc(peakCPU :: Float64, peakMem :: Float64) +function rooflineCalc(peakCPU :: Number, peakMem :: Number) return (opint) -> min(peakCPU, opint * peakMem) end @@ -42,6 +42,7 @@ function onRooflineDefinition(formula :: ExtendedExpr, ctx :: Context, info) end info[:test_flop] = false end + info[:mem_benchmark] = metricID(info[:mem_benchmark]) #Is there a defined ratio as the test threshold? if haskey(info, :target_ratio) else @@ -80,8 +81,12 @@ function buildRoofline(context::Context)::Expr let opint = test_res.metrics[:opInt].value flop_s = test_res.metrics[:attainedFLOPS].value + flop_peak = $(haskey(info, :cpu_peak) ? info[:cpu_peak] : quote _PRFT_GLOBALS.builtins[:CPU_FLOPS_PEAK] end) + mem_peak = $(haskey(info, :membw_peak) ? info[:membw_peak] : quote _PRFT_GLOBALS.builtins[$(QuoteNode(info[:mem_benchmark]))] end) - roof = PerfTest.rooflineCalc(_PRFT_GLOBALS.builtins[:CPU_FLOPS_PEAK], _PRFT_GLOBALS.builtins[:MEM_STREAM_COPY]) + $(context._global.uses_benchmarks = union(context._global.uses_benchmarks, [:CPU_FLOPS_PEAK, :MEM_STREAM_DAXPY]); nothing) + # Check if the user wants to use a hardcoded memory bandwidth or cpu peak + roof = PerfTest.rooflineCalc(flop_peak, mem_peak) result_flop_ratio = newMetricResult( $mode, @@ -93,7 +98,6 @@ function buildRoofline(context::Context)::Expr methodology_res = Methodology_Result( name="Roofline Model" ) - $(if info[:test_flop] quote success_flop = result_flop_ratio.value >= $(info[:target_ratio]) * 100 @@ -122,13 +126,13 @@ function buildRoofline(context::Context)::Expr $mode, name="Peak empirical bandwidth", units="B/s", - value=_PRFT_GLOBALS.builtins[$(QuoteNode(info[:mem_benchmark]))] + value=mem_peak ) aux_flops = newMetricResult( $mode, name="Peak empirical flops", units="FLOP/s", - value=_PRFT_GLOBALS.builtins[:CPU_FLOPS_PEAK] + value=flop_peak ) aux_rcorner = newMetricResult( $mode, @@ -145,16 +149,16 @@ function buildRoofline(context::Context)::Expr methodology_res.custom_elements[:plot] = PerfTest.printFullRoofline - # Printing - #if $(Configuration.CONFIG["general"]["verbose"]) || !(flop_test.succeeded) - # PerfTest.printMethodology(methodology_res, $(length(context._local.depth_record)), $(Configuration.CONFIG["general"]["plotting"])) - #end # Testing try PerfTest.@_prftest flop_test.succeeded saveMethodologyData(test_res.name, methodology_res) - catch end + catch e + @error "Roofline test failed with error: $e" + + + end end end end diff --git a/src/transform/metrics/custom.jl b/src/transform/metrics/custom.jl index 1937065..26030f5 100644 --- a/src/transform/metrics/custom.jl +++ b/src/transform/metrics/custom.jl @@ -80,3 +80,12 @@ function defineCustomBenchmark(ctx::Context, info) :: Expr end end + +function isCustomMetricDefined(ctx::Context, name :: Symbol) + for custom_metric in Iterators.flatten(ctx._local.custom_metrics) + if custom_metric.symbol == name + return true + end + end + return false +end diff --git a/src/transform/parsing/formula_transform.jl b/src/transform/parsing/formula_transform.jl index e90b62b..ed5120d 100644 --- a/src/transform/parsing/formula_transform.jl +++ b/src/transform/parsing/formula_transform.jl @@ -29,7 +29,7 @@ formula_rules = ASTRule[ quote test_res.primitives[$x] end - ) : x), + ) : SBMID(x.value)), ] @@ -48,12 +48,38 @@ function exportVars(symbols::Set{Symbol}, context::Context)::Expr return expr end -function transformFormula(form_expr :: ExtendedExpr, context :: Context) :: ExtendedExpr - x = MacroTools.prettify(MacroTools.postwalk(ruleSet(context, formula_rules), form_expr)) - # There is the edge case of having just a basic type, the else branch deals with it - if x isa ExtendedExpr - return x - else - return :(:($$x)) - end +# A unique wrapper type to mark "do not transform this QuoteNode" +# The thing is, transforming symbols can get quite messy cause some operators like a.b trigger it as well, so we need to mark them otherwise the fields will be treated as symbols. +struct DotFieldNode + inner::QuoteNode end + +function transformFormula(form_expr::ExtendedExpr, context::Context)::ExtendedExpr + + # prewalk: wrap QuoteNodes that are dot-access field names + protected = MacroTools.prewalk(form_expr) do node + if node isa Expr && node.head === :. && + length(node.args) == 2 && node.args[2] isa QuoteNode + # Replace the QuoteNode child with our sentinel + Expr(:., node.args[1], DotFieldNode(node.args[2])) + else + node + end + end + + # Ordinary context independent transformations + walked = MacroTools.postwalk(ruleSet(context, formula_rules), protected) + + # postwalk: unwrap sentinels back to QuoteNodes + result = MacroTools.postwalk(walked) do node + if node isa Expr && node.head === :. && + length(node.args) == 2 && node.args[2] isa DotFieldNode + Expr(:., node.args[1], node.args[2].inner) + else + node + end + end + + x = MacroTools.prettify(result) + return x isa ExtendedExpr ? x : :(:($$x)) +end \ No newline at end of file diff --git a/src/transform/parsing/hierarchy_transform.jl b/src/transform/parsing/hierarchy_transform.jl index a4f1cc6..4fe62db 100644 --- a/src/transform/parsing/hierarchy_transform.jl +++ b/src/transform/parsing/hierarchy_transform.jl @@ -8,7 +8,9 @@ function transformTestset(input_expr::Expr, context::Context) depth = length(context.test_tree_expr_builder) # Concatenate expressions of the current level into a new node on the upper level - concat = :(begin end) + concat = :( + begin end + ) # for expr in context.test_tree_expr_builder[depth] @@ -32,10 +34,12 @@ function transformTestset(input_expr::Expr, context::Context) push!(context._local.enabled_methodologies, MethodologyParameters[]) # LOGINFO - addLog("hierarchy", "[BNCH] New Group: $([i.set_name for i in context._local.depth_record])") + addLog("hierarchy", "[TESTSET] New Group: $([i.set_name for i in context._local.depth_record])") - # Launch regression methodology by default - onRegressionDefinition(quote end, context, Dict()) + # Launch regression methodology by default for :median_time on root testset + if Configuration.CONFIG["regression"]["enabled"] && length(context._local.depth_record) == 1 + onRegressionDefinition(quote end, context, Dict()) + end outerset = length(context._local.depth_record) <= 1 @@ -43,7 +47,7 @@ function transformTestset(input_expr::Expr, context::Context) result = quote $(outerset ? :(:__PERFTEST_FW__) : begin end) - TS = @perftestset PerfTestSet $name for $a in $b + TS = @perftestset PerfTestSet $name for $a in $b local ts = Test.get_testset() ts.iterator = $a $(outerset ? :(ts.old_test_results = _PRFT_GLOBALS.old) : begin end) @@ -56,7 +60,7 @@ function transformTestset(input_expr::Expr, context::Context) else result = quote $(outerset ? :(:__PERFTEST_FW__) : begin end) - TS = @perftestset PerfTestSet $name begin + TS = @perftestset PerfTestSet $name begin local ts = Test.get_testset() $(outerset ? :(ts.old_test_results = _PRFT_GLOBALS.old) : begin end) @@ -84,11 +88,15 @@ function transformPerftest(input_expr::Expr, context::Context) name = "Test $num" # LOGINFO - addLog("hierarchy", "[BNCH] New Test: $name \"$expr\" @ $([i.set_name for i in context._local.depth_record])") + addLog("hierarchy", "[PERFTEST] New Test: $name \"$expr\" @ $([i.set_name for i in context._local.depth_record])") # Return the transformed expression, in the following quote ts means the current testset return quote # Run the benchmark - ts.benchmarks[$name] = @PRFTBenchmark(($parsed_target), $(prop...)) + $(if mode == NormalMode + quote ts.benchmarks[$name] = @PRFTBenchmark(($parsed_target), $(prop...)) end + else + quote ts.benchmarks[$name] = @PRFTBenchmark(($parsed_target; MPI.Barrier(MPI.COMM_WORLD)), $(prop...)) end + end) # Create Test_Result struct to save test data test_res = Test_Result($name) @@ -127,15 +135,13 @@ function transformPerftest(input_expr::Expr, context::Context) buildPrimitiveMetrics!($mode, ts, test_res) $(buildCustomMetrics(context._local.custom_metrics)) - # Compute performance test based on enabled methodologies - $(buildMemTRPTMethodology(context)) - $(buildRoofline(context)) - $(buildPerfcmp(context)) - $(buildRegression(context)) - - PerfTest.printAuxiliaries(test_res.auxiliar, Test.get_testset_depth()); + if main_rank($mode) + $(buildMemTRPTMethodology(context)) + $(buildRoofline(context)) + $(buildPerfcmp(context)) + $(buildRegression(context)) + end - nothing end end diff --git a/src/transform/parsing/hierarchy_transform_benchmark_region.jl b/src/transform/parsing/hierarchy_transform_benchmark_region.jl index a287c64..f4b3de8 100644 --- a/src/transform/parsing/hierarchy_transform_benchmark_region.jl +++ b/src/transform/parsing/hierarchy_transform_benchmark_region.jl @@ -107,7 +107,7 @@ function backTokenToContextUpdate!(input_expr::QuoteNode, context::Context) pop!(context._local.custom_metrics) pop!(context._local.enabled_methodologies) - addLog("hierarchy", "[BNCH] Exiting group") + addLog("hierarchy", "[TESTSET] Exiting group") return nothing; end diff --git a/src/transform/parsing/hierarchy_transform_test_region.jl b/src/transform/parsing/hierarchy_transform_test_region.jl index cb01d05..339bf31 100644 --- a/src/transform/parsing/hierarchy_transform_test_region.jl +++ b/src/transform/parsing/hierarchy_transform_test_region.jl @@ -1,4 +1,3 @@ - # The tree builder is a stack used to construct the test region # 3 DIRECTIONS # DOWNWARDS : a new testset level is parsed -> a new test level is created (new top of stack) @@ -140,7 +139,6 @@ function updateTestTreeSideways!(context::Context, name::String) $(buildPerfcmp(context)) $(buildRegression(context)) - PerfTest.printAuxiliaries(_PRFT_LOCAL[:metrics], length(_PRFT_LOCAL[:depth])) end end)) end diff --git a/src/transform/parsing/rules.jl b/src/transform/parsing/rules.jl index a5aa21a..a3ccf13 100644 --- a/src/transform/parsing/rules.jl +++ b/src/transform/parsing/rules.jl @@ -73,7 +73,7 @@ perftest_macro_rule = ASTRule( ) perftest_begin_macro_rule = ASTRule( - x -> @capture(x, @benchmark __) || (escCaptureGetblock(x, Symbol("@count_ops")) != nothing), + x -> @capture(x, @benchmark __) || (escCaptureGetblock(x, Symbol("@count_ops")) !== nothing), no_validation, (x, ctx, info) -> (ctx.env_flags.inside_target = true; x) ) @@ -117,6 +117,8 @@ back_macro_rule = ASTRule( (x, ctx, info) -> backTokenToContextUpdate!(x, ctx) ) +# SECOND PASS TRIGGER TOKENS + prefix_macro_rule = ASTRule( x -> (x == :(:__PERFTEST_FW__)), no_validation, @@ -137,6 +139,11 @@ config_macro_rule = ASTRule( (x, ctx, info) -> (parseConfigurationMacro(x, ctx, info)) ) +threads_macro_rule = ASTRule( + x -> (@capture(x, @m_ __); m == Symbol("@perftest_threads")), + threads_validation, + (x, ctx, info) -> (parseThreadsMacro(x, ctx, info)) +) # CONDITIONAL EXECUTION @@ -149,7 +156,9 @@ on_perftest_exec_rule = ASTRule( on_perftest_ignore_rule = ASTRule( x -> escCaptureGetblock(x, Symbol("@on_perftest_ignore")) !== nothing, on_perftest_ignore_validation, - (x, ctx, info) -> :(begin end) + (x, ctx, info) -> :( + begin end + ) ) # CUSTOM METRICS @@ -160,6 +169,12 @@ define_memory_throughput_rule = ASTRule( (x, ctx, info) -> onMemoryThroughputDefinition(transformFormula(info[Symbol("")], ctx), ctx, info) ) +regression_macro_rule = ASTRule( + x -> @capture(x, @regression __), + validateBlocklessMacro(define_regression_validation), + (x, ctx, info) -> onRegressionDefinition(x, ctx, info) +) + define_metric_rule = ASTRule( x -> escCaptureGetblock(x, Symbol("@define_metric")) !== nothing, define_metric_validation, @@ -200,7 +215,7 @@ manual_macro_rule = ASTRule( (x, ctx, info) -> onPerfcmpDefinition(x, ctx, info) ) -function treeRunRecursive!(path::AbstractString, parent_context :: Context)::ExtendedExpr +function treeRunRecursive!(path::AbstractString, parent_context::Context)::ExtendedExpr addLog("hierarchy", "[RECURSIVE] Recursivity is enabled, entering \"$path\"") # The transformation treats child files as if it were into the main one plugging their testsets into the global hierarchy @@ -218,7 +233,7 @@ function treeRunRecursive!(path::AbstractString, parent_context :: Context)::Ext Configuration.CONFIG = pop!(Configuration.PARENT_CONFIGS) - addLog("hierarchy", "[RECURSIVE] \"$path\" has been processed, $(num_errors(ctx._global.errors)) errors found") + addLog("hierarchy", "[RECURSIVE] \"$path\" has been processed, $(num_errors(ctx)) errors found") return middle end @@ -227,13 +242,15 @@ end recursive_rule = ASTRule( x -> @capture(x, include(path_)), always_true, - (x, ctx, info) -> (begin - if Configuration.CONFIG["general"]["recursive"] - @capture(x, include(path_)) - measure_expr = treeRunRecursive!(joinpath(dirname(ctx._global.original_file_path), path), ctx) - measure_expr - else - quote end + (x, ctx, info) -> ( + begin + if Configuration.CONFIG["general"]["recursive"] + @capture(x, include(path_)) + measure_expr = treeRunRecursive!(joinpath(dirname(ctx._global.original_file_path), path), ctx) + measure_expr + else + quote end + end end - end) + ) ) diff --git a/src/transform/prefix.jl b/src/transform/prefix.jl index 3ad3749..46d3fa5 100644 --- a/src/transform/prefix.jl +++ b/src/transform/prefix.jl @@ -1,6 +1,6 @@ ### PREFIX FILLER -function perftestprefix(ctx :: Context)::Expr +function perftestprefix(ctx::Context)::Expr suite_name = "$(basename(ctx._global.original_file_path))_PERFORMANCE" if isdir("./$(Configuration.CONFIG["general"]["save_folder"])") @@ -10,12 +10,12 @@ function perftestprefix(ctx :: Context)::Expr return quote using Test, Dates - using PerfTest: DepthRecord,Metric_Test,Methodology_Result,StrOrSym,Metric_Result, magnitudeAdjust, MPISetup, newMetricResult, buildPrimitiveMetrics!, measureCPUPeakFlops!,measureMemBandwidth!,addLog,@PRFTBenchmark,PRFTBenchmarkGroup,@PRFTCapture_out,@PRFTCount_ops,PRFTflop,@PRFTSuppress,Test_Result,by_index,regression,Suite_Execution_Result,savePrimitives,main_rank,GlobalSuiteData,@perftestset,PerfTestSet,extractTestResults,saveMethodologyData,Configuration + $(mode == MPIMode ? :(using MPI; PerfTest.MPISetup(PerfTest.MPIMode)) : :(nothing)) + using PerfTest: DepthRecord, Metric_Test, Methodology_Result, StrOrSym, Metric_Result, magnitudeAdjust, MPISetup, newMetricResult, buildPrimitiveMetrics!, measureCPUPeakFlops!, measureMemBandwidth!, addLog, @PRFTBenchmark, PRFTBenchmarkGroup, @PRFTCapture_out, @PRFTCount_ops, PRFTflop, @PRFTSuppress, Test_Result, by_index, regression, Suite_Execution_Result, savePrimitives, main_rank, GlobalSuiteData, @perftestset, PerfTestSet, extractTestResults, saveMethodologyData, Configuration _t_begin = time() + _GLOBAL_MODE = $mode - MPISetup($mode) - if main_rank($mode) # Used to save data about this test suite if needed path = $("./$(Configuration.CONFIG["general"]["save_folder"])/$(suite_name).JLD2") @@ -26,28 +26,32 @@ function perftestprefix(ctx :: Context)::Expr datafile = PerfTest.openDataFile(path) else datafile = PerfTest.Perftest_Datafile_Root(PerfTest.Suite_Execution_Result[]) - - PerfTest.p_yellow("[!]") - println("Regression: No previous performance reference for this configuration has been found, measuring performance without evaluation.") end - + # Regression data - regression_path = Configuration.CONFIG["regression"]["custom_file"] + regression_path = Configuration.CONFIG["regression"]["dedicated_reference_file"] # If absent use default data file, otherwise check if exists and open custom file if regression_path != "" && isfile(regression_path) regression_file = PerfTest.openDataFile(regression_path) else if regression_path != "" - @error "Regression data file $regression_file could not be opened or found" + regression_file = PerfTest.Perftest_Datafile_Root(PerfTest.Suite_Execution_Result[]) else regression_file = datafile + regression_path = path end end - _PRFT_GLOBALS = GlobalSuiteData(datafile,path,$(ctx._global.original_file_path)) + _PRFT_GLOBALS = GlobalSuiteData(datafile, path, $(ctx._global.original_file_path)) if length(regression_file.results) > 0 - _PRFT_GLOBALS.old = regression_file.results[end].perftests + for i in length(regression_file.results):-1:1 + if length(retrievePerfTests(regression_path, get=:tests, where_pred=testFailed, execution=i)) == 0 + _PRFT_GLOBALS.old = regression_file.results[i].perftests + println("Regression: Previous performance reference for this configuration has been found, regression tests could be performed. Regression is currenctly $($(Configuration.CONFIG["regression"]["enabled"]) ? "enabled" : "disabled")).") + break + end + end else _PRFT_GLOBALS.old = nothing end @@ -55,9 +59,8 @@ function perftestprefix(ctx :: Context)::Expr _PRFT_GLOBALS = GlobalSuiteData() end - # Do machine specs - # Will compute peak flops and peak bandwidth and populate - $(machineBenchmarks(mode)) + # Do machine probe + $(machineBenchmarks(mode, ctx)) # Methodology prefixes #$(regressionPrefix(ctx)) diff --git a/src/transform/suffix.jl b/src/transform/suffix.jl index e73b139..776927f 100644 --- a/src/transform/suffix.jl +++ b/src/transform/suffix.jl @@ -153,19 +153,6 @@ end function perftestsuffix(context :: Context) return quote if main_rank($mode) - # Deal with recorder results - let - res_num = length(_PRFT_GLOBALS.datafile.results) - - if (excess = $(Configuration.CONFIG["general"]["max_saved_results"]) - res_num) <= 0 - PerfTest.p_yellow("[ℹ]") - println(" Regression: Exceeded maximum recorded results. The oldest $(-1*excess + 1) result/s will be removed.") - for i in 1:(-1*excess+1) - popfirst!(_PRFT_GLOBALS.datafile.results) - end - end - - end testresdict = Dict{String,Union{Dict,Test_Result}}() if TS isa Vector @@ -193,14 +180,42 @@ function perftestsuffix(context :: Context) perftests = testresdict ) end - push!(_PRFT_GLOBALS.datafile.results, newres) - # No fails no errors - if sum(Test.get_test_counts(TS)[2:3]) == 0 - PerfTest.saveDataFile(_PRFT_GLOBALS.datafile_path, _PRFT_GLOBALS.datafile) - # Export as json - #BencherInterface.exportToJSON(_PRFT_GLOBALS.datafile_path * ".json", newres) + # No fails no errors, regression enabled and different regression file path + if PerfTest.testsSucceeded(TS) && Configuration.CONFIG["regression"]["enabled"] && regression_path != _PRFT_GLOBALS.datafile_path + println("All performance tests have passed. Values will be registered as reference for regression testing.") + push!(regression_file.results, newres) + # Deal with recorded results + let + res_num = length(regression_file.results) + if (excess = $(Configuration.CONFIG["general"]["max_saved_results"]) - res_num) <= 0 + PerfTest.p_yellow("[ℹ]") + println(" Regression: Exceeded maximum recorded results. The oldest $(-1*excess + 1) result/s will be removed.") + for i in 1:(-1*excess+1) + popfirst!(regression_file.results) + end + end + end + PerfTest.saveDataFile(regression_path, regression_file) + else + println("Some tests failed or errored.") + end + # Deal with recorded results + let + push!(_PRFT_GLOBALS.datafile.results, newres) + res_num = length(_PRFT_GLOBALS.datafile.results) + + if (excess = $(Configuration.CONFIG["general"]["max_saved_results"]) - res_num) <= 0 + PerfTest.p_yellow("[ℹ]") + println(" Results File: Exceeded maximum recorded results. The oldest $(-1*excess + 1) result/s will be removed.") + for i in 1:(-1*excess+1) + popfirst!(_PRFT_GLOBALS.datafile.results) + end + end end + # Export as json + #BencherInterface.exportToJSON(_PRFT_GLOBALS.datafile_path * ".json", newres) + PerfTest.saveDataFile(_PRFT_GLOBALS.datafile_path, _PRFT_GLOBALS.datafile) println("[✓] $path Performance tests have been finished (elapsed $(newres.elapsed) s)") # Bencher export using the REST API diff --git a/src/transform/validation/errors.jl b/src/transform/validation/errors.jl index 1affe82..e5bd469 100644 --- a/src/transform/validation/errors.jl +++ b/src/transform/validation/errors.jl @@ -7,14 +7,20 @@ function printError(e :: ParsingErrorInfo, l :: String) println("") end -function num_errors(e :: VecErrorCollection) :: Int - return length(e.errors) +function num_errors(c :: Context) :: Int + return length(c._global.errors.errors) +end +function num_errors(c :: VecErrorCollection) :: Int + return length(c.errors) +end +function num_errors() :: Int + return ctx._global.errors.errors |> length end function pushError!(error :: ParsingErrorInfo, collection :: VecErrorCollection, depth :: AbstractArray{DepthEntry}) push!(collection.errors, error) push!(collection.loc, "| " * foldl(*, [e.set_name * " > " for e in depth])) - addLog("general", error.name) + addLog("general", Base.text_colors[:red] * "[ERROR] " * error.name * Base.text_colors[:default]) end # Abbreviations for ASTRule @@ -25,8 +31,8 @@ function throwParseError!(num, name, context) pushError!(ParsingErrorInfo(num, name), context._global.errors, context._local.depth_record) end -function printErrors(collection :: VecErrorCollection) - for e in zip(collection.errors, collection.loc) +function printErrors(context :: Context) + for e in zip(context._global.errors.errors, context._global.errors.loc) printError(e...) end end diff --git a/src/transform/validation/formula.jl b/src/transform/validation/formula.jl index 6e1fbc6..23adb83 100644 --- a/src/transform/validation/formula.jl +++ b/src/transform/validation/formula.jl @@ -12,3 +12,20 @@ formula_symbols = Set([ :peak_bandwidth, ]) +function SBMID(metric :: metricID()) + sym = metricID(metric) + if sym in formula_symbols + return quote _PRFT_GLOBALS.builtins[$(QuoteNode(sym))] end + else + return quote (haskey(_PRFT_GLOBALS.custom_benchmarks,$(QuoteNode(sym))) ? + _PRFT_GLOBALS.custom_benchmarks[$(QuoteNode(sym))].value : + haskey(_PRFT_GLOBALS.builtins, $(QuoteNode(sym))) ? + _PRFT_GLOBALS.builtins[$(QuoteNode(sym))] : + haskey(test_res.metrics,$(QuoteNode(sym))) ? + test_res.metrics[$(QuoteNode(sym))].value : + haskey(test_res.auxiliar,$(QuoteNode(sym))) ? + test_res.auxiliar[$(QuoteNode(sym))].value : + error("Undefined $($(QuoteNode(sym))), wrong spelling or not defined in the current context?")) + end + end +end diff --git a/src/transform/validation/macro.jl b/src/transform/validation/macro.jl index 35fd2eb..ace9f46 100644 --- a/src/transform/validation/macro.jl +++ b/src/transform/validation/macro.jl @@ -61,7 +61,7 @@ function validateMacro(macro_param :: Dict{Symbol, MacroParameter}) param_info = macro_param[Symbol("")] param = args[end] @matchast param quote - ($a = $b) => (throwParseError!("Last parameter cannot be a keyword parameter on macro $m",context); return nothing) + ($a = $b) => (throwParseError!("Last parameter cannot be a keyword ($a) parameter on macro $m",context); return nothing) $_ => nothing end if typeof(param) <: param_info.type @@ -102,7 +102,6 @@ function validateBlocklessMacro(macro_param::Dict{Symbol,MacroParameter}) @capture(macro_expr, @m_ args__) - mandatory = sum([a.second.mandatory for a in macro_param]) _all = length(macro_param) # Invalid parameter numbers diff --git a/test/mock3-roofline.jl b/test/mock3-roofline.jl index 5a213b5..163b3cd 100644 --- a/test/mock3-roofline.jl +++ b/test/mock3-roofline.jl @@ -7,7 +7,7 @@ using PerfTest [regression] enabled = false [general] -verbose = true +verbose = 3 " diff --git a/test/newtest.jl b/test/newtest.jl deleted file mode 100644 index e28d90d..0000000 --- a/test/newtest.jl +++ /dev/null @@ -1,104 +0,0 @@ -module __PERFTEST__ -using Test -using PerfTest -PerfTest._perftest_config("[perfcompare]\nenabled = true\n\n[MPI]\nmode = \"reduce\"\nenabled = false\n\n[general]\nrecursive = true\nsave_results = true\nautoflops = true\nsuppress_output = true\nplotting = true\nsave_folder = \".perftests\"\nmax_saved_results = 20\nverbose = true\nlogs_enabled = true\nsafe_formulas = false\n\n[regression]\nenabled = false\ndefault_threshold = 0.05\n\n[roofline]\nenabled = true\ndefault_threshold = 0.5\n\n[memory_bandwidth]\nenabled = true\ndefault_threshold = 0.5\n\n[machine_benchmarking]\nmemory_bandwidth_test_buffer_size = false\n") -function testfun(a::Int) - c = 1 - for i = 1:a - c = c + i ^ 2 / c - end - return c -end -using Test, Dates -using PerfTest: DepthRecord, Metric_Test, Methodology_Result, StrOrSym, Metric_Result, magnitudeAdjust, MPISetup, newMetricResult, buildPrimitiveMetrics!, measureCPUPeakFlops!, measureMemBandwidth!, addLog, @PRFTBenchmark, PRFTBenchmarkGroup, @PRFTCapture_out, @PRFTCount_ops, PRFTflop, @PRFTSuppress, Test_Result, by_index, regression, Suite_Execution_Result, savePrimitives, main_rank, GlobalSuiteData, @perftestset, PerfTestSet, extractTestResults, saveMethodologyData -if main_rank() - path = "./.perftests/ex2-effmemtp.jl_PERFORMANCE.JLD2" - nofile = true - if isfile(path) - nofile = false - datafile = PerfTest.openDataFile(path) - else - datafile = PerfTest.Perftest_Datafile_Root(PerfTest.Suite_Execution_Result[]) - PerfTest.p_yellow("[!]") - println("Regression: No previous performance reference for this configuration has been found, measuring performance without evaluation.") - end - _PRFT_GLOBALS = GlobalSuiteData(datafile, path, "ex2-effmemtp.jl") - MPISetup(PerfTest.NormalMode, _PRFT_GLOBALS) -end -let - size = try - CpuId.cachesize() - catch - addLog("machine", "[MACHINE] CpuId failed, using default cache size") - [1024 * 1024 * 16] - end - global _PRFT_GLOBALS.builtins[:MEM_CACHE_SIZES] = size - addLog("machine", "[MACHINE] Memory buffer size for benchmarking = $((size ./ 1024) ./ 1024) [MB]") - measureCPUPeakFlops!(PerfTest.NormalMode, _PRFT_GLOBALS) - measureMemBandwidth!(PerfTest.NormalMode, _PRFT_GLOBALS) -end -TS = @perftestset(PerfTestSet, "FIRST LEVEL", begin - local ts = Test.get_testset() - nothing - TS = @perftestset(PerfTestSet, "SECOND LEVEL", begin - - local ts = Test.get_testset() - # BENCH - ts.benchmarks["Test 1"] = @PRFTBenchmark(($testfun)(10)) - test_res = Test_Result("Test 1") - ts.test_results["Test 1"] = test_res - test_res.primitives[:autoflop] = PRFTflop(@PRFTCount_ops(($testfun)(10))) - test_res.primitives[:printed_output] = @PRFTCapture_out(test_res.primitives[:ret_value] = (x = testfun(10))) - # METRICS - buildPrimitiveMetrics!(PerfTest.NormalMode, ts, test_res) - test_res.metrics[:effMemTP] = newMetricResult(PerfTest.NormalMode, name = "Effective memory throughput", units = "GB/s", value = 2.0 + 5.0, auxiliary = false) - # METHIDOLOGY - let - reference_benchmark = _PRFT_GLOBALS.builtins[:MEM_STREAM_COPY] - value = (test_res.metrics[:effMemTP]).value / reference_benchmark - success = value >= 0.01 - result = newMetricResult(PerfTest.NormalMode, name = "Effective Throughput Ratio", units = "%", value = value * 100) - test = Metric_Test(reference = 100, threshold_min_percent = 0.01, threshold_max_percent = 1.0, low_is_bad = true, succeeded = success, custom_plotting = Symbol[], full_print = true) - aux_abs_value = newMetricResult(PerfTest.NormalMode, name = "Attained Bandwidth", units = "B/s", value = value) - aux_ref_value = newMetricResult(PerfTest.NormalMode, name = "Peak empirical bandwidth", units = "B/s", value = reference_benchmark) - methodology_res = Methodology_Result(name = "Effective Memory Throughput") - push!(methodology_res.metrics, result => test) - methodology_res.custom_elements[:abs] = magnitudeAdjust(aux_abs_value) - methodology_res.custom_elements[:abs_ref] = magnitudeAdjust(aux_ref_value) - if true || !(test.succeeded) - PerfTest.printMethodology(methodology_res, 2, true) - end - saveMethodologyData(test_res.name, methodology_res) - end - # PRINT - PerfTest.printAuxiliaries(test_res.auxiliar, Test.get_testset_depth()) - nothing - nothing - end) - nothing - nothing - end) -if main_rank() - let - res_num = length(_PRFT_GLOBALS.datafile.results) - if (excess = 20 - res_num) <= 0 - PerfTest.p_yellow("[ℹ]") - println(" Regression: Exceeded maximum recorded results. The oldest $(-1 * excess + 1) result/s will be removed.") - for i = 1:-1 * excess + 1 - popfirst!(_PRFT_GLOBALS.datafile.results) - end - end - if length(_PRFT_GLOBALS.datafile.results) > 0 - _PRFT_GLOBALS.old = (_PRFT_GLOBALS.datafile.results[end]).perftests - else - _PRFT_GLOBALS.old = nothing - end - end - newres = Suite_Execution_Result(timestamp = datetime2unix(now()), benchmarks = TS.benchmarks, perftests = extractTestResults(TS)) - push!(_PRFT_GLOBALS.datafile.results, newres) - if sum((Test.get_test_counts(TS))[2:3]) == 0 - PerfTest.saveDataFile(_PRFT_GLOBALS.datafile_path, _PRFT_GLOBALS.datafile) - end - println("[✓] $(path) Performance tests have been finished") -end -end diff --git a/test/perftest_config.toml b/test/perftest_config.toml index cbb294c..06d97b2 100644 --- a/test/perftest_config.toml +++ b/test/perftest_config.toml @@ -6,22 +6,24 @@ mode = "reduce" enabled = false [general] -recursive = true save_results = true -autoflops = true +max_saved_results = 20 suppress_output = true -plotting = true save_folder = ".perftests" -max_saved_results = 20 -verbose = true +threads_per_numa = "single" +recursive = true +verbose = 0 +autoflops = true +plotting = true +numas = "single" logs_enabled = true safe_formulas = false [regression] enabled = true -default_threshold = 0.9 +dedicated_reference_file = "" +default_threshold = 1.1 use_bencher = false -custom_file = "" [roofline] enabled = true @@ -35,8 +37,8 @@ default_threshold = 0.5 memory_bandwidth_test_buffer_size = 0 [bencher] -api_key = "FILL THIS IF IN USE" -organization = "FILL THIS IF IN USE" +api_key = "" +organization = "" api_url = "https://api.bencher.dev" -project_name = "FILL THIS IF IN USE" +project_name = "" custom_testbed_name = "" diff --git a/test/runtests.jl b/test/runtests.jl index ad29186..95c5878 100644 --- a/test/runtests.jl +++ b/test/runtests.jl @@ -1,12 +1,39 @@ using Test using PerfTest +excludedfiles = Set{String}([ +]) + if !(PerfTest.Configuration.load_config() isa Nothing) - include("t1-validation-formula.jl") - include("t2-validation-macro.jl") - include("t3-transforms.jl") + @info "This may take a couple of minutes..." + exename = joinpath(Sys.BINDIR, Base.julia_exename()) + testdir = pwd() + istest(f) = endswith(f, ".jl") && startswith(basename(f), "t") + testfiles = sort(filter(istest, vcat([joinpath.(root, files) for (root, dirs, files) in walkdir(testdir)]...))) + + global nfail = 0 + printstyled("Testing package PerfTest.jl\n"; bold=true, color=:white) + + for f in testfiles + println("") + if f ∈ excludedfiles + println("Test Skip:") + println("$f") + continue + end + try + run(`$exename -O3 --startup-file=no $(joinpath(testdir, f))`) + catch ex + nfail += 1 + end + end rm(".perftest_logs", recursive=true, force=true) rm(".perftests", recursive=true,force=true) + + if nfail == 0 + else + printstyled("\n$nfail test(s) failed.\n"; bold=true, color=:red) + end end \ No newline at end of file diff --git a/test/t.jl b/test/t.jl deleted file mode 100644 index ced6fb0..0000000 --- a/test/t.jl +++ /dev/null @@ -1,23 +0,0 @@ -using Test -using PerfTest - - -# Note the config here disables verbosity but the nested files enable it, inside the nested file its config has priority -@perftest_config " -[regression] -enabled = false -[general] -verbose = false -" - -@testset "A" begin - @testset "A.1" begin - # Check that time elapsed is less than one second, applies to the targets inside this testset - @perfcompare :median_time < 1 - # Being "mock3-roofline.jl" a file with the roofline mock example source code. - include("mock3-roofline.jl") - end - include("mock3-roofline.jl") -end - - diff --git a/test/t1-validation-formula.jl b/test/t1-validation-formula.jl index 878ea81..faf27b4 100644 --- a/test/t1-validation-formula.jl +++ b/test/t1-validation-formula.jl @@ -1,5 +1,6 @@ using MacroTools - +using Test +using PerfTest @testset "Formula validation tests" begin @@ -13,16 +14,34 @@ using MacroTools r = PerfTest.transformFormula(form, ctx) @test r == MacroTools.prettify(quote - a = 54 + a = 54 test_res.primitives[:autoflop] / test_res.primitives[:min_time] * a end) + form = quote + a.b = 54 + a.c = :min_time + end + r = PerfTest.transformFormula(form, ctx) + @test r == MacroTools.prettify(quote + a.b = 54 + a.c = test_res.primitives[:min_time] + end) + + form = quote + A.b(C.D) + end + r = PerfTest.transformFormula(form, ctx) + @test r == MacroTools.prettify(quote + A.b(C.D) + end) + #= # illegal symbol form = quote :aflops end PerfTest.transformFormula(form, ctx) - @test PerfTest.num_errors(ctx._global.errors) == 1 + @test PerfTest.num_errors(ctx) == 1 # For now admitted, may be restricted in the future form = quote @@ -30,7 +49,7 @@ using MacroTools :autoflop end PerfTest.transformFormula(form, ctx) - @test PerfTest.num_errors(ctx._global.errors) == 1 =# + @test PerfTest.num_errors(ctx) == 1 =# - PerfTest.printErrors(ctx._global.errors) + PerfTest.printErrors(ctx) end diff --git a/test/t2-validation-macro.jl b/test/t2-validation-macro.jl index 1dba3b2..6fd0394 100644 --- a/test/t2-validation-macro.jl +++ b/test/t2-validation-macro.jl @@ -1,3 +1,4 @@ +using Test, PerfTest # Here the macro parser is tested, common errors are put to check if they are caught @testset "Macro validation tests" begin @@ -18,7 +19,7 @@ f(expr, ctx) - @test PerfTest.num_errors(ctx._global.errors) == 0 + @test PerfTest.num_errors(ctx) == 0 # VALID @@ -28,7 +29,7 @@ parsed = f(expr, ctx) - @test PerfTest.num_errors(ctx._global.errors) == 0 + @test PerfTest.num_errors(ctx) == 0 # Test parameter parsing @test parsed[:aparam] == 3.45 @@ -41,7 +42,7 @@ end parsed = f(expr, ctx) - @test PerfTest.num_errors(ctx._global.errors) == 1 + @test PerfTest.num_errors(ctx) == 1 # Test parameter parsing @test parsed[Symbol("")] == :(4+5) @@ -54,7 +55,7 @@ end f(expr, ctx) - @test PerfTest.num_errors(ctx._global.errors) == 2 + @test PerfTest.num_errors(ctx) == 2 # missing all arguments expr = quote @@ -62,7 +63,7 @@ end f(expr, ctx) - @test PerfTest.num_errors(ctx._global.errors) == 3 + @test PerfTest.num_errors(ctx) == 3 # bparam should be positive expr = quote @@ -70,7 +71,7 @@ end f(expr, ctx) - @test PerfTest.num_errors(ctx._global.errors) == 4 + @test PerfTest.num_errors(ctx) == 4 # expression should be a sum expr = quote @@ -78,7 +79,7 @@ end f(expr, ctx) - @test PerfTest.num_errors(ctx._global.errors) == 5 + @test PerfTest.num_errors(ctx) == 5 # non-existent parameter expr = quote @@ -86,9 +87,9 @@ end f(expr, ctx) - @test PerfTest.num_errors(ctx._global.errors) == 6 + @test PerfTest.num_errors(ctx) == 6 # Uncomment to see error messages - #PerfTest.printErrors(ctx._global.errors) + #PerfTest.printErrors(ctx) end diff --git a/test/t3-transforms.jl b/test/t3-transforms.jl index 94d49ab..594884a 100644 --- a/test/t3-transforms.jl +++ b/test/t3-transforms.jl @@ -1,4 +1,8 @@ +using Test, PerfTest + +prefix = "test-recipes/" + sources = [ "ex1-hierarchy-basic.jl", "ex2-effmemtp.jl", @@ -6,70 +10,27 @@ sources = [ "ex4-perfcmp.jl", "ex5-recursive.jl" ] +sources = [prefix * s for s in sources] checks = [ [ - "[BNCH] New Group: [\"FIRST LEVEL\"]", - - "[BNCH] New Group: [\"FIRST LEVEL\", \"SECOND LEVEL\"]", - - "[BNCH] New Test: Test 1 \"testfun(10)\" @ [\"FIRST LEVEL\", \"SECOND LEVEL\"]", - - "[BNCH] Exiting group", - - "[BNCH] Exiting group" + "[TESTSET] New Group: [\"FIRST LEVEL\"]", "[TESTSET] New Group: [\"FIRST LEVEL\", \"SECOND LEVEL\"]", "[PERFTEST] New Test: Test 1 \"testfun(10)\" @ [\"FIRST LEVEL\", \"SECOND LEVEL\"]", "[TESTSET] Exiting group", "[TESTSET] Exiting group" ], [ - "[BNCH] New Group: [\"FIRST LEVEL\"]", - - "[BNCH] New Group: [\"FIRST LEVEL\", \"SECOND LEVEL\"]", - - "[BNCH] New Test: Test 1 \"x = testfun(10)\" @ [\"FIRST LEVEL\", \"SECOND LEVEL\"]", - - "[BNCH] Exiting group", - - "[BNCH] Exiting group", + "[TESTSET] New Group: [\"FIRST LEVEL\"]", "[TESTSET] New Group: [\"FIRST LEVEL\", \"SECOND LEVEL\"]", "[PERFTEST] New Test: Test 1 \"x = testfun(10)\" @ [\"FIRST LEVEL\", \"SECOND LEVEL\"]", "[TESTSET] Exiting group", "[TESTSET] Exiting group", ], [ - "[BNCH] New Group: [\"FIRST LEVEL\"]", - - "[BNCH] New Group: [\"FIRST LEVEL\", \"SECOND LEVEL\"]", - - "[BNCH] New Test: Test 1 \"testfun(10)\" @ [\"FIRST LEVEL\", \"SECOND LEVEL\"]", - "[METHODOLOGY] Defined ROOFLINE MODEL on [\"FIRST LEVEL\", \"SECOND LEVEL\"]", - "Building Operational intensity", - "Building Attained Flops", - - "[BNCH] Exiting group", - - "[BNCH] Exiting group", + "[TESTSET] New Group: [\"FIRST LEVEL\"]", "[TESTSET] New Group: [\"FIRST LEVEL\", \"SECOND LEVEL\"]", "[PERFTEST] New Test: Test 1 \"testfun(10)\" @ [\"FIRST LEVEL\", \"SECOND LEVEL\"]", + "[METHODOLOGY] Defined ROOFLINE MODEL on [\"FIRST LEVEL\", \"SECOND LEVEL\"]", + "Building Operational intensity", + "Building Attained Flops", "[TESTSET] Exiting group", "[TESTSET] Exiting group", ], [ - "[BNCH] New Group: [\"FIRST LEVEL\"]", - - "[BNCH] New Group: [\"FIRST LEVEL\", \"SECOND LEVEL\"]", - - "[BNCH] New Test: Test 1 \"testfun(10)\" @ [\"FIRST LEVEL\", \"SECOND LEVEL\"]", - - "[BNCH] Exiting group", - - "[BNCH] Exiting group", + "[TESTSET] New Group: [\"FIRST LEVEL\"]", "[TESTSET] New Group: [\"FIRST LEVEL\", \"SECOND LEVEL\"]", "[PERFTEST] New Test: Test 1 \"testfun(10)\" @ [\"FIRST LEVEL\", \"SECOND LEVEL\"]", "[TESTSET] Exiting group", "[TESTSET] Exiting group", ], [ - "[BNCH] New Group: [\"RECURSIVE\"]", - - "[RECURSIVE] Recursivity is enabled, entering \"ex4-perfcmp.jl\"", - "[BNCH] New Group: [\"RECURSIVE\", \"FIRST LEVEL\"]", - - "[BNCH] New Group: [\"RECURSIVE\", \"FIRST LEVEL\", \"SECOND LEVEL\"]", - - "[BNCH] New Test: Test 1 \"testfun(10)\" @ [\"RECURSIVE\", \"FIRST LEVEL\", \"SECOND LEVEL\"]", - - "[BNCH] Exiting group", - - "[BNCH] Exiting group", - "[RECURSIVE] \"ex4-perfcmp.jl\" has been processed, 0 errors found", - - "[BNCH] Exiting group" + "[TESTSET] New Group: [\"RECURSIVE\"]", "[RECURSIVE] Recursivity is enabled, entering \"test-recipes/ex4-perfcmp.jl\"", + "[TESTSET] New Group: [\"RECURSIVE\", \"FIRST LEVEL\"]", "[TESTSET] New Group: [\"RECURSIVE\", \"FIRST LEVEL\", \"SECOND LEVEL\"]", "[PERFTEST] New Test: Test 1 \"testfun(10)\" @ [\"RECURSIVE\", \"FIRST LEVEL\", \"SECOND LEVEL\"]", "[TESTSET] Exiting group", "[TESTSET] Exiting group", + "[RECURSIVE] \"test-recipes/ex4-perfcmp.jl\" has been processed, 0 errors found", "[TESTSET] Exiting group" ], ] diff --git a/test/t4-measure.jl b/test/t4-measure.jl new file mode 100644 index 0000000..550ef0b --- /dev/null +++ b/test/t4-measure.jl @@ -0,0 +1,57 @@ +using Test +using PerfTest + +view = false + +@testset "Measure - Full pass" begin + + @testset "Run transformation" begin + e = PerfTest.transform("test-recipes/ex7-measure-tests.jl") + + PerfTest.saveExprAsFile(e, "_t4_tmp_t4_tmp.jl") + + @test PerfTest.num_errors() == 0 + end + + @testset "Execution" begin + if view + include("_t4_tmp_t4_tmp.jl") + else + redirect_stdout(devnull) do + include("_t4_tmp_t4_tmp.jl") + end + end + @test length(PerfTest.get_testset().results) > 0 + Test.get_testset().results = [] + end + + @testset "Data retrieval" begin + passed = PerfTest.retrievePerfTests(".perftests/ex7-measure-tests.jl_PERFORMANCE.JLD2", get=:tests, where_pred=PerfTest.testPassed, pathonly=true) + failed = PerfTest.retrievePerfTests(".perftests/ex7-measure-tests.jl_PERFORMANCE.JLD2", get=:tests, where_pred=PerfTest.testFailed, pathonly=true) + + @test all([occursin("pass", t) for t in passed]) + @test all([occursin("fail", t) for t in failed]) + @test length(passed) == 5 + @test length(failed) == 5 + + methodologies = PerfTest.retrievePerfTests(".perftests/ex7-measure-tests.jl_PERFORMANCE.JLD2", get=:methodologies) + @test length(methodologies) == 10 + @test methodologies[1].name == "Effective Memory Throughput" + @test 1.95 <= methodologies[1].custom_elements[:abs].value <= 2.0 + @test methodologies[1].custom_elements[:abs_ref].value == 2.0 + + @test methodologies[2].name == "Performance Assertion" + @test methodologies[2].metrics[1][1].value == true + @test methodologies[2].metrics[1][2].succeeded == true + @test methodologies[2].metrics[1][1].name == ":(duration * 0.9 < :median_time < duration * 1.1)" + end + + # Cleanup + try + rm("_t4_tmp_t4_tmp.jl") + rm(".perftests/ex7-measure-tests.jl_PERFORMANCE.JLD2") + rm(".perftests", recursive=true) + catch + @warn "Automatic file cleanup failed." + end +end \ No newline at end of file diff --git a/test/t5-regression.jl b/test/t5-regression.jl new file mode 100644 index 0000000..a255210 --- /dev/null +++ b/test/t5-regression.jl @@ -0,0 +1,56 @@ +using Test,PerfTest + +view = false + +@testset "Regression - Full pass" begin + + @testset "Run transformation" begin + e = PerfTest.transform("test-recipes/ex8-regression-test.jl") + PerfTest.saveExprAsFile(e, "_t5_tmp_t5_tmp.jl") + + @test PerfTest.num_errors() == 0 + end + + @testset "Execution" begin + for i in 1:3 + + @testset "PerfTest env" begin + if view + include("_t5_tmp_t5_tmp.jl") + else + redirect_stdout(devnull) do + include("_t5_tmp_t5_tmp.jl") + end + end + end + Test.get_testset().results = [] + + + passed = PerfTest.retrievePerfTests(".perftests/ex8-regression-test.jl_PERFORMANCE.JLD2", get=:tests, where_pred=PerfTest.testPassed, pathonly=true) + failed = PerfTest.retrievePerfTests(".perftests/ex8-regression-test.jl_PERFORMANCE.JLD2", get=:tests, where_pred=PerfTest.testFailed, pathonly=true) + if i == 1 + @test length(passed) == 4 + @test length(failed) == 0 + else + @test length(passed) == 2 + for p in passed + @test contains(p, "pass") + end + @test length(failed) == 2 + for f in failed + @test contains(f, "fail") + end + end + end + end + + + # Cleanup + try + rm("_t5_tmp_t5_tmp.jl") + rm(".perftests/ex8-regression-test.jl_PERFORMANCE.JLD2") + rm(".perftests", recursive=true) + catch + @warn "Automatic file cleanup failed." + end +end \ No newline at end of file diff --git a/test/t6-no-perftests.jl b/test/t6-no-perftests.jl new file mode 100644 index 0000000..17a0330 --- /dev/null +++ b/test/t6-no-perftests.jl @@ -0,0 +1,8 @@ + +using Test, PerfTest + +@testset "No perf tests" begin + include("test-recipes/ex9-no-perftests.jl") + + @test all([r isa Test.DefaultTestSet for r in Test.get_testset().results]) +end \ No newline at end of file diff --git a/test/ex1-hierarchy-basic.jl b/test/test-recipes/ex1-hierarchy-basic.jl similarity index 100% rename from test/ex1-hierarchy-basic.jl rename to test/test-recipes/ex1-hierarchy-basic.jl diff --git a/test/test-recipes/ex10-mpi-test.jl b/test/test-recipes/ex10-mpi-test.jl new file mode 100644 index 0000000..21b3a5f --- /dev/null +++ b/test/test-recipes/ex10-mpi-test.jl @@ -0,0 +1,28 @@ +#using MPI +using Test,PerfTest + +@perftest_config " +[regression] +enabled = false +[general] +verbose = 3 +" + +@testset "MPI Mem bandwidth" begin + + @testset "This should pass" begin + @on_perftest_exec begin + s = _PRFT_GLOBALS.builtins[:MEM_BENCH_SDAXPY] + end + + @define_metric name = "space" units = "Bytes" begin + s + end + time = 0.1 + + @define_eff_memory_throughput ratio = 0.97 mem_benchmark="MEM_BENCH_SDAXPY" begin + :space + end + @perftest samples = 5 sleep(time) + end +end \ No newline at end of file diff --git a/test/ex2-effmemtp.jl b/test/test-recipes/ex2-effmemtp.jl similarity index 96% rename from test/ex2-effmemtp.jl rename to test/test-recipes/ex2-effmemtp.jl index ce7ce4a..eaa100b 100644 --- a/test/ex2-effmemtp.jl +++ b/test/test-recipes/ex2-effmemtp.jl @@ -5,7 +5,7 @@ using PerfTest [regression] enabled = false [general] -verbose = true +verbose = 0 " function testfun(a :: Int) diff --git a/test/ex3-roofline.jl b/test/test-recipes/ex3-roofline.jl similarity index 100% rename from test/ex3-roofline.jl rename to test/test-recipes/ex3-roofline.jl diff --git a/test/ex4-perfcmp.jl b/test/test-recipes/ex4-perfcmp.jl similarity index 100% rename from test/ex4-perfcmp.jl rename to test/test-recipes/ex4-perfcmp.jl diff --git a/test/ex5-recursive.jl b/test/test-recipes/ex5-recursive.jl similarity index 100% rename from test/ex5-recursive.jl rename to test/test-recipes/ex5-recursive.jl diff --git a/test/ex6-basic-regression.jl b/test/test-recipes/ex6-basic-regression.jl similarity index 92% rename from test/ex6-basic-regression.jl rename to test/test-recipes/ex6-basic-regression.jl index 5a06aaf..e454df4 100644 --- a/test/ex6-basic-regression.jl +++ b/test/test-recipes/ex6-basic-regression.jl @@ -5,7 +5,7 @@ using PerfTest @perftest_config " [regression] enabled = true -use_bencher = true +use_bencher = false " function testfun() diff --git a/test/test-recipes/ex7-measure-tests.jl b/test/test-recipes/ex7-measure-tests.jl new file mode 100644 index 0000000..cd8b001 --- /dev/null +++ b/test/test-recipes/ex7-measure-tests.jl @@ -0,0 +1,128 @@ +# Careful changing testset names (used for parsing) + +using Test +using PerfTest + +@perftest_config " +[regression] +enabled = false +" + +@testset "Time Measurements" begin + # Test that sleep duration is measured correctly + sleep_duration = 0.1 + + # Test multiple sleep calls with varying durations + @testset "This should pass" begin + + @testset "Perfcompare" for duration in [0.1, 0.5] + @auxiliary_metric name="Duration" units="" begin + :median_time + end + @perfcompare duration * 0.9 < :median_time < duration * 1.1 + @perftest samples = 5 sleep(duration) + end + + @testset "Effective mem throughput" begin + space = 1_000_000_000 + time = 0.5 + + @define_benchmark name = "dummy" units = "Byte/s" begin + space / time + end + @define_eff_memory_throughput ratio = 0.97 custom_benchmark=dummy begin + space / :median_time + end + @perftest samples = 5 sleep(time) + end + + @testset "Custom metric" begin + @define_metric name = "apples" units = "apples" begin + 5 + end + + @auxiliary_metric name = "pears" units = "pears" begin + 8 + end + + @perfcompare (:apples < :pears) + @perftest sleep(0.1) + end + + @testset "Roofline hardcoded" begin + @roofline actual_flops=10 target_ratio=0.8 cpu_peak=120 membw_peak=100 begin + 4 / 4 + end + @perftest sleep(0.1) + end + + + @testset "Roofline perfect" begin + @on_perftest_exec begin + global _flops = _PRFT_GLOBALS.builtins[:CPU_FLOPS_PEAK]/10 + global _opint = 10 + end + + @roofline actual_flops=_flops target_ratio=0.9 begin + _opint + end + + @perftest sleep(0.1) + end + end + + @testset "This should fail" begin + + @testset "Perfcompare" for duration in [0.1, 0.5] + @perfcompare duration * 0.9 > :median_time + @perfcompare duration * 1.1 < :median_time + @perftest samples = 5 sleep(duration) + end + + @testset "Effective mem throughput" begin + space = 1_000_000_000 + time = 0.5 + + @define_benchmark name = "dummy" units = "Byte/s" begin + space / time * 2 + end + @define_eff_memory_throughput ratio = 0.97 custom_benchmark=dummy begin + space / :median_time + end + @perftest samples = 5 sleep(time) + end + + @testset "Custom metric" begin + @define_metric name = "apples" units = "apples" begin + 5 + end + + @auxiliary_metric name = "pears" units = "pears" begin + 8 + end + + @perfcompare :apples > :pears + @perftest sleep(0.1) + end + + @testset "Roofline hardcoded" begin + @roofline actual_flops=10 target_ratio=0.95 cpu_peak=120 membw_peak=100 begin + 6 / 4 + end + @perftest sleep(0.1) + end + + @testset "Roofline perfect" begin + @on_perftest_exec begin + global _flops = _PRFT_GLOBALS.builtins[:CPU_FLOPS_PEAK]/10/2 + global _opint = 10 + end + + @roofline actual_flops=_flops target_ratio=0.9 begin + _opint + end + + @perftest sleep(0.1) + end + end +end diff --git a/test/test-recipes/ex8-regression-test.jl b/test/test-recipes/ex8-regression-test.jl new file mode 100644 index 0000000..2250b49 --- /dev/null +++ b/test/test-recipes/ex8-regression-test.jl @@ -0,0 +1,50 @@ +# Careful changing testset names (used for parsing) +using Test, PerfTest + +@perftest_config " +[regression] +enabled = true +" + +execution_number = try length(PerfTest.getExecutionTimestamps(".perftests/ex8-regression-test.jl_PERFORMANCE.JLD2")) + 1 catch; 1 end + +@testset "Regression Tests" begin + + @testset "Default :median_time" begin + + @testset "This should pass" begin + time = 0.5 / execution_number + + @perftest samples = 5 sleep(time) + + end + + @testset "This should fail" begin + time = 0.1 * execution_number + + @perftest samples = 5 sleep(time) + end + + end + + @testset "Custom metric" begin + + @define_metric name = "custom_time" units = "1/s" begin + 1 / :median_time + end + # Defining regression on a parent testset will apply it to all child testsets, but it can be redefined on a child testset to apply different regression criteria + @regression threshold=0.9 low_is_bad=true metrics="custom_time" + + @testset "This should pass" begin + time = 0.5 / execution_number + + @perftest samples = 5 sleep(time) + end + + @testset "This should fail" begin + time = 0.1 * execution_number + + @perftest samples = 5 sleep(time) + end + end +end \ No newline at end of file diff --git a/test/test-recipes/ex9-no-perftests.jl b/test/test-recipes/ex9-no-perftests.jl new file mode 100644 index 0000000..e4f4f6d --- /dev/null +++ b/test/test-recipes/ex9-no-perftests.jl @@ -0,0 +1,31 @@ +using Test,PerfTest + +@perftest_config " +[regression] +enabled = false +[general] +verbose = 0 +" + +@define_benchmark name = "dummy" units = "units" begin + 42 +end + +@testset "No perf tests" begin + + @regression low_is_bad=false + + @perfcompare :dummy > 40 + + @roofline actual_flops=:autoflop target_ratio=0.1 begin + :autoflop / 42 + end + + @perftest begin + sleep(0.1) + end + + @testset "This should pass" begin + @test 1 + 1 == 2 + end +end \ No newline at end of file