Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -5,4 +5,6 @@
/docs/build/
# /docs/Manifest.toml
/docs/src/index.md
.vscode/
.vscode/
/benchmark/Manifest.toml
/benchmark/.CondaPkg/
7 changes: 7 additions & 0 deletions benchmark/CondaPkg.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
[deps]
python = "3.12.*"

[pip.deps]
graspologic = "==3.4.4"
networkx = ">=3.0"
scipy = ">=1.10"
24 changes: 24 additions & 0 deletions benchmark/Project.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
[deps]
CairoMakie = "13f3f980-e62b-5c42-98c6-ff1f3baf88f0"
CondaPkg = "992eb4ea-22a4-4c89-a5bb-47a3300528ab"
Graphs = "86223c79-3864-5bf0-83f7-82e725a168b6"
GraphsOptim = "e79ef3ae-79c8-4b97-b165-63f338db30c2"
HiGHS = "87dc4568-4c63-4d18-b0c0-bb2238e4078b"
JuMP = "4076af6c-e467-56ae-b986-b466b2749572"
LinearAlgebra = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e"
PythonCall = "6099a3de-0909-46bc-b1f4-468b9a2dfc0d"
Random = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c"
SparseArrays = "2f01184e-e22b-5df5-ae63-d93ebab69eaf"
Statistics = "10745b16-79ce-11e8-11f9-7d13ad32a3b2"

[sources]
GraphsOptim = {path = ".."}

[compat]
CairoMakie = "0.15"
CondaPkg = "0.2"
Graphs = "1.15"
HiGHS = "1"
JuMP = "1"
PythonCall = "0.9"
julia = "1.11"
25 changes: 25 additions & 0 deletions benchmark/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
# Benchmarks

Benchmarks of GraphsOptim.jl against other Julia and Python libraries (Graphs.jl, graspologic, NetworkX, SciPy), in terms of running time and solution quality.
The results are shown on the [Benchmarks](https://juliagraphs.org/GraphsOptim.jl/dev/benchmarks/) page of the documentation, which also describes the methodology.

| File | Content |
| :--- | :--- |
| `run.jl` | entry point, runs the suites and draws the figures |
| `suites/*.jl` | one suite per family of problems: instances, methods and metrics |
| `common.jl` | shared tools: timing, calls to Python, CSV files |
| `plots.jl` | figures, saved in `docs/src/assets/benchmarks/` |
| `results/*.csv` | raw results |

To update the results and figures, run from the root of the repository:

```bash
julia --project=benchmark -e 'using Pkg; Pkg.instantiate()'
julia --project=benchmark benchmark/run.jl # all suites
julia --project=benchmark benchmark/run.jl cliques # a single suite
julia --project=benchmark benchmark/run.jl --plots-only # only redraw the figures
```

The Python libraries are installed in `benchmark/.CondaPkg` by [CondaPkg.jl](https://github.com/JuliaPy/CondaPkg.jl), and called through [PythonCall.jl](https://github.com/JuliaPy/PythonCall.jl).

To add a suite, create `suites/<name>.jl` with a function `run_<name>()` writing its CSV files, add a `plot_<name>()` function to `plots.jl`, register both in `SUITES` in `run.jl`, and add a section to `docs/src/benchmarks.md`.
158 changes: 158 additions & 0 deletions benchmark/common.jl
Original file line number Diff line number Diff line change
@@ -0,0 +1,158 @@
# Tools shared by all benchmark suites.

# All libraries use a single BLAS thread for a fair comparison.
# This must be set before Python and NumPy are loaded.
ENV["OPENBLAS_NUM_THREADS"] = "1"

using Graphs
using GraphsOptim
using HiGHS
using JuMP
using LinearAlgebra
using PythonCall
using Random
using SparseArrays
using Statistics

BLAS.set_num_threads(1)

const RESULTS_DIR = joinpath(@__DIR__, "results")
const FIGURES_DIR = joinpath(dirname(@__DIR__), "docs", "src", "assets", "benchmarks")

"Time limit (in seconds) given to the optimizers and to the exponential-time algorithms."
const TIME_LIMIT = 60.0

"HiGHS with the benchmark time limit."
const OPTIMIZER = optimizer_with_attributes(HiGHS.Optimizer, "time_limit" => TIME_LIMIT)

## Timing

"""
measure(f; min_time=1.0, max_reps=7)

Run `f()` once (after compilation by a previous call) and repeat it until `min_time` seconds or `max_reps` runs are reached.
Return the median running time and the result of the last run, or `(NaN, nothing)` if `f` fails, e.g. because a time limit was hit.
"""
function measure(f; min_time=1.0, max_reps=7)
times = Float64[]
local result
try
while length(times) < max_reps && (isempty(times) || sum(times) < min_time)
t = @elapsed result = f()
push!(times, t)
end
catch e
e isa InterruptException && rethrow()
@warn "Run failed" exception = (e, catch_backtrace())
return NaN, nothing
end
return median(times), result
end

pyexec(
"""
import time

def measure(f, args, min_time=1.0, max_reps=7, **kwargs):
times = []
while len(times) < max_reps and (not times or sum(times) < min_time):
t = time.perf_counter()
result = f(*args, **kwargs)
times.append(time.perf_counter() - t)
times.sort()
n = len(times)
median = times[n // 2] if n % 2 else (times[n // 2 - 1] + times[n // 2]) / 2
return median, result
""",
Main,
)

"""
pymeasure(f, args...; kwargs...)

Same as [`measure`](@ref) for the Python function `f`, timed on the Python side so that the conversion of the arguments is not counted.
"""
function pymeasure(f, args...; kwargs...)
t, result = pyeval("measure", Main)(f, pytuple(args); kwargs...)
return pyconvert(Float64, t), result
end

## Conversions

"Convert an undirected or directed `Graphs.jl` graph into a NetworkX graph with vertices `0:n-1`."
function to_networkx(g::AbstractGraph; vertex_weights=nothing)
nx = pyimport("networkx")
h = is_directed(g) ? nx.DiGraph() : nx.Graph()
h.add_nodes_from(0:(nv(g) - 1))
h.add_edges_from([(src(e) - 1, dst(e) - 1) for e in edges(g)])
if !isnothing(vertex_weights)
for v in vertices(g)
h.nodes[v - 1]["weight"] = vertex_weights[v]
end
end
return h
end

## Results

function write_csv(file, rows)
mkpath(RESULTS_DIR)
open(joinpath(RESULTS_DIR, file), "w") do io
println(io, join(keys(first(rows)), ","))
for row in rows
println(io, join(values(row), ","))
end
end
return nothing
end

function read_csv(file)
lines = readlines(joinpath(RESULTS_DIR, file))
header = Tuple(Symbol.(split(lines[1], ",")))
parsevalue(x) = something(tryparse(Int, x), tryparse(Float64, x), String(x))
return [
NamedTuple{header}(Tuple(parsevalue.(split(line, ",")))) for line in lines[2:end]
]
end

function write_environment()
commit = try
readchomp(`git -C $(@__DIR__) describe --always --dirty`)
catch
"unknown"
end
version(mod) = pyconvert(String, pyimport(mod).__version__)
open(joinpath(RESULTS_DIR, "environment.md"), "w") do io
println(io, "- CPU: ", Sys.cpu_info()[1].model, " (single BLAS thread)")
println(io, "- Julia ", VERSION, ", GraphsOptim.jl at commit `", commit, "`")
println(
io,
"- Graphs.jl ",
pkgversion(Graphs),
", JuMP.jl ",
pkgversion(JuMP),
", HiGHS.jl ",
pkgversion(HiGHS),
)
println(
io,
"- Python ",
pyconvert(String, pyimport("platform").python_version()),
", graspologic ",
version("graspologic"),
", NetworkX ",
version("networkx"),
", SciPy ",
version("scipy"),
)
end
return nothing
end

"""
too_slow(time)

Whether a method should be skipped on larger instances, because it failed or took more than a tenth of the time limit.
Exponential-time algorithms written in Python cannot be interrupted, so they are stopped early.
"""
too_slow(time) = isnan(time) || time > TIME_LIMIT / 10
Loading
Loading