Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,7 @@ G.add_edge("B", "C", weight=2)
G.add_edge("C", "D", weight=1)

# One terminal group = Steiner tree; multiple groups = Steiner forest
# Edge/arc costs must be non-negative.
solution = SteinerProblem(G, [["A", "D"]]).get_solution()

print(f"Optimal cost: {solution.objective}")
Expand Down
14 changes: 14 additions & 0 deletions benchmarks/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -46,6 +46,20 @@ skipped and rows are appended, so a preempted sweep is restarted with the same
command. Per-instance solve time is bounded by `--time-limit`; for hard
wall-clock isolation on a cluster, the Slurm array runs one instance per task.

### Dreyfus-Wagner memory benchmark

`benchmark_dreyfus_memory.py` compares the few-terminal dynamic program across
source checkouts with fixed generated graphs. Every repetition runs in a fresh
process so peak RSS is isolated, and the CSV records the Python, NetworkX, and
SciPy versions used:

```bash
python benchmarks/benchmark_dreyfus_memory.py \
--source-root /path/to/checkout --label candidate \
--k 6,7,8,9,10 --sizes 300,1200 --densities 3,8 --repeats 3 \
--output candidate.csv
```

## Output columns

`instance, nodes, edges, terminals, opt, base_obj, base_rt, base_gap,
Expand Down
196 changes: 196 additions & 0 deletions benchmarks/benchmark_dreyfus_memory.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,196 @@
"""Reproducible Dreyfus-Wagner runtime and peak-RSS benchmark.

Each repetition runs in a fresh subprocess so ``ru_maxrss`` is isolated. Use
the same interpreter with two source checkouts to compare implementations.
"""

from __future__ import annotations

import argparse
import csv
import json
import os
import resource
import statistics
import subprocess
import sys
import time
from pathlib import Path


DEFAULT_K = (6, 7, 8, 9, 10)
DEFAULT_SIZES = (300, 1200)
DEFAULT_DENSITIES = (3, 8)


def _connected_graph(nx, n: int, edge_factor: int, k: int, seed: int):
"""Deterministic connected sparse graph with exactly about factor*n edges."""
import random

rng = random.Random(seed)
graph = nx.Graph()
graph.add_nodes_from(range(n))
order = list(range(n))
rng.shuffle(order)
for i in range(1, n):
parent = order[rng.randrange(i)]
graph.add_edge(order[i], parent, weight=rng.randint(1, 20))

target = min(n * (n - 1) // 2, edge_factor * n)
while graph.number_of_edges() < target:
u, v = rng.sample(range(n), 2)
if not graph.has_edge(u, v):
graph.add_edge(u, v, weight=rng.randint(1, 20))
terminals = rng.sample(range(n), k)
return graph, terminals


def _worker(args) -> None:
sys.path.insert(0, str(Path(args.source_root).resolve()))
import networkx as nx
import scipy

from steinerpy.dreyfus_wagner import dreyfus_wagner

graph, terminals = _connected_graph(
nx, args.nodes, args.edge_factor, args.k, args.seed
)
started = time.perf_counter()
objective, edges = dreyfus_wagner(graph, terminals)
runtime = time.perf_counter() - started

selected = nx.Graph()
selected.add_nodes_from(terminals)
selected.add_edges_from(edges)
feasible = len(terminals) <= 1 or all(
nx.has_path(selected, terminals[0], terminal) for terminal in terminals[1:]
)
raw_rss = resource.getrusage(resource.RUSAGE_SELF).ru_maxrss
rss_bytes = raw_rss if sys.platform == "darwin" else raw_rss * 1024
print(
json.dumps(
{
"runtime_s": runtime,
"peak_rss_mib": rss_bytes / (1024 * 1024),
"objective": objective,
"edge_count": len(edges),
"feasible": feasible,
"python": sys.version.split()[0],
"networkx": nx.__version__,
"scipy": scipy.__version__,
}
)
)


def _parse_ints(value: str):
return tuple(int(item) for item in value.split(",") if item)


def _coordinator(args) -> None:
rows = []
script = str(Path(__file__).resolve())
for k in args.k:
for n in args.sizes:
for factor in args.densities:
runtimes = []
peaks = []
reference = None
for _repetition in range(args.repeats):
seed = args.seed + 100_000 * k + 1_000 * n + factor
command = [
sys.executable,
script,
"--worker",
"--source-root",
args.source_root,
"--k",
str(k),
"--nodes",
str(n),
"--edge-factor",
str(factor),
"--seed",
str(seed),
]
completed = subprocess.run(
command,
check=True,
text=True,
capture_output=True,
env={**os.environ, "PYTHONHASHSEED": "0"},
)
result = json.loads(completed.stdout.strip().splitlines()[-1])
if not result["feasible"]:
raise RuntimeError(
f"infeasible reconstruction for k={k}, n={n}, "
f"factor={factor}"
)
signature = (result["objective"], result["edge_count"])
if reference is None:
reference = signature
elif signature != reference:
raise RuntimeError(
f"non-deterministic result for k={k}, n={n}, "
f"factor={factor}: {signature} != {reference}"
)
runtimes.append(result["runtime_s"])
peaks.append(result["peak_rss_mib"])

assert reference is not None
rows.append(
{
"label": args.label,
"k": k,
"nodes": n,
"edge_factor": factor,
"edges": min(n * (n - 1) // 2, factor * n),
"seed": seed,
"repeats": args.repeats,
"median_runtime_s": statistics.median(runtimes),
"median_peak_rss_mib": statistics.median(peaks),
"objective": reference[0],
"python": result["python"],
"networkx": result["networkx"],
"scipy": result["scipy"],
}
)

fieldnames = list(rows[0])
destination = Path(args.output) if args.output else None
handle = destination.open("w", newline="") if destination else sys.stdout
try:
writer = csv.DictWriter(handle, fieldnames=fieldnames)
writer.writeheader()
writer.writerows(rows)
finally:
if destination:
handle.close()


def _parser():
parser = argparse.ArgumentParser()
parser.add_argument("--worker", action="store_true")
parser.add_argument("--source-root", default=".")
parser.add_argument("--label", default="current")
parser.add_argument("--output")
parser.add_argument("--repeats", type=int, default=3)
parser.add_argument("--k", type=_parse_ints, default=DEFAULT_K)
parser.add_argument("--sizes", type=_parse_ints, default=DEFAULT_SIZES)
parser.add_argument("--densities", type=_parse_ints, default=DEFAULT_DENSITIES)
parser.add_argument("--nodes", type=int)
parser.add_argument("--edge-factor", type=int)
parser.add_argument("--seed", type=int, default=20260904)
return parser


if __name__ == "__main__":
args = _parser().parse_args()
if args.worker:
if isinstance(args.k, tuple):
if len(args.k) != 1:
raise SystemExit("--worker requires one --k")
args.k = args.k[0]
_worker(args)
else:
_coordinator(args)
18 changes: 18 additions & 0 deletions docs/source/guide/solvers.md
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,24 @@ solution = SteinerProblem(graph, terminal_groups).get_solution(solver="gurobi")
Both solvers implement the same cut-based (DO-D) formulation from Markhorst et al. (2025) and produce identical optimal solutions.
Gurobi may be faster on larger instances because callbacks avoid repeated re-solves from scratch.

## Input and certificate guarantees

All edge-cost Steiner variants require non-negative values in the selected
edge/arc weight attribute. This is required by the shortest-path reductions
and exact formulations.
Negative node weights remain valid for maximum-weight connected-subgraph
problems. Inputs with a negative edge or arc cost raise `ValueError` at
construction time instead of risking a disconnected negative-cost cycle in the
reported solution.

A time limit is not an optimality certificate. If a solve stops early with an
independently validated feasible incumbent, `get_solution()` may return it
with a nonzero or unknown gap (`math.inf`). If no valid incumbent exists—or
the last incumbent predates newly added connectivity cuts—the values are
discarded and the public solve raises `RuntimeError`. In particular,
`solution.gap == 0` is reported only after the solver or an independent exact
bound proves optimality.

## Enumerating multiple optimal solutions

`get_solution()` returns exactly one optimal Steiner tree. When multiple
Expand Down
8 changes: 8 additions & 0 deletions docs/source/guide/variants.md
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,14 @@ Simply pass a list of terminal lists as `terminal_groups` — one list for a tre

Several of these variants implement the "further related problems" of Chapter 5 of D. Rehfeldt's PhD thesis (*Faster algorithms for Steiner tree and related problems*, TU Berlin 2021), reusing the same directed-cut kernel by transformation.

All edge and arc costs used by these exact Steiner formulations must be
non-negative. This restriction does not apply to node *weights* in
`MaxWeightConnectedSubgraph` or
`BudgetedMaxWeightConnectedSubgraph`, where negative values mean that a node
is an undesirable connector and are part of the problem semantics. MWCSPB also
treats edge attributes as topology-only; its objective and budget use node
weights and node costs.

```python
from steinerpy import (
PartialTerminalSteinerProblem, FullTerminalSteinerProblem, GroupSteinerProblem,
Expand Down
Loading
Loading