Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
27 changes: 23 additions & 4 deletions .github/scripts/compare_regression.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,8 +17,12 @@
)


COUNTER = "jit_kernel_compiles"


def load_runs(path):
runs = {}
counters = {}
failed = set()
with open(path, newline="", encoding="utf-8") as source:
for row in csv.DictReader(source):
Expand All @@ -29,12 +33,20 @@ def load_runs(path):
failed.add(key)
continue
runs[key] = float(row["time"])
return runs, failed - set(runs)
if row.get(COUNTER):
counters[key] = float(row[COUNTER])
return runs, counters, failed - set(runs)


def report(base_path, candidate_path, threshold):
base, _ = load_runs(base_path)
candidate, candidate_failed = load_runs(candidate_path)
base, base_counters, _ = load_runs(base_path)
candidate, candidate_counters, candidate_failed = load_runs(candidate_path)
# Deterministic, so any increase is a real change -- no threshold needed.
recompiles = [
(key, base_counters[key], candidate_counters[key])
for key in sorted(candidate_counters)
if key in base_counters and candidate_counters[key] > base_counters[key]
]
# Not fatal: a PR that fixes a crash on base would always fail here.
missing = sorted(set(candidate) - set(base))

Expand All @@ -57,6 +69,10 @@ def report(base_path, candidate_path, threshold):
if change > threshold:
regressions.append((key, change))

if recompiles:
lines.extend(["", "**More JIT kernels compiled than base:**"])
lines.extend(f"- {k[0]}/{k[1]}: {int(b)} -> {int(c)}"
for k, b, c in recompiles)
for label, keys in (("Failed on this PR", candidate_failed),
("Not compared (no base result)", missing)):
if keys:
Expand All @@ -76,7 +92,10 @@ def report(base_path, candidate_path, threshold):
print(f"failed on this PR: {key[0]}/{key[1]}", file=sys.stderr)
for key, change in regressions:
print(f"regression: {key[0]}/{key[1]} is {change:.1f}% slower", file=sys.stderr)
return 1 if (regressions or candidate_failed) else 0
for key, b, c in recompiles:
print(f"more JIT compiles: {key[0]}/{key[1]} {int(b)} -> {int(c)}",
file=sys.stderr)
return 1 if (regressions or candidate_failed or recompiles) else 0


def main():
Expand Down
25 changes: 25 additions & 0 deletions .github/scripts/regression-dynamic.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
# Dynamic suite: separate config because benchmark_root is global per config.
# Rows append to the same CSVs as regression.toml, so one comparison covers both.
[runner]
benchmark_root = "dynamic-benchmarks"
dpus = [64]
warmup = 0 # dynamic workloads recompile; cold start is the measurement
iterations = 20
ntrials = 1
seed = 1
variants = ["polymerpim"]

[setup]
variants = ["polymerpim"]
commands = [
"make -C \"{repo}\" install DESTDIR=\"{install}\" PIPELINE=1 JIT=1 LOGGING=3 TRACE=0 JULIA=\"{julia}\" {fusion_flags}",
]

[[adaptive_image]]
elements_per_dpu = [262144]
parameters = { channels = 4, check_interval = 2, tolerance = 2 }

[[dynamic_query]]
iterations = 10
elements_per_dpu = [16384]
parameters = { columns = 4, projections = 4, batches_per_query = 5, query_ops = 6 }
8 changes: 8 additions & 0 deletions .github/workflows/benchmark-regression.yml
Original file line number Diff line number Diff line change
Expand Up @@ -92,6 +92,10 @@ jobs:
--default-params --runner \
--state "$RUNNER_TEMP/benchmark-regression/base/state.toml" \
--csv "$RUNNER_TEMP/benchmark-regression/base/runs.csv"
./run.sh --config "$GITHUB_WORKSPACE/candidate/.github/scripts/regression-dynamic.toml" \
--default-params --runner \
--state "$RUNNER_TEMP/benchmark-regression/base/state-dynamic.toml" \
--csv "$RUNNER_TEMP/benchmark-regression/base/runs.csv"

- name: Run candidate benchmarks
working-directory: candidate/benchmarks
Expand All @@ -100,6 +104,10 @@ jobs:
--default-params --check --runner \
--state "$RUNNER_TEMP/benchmark-regression/candidate/state.toml" \
--csv "$RUNNER_TEMP/benchmark-regression/candidate/runs.csv"
./run.sh --config "$GITHUB_WORKSPACE/candidate/.github/scripts/regression-dynamic.toml" \
--default-params --runner \
--state "$RUNNER_TEMP/benchmark-regression/candidate/state-dynamic.toml" \
--csv "$RUNNER_TEMP/benchmark-regression/candidate/runs.csv"

- name: Compare performance
run: |
Expand Down
7 changes: 7 additions & 0 deletions benchmarks/src/timings.jl
Original file line number Diff line number Diff line change
Expand Up @@ -2,9 +2,12 @@ const APP_TIME_FORMAT = "APP_TIME real=%e user=%U sys=%S maxrss=%M"
const APP_TIME_RE = r"APP_TIME real=([\d.]+) user=([\d.]+) sys=([\d.]+) maxrss=(\d+)"
const STAGE_NAMES = ("alloc", "load", "transpose", "init", "write", "kernel",
"read", "merge", "query_first", "query_reuse")
const RUNTIME_COUNTERS = ("compute_launches", "jit_kernel_compiles",
"binary_switches")
const MEASURE_COLUMNS = [
"time", "stddev", "min", "max", "warmup_ms",
"real_s", "user_s", "sys_s", "max_rss_kb",
RUNTIME_COUNTERS...,
]
const RUN_COLUMNS = [
"timestamp", "invocation", "benchmark", "variant", "phase", "status",
Expand Down Expand Up @@ -115,6 +118,10 @@ function parse_timings(label::AbstractString, out::AbstractString,
values["$(stage)_cold_ms"] = parsed_value(
Regex("$(escaped)_cold_stage_$stage\\s*\\(ms\\):\\s*([0-9.]+)"), out)
end
# Deterministic run-to-run, unlike the timings, so a change here is signal.
for counter in RUNTIME_COUNTERS
values[counter] = parsed_value(Regex("\\b$counter=([0-9]+)"), out)
end
return values
end

Expand Down
2 changes: 1 addition & 1 deletion benchmarks/test/tests/config_cli.jl
Original file line number Diff line number Diff line change
Expand Up @@ -53,7 +53,7 @@

modes = load_config(joinpath(
BENCHMARKS, "main-benchmarks", "polymerpim-modes.toml"))
@test modes.benchmark_names == ["elementwise", "knn", "linreg"]
@test modes.benchmark_names == ["knn", "linreg"]
@test modes.defaults.variants ==
["polymerpim-jit", "polymerpim-pipeline", "polymerpim-eager"]
@test modes.defaults.group_by_variant
Expand Down
2 changes: 1 addition & 1 deletion benchmarks/test/tests/execution.jl
Original file line number Diff line number Diff line change
Expand Up @@ -75,7 +75,7 @@ end
@testset "mode suite groups setup" begin
text = captured_output() do
run_cli([
"elementwise", "--config",
"knn", "--config",
joinpath(BENCHMARKS, "main-benchmarks", "polymerpim-modes.toml"),
"--dpus", "2", "--elements-per-dpu", "64,128",
"--warmup", "0", "--iterations", "1", "--ntrials", "1",
Expand Down
14 changes: 13 additions & 1 deletion host/polymerpim.cc
Original file line number Diff line number Diff line change
Expand Up @@ -41,6 +41,8 @@ struct Node {
// on its value: that would make the program shape depend on the data and
// force a JIT recompile whenever the value changes.
bool structural = false;
// Cached subtree op count; the deferral cap reads it once per op.
size_t ops = 0;
std::vector<std::shared_ptr<Node>> children;
};

Expand All @@ -49,6 +51,10 @@ using NodeRef = std::shared_ptr<Node>;
NodeRef node(ExprOp op, std::vector<NodeRef> children = {}) {
auto result = std::make_shared<Node>();
result->op = op;
result->ops = (op == ExprOp::input || op == ExprOp::scalar) ? 0 : 1;
for (const auto& child : children) {
if (child) result->ops += child->ops;
}
result->children = std::move(children);
return result;
}
Expand Down Expand Up @@ -103,6 +109,9 @@ uint8_t binary_opcode(ExprOp op) {
}
}

// Ops a pending expression contributes to a fused program.
size_t expression_ops(const NodeRef& value) { return value ? value->ops : 0; }

uint8_t scalar_opcode(ExprOp op) {
switch (op) {
case ExprOp::add:
Expand Down Expand Up @@ -578,7 +587,10 @@ size_t DPUVector<T>::size() const { return impl_ ? impl_->size() : 0; }

DPUVector<T>::operator DpuLazy<T>() const {
if (!impl_) return {};
if (impl_->pending && !impl_->consumed) {
// Past the cap the chain cannot fuse into one kernel, so extending it only
// mints another program shape for the JIT to compile.
if (impl_->pending && !impl_->consumed &&
expression_ops(impl_->pending) < MAX_VFUSE_OPS) {
impl_->consumed = true;
return DpuLazy<T>(std::make_shared<DpuLazy<T>::Impl>(impl_->pending));
}
Expand Down
3 changes: 3 additions & 0 deletions host/polymerpim.h
Original file line number Diff line number Diff line change
Expand Up @@ -261,6 +261,9 @@ void init(uint32_t dpus);
uint32_t ndpus();
uint32_t ntasklets();
void sync();
// sync() drains queued work; it does not materialise a vector's pending
// expression. `x = f(x)` in a loop therefore extends one chain, and each depth
// is a distinct JIT kernel -- fence(x) per iteration keeps it to one.
void fence(DPUVector<int32_t>& vector);
void shutdown();

Expand Down
Loading