From 80702d539f27667c3ffd34410f95aba51bc94c05 Mon Sep 17 00:00:00 2001 From: Frank McSherry Date: Sun, 6 Sep 2026 20:19:45 -0400 Subject: [PATCH 1/8] Add a reproducible LDBC-derived workload through ddir_server --- .github/workflows/test.yml | 9 + interactive/server/README.md | 6 + interactive/server/bench/ldbc/.gitignore | 2 + interactive/server/bench/ldbc/README.md | 151 +++++++++++++++ interactive/server/bench/ldbc/bi11.ddp | 18 ++ interactive/server/bench/ldbc/bi5.ddp | 22 +++ interactive/server/bench/ldbc/client.py | 165 ++++++++++++++++ interactive/server/bench/ldbc/compare.py | 47 +++++ interactive/server/bench/ldbc/graph.ddp | 28 +++ interactive/server/bench/ldbc/ic6.ddp | 23 +++ interactive/server/bench/ldbc/is3.ddp | 9 + interactive/server/bench/ldbc/run.py | 226 ++++++++++++++++++++++ interactive/server/bench/ldbc/tiny.json | 9 + interactive/server/bench/ldbc/workload.py | 185 ++++++++++++++++++ 14 files changed, 900 insertions(+) create mode 100644 interactive/server/bench/ldbc/.gitignore create mode 100644 interactive/server/bench/ldbc/README.md create mode 100644 interactive/server/bench/ldbc/bi11.ddp create mode 100644 interactive/server/bench/ldbc/bi5.ddp create mode 100644 interactive/server/bench/ldbc/client.py create mode 100644 interactive/server/bench/ldbc/compare.py create mode 100644 interactive/server/bench/ldbc/graph.ddp create mode 100644 interactive/server/bench/ldbc/ic6.ddp create mode 100644 interactive/server/bench/ldbc/is3.ddp create mode 100644 interactive/server/bench/ldbc/run.py create mode 100644 interactive/server/bench/ldbc/tiny.json create mode 100644 interactive/server/bench/ldbc/workload.py diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index e774c3b18..d873e0617 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -39,6 +39,15 @@ jobs: run: cargo test --workspace --all-targets - name: Cargo doc test run: cargo test --doc + # One small server-backed workload; no data download or timing thresholds. + - name: LDBC-derived correctness smoke + if: matrix.os == 'ubuntu' && matrix.toolchain == 'stable' + run: | + cargo build -p ddir-server + for workers in 1 4; do + python3 interactive/server/bench/ldbc/run.py --server target/debug/ddir_server \ + --workers "$workers" --rounds 1 --warmup 0 --batch-size 9 --changes 1 + done # The explanation suite's heavier sweeps (soundness regression, fuzz) # are #[ignore]d for local debug runs but must gate merges. The metric # report is skipped (it prints, it does not assert). diff --git a/interactive/server/README.md b/interactive/server/README.md index 9e5bc7209..e9aa9c44f 100644 --- a/interactive/server/README.md +++ b/interactive/server/README.md @@ -170,3 +170,9 @@ fronting proxy if a deployment ever needs one. support belongs on the scope-tree explanation machinery, and until that lands the server reports an error rather than giving those commands an improvised meaning. + +## Benchmarking + +The [LDBC-derived workload](bench/ldbc/README.md) runs parameterized interactive +queries and maintained BI queries together through this server. It includes a +tiny correctness fixture and can read an existing SNB CSV snapshot for timing. diff --git a/interactive/server/bench/ldbc/.gitignore b/interactive/server/bench/ldbc/.gitignore new file mode 100644 index 000000000..2b427312d --- /dev/null +++ b/interactive/server/bench/ldbc/.gitignore @@ -0,0 +1,2 @@ +__pycache__/ +results/ diff --git a/interactive/server/bench/ldbc/README.md b/interactive/server/bench/ldbc/README.md new file mode 100644 index 000000000..ab191e79b --- /dev/null +++ b/interactive/server/bench/ldbc/README.md @@ -0,0 +1,151 @@ +# LDBC-derived server workload + +A small performance/correctness baseline for DDIR, through the real +`ddir_server` TCP interface. No server extensions, external database, generator, +or Python packages are required. Python 3.9+ and `ps` are required by the runner. + +This is **not an official LDBC benchmark result**. It implements a four-query +panel with a controlled update/request schedule, not the LDBC driver's arrival +distribution, update stream, conformance checks, or throughput scoring. + +## Run + +From the repository root: + +```sh +cargo build --release -p ddir-server +python3 interactive/server/bench/ldbc/run.py --server target/release/ddir_server +``` + +The default uses the 50-row hand-authored fixture, both backends (separate +processes), one worker, one warmup round and five measured rounds. It prints +phase medians and the location of a JSON report and server logs. This fixture +is for correctness and overhead, not evidence of columnar throughput. + +Use an existing SNB BI **CSV composite-merged-fk** initial snapshot for timing: + +```sh +python3 interactive/server/bench/ldbc/run.py --server target/release/ddir_server \ + --snapshot /path/to/graphs/csv/bi/composite-merged-fk/initial_snapshot \ + --backend corgi --workers 4 --rounds 10 --batch-size 8 --changes 32 \ + --output /tmp/ldbc-baseline +``` + +`--output` must be a new directory. The adapter reads existing files only; it +does not acquire data or modify a database. It accepts the same layout at small +and large scale factors. Missing table directories fail rather than silently +becoming empty tables. Only seven relations/needed columns are loaded; the +reported projected row counts and data hash define the actual input. + +`--queries ic6` isolates one query over the same common graph substrate; +`--queries bi5 bi11` measures BI maintenance without interactive consumers. +Compare those controls with the default concurrent panel. `--batch-size 1` +measures single bindings **per interactive query**; in the default panel that +still means two simultaneous requests per tick. Increase batch size separately +from data size to study dispatch amortization. Defaults enforce a 120-second +command deadline and a sampled 6-GiB **server** RSS ceiling. The latter is not +a hard memory limit and excludes the Python loader/reference process. + +## Queries and lifecycle + +| Plan | Work exercised | Binding lifetime | +| --- | --- | --- | +| IS3 | Full friend list, names, descending creation-date order | One request batch | +| IC6 | Distinct one/two-hop authors, post/tag joins, grouped counts, top ten | One request batch | +| BI5 | Tagged posts/comments, immediate replies and likes, score aggregation, top 100 | Whole run | +| BI11 | Country/date-filtered unordered friendship triangles, including a zero result | Whole run | + +The readable `.ddp` files specify the plans; there is no hidden query compiler. +The query semantics follow the SNB read definitions. In particular BI5 counts +comments as well as posts, and all three BI11 friendship dates must fall in the +inclusive interval. Reference implementations: +[BI5](https://github.com/ldbc/ldbc_snb_bi/blob/47dd38b40844ecdb0e42e5a610c369535304786d/umbra/queries/bi-5.sql), +[BI11](https://github.com/ldbc/ldbc_snb_bi/blob/47dd38b40844ecdb0e42e5a610c369535304786d/umbra/queries/bi-11.sql). + +All selected plans are installed once and import a shared graph program. +Their common substrate symmetrizes friendships, attaches tag names, and maps +people to countries; its load and maintenance costs are included. Even isolated +controls retain this fixed substrate. This uses existing named-import semantics; +it does not introduce native Corgi trace sharing or a new prepared-query feature. + +Each round runs requests on the initial graph, retracts those bindings, removes +a deterministic bounded set of friendship/tag/like edges, runs new requests on +the changed graph, retracts them, and restores the exact removed edges. The +dataset oscillates between two states: **bounded churn, not growth or a +time-bounded saturation run**. No node deletion/cascade semantics are assumed. +Every `feed` group ends with `tick`; acknowledgement of that tick, not of feed +admission, establishes completion. + +BI parameters remain bound throughout; their maintenance is paid even when +unread. Interactive bindings are absent during graph mutations. Their parameter +bank spans degree quantiles and an absent person. IC6 tags are chosen from +reachable multi-tag posts when possible, to exercise joins and ranking rather +than mostly empty lookups. It is a data-derived bank, not the official parameter +distribution; the runner prints its nonempty-answer coverage. +Bindings vary by round and state; request IDs are reused after retraction. +They are not maintained answers to future requests. Batch size controls how +many bindings coexist. The report records the exact bank, bindings and deltas. +This is a closed-loop single-client workload, not a maximum-QPS claim. + +An independent Python graph traversal/counting oracle checks every returned +row, order, rank and multiplicity, including empty results after release and BI +results after mutation/restoration. Oracle construction/validation is untimed. +The tiny fixture also has hand-counted reference answers. Larger snapshots use +the same oracle and plans, not a row-count-only check. This is still not a full +LDBC conformance suite. + +## Read the measurements + +- Setup records plan install/parse, input admission and initial `tick` separately. +- `bind`: feed all interactive bindings and wait for `tick`. +- `read:*`: `peek` and decode the full answer. No top-k work is done by the client. +- `batch_answers`: sum of bind and read phases, excluding reference checks; + this is the complete batch-answer latency, **not the bind time alone**. +- `release`: retract all bindings and wait for `tick`. Do not omit this from + a sustained-work accounting. Subsequent `empty:*` reads check cleanup. +- `update` / `restore`: graph feeds and `tick`, including maintenance of both + standing BI results and unbound interactive plans. `maintained:*` reads are + separate; subtracting them does not remove the maintenance cost. + +Events split command preparation, encoding, wire wait/receipt and Value decoding. +Wire time includes server work and TCP overhead; it is not a kernel profile. +Medians/min/max exclude warmup; raw samples, result bytes, reference answers, +sampled server RSS, machine metadata, binary/source/data hashes, repository +revision/status and the available Cargo lockfile accompany the report. Keep the +same data, plans, workers and schedule when comparing binaries. Use release +builds and an otherwise quiet machine. The repository's pinned dependencies are +used; no patch to an external Corgi checkout is required. + +After running a candidate binary with the same arguments, compare the reports: + +```sh +python3 interactive/server/bench/ldbc/compare.py \ + /tmp/ldbc-baseline/report.json /tmp/ldbc-candidate/report.json +``` + +The comparison rejects failed runs, changed data/parameters/schedules/answers, +different worker counts or environments, and changed measurement code. Changed +DDP plans are allowed and called out: plan improvements are benchmark subjects +too. Ratios above one mean faster. Keep setup and release costs in view, and +repeat runs before treating small differences as improvements. + +## CI and follow-ups + +CI runs only the fixture, both backends, at one and four workers, with the whole +parameter bank. It asserts answers, not timing thresholds, and uses this runner +rather than a separate simulation. The equivalent local command is: + +```sh +python3 interactive/server/bench/ldbc/run.py --server target/debug/ddir_server \ + --workers 4 --rounds 1 --warmup 0 --batch-size 9 --changes 1 +``` + +First optimization controls: IC6's expand-before-tag-filter plan, BI5's +`collect`/`fold` sums, repeated scalar compilation/dispatch, and shared-import +row/column conversion. Keep their patches independent and compare against this +baseline before stacking them. Add the remaining queries in small panels, +especially recursive, optional/SUM, temporal and large-output cases. This panel +does not yet measure ad-hoc re-planning, parameterized BI, simultaneous bound-IC +updates, or the official update stream. Empty strings are rejected by this narrow +adapter: general nullable/message-content queries need typed server admission, +not padding the data to evade the current shape-inference limitation. diff --git a/interactive/server/bench/ldbc/bi11.ddp b/interactive/server/bench/ldbc/bi11.ddp new file mode 100644 index 000000000..3b99ed61c --- /dev/null +++ b/interactive/server/bench/ldbc/bi11.ddp @@ -0,0 +1,18 @@ +-- BI11: standing request(request_id, country_name, start, end), inclusive dates. +-- Count each unordered friendship triangle once; a missing triangle count is 0. +let request = input 0; +let resident = import "resident"; +let edge = import "edge"; +let people = request | key($0[1] ; $0[0], $0[2], $0[3]) + | join(resident, ($2[0] ; $1[0], $1[1], $1[2])); +let candidates = people | join(edge, ($1[0], $0[0], $2[0], $2[1], $1[1], $1[2] ;)) + | filter($0[1] < $0[2] && $0[3] >= $0[4] && $0[3] <= $0[5]); +let membership = people | map($1[0], $0[0] ;) | distinct; +let pairs = candidates | key($0[0], $0[2] ; $0[1]) + | join(membership, ($0[0], $1[0], $0[1] ;)) | distinct; +let paths = pairs | key($0[0], $0[2] ; $0[1]) + | join(pairs | key($0[0], $0[1] ; $0[2]), ($0[0], $1[0], $2[0] ; $0[1])); +let counts = paths | join(pairs, ($0[0] ;)) | count; +let zero = request | map($0[0] ; 0); +let missing = zero - (zero | join(counts, ($0 ; 0))); +export "bi11.answer" = (counts + missing) | map($0[0], 0 ; $1[0]); diff --git a/interactive/server/bench/ldbc/bi5.ddp b/interactive/server/bench/ldbc/bi5.ddp new file mode 100644 index 000000000..b68aba7e9 --- /dev/null +++ b/interactive/server/bench/ldbc/bi5.ddp @@ -0,0 +1,22 @@ +-- BI5: standing request(request_id, tag_name). Tagged messages and their +-- immediate replies/likes contribute 1/2/10 points to the message's author. +let request = input 0; +let tagged = import "tagged"; +let message = import "message"; +let likes = import "likes"; +let selected = request | key($0[1] ; $0[0]) + | join(tagged | key($1[0] ; $0[0]), ($2[0] ; $1[0])) + | join(message, ($1[0], $0[0], $2[1] ;)) | distinct; +-- Contributions are (messages, replies, likes). Collect+fold expresses a +-- sum without introducing a new reducer or optimizer in this benchmark. +let own = selected | map($0[0], $0[2] ; 1, 0, 0); +let liked = selected | key($0[1] ; $0[0], $0[2]) + | join(likes, ($1[0], $1[1] ; 0, 0, 1)); +let comments = message | filter($1[0] == 1) | key($1[2] ; $0[0]); +let replied = selected | key($0[1] ; $0[0], $0[2]) + | join(comments, ($1[0], $1[1] ; 0, 1, 0)); +let totals = (own + liked + replied) | collect + | map($0 ; fold($1, tuple(0, 0, 0), tuple(^1[0]+^0[0], ^1[1]+^0[1], ^1[2]+^0[2]))); +let scores = totals | map($0[0] ; -($1[0][0]+2*$1[0][1]+10*$1[0][2]), $0[1], $1[0]); +export "bi5.answer" = scores | collect | flatmap($1) | filter($1[0] < 100) + | map($0[0], $1[0] ; $1[1][1], $1[1][2][1], $1[1][2][2], $1[1][2][0], -$1[1][0]); diff --git a/interactive/server/bench/ldbc/client.py b/interactive/server/bench/ldbc/client.py new file mode 100644 index 000000000..f4989e91f --- /dev/null +++ b/interactive/server/bench/ldbc/client.py @@ -0,0 +1,165 @@ +"""Small, bounded TCP client for a private, real ddir_server process.""" +import ast +import os +import re +import socket +import subprocess +import threading +import time + + +def term(value): + if type(value) is int: + return str(value) + if isinstance(value, str): + return 'list(' + ','.join(map(str, value.encode())) + ')' + if isinstance(value, (tuple, list)): + return 'tuple(' + ','.join(map(term, value)) + ')' + raise TypeError(value) + + +def decode(lines): + def value(text): + def parse(node): + if isinstance(node, ast.Constant) and type(node.value) is int: + return node.value + if isinstance(node, ast.UnaryOp) and isinstance(node.op, ast.USub): + arg = parse(node.operand) + if type(arg) is int: + return -arg + if isinstance(node, ast.List): + return [parse(v) for v in node.elts] + if isinstance(node, ast.Call) and isinstance(node.func, ast.Name) and not node.keywords and len(node.args) == 1: + arg = parse(node.args[0]) + if node.func.id == 'Int' and type(arg) is int: + return arg + if node.func.id in ('Tuple', 'List') and isinstance(arg, list): + return arg + raise ValueError('unsupported server value: ' + text[:200]) + return parse(ast.parse(text, mode='eval').body) + result = [] + for line in lines: + match = re.fullmatch(r'diff=(-?\d+) key=(.+) val=(.+)', line) + if not match: + raise ValueError(line) + result.append([value(match[2]), value(match[3]), int(match[1])]) + return sorted(result, key=lambda r: r[0]) + + +class Server: + def __init__(self, binary, log, backend, workers, timeout, max_rss_gib): + self.timeout, self.counter = timeout, 0 + self.stream = self.reader = None + self.log = log.open('wb') + self.peak_rss = 0 + self.failure = None + with socket.socket() as reservation: + reservation.bind(('127.0.0.1', 0)) + port = reservation.getsockname()[1] + env = dict(os.environ, DDIR_BIND=f'127.0.0.1:{port}', DDIR_WS_BIND='127.0.0.1:0', + DDIR_DIAGNOSTICS='0', DDIR_WORKERS=str(workers), DDIR_BACKEND=backend) + self.process = subprocess.Popen([str(binary)], stdin=subprocess.PIPE, + stdout=subprocess.DEVNULL, stderr=self.log, env=env) + self.stop_monitor = threading.Event() + + def monitor(): + while not self.stop_monitor.wait(1): + try: + ps = subprocess.run(['ps', '-o', 'rss=', '-p', str(self.process.pid)], + capture_output=True, text=True, timeout=2, check=False) + rss = int(ps.stdout.strip() or 0) * 1024 + self.peak_rss = max(self.peak_rss, rss) + if rss > max_rss_gib * 1024**3: + self.failure = f'server RSS exceeded {max_rss_gib} GiB' + self.process.kill() + return + except (OSError, ValueError, subprocess.TimeoutExpired) as error: + self.failure = f'RSS monitor failed: {error}' + if self.process.poll() is None: + self.process.kill() + return + self.monitor = threading.Thread(target=monitor, daemon=True) + self.monitor.start() + try: + deadline = time.monotonic() + timeout + while True: + try: + self.stream = socket.create_connection(('127.0.0.1', port), timeout=1) + break + except OSError: + if self.process.poll() is not None or time.monotonic() > deadline: + raise RuntimeError(f'server did not start; see {log}') from None + time.sleep(0.02) + self.stream.setsockopt(socket.IPPROTO_TCP, socket.TCP_NODELAY, 1) + self.reader = self.stream.makefile('rb') + except BaseException: + self.close() + raise + + def commands(self, commands): + start = time.perf_counter() + ids, payload = [], [] + for command in commands: + self.counter += 1 + rid = f'r{self.counter}' + ids.append(rid) + payload.append(rid + ' ' + command.replace('{rid}', rid) + '\n') + wire = ''.join(payload).encode() + replies = {rid: [] for rid in ids} + pending = set(ids) + encoded = time.perf_counter() + deadline = encoded + self.timeout + self.stream.settimeout(self.timeout) + self.stream.sendall(wire) + received = 0 + while pending: + remaining = deadline - time.perf_counter() + if remaining <= 0: + raise TimeoutError('command deadline exceeded') + self.stream.settimeout(remaining) + raw = self.reader.readline() + received += len(raw) + if not raw: + raise RuntimeError(self.failure or 'server disconnected; see server log') + rid, kind, *body = raw.decode().rstrip('\r\n').split(' ', 2) + if rid not in pending: + raise RuntimeError(f'unexpected response: {raw[:200]!r}') + body = body[0] if body else '' + if kind == 'data': + replies[rid].append(body) + elif kind == 'ok': + pending.remove(rid) + else: + raise RuntimeError(body) + end = time.perf_counter() + return dict(encode_ms=1000*(encoded-start), wire_ms=1000*(end-encoded), + bytes_sent=len(wire), bytes_received=received), [replies[rid] for rid in ids] + + @staticmethod + def feed(program, index, rows, diff): + body = '\n'.join(term(row) + f' diff={diff}' for row in rows) + return f'feed {program} {index} begin\n{body}\n{{rid}} end-feed' + + def close(self): + self.stop_monitor.set() + self.monitor.join(timeout=3) + if self.reader: + self.reader.close() + if self.stream: + self.stream.close() + if self.process.poll() is None: + try: + self.process.stdin.write(b'exit\n') + self.process.stdin.flush() + self.process.wait(timeout=5) + except (BrokenPipeError, subprocess.TimeoutExpired): + self.process.kill() + self.process.wait() + self.process.stdin.close() + self.log.close() + + def __enter__(self): + return self + + def __exit__(self, *_): + self.close() diff --git a/interactive/server/bench/ldbc/compare.py b/interactive/server/bench/ldbc/compare.py new file mode 100644 index 000000000..9d8f190fb --- /dev/null +++ b/interactive/server/bench/ldbc/compare.py @@ -0,0 +1,47 @@ +#!/usr/bin/env python3 +"""Compare matching, successful workload runs. Ratios above 1 mean faster.""" +import argparse +import json +from pathlib import Path + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('baseline', type=Path) + parser.add_argument('candidate', type=Path) + args = parser.parse_args() + before, after = [json.loads(path.read_text()) for path in (args.baseline, args.candidate)] + for report in (before, after): + if report['status'] != 'passed' or report['format_version'] != 1: + parser.error('both reports must be successful format-version-1 runs') + for key in ('data', 'parameter_bank', 'standing', 'retracted_rows', 'reference_answers', + 'platform', 'python', 'logical_cpus'): + if before[key] != after[key]: + parser.error(f'incomparable {key}') + for key in ('workers', 'queries', 'rounds', 'warmup', 'batch_size', 'changes'): + if before['config'][key] != after['config'][key]: + parser.error(f'incomparable configuration: {key}') + for source in ('client.py', 'workload.py', 'run.py'): + if before['sources_sha256'][source] != after['sources_sha256'][source]: + parser.error(f'measurement/reference code changed: {source}; establish a fresh baseline') + changed_plans = [name for name, sha in before['sources_sha256'].items() + if name.endswith('.ddp') and sha != after['sources_sha256'].get(name)] + if changed_plans: + print('Changed plans (answers checked): ' + ', '.join(changed_plans)) + a = {r['backend']: r for r in before['runs']} + b = {r['backend']: r for r in after['runs']} + if a.keys() != b.keys(): + parser.error('backend sets differ') + for backend in a: + if a[backend]['requests'] != b[backend]['requests']: + parser.error(f'{backend}: request schedule changed') + print(f'\n{backend}: milliseconds, baseline / candidate; >1x is faster') + print(f'{"phase":36} {"baseline":>10} {"candidate":>10} {"ratio":>8}') + for phase, old in a[backend]['summary'].items(): + new = b[backend]['summary'][phase] + ratio = old['median_ms'] / new['median_ms'] + print(f'{phase:36} {old["median_ms"]:10.3f} {new["median_ms"]:10.3f} {ratio:7.2f}x') + + +if __name__ == '__main__': + main() diff --git a/interactive/server/bench/ldbc/graph.ddp b/interactive/server/bench/ldbc/graph.ddp new file mode 100644 index 000000000..5962a3418 --- /dev/null +++ b/interactive/server/bench/ldbc/graph.ddp @@ -0,0 +1,28 @@ +-- Positional input schema (all rows arrive as keys, with unit values): +-- 0 person(id, first, last, city); 1 knows(src, dst, created); +-- 2 message(id, kind, creator, parent), where post=0, comment=1, no parent=-1; +-- 3 tag(id, name); 4 message_tag(message, tag); 5 likes(person, message); +-- 6 place(id, name, parent). Strings are UTF-8 List(int). +let person = input 0; +let knows = input 1; +let message = input 2; +let tag = input 3; +let message_tag = input 4; +let likes = input 5; +let place = input 6; + +export "person" = person | key($0[0] ; $0[1], $0[2], $0[3]) | arrange; +-- The input contains each undirected friendship once. +let edges = ((knows | map($0[0], $0[1], $0[2] ;)) + + (knows | map($0[1], $0[0], $0[2] ;))) | distinct; +export "edge" = edges | key($0[0] ; $0[1], $0[2]) | arrange; +export "message" = message | key($0[0] ; $0[1], $0[2], $0[3]) | arrange; +export "likes" = likes | key($0[1] ; $0[0]) | arrange; +let tags = tag | key($0[0] ; $0[1]); +export "tagged" = message_tag | key($0[1] ; $0[0]) + | join(tags, ($1[0] ; $2[0])) | arrange; +let places = place | key($0[0] ; $0[1], $0[2]); +-- City -> country is one parent step in SNB. +export "resident" = person | key($0[3] ; $0[0]) + | join(places, ($2[1] ; $1[0])) + | join(places, ($2[0] ; $1[0])) | arrange; diff --git a/interactive/server/bench/ldbc/ic6.ddp b/interactive/server/bench/ldbc/ic6.ddp new file mode 100644 index 000000000..774424c98 --- /dev/null +++ b/interactive/server/bench/ldbc/ic6.ddp @@ -0,0 +1,23 @@ +-- IC6: request(request_id, person_id, tag_name). Two-hop tag co-occurrence. +-- This explicit plan expands friends and posts before testing the bound tag. +let request = input 0; +let edge = import "edge"; +let message = import "message"; +let tagged = import "tagged"; +-- Reachable rows: (request_id, origin, author, tag_name), deduplicated across paths. +let one = request | key($0[1] ; $0[0], $0[2]) + | join(edge, ($1[0], $0[0], $2[0], $1[1] ;)); +let two = one | key($0[2] ; $0[0], $0[1], $0[3]) + | join(edge, ($1[0], $1[1], $2[0], $1[2] ;)); +let authors = (one + two) | filter($0[1] != $0[2]) | distinct; +let posts = message | filter($1[0] == 0) | key($1[1] ; $0[0]); +let candidates = authors | key($0[2] ; $0[0], $0[3]) + | join(posts, ($2[0] ; $1[0], $1[1])); +let selected = candidates | join(tagged, ($1[0], $0[0], $1[1], $2[0] ;)) + | filter($0[2] == $0[3]) | map($0[0], $0[1], $0[2] ;) | distinct; +let counts = selected | key($0[1] ; $0[0], $0[2]) + | join(tagged, ($1[0], $2[0], $1[1] ;)) + | filter($0[1] != $0[2]) | map($0[0], $0[1] ;) | count; +export "ic6.answer" = counts | key($0[0] ; -$1[0], $0[1]) + | collect | flatmap($1) | filter($1[0] < 10) + | map($0[0], $1[0] ; $1[1][1], -$1[1][0]); diff --git a/interactive/server/bench/ldbc/is3.ddp b/interactive/server/bench/ldbc/is3.ddp new file mode 100644 index 000000000..17793e96a --- /dev/null +++ b/interactive/server/bench/ldbc/is3.ddp @@ -0,0 +1,9 @@ +-- IS3: request(request_id, person_id). Full friend list, newest friendship first. +let request = input 0; +let edge = import "edge"; +let person = import "person"; +let friends = request | key($0[1] ; $0[0]) + | join(edge, ($2[0] ; $1[0], $2[1])) + | join(person, ($1[0] ; -$1[1], $0[0], $2[0], $2[1])); +export "is3.answer" = friends | collect | flatmap($1) + | map($0[0], $1[0] ; $1[1][1], $1[1][2], $1[1][3], -$1[1][0]); diff --git a/interactive/server/bench/ldbc/run.py b/interactive/server/bench/ldbc/run.py new file mode 100644 index 000000000..8430544aa --- /dev/null +++ b/interactive/server/bench/ldbc/run.py @@ -0,0 +1,226 @@ +#!/usr/bin/env python3 +"""Reproducible maintained/parameterized LDBC-derived server workload (stdlib only).""" +import argparse +from collections import defaultdict, deque +import hashlib +import json +import os +from pathlib import Path +import platform +import shutil +import statistics +import subprocess +import sys +import tempfile +import time + +from client import Server, decode +from workload import HERE, TABLES, QUERIES, Reference, changes, expected, fingerprint, load, parameters + + +def digest(path): + h = hashlib.sha256() + with path.open('rb') as source: + for block in iter(lambda: source.read(1024 * 1024), b''): + h.update(block) + return h.hexdigest() + + +def positive(text): + value = int(text) + if value < 1: + raise argparse.ArgumentTypeError('must be positive') + return value + + +def checked(actual, want, label): + if actual != want: + mismatch = next((i for i, pair in enumerate(zip(actual, want)) if pair[0] != pair[1]), + min(len(actual), len(want))) + raise AssertionError(f'{label}: row {mismatch}: expected {want[mismatch:mismatch+1]!r} ' + f'({len(want)} rows), got {actual[mismatch:mismatch+1]!r} ({len(actual)} rows)') + + +def run_server(args, backend, graph, delta, bank, standing, answers, report): + interactive = [q for q in args.queries if q.startswith('i')] + bi = [q for q in args.queries if q.startswith('b')] + record = dict(backend=backend, workers=args.workers, events=[], requests=[]) + report['runs'].append(record) + with Server(args.server, args.output / f'{backend}.log', backend, args.workers, + args.timeout, args.max_rss_gib) as server: + context = dict(round=-1, warmup=True, state='setup') + + def command(phase, make_commands): + start = time.perf_counter() + commands = make_commands() + prepared = time.perf_counter() + metrics, replies = server.commands(commands) + metrics['prepare_ms'] = 1000*(prepared-start) + metrics['client_ms'] = 1000*(time.perf_counter()-start) + record['events'].append(dict(context, phase=phase, **metrics)) + return replies + + def read(name, want, phase='read'): + lines, = command(f'{phase}:{name}', lambda: [f'peek {name}.answer']) + start = time.perf_counter() + actual = decode(lines) + event = record['events'][-1] + event['decode_ms'] = 1000*(time.perf_counter()-start) + event['client_ms'] += event['decode_ms'] + event['rows'] = len(actual) + # Validation/reference work is deliberately outside all timings. + checked(actual, want, f'{backend}/{context}/{phase}/{name}') + + for name in ('graph', *args.queries): + program = (HERE / f'{name}.ddp').read_text() + command(f'install:{name}', lambda: [f'load {name} begin\n{program}\n{{rid}} end-load', 'tick']) + for index, name in enumerate(TABLES): + rows = sorted(graph[name]) + for offset in range(0, len(rows), 1000): + chunk = rows[offset:offset+1000] + command(f'load:{name}', lambda: [Server.feed('graph', index, chunk, 1)]) + for name in bi: + command(f'standing:{name}', lambda: [Server.feed(name, 0, [(0, *standing[name])], 1)]) + command('initial_tick', lambda: ['tick']) + for name in bi: + read(name, answers['initial'][name][standing[name]]) + for name in interactive: + read(name, [], 'empty') + + for cycle in range(args.warmup + args.rounds): + context.update(round=cycle-args.warmup, warmup=cycle < args.warmup) + for state_index, state in enumerate(('initial', 'changed')): + context['state'] = state + if state == 'changed': + command('update', lambda: [Server.feed('graph', TABLES.index(t), rows, -1) + for t, rows in delta.items() if rows] + ['tick']) + for name in bi: + read(name, answers[state][name][standing[name]], 'maintained') + bindings = {name: [(rid, *bank[name][((2*cycle+state_index)*args.batch_size+rid) % len(bank[name])]) + for rid in range(args.batch_size)] for name in interactive} + record['requests'].append(dict(context, bindings=bindings)) + if interactive: + first_event = len(record['events']) + command('bind', lambda: [Server.feed(name, 0, bindings[name], 1) for name in interactive] + ['tick']) + for name in interactive: + want = [] + for rid, *params in bindings[name]: + for key, value, diff in answers[state][name][tuple(params)]: + want.append([[rid, key[1]], value, diff]) + read(name, want) + # Sum measured phases, excluding the intervening answer checks. + elapsed = sum(e['client_ms'] for e in record['events'][first_event:]) + record['events'].append(dict(context, phase='batch_answers', client_ms=elapsed, + derived=True, requests=len(interactive)*args.batch_size)) + command('release', lambda: [Server.feed(name, 0, bindings[name], -1) for name in interactive] + ['tick']) + for name in interactive: + read(name, [], 'empty') + context['state'] = 'restored' + command('restore', lambda: [Server.feed('graph', TABLES.index(t), rows, 1) + for t, rows in delta.items() if rows] + ['tick']) + for name in bi: + read(name, answers['initial'][name][standing[name]], 'maintained') + print(f'{backend}: round {cycle-args.warmup + 1}/{args.rounds}, answers verified', flush=True) + record['peak_server_rss_bytes_sampled'] = server.peak_rss + buckets = defaultdict(list) + for event in record['events']: + if not event['warmup']: + buckets[(event['state'], event['phase'])].append(event['client_ms']) + record['summary'] = {f'{state}/{phase}': dict(samples=len(values), median_ms=statistics.median(values), + min_ms=min(values), max_ms=max(values)) + for (state, phase), values in sorted(buckets.items())} + for label, summary in record['summary'].items(): + if not label.split('/')[-1].startswith('empty'): + print(f' {label}: {summary["median_ms"]:.3f} ms median ({summary["samples"]} samples)') + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--server', type=Path, required=True, help='existing ddir_server binary (release for timing)') + parser.add_argument('--snapshot', type=Path, help='SNB BI composite-merged-fk initial_snapshot directory; default: tiny fixture') + parser.add_argument('--backend', choices=('vec', 'corgi', 'both'), default='both') + parser.add_argument('--workers', type=positive, default=1) + parser.add_argument('--queries', nargs='+', choices=QUERIES, default=list(QUERIES)) + parser.add_argument('--rounds', type=positive, default=5) + parser.add_argument('--warmup', type=int, default=1) + parser.add_argument('--batch-size', type=positive, default=4, help='bindings per interactive query per tick') + parser.add_argument('--changes', type=positive, default=4, help='maximum deleted rows per mutable edge table') + parser.add_argument('--timeout', type=positive, default=120, help='seconds per command group') + parser.add_argument('--max-rss-gib', type=positive, default=6, help='sampled server RSS ceiling, not a hard OS limit') + parser.add_argument('--output', type=Path, help='new result directory; default: a fresh temporary directory') + args = parser.parse_args() + if args.warmup < 0 or len(set(args.queries)) != len(args.queries): + parser.error('warmup must be non-negative and queries must be unique') + args.server = args.server.resolve(strict=True) + if args.snapshot: + args.snapshot = args.snapshot.resolve(strict=True) + if args.output: + args.output = args.output.resolve() + args.output.mkdir(parents=True, exist_ok=False) + else: + args.output = Path(tempfile.mkdtemp(prefix='ddir-ldbc-')) + print(f'Artifacts: {args.output}', flush=True) + repo = HERE.parents[3] + def git(*command): + return subprocess.run(['git', '-C', str(repo), *command], capture_output=True, + text=True, check=False).stdout.strip() + report = dict(format_version=1, status='running', config={k: str(v) if isinstance(v, Path) else v for k, v in vars(args).items()}, + platform=platform.platform(), python=platform.python_version(), logical_cpus=os.cpu_count(), + revision=git('rev-parse', 'HEAD'), worktree=git('status', '--porcelain'), + binary_sha256=digest(args.server), + sources_sha256={p.name: digest(p) for p in sorted(HERE.iterdir()) if p.suffix in ('.py', '.ddp', '.json')}, runs=[]) + lock = repo / 'Cargo.lock' + if lock.exists(): + shutil.copyfile(lock, args.output / 'Cargo.lock') + report['cargo_lock_sha256'] = digest(lock) + try: + started = time.perf_counter() + graph = load(args.snapshot) + report['data'] = dict(rows={t: len(rows) for t, rows in graph.items()}, sha256=fingerprint(graph)) + reference = Reference(graph) + if args.snapshot is None: + # Anchor the independent oracle in a few hand-countable fixture facts. + checked(reference.answer('is3', (1,)), [(3, 'Cy', 'Q', 200), (2, 'Bo', 'Long', 100)], 'tiny IS3') + checked(reference.answer('ic6', (1, 'Topic')), [('A', 2), ('Alphabet', 2), ('Z', 2), ('é', 1)], 'tiny IC6') + checked(reference.answer('bi5', ('Topic',))[:1], [(2, 2, 2, 1, 25)], 'tiny BI5') + checked(reference.answer('bi11', ('CountryA', 100, 700)), [(2,)], 'tiny BI11') + bank, standing = parameters(reference) + delta = changes(graph, args.changes) + report.update(parameter_bank=bank, standing=standing, retracted_rows=delta) + answers = {} + for state in ('initial', 'changed'): + if state == 'changed': + altered = {t: rows - set(delta[t]) if t in delta else rows for t, rows in graph.items()} + reference = Reference(altered) + del altered + answers[state] = {name: {params: expected(reference, name, [(0, *params)]) + for params in (bank[name] if name.startswith('i') else [standing[name]])} + for name in args.queries} + del reference + report['reference_ms'] = 1000*(time.perf_counter()-started) + report['reference_answers'] = {state: {name: [dict(parameters=p, rows=rows) for p, rows in values.items()] + for name, values in queries.items()} for state, queries in answers.items()} + print(f'Projected rows: {report["data"]["rows"]}', flush=True) + for name, cases in answers['initial'].items(): + nonempty = sum(bool(rows) for rows in cases.values()) + print(f'{name}: {nonempty}/{len(cases)} distinct parameter cases have nonempty answers', flush=True) + if 'bi11' in args.queries: + print(f'bi11: {answers["initial"]["bi11"][standing["bi11"]][0][1][0]} initial triangles', flush=True) + for backend in ('vec', 'corgi') if args.backend == 'both' else (args.backend,): + run_server(args, backend, graph, delta, bank, standing, answers, report) + report['status'] = 'passed' + except BaseException as error: + report.update(status='failed', error=f'{type(error).__name__}: {error}') + for log in args.output.glob('*.log'): + with log.open(errors='replace') as source: + tail = ''.join(deque(source, maxlen=20))[-8000:] + if tail: + print(f'{log.name} (tail):\n{tail}', file=sys.stderr) + raise + finally: + (args.output / 'report.json').write_text(json.dumps(report, indent=2) + '\n') + print(f'Report: {args.output / "report.json"}', flush=True) + + +if __name__ == '__main__': + main() diff --git a/interactive/server/bench/ldbc/tiny.json b/interactive/server/bench/ldbc/tiny.json new file mode 100644 index 000000000..c8538f17f --- /dev/null +++ b/interactive/server/bench/ldbc/tiny.json @@ -0,0 +1,9 @@ +{ + "person": [[1,"Ada","L",10],[2,"Bo","Long",10],[3,"Cy","Q",10],[4,"Di","R",11],[5,"Eve","S",10],[6,"Fay","T",11]], + "knows": [[1,2,100],[1,3,200],[2,3,300],[2,4,400],[3,5,500],[2,5,600],[4,6,700]], + "message": [[100,0,2,-1],[101,0,3,-1],[102,0,4,-1],[103,0,5,-1],[104,0,1,-1],[200,1,3,100],[201,1,4,100],[202,1,2,200]], + "tag": [[1,"Topic"],[2,"Z"],[3,"Alphabet"],[4,"A"],[5,"é"]], + "message_tag": [[100,1],[100,2],[100,3],[101,1],[101,2],[101,4],[102,1],[102,3],[102,5],[103,1],[103,4],[104,1],[104,5],[200,1],[200,3]], + "likes": [[1,100],[3,100],[4,101],[2,102],[5,200]], + "place": [[10,"CityA",20],[11,"CityB",21],[20,"CountryA",-1],[21,"CountryB",-1]] +} diff --git a/interactive/server/bench/ldbc/workload.py b/interactive/server/bench/ldbc/workload.py new file mode 100644 index 000000000..1fde08b83 --- /dev/null +++ b/interactive/server/bench/ldbc/workload.py @@ -0,0 +1,185 @@ +"""Narrow SNB snapshot adapter and independent, untimed reference queries.""" +from collections import Counter, defaultdict +import csv +from datetime import datetime +import hashlib +import json +from pathlib import Path + +HERE = Path(__file__).resolve().parent +TABLES = ('person', 'knows', 'message', 'tag', 'message_tag', 'likes', 'place') +QUERIES = ('is3', 'ic6', 'bi5', 'bi11') + + +def millis(text): + return int(datetime.fromisoformat(text.replace('Z', '+00:00')).timestamp() * 1000) + + +def load(snapshot): + if snapshot is None: + raw = json.loads((HERE / 'tiny.json').read_text()) + return {name: {tuple(row) for row in raw[name]} for name in TABLES} + # Accept the initial_snapshot directory, not an ambiguous graph/history root. + result = {name: set() for name in TABLES} + entities = { + 'Person': 'person', 'Person_knows_Person': 'knows', + 'Post': 'message', 'Comment': 'message', 'Tag': 'tag', + 'Post_hasTag_Tag': 'message_tag', 'Comment_hasTag_Tag': 'message_tag', + 'Person_likes_Post': 'likes', 'Person_likes_Comment': 'likes', 'Place': 'place', + } + for entity, table in entities.items(): + category = 'static' if entity in ('Tag', 'Place') else 'dynamic' + paths = sorted((snapshot / category / entity).glob('*.csv')) + if not paths: + raise ValueError(f'missing CSV partition(s): {category}/{entity}') + for path in paths: + with path.open(newline='', encoding='utf-8') as source: + for r in csv.DictReader(source, delimiter='|', quoting=csv.QUOTE_NONE): + if table == 'person': + row = (int(r['id']), r['firstName'], r['lastName'], int(r['LocationCityId'])) + elif table == 'knows': + a, b = sorted((int(r['Person1Id']), int(r['Person2Id']))) + row = (a, b, millis(r['creationDate'])) + elif table == 'message': + parent = r.get('ParentPostId') or r.get('ParentCommentId') or '-1' + row = (int(r['id']), int(entity == 'Comment'), int(r['CreatorPersonId']), int(parent)) + elif table == 'tag': + row = (int(r['id']), r['name']) + elif table == 'place': + row = (int(r['id']), r['name'], int(r.get('PartOfPlaceId') or '-1')) + else: + mid = int(r['PostId'] if 'PostId' in r else r['CommentId']) + row = (mid, int(r['TagId'])) if table == 'message_tag' else (int(r['PersonId']), mid) + if any(isinstance(v, str) and not v for v in row): + raise ValueError(f'{path}: empty strings require typed server inputs; not padded here') + result[table].add(row) + return result + + +def fingerprint(graph): + digest = hashlib.sha256() + for name in TABLES: + digest.update(name.encode()) + for row in sorted(graph[name]): + digest.update((json.dumps(row, ensure_ascii=True) + '\n').encode()) + return digest.hexdigest() + + +class Reference: + """Direct graph traversal/counting, deliberately independent of the DDP plans.""" + def __init__(self, graph): + self.people = {p[0]: p[1:] for p in graph['person']} + self.messages = {m[0]: m[1:] for m in graph['message']} + self.places = {p[0]: p[1:] for p in graph['place']} + self.tag_names = {t[0]: t[1] for t in graph['tag']} + self.edges = defaultdict(dict) + for a, b, created in graph['knows']: + if a == b or a not in self.people or b not in self.people: + raise ValueError('friendship must connect two distinct existing people') + if b in self.edges[a]: + raise ValueError('multiple creation dates for a friendship') + self.edges[a][b] = self.edges[b][a] = created + self.tags = defaultdict(set) + for mid, tag in graph['message_tag']: + if mid not in self.messages: + raise ValueError(f'tag on missing message {mid}') + self.tags[mid].add(self.tag_names[tag]) + self.liked = Counter(mid for _, mid in graph['likes']) + self.replied = Counter(parent for kind, _, parent in self.messages.values() if kind == 1) + self.posts = defaultdict(list) + for mid, (kind, author, _) in self.messages.items(): + if kind == 0: + self.posts[author].append(mid) + + def country(self, pid): + city = self.people[pid][2] + return self.places[self.places[city][1]][0] + + def answer(self, name, params): + if name == 'is3': + pid, = params + rows = [(friend, *self.people[friend][:2], date) for friend, date in self.edges[pid].items()] + return sorted(rows, key=lambda r: (-r[3], r[0])) + if name == 'ic6': + pid, tag = params + authors = set(self.edges[pid]) + for friend in self.edges[pid]: + authors.update(self.edges[friend]) + authors.discard(pid) + counts = Counter() + for author in authors: + for mid in self.posts[author]: + if tag in self.tags[mid]: + counts.update(self.tags[mid] - {tag}) + return sorted(counts.items(), key=lambda r: (-r[1], r[0].encode()))[:10] + if name == 'bi5': + tag, = params + totals = defaultdict(lambda: [0, 0, 0]) + for mid, names in self.tags.items(): + if tag in names: + total = totals[self.messages[mid][1]] + total[0] += self.replied[mid] + total[1] += self.liked[mid] + total[2] += 1 + rows = [(pid, replies, likes, messages, messages + 2*replies + 10*likes) + for pid, (replies, likes, messages) in totals.items()] + return sorted(rows, key=lambda r: (-r[4], r[0]))[:100] + if name == 'bi11': + country, start, end = params + residents = {p for p in self.people if self.country(p) == country} + neighbors = {p: {f for f, d in self.edges[p].items() + if f in residents and start <= d <= end} for p in residents} + count = sum(1 for a in residents for b in neighbors[a] if a < b + for c in neighbors[a] & neighbors[b] if b < c) + return [(count,)] + raise ValueError(name) + + +def parameters(reference): + # Stratify people by degree. Choose tags present on reachable multi-tag + # posts when possible, so a small snapshot is not mostly empty IC6 lookups. + # This is data-derived parameter generation, not the official distribution. + people = sorted(reference.people, key=lambda p: (len(reference.edges[p]), p)) + if not people or not reference.tags: + raise ValueError('workload needs people and tagged messages') + popular = Counter(t for tags in reference.tags.values() for t in tags) + tags = sorted(popular, key=lambda t: (-popular[t], t))[:8] + bank = [people[i * (len(people)-1) // 7] for i in range(8)] + absent = max(people) + 1 + tagged_requests = [] + for i, pid in enumerate(bank): + authors = set(reference.edges[pid]) + for friend in reference.edges[pid]: + authors.update(reference.edges[friend]) + authors.discard(pid) + nearby = Counter(tag for author in authors for mid in reference.posts[author] + if len(reference.tags[mid]) > 1 for tag in reference.tags[mid]) + choices = sorted(nearby, key=lambda tag: (-nearby[tag], tag))[:8] or tags + tagged_requests.append((pid, choices[i % len(choices)])) + requests = {'is3': [(p,) for p in bank] + [(absent,)], + 'ic6': tagged_requests + [(absent, tags[0])]} + countries = Counter(reference.country(p) for p in people) + country = min(countries, key=lambda c: (-countries[c], c)) + dates = [d for neighbors in reference.edges.values() for d in neighbors.values()] + standing = {'bi5': (tags[0],), 'bi11': (country, min(dates, default=0), max(dates, default=0))} + return requests, standing + + +def changes(graph, limit): + # Bounded churn: detach selected friendship, tag and like edges; then restore + # the exact same rows. No node deletion/cascades or synthetic timestamps. + return {name: sorted(graph[name])[:limit] for name in ('knows', 'message_tag', 'likes')} + + +def wire_value(value): + if isinstance(value, str): + return list(value.encode()) + if isinstance(value, (tuple, list)): + return [wire_value(v) for v in value] + return value + + +def expected(reference, name, requests): + return [[[rid, rank], wire_value(row), 1] + for rid, *params in requests + for rank, row in enumerate(reference.answer(name, params))] From aa0584342f00da6ef4f1dfc214c86341d05576f8 Mon Sep 17 00:00:00 2001 From: Frank McSherry Date: Sun, 6 Sep 2026 20:27:38 -0400 Subject: [PATCH 2/8] Target standing BI footprints in the benchmark churn schedule --- interactive/server/bench/ldbc/README.md | 6 ++++- interactive/server/bench/ldbc/run.py | 5 +++- interactive/server/bench/ldbc/workload.py | 31 ++++++++++++++++++++--- 3 files changed, 36 insertions(+), 6 deletions(-) diff --git a/interactive/server/bench/ldbc/README.md b/interactive/server/bench/ldbc/README.md index ab191e79b..471a32597 100644 --- a/interactive/server/bench/ldbc/README.md +++ b/interactive/server/bench/ldbc/README.md @@ -70,7 +70,11 @@ it does not introduce native Corgi trace sharing or a new prepared-query feature Each round runs requests on the initial graph, retracts those bindings, removes a deterministic bounded set of friendship/tag/like edges, runs new requests on -the changed graph, retracts them, and restores the exact removed edges. The +the changed graph, retracts them, and restores the exact removed edges. Churn +targets edges in the standing BI11 country's triangles and the tagged messages +of BI5's top-100 authors, filling shortfalls with other edges. This is deliberate +**footprint-targeted stress**, not a random or official update distribution. +The report states whether each displayed BI answer actually changed. The dataset oscillates between two states: **bounded churn, not growth or a time-bounded saturation run**. No node deletion/cascade semantics are assumed. Every `feed` group ends with `tick`; acknowledgement of that tick, not of feed diff --git a/interactive/server/bench/ldbc/run.py b/interactive/server/bench/ldbc/run.py index 8430544aa..8c2503583 100644 --- a/interactive/server/bench/ldbc/run.py +++ b/interactive/server/bench/ldbc/run.py @@ -185,7 +185,7 @@ def git(*command): checked(reference.answer('bi5', ('Topic',))[:1], [(2, 2, 2, 1, 25)], 'tiny BI5') checked(reference.answer('bi11', ('CountryA', 100, 700)), [(2,)], 'tiny BI11') bank, standing = parameters(reference) - delta = changes(graph, args.changes) + delta = changes(graph, args.changes, reference, standing) report.update(parameter_bank=bank, standing=standing, retracted_rows=delta) answers = {} for state in ('initial', 'changed'): @@ -206,6 +206,9 @@ def git(*command): print(f'{name}: {nonempty}/{len(cases)} distinct parameter cases have nonempty answers', flush=True) if 'bi11' in args.queries: print(f'bi11: {answers["initial"]["bi11"][standing["bi11"]][0][1][0]} initial triangles', flush=True) + report['changed_bi_answers'] = {name: answers['initial'][name] != answers['changed'][name] + for name in args.queries if name.startswith('b')} + print(f'Churn changes displayed BI answers: {report["changed_bi_answers"]}', flush=True) for backend in ('vec', 'corgi') if args.backend == 'both' else (args.backend,): run_server(args, backend, graph, delta, bank, standing, answers, report) report['status'] = 'passed' diff --git a/interactive/server/bench/ldbc/workload.py b/interactive/server/bench/ldbc/workload.py index 1fde08b83..80d07e0db 100644 --- a/interactive/server/bench/ldbc/workload.py +++ b/interactive/server/bench/ldbc/workload.py @@ -165,10 +165,33 @@ def parameters(reference): return requests, standing -def changes(graph, limit): - # Bounded churn: detach selected friendship, tag and like edges; then restore - # the exact same rows. No node deletion/cascades or synthetic timestamps. - return {name: sorted(graph[name])[:limit] for name in ('knows', 'message_tag', 'likes')} +def changes(graph, limit, reference, standing): + # Target standing BI footprints instead of measuring mostly irrelevant + # changes. This is an explicit stress schedule, not LDBC's update stream. + country, start, end = standing['bi11'] + residents = {p for p in reference.people if reference.country(p) == country} + neighbors = {p: {f for f, d in reference.edges[p].items() + if f in residents and start <= d <= end} for p in residents} + triangles = [row for row in graph['knows'] if row[0] in neighbors + and row[1] in neighbors[row[0]] and neighbors[row[0]] & neighbors[row[1]]] + topic, = standing['bi5'] + leaders = {row[0] for row in reference.answer('bi5', (topic,))} + tagged = {mid for mid, tags in reference.tags.items() + if topic in tags and reference.messages[mid][1] in leaders} + candidates = { + 'knows': triangles, + 'message_tag': [row for row in graph['message_tag'] if row[0] in tagged and reference.tag_names[row[1]] == topic], + 'likes': [row for row in graph['likes'] if row[1] in tagged], + } + # Fill any shortfall with other edges, so even a zero-triangle case churns. + result = {} + for name, rows in candidates.items(): + selected = sorted(rows)[:limit] + remaining = limit - len(selected) + if remaining: + selected += sorted(graph[name] - set(selected))[:remaining] + result[name] = sorted(selected) + return result def wire_value(value): From 8b08fb723686fa93f7a456d114bd0705bfc3009d Mon Sep 17 00:00:00 2001 From: Frank McSherry Date: Sun, 6 Sep 2026 20:42:11 -0400 Subject: [PATCH 3/8] Document SF1 benchmark memory headroom --- interactive/server/bench/ldbc/README.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/interactive/server/bench/ldbc/README.md b/interactive/server/bench/ldbc/README.md index 471a32597..d6fa2687f 100644 --- a/interactive/server/bench/ldbc/README.md +++ b/interactive/server/bench/ldbc/README.md @@ -45,6 +45,8 @@ still means two simultaneous requests per tick. Increase batch size separately from data size to study dispatch amortization. Defaults enforce a 120-second command deadline and a sampled 6-GiB **server** RSS ceiling. The latter is not a hard memory limit and excludes the Python loader/reference process. +SF1 can exceed the default ceiling during loading; use `--max-rss-gib 8` +if the host has sufficient memory for both processes. ## Queries and lifecycle From 43a516a033561caa2eab2a5ac82d78f63bdc629a Mon Sep 17 00:00:00 2001 From: Frank McSherry Date: Sun, 6 Sep 2026 23:07:56 -0400 Subject: [PATCH 4/8] ddir: run all 41 SNB reads through the server benchmark --- .github/workflows/test.yml | 10 + interactive/server/bench/ldbc/README.md | 21 +- interactive/server/bench/ldbc/SNB.md | 198 +++++++ interactive/server/bench/ldbc/client.py | 5 + interactive/server/bench/ldbc/compare.py | 38 +- interactive/server/bench/ldbc/snb/__init__.py | 1 + interactive/server/bench/ldbc/snb/data.py | 130 +++++ .../server/bench/ldbc/snb/parameters.py | 118 ++++ interactive/server/bench/ldbc/snb/queries.py | 515 ++++++++++++++++++ interactive/server/bench/ldbc/snb/rel.py | 310 +++++++++++ interactive/server/bench/ldbc/snb/witness.py | 89 +++ interactive/server/bench/ldbc/suite.py | 278 ++++++++++ interactive/server/bench/ldbc/test_snb.py | 73 +++ 13 files changed, 1769 insertions(+), 17 deletions(-) create mode 100644 interactive/server/bench/ldbc/SNB.md create mode 100644 interactive/server/bench/ldbc/snb/__init__.py create mode 100644 interactive/server/bench/ldbc/snb/data.py create mode 100644 interactive/server/bench/ldbc/snb/parameters.py create mode 100644 interactive/server/bench/ldbc/snb/queries.py create mode 100644 interactive/server/bench/ldbc/snb/rel.py create mode 100644 interactive/server/bench/ldbc/snb/witness.py create mode 100644 interactive/server/bench/ldbc/suite.py create mode 100644 interactive/server/bench/ldbc/test_snb.py diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index d873e0617..fbc72599c 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -48,6 +48,16 @@ jobs: python3 interactive/server/bench/ldbc/run.py --server target/debug/ddir_server \ --workers "$workers" --rounds 1 --warmup 0 --batch-size 9 --changes 1 done + - name: Complete SNB read catalogue + if: matrix.os == 'ubuntu' && matrix.toolchain == 'stable' + run: | + python3 interactive/server/bench/ldbc/test_snb.py + for workers in 1 4; do + python3 interactive/server/bench/ldbc/suite.py --server target/debug/ddir_server \ + --workers "$workers" --rounds 1 --warmup 0 + done + python3 interactive/server/bench/ldbc/suite.py --server target/debug/ddir_server \ + --mode maintained --workers 4 --rounds 1 --warmup 0 # The explanation suite's heavier sweeps (soundness regression, fuzz) # are #[ignore]d for local debug runs but must gate merges. The metric # report is skipped (it prints, it does not assert). diff --git a/interactive/server/bench/ldbc/README.md b/interactive/server/bench/ldbc/README.md index d6fa2687f..63a02a4ec 100644 --- a/interactive/server/bench/ldbc/README.md +++ b/interactive/server/bench/ldbc/README.md @@ -1,4 +1,15 @@ -# LDBC-derived server workload +# LDBC-derived server benchmarks + +The **[complete read suite](SNB.md)** is in `suite.py`: IS1–7, IC1–14 (v2), +and BI1–20, with readable named-field definitions, generated DDP, shared server +inputs, varying interactive bindings, and maintained BI results. Start there +for full query coverage. It requires no downloaded data for its CI fixture. + +The four-query `run.py` panel documented below remains a smaller control with +independent graph oracles and hand-written physical plans. Its measurements +are not directly interchangeable with the full suite's wider schema/plans. + +## Four-query control panel A small performance/correctness baseline for DDIR, through the real `ddir_server` TCP interface. No server extensions, external database, generator, @@ -149,9 +160,9 @@ python3 interactive/server/bench/ldbc/run.py --server target/debug/ddir_server \ First optimization controls: IC6's expand-before-tag-filter plan, BI5's `collect`/`fold` sums, repeated scalar compilation/dispatch, and shared-import row/column conversion. Keep their patches independent and compare against this -baseline before stacking them. Add the remaining queries in small panels, -especially recursive, optional/SUM, temporal and large-output cases. This panel +baseline before stacking them. The [full suite](SNB.md) adds recursive, +optional, temporal and large-output cases on a separate shared substrate. This panel does not yet measure ad-hoc re-planning, parameterized BI, simultaneous bound-IC updates, or the official update stream. Empty strings are rejected by this narrow -adapter: general nullable/message-content queries need typed server admission, -not padding the data to evade the current shape-inference limitation. +adapter; the full suite handles them through explicit source-shape contracts, +without padding the data to evade shape-inference limitations. diff --git a/interactive/server/bench/ldbc/SNB.md b/interactive/server/bench/ldbc/SNB.md new file mode 100644 index 000000000..6ebcedf89 --- /dev/null +++ b/interactive/server/bench/ldbc/SNB.md @@ -0,0 +1,198 @@ +# Complete SNB read suite + +This is a runnable DDIR benchmark catalogue, not an official LDBC score or a +conformance claim. It contains every read in the pinned specification: +[IS1–7](https://github.com/ldbc/ldbc_snb_docs/blob/b2269610f433da72e7c97041f01680aae369a903/interactive-short-reads.tex), +[IC1–14 v2](https://github.com/ldbc/ldbc_snb_docs/blob/b2269610f433da72e7c97041f01680aae369a903/interactive-v2-complex-reads.tex), +and [BI1–20](https://github.com/ldbc/ldbc_snb_docs/blob/b2269610f433da72e7c97041f01680aae369a903/bi-reads.tex). +There are **41 reads**, not 25 per family. No selected query is silently skipped. +IC14 is the v2 cheapest-interaction-path query, not v1's all-shortest-path query. + +## Run + +From the repository root, with Python 3.9+ and `ps`: + +```sh +cargo build --release -p ddir-server +python3 interactive/server/bench/ldbc/suite.py --server target/release/ddir_server +``` + +The default starts a private real TCP server for each backend, installs all 41 +queries concurrently, loads the same shared graph, and checks answers throughout +a bounded update/request cycle. It uses a hand-authored 199-row witness graph +(8 people, 35 messages), one worker, one warmup round, and three measured rounds. +No generator, external database, downloaded data, Python packages, or custom +Rust benchmark executable is needed. + +The query definitions are in [snb/queries.py](snb/queries.py), with named fields +and relational steps. [snb/rel.py](snb/rel.py) is a small authoring notation that +emits ordinary DDP. It is not another engine backend: the server does all joins, +recursion, aggregation, nested collection construction, and ranking. Each run +saves the emitted `.ddp` plans and a catalogue of parameters/output fields. To +inspect them without starting a server: + +```sh +python3 interactive/server/bench/ldbc/suite.py --list +python3 interactive/server/bench/ldbc/suite.py --emit --output /tmp/snb-plans +``` + +Use a subset or a different lifecycle without editing query definitions: + +```sh +python3 interactive/server/bench/ldbc/suite.py --server target/release/ddir_server \ + --queries ic1 ic6 ic14 bi1 bi15 --workers 4 --rounds 10 +python3 interactive/server/bench/ldbc/suite.py --server target/release/ddir_server \ + --queries bi --mode maintained --isolated +``` + +`--queries` accepts `all`, `is`, `ic`, `bi`, and individual names. `--isolated` +starts a fresh server for each query, retaining the same raw graph substrate; +this is an attribution control, not concurrent throughput. `--output` must be +empty or new. Server logs and a JSON report survive failures. + +## Data and parameters + +To use existing generated data: + +```sh +python3 interactive/server/bench/ldbc/suite.py --server target/release/ddir_server \ + --snapshot /path/to/graphs/csv/bi/composite-merged-fk/initial_snapshot \ + --backend corgi --workers 4 --batch-size 3 --rounds 5 +``` + +The adapter reads all 18 required tables in SNB BI CSV `composite-merged-fk` +layout, including `.csv.gz` partitions. It projects 17 relations: people, +posts/comments, forums, friendships, memberships, tags/classes, places, +organisations, interests, message/forum tags, likes, education, employment, +email, and languages. Missing files fail explicitly. All reads have their +required source fields; BI data is used for Interactive semantics too, not the +official Interactive generator/driver workload. No files are downloaded or +modified. Calendar fields are derived in UTC; query results are not precomputed +by the adapter. + +Strings are UTF-8 byte lists, including genuinely empty strings. The current +leaves are 64-bit integers, not byte-packed or dictionary-encoded strings. +Post and Comment share a tuple with a kind field; they are **not** an enum-layout +optimization experiment. Source shape ascriptions supply the element encoding +of empty lists and inactive sum lanes. They are execution-time contracts, not +transactional validation at `feed` admission. + +Default bindings are deterministic smoke parameters selected from input facts, +not LDBC's parameter generator. The fixture has nonempty witnesses for all 41 +reads; generated datasets may legitimately give empty answers. Actual values +(people, messages, tags, dates) vary between interactive request batches, not +just request IDs. Equal bindings under distinct IDs are also exercised at the +default batch size. The full bank and exact bindings are recorded. + +Override bindings with `--parameters /path/to/bindings.json`. Its format is a +query-name object containing a nonempty list of parameter objects; omitted +fields use the smoke defaults. Integer dates are milliseconds since the epoch: + +```json +{"ic6": [{"pid": 1, "tagname": "Topic"}, {"pid": 2, "tagname": "Other"}], + "bi12": [{"language": ["en", "fr"]}]} +``` + +BI12's language list is represented by rows sharing a request ID. An empty set +uses a disabled row, so the request still exists and returns the zero-count +distribution. BI13's calendar-month parameter is derived from its end date. +In maintained mode the first binding for each query remains standing; later +bank entries are for transient request batches, not simultaneous BI bindings. + +## Lifecycle and accounting + +| Mode | BI bindings | IS/IC bindings | Graph changes | +| --- | --- | --- | --- | +| `mixed` (default) | Standing until final retirement | Bound per batch, then retracted | Paid with BI live and interactive parameter inputs empty | +| `maintained` | Standing | Standing | All bound answers maintained and checked | + +All query programs are installed once. Every query imports the shared raw +graph program and has a separate request input. Parameter-independent parts +of an installed query remain as dataflows even when its request input is empty; +their maintenance is included. Named imports currently cross the server's row +boundary: this is not native Corgi arrangement sharing. Derived relations are +not factored across different query programs. + +Each round visits the initial graph, changes it, and restores it. The witness +removes a friendship and leaf post (including tags) and renames a person. CSV +mode retracts a bounded number of friendship, membership, and like rows, then +restores them. This is **churn between two fixed snapshots**, not growing data, +an official update stream, or a time-bounded saturation run. It does not claim +general entity-deletion/cascade semantics. Interactive request IDs are reused +after retraction with new values; they do not name pre-maintained future answers. + +Each feed group is completed by `tick`; the feed acknowledgement alone means +admission. An empty `peek` after that tick is a completed empty result. The +report separates install/parse, graph admission, initial settlement, bind/tick, +answer retrieval, release/tick, graph update/tick, and restoration. `read:*` +includes full-result wire transfer and client decoding, not just a handle to an +answer. Ranking/top-k work stays in DDIR. `batch_answers` sums binding and answer +retrieval, excluding reference checks between them. +Standing-result reads are `maintained:*`; cleanup checks are `empty:*`, so empty +reads do not dilute the latency summary for bound requests. + +Per-event command preparation, encoding, wire wait/receipt, and decoding are +recorded. Wire time includes server work and TCP overhead; it is not a kernel +profile. Shared update/bind time cannot be attributed to one query by inspecting +its later `peek`. Use `--isolated` and selected concurrent panels for attribution. +Parsing/planning is setup cost here: **this is not an ad-hoc re-planning test**. +Do not omit release or standing maintenance when assessing sustained work. + +The JSON report contains raw timings and non-warmup medians, result hashes/row +counts, source/data/binary/plan hashes, exact deltas, machine/revision metadata, +sampled server RSS, and a copy/hash of the available Cargo lockfile. Compare +matching reports using the same `compare.py` as the four-query panel: + +```sh +python3 interactive/server/bench/ldbc/compare.py /tmp/snb-before/report.json /tmp/snb-after/report.json +``` + +Changed logical/DDP plans are permitted but called out; returned answers and +schedules must still match. Changed measurement/data-adapter code requires a +fresh baseline. Four-query and full-suite reports are not interchangeable. + +## Validation, limits, and optimization controls + +Every returned row, rank, multiplicity, and empty result is checked against +fresh Python evaluation of the **same logical plan**, outside the timers. This +checks DDP lowering and incremental execution; it is not an independent query +specification oracle. [test_snb.py](test_snb.py) adds hand-counted witnesses for +paths, ranking/ties, optional nested collections, triangles, propagation, +recruitment, language sets, and floating aggregates. The older four-query panel +retains independent traversal/counting oracles. Official conformance validation +is still separate work. + +Floating results use explicit IEEE f64 newtypes with total ordering, not the +specification's Float32 API representation. They are preserved as tagged values +in reports. Shortest-path maintenance uses a hop-indexed bound of `|V|-1` so +deletions terminate; this can be expensive on large connected graphs. General +aggregates currently use `collect`/`fold`; nested and ranked outputs are fully +materialized. These are deliberate visible baseline costs, not tuned plans. + +The complete suite is verified on the 199-row fixture and the generated +SF0.003 snapshot (35,588 projected rows) on both backends. CI runs only the +fixture: mixed at one/four workers and maintained at four workers. It asserts +answers, not timing thresholds, and does not fetch data. The local equivalent is: + +```sh +python3 interactive/server/bench/ldbc/test_snb.py +python3 interactive/server/bench/ldbc/suite.py --server target/debug/ddir_server \ + --workers 4 --rounds 1 --warmup 0 +python3 interactive/server/bench/ldbc/suite.py --server target/debug/ddir_server \ + --mode maintained --workers 4 --rounds 1 --warmup 0 +``` + +Start scaling with selected queries. Neither the Python snapshot evaluator nor +the hop-indexed path plans are intended to make all-query SF1 runs cheap. The +default 120-second command deadline and sampled 6-GiB server RSS ceiling exclude +Python and are not a hard OS memory limit. Small debug runs establish correctness, +not competitive performance; use release builds and a quiet machine for timing. + +Useful initial attribution groups (not official choke-point classifications): +IC1/IC12 for optional/nested output; IC2/IC6 for selective joins and ranking; +BI1/BI5/BI12/BI13 for aggregation; BI11/BI18 for graph motifs; IS2/IS6/BI9 for +thread recursion; IC13/IC14/BI10/BI15/BI19/BI20 for distance/path maintenance; +BI17 for broad temporal joins. Keep optimizer and kernel changes separate from +these definitions so each improvement can be measured against an unchanged +workload. In particular, the IC6 baseline has not been hand-rewritten to push +the tag filter ahead of friendship/message expansion. diff --git a/interactive/server/bench/ldbc/client.py b/interactive/server/bench/ldbc/client.py index f4989e91f..2dba8b2ca 100644 --- a/interactive/server/bench/ldbc/client.py +++ b/interactive/server/bench/ldbc/client.py @@ -29,6 +29,11 @@ def parse(node): return -arg if isinstance(node, ast.List): return [parse(v) for v in node.elts] + if (isinstance(node, ast.Call) and isinstance(node.func, ast.Name) + and node.func.id == 'Variant' and not node.keywords and len(node.args) == 2): + tag, payload = map(parse, node.args) + if type(tag) is int and tag >= 0: + return dict(tag=tag, payload=payload) if isinstance(node, ast.Call) and isinstance(node.func, ast.Name) and not node.keywords and len(node.args) == 1: arg = parse(node.args[0]) if node.func.id == 'Int' and type(arg) is int: diff --git a/interactive/server/bench/ldbc/compare.py b/interactive/server/bench/ldbc/compare.py index 9d8f190fb..7f79d77b7 100644 --- a/interactive/server/bench/ldbc/compare.py +++ b/interactive/server/bench/ldbc/compare.py @@ -11,35 +11,49 @@ def main(): parser.add_argument('candidate', type=Path) args = parser.parse_args() before, after = [json.loads(path.read_text()) for path in (args.baseline, args.candidate)] + suite = before['format_version'] == 'snb-suite-1' for report in (before, after): - if report['status'] != 'passed' or report['format_version'] != 1: - parser.error('both reports must be successful format-version-1 runs') - for key in ('data', 'parameter_bank', 'standing', 'retracted_rows', 'reference_answers', - 'platform', 'python', 'logical_cpus'): + if report['status'] != 'passed' or report['format_version'] != ('snb-suite-1' if suite else 1): + parser.error('both reports must be successful runs of the same supported format') + fields = ('catalogue', 'spec_commit', 'changed_sha256', 'changes') if suite else ('standing', 'retracted_rows', 'reference_answers') + for key in ('data', 'parameter_bank', 'platform', 'python', 'logical_cpus', *fields): if before[key] != after[key]: parser.error(f'incomparable {key}') - for key in ('workers', 'queries', 'rounds', 'warmup', 'batch_size', 'changes'): + for key in ('workers', 'queries', 'rounds', 'warmup', 'batch_size', 'changes', *(('mode', 'isolated') if suite else ())): if before['config'][key] != after['config'][key]: parser.error(f'incomparable configuration: {key}') - for source in ('client.py', 'workload.py', 'run.py'): + sources = ('client.py', 'run.py', 'suite.py', 'snb/data.py', 'snb/parameters.py', 'snb/witness.py') if suite else ('client.py', 'workload.py', 'run.py') + for source in sources: if before['sources_sha256'][source] != after['sources_sha256'][source]: parser.error(f'measurement/reference code changed: {source}; establish a fresh baseline') - changed_plans = [name for name, sha in before['sources_sha256'].items() - if name.endswith('.ddp') and sha != after['sources_sha256'].get(name)] + plan_field = 'plans_sha256' if suite else 'sources_sha256' + changed_plans = [name for name, sha in before[plan_field].items() + if name.endswith('.ddp') and sha != after[plan_field].get(name)] if changed_plans: print('Changed plans (answers checked): ' + ', '.join(changed_plans)) - a = {r['backend']: r for r in before['runs']} - b = {r['backend']: r for r in after['runs']} + def run_key(r): + return r['backend'] + ('/' + ','.join(r['queries']) if suite else '') + a = {run_key(r): r for r in before['runs']} + b = {run_key(r): r for r in after['runs']} if a.keys() != b.keys(): parser.error('backend sets differ') for backend in a: - if a[backend]['requests'] != b[backend]['requests']: + bindings = 'bindings' if suite else 'requests' + if a[backend][bindings] != b[backend][bindings]: parser.error(f'{backend}: request schedule changed') + if suite: + def answers(run): + return [{k: e[k] for k in ('round', 'state', 'phase', 'rows', 'answer_sha256')} + for e in run['events'] if 'answer_sha256' in e] + if answers(a[backend]) != answers(b[backend]): + parser.error(f'{backend}: answers changed') + if a[backend]['summary'].keys() != b[backend]['summary'].keys(): + parser.error(f'{backend}: measured phases changed') print(f'\n{backend}: milliseconds, baseline / candidate; >1x is faster') print(f'{"phase":36} {"baseline":>10} {"candidate":>10} {"ratio":>8}') for phase, old in a[backend]['summary'].items(): new = b[backend]['summary'][phase] - ratio = old['median_ms'] / new['median_ms'] + ratio = old['median_ms'] / new['median_ms'] if new['median_ms'] else float('inf') print(f'{phase:36} {old["median_ms"]:10.3f} {new["median_ms"]:10.3f} {ratio:7.2f}x') diff --git a/interactive/server/bench/ldbc/snb/__init__.py b/interactive/server/bench/ldbc/snb/__init__.py new file mode 100644 index 000000000..89361caeb --- /dev/null +++ b/interactive/server/bench/ldbc/snb/__init__.py @@ -0,0 +1 @@ +"""Readable SNB read-query definitions and benchmark authoring support.""" diff --git a/interactive/server/bench/ldbc/snb/data.py b/interactive/server/bench/ldbc/snb/data.py new file mode 100644 index 000000000..3ce5f106f --- /dev/null +++ b/interactive/server/bench/ldbc/snb/data.py @@ -0,0 +1,130 @@ +"""Project the 18 SNB BI composite-merged-fk snapshot tables; no query results.""" +import csv +from datetime import datetime, timezone +import gzip + +# Field names are shared with the named relational query definitions. +SCHEMA = { + 'person': 'id:int first:bytes last:bytes gender:bytes birthday:int created:int ip:bytes browser:bytes city:int birthmonth:int birthdaynum:int creationmonth:int', + 'message': 'id:int kind:int created:int creator:int country:int content:bytes image:bytes length:int language:bytes forum:int parent:int year:int month:int day:int', + 'forum': 'id:int title:bytes created:int moderator:int', + 'knows': 'src:int dst:int created:int', + 'member': 'forum:int person:int created:int', + 'tag': 'id:int name:bytes class:int', + 'tagclass': 'id:int name:bytes parent:int', + 'place': 'id:int name:bytes type:bytes parent:int', + 'org': 'id:int name:bytes type:bytes place:int', + 'interest': 'person:int tag:int', + 'mtag': 'message:int tag:int', + 'ftag': 'forum:int tag:int', + 'likes': 'person:int message:int created:int', + 'study': 'person:int org:int year:int', + 'work': 'person:int org:int year:int', + 'email': 'person:int value:bytes', + 'language': 'person:int value:bytes', +} +SCHEMA = {name: dict(field.split(':') for field in fields.split()) for name, fields in SCHEMA.items()} +ENTITIES = ('Person', 'Forum', 'Post', 'Comment', 'Person_knows_Person', + 'Forum_hasMember_Person', 'Forum_hasTag_Tag', 'Person_hasInterest_Tag', + 'Person_likes_Comment', 'Person_likes_Post', 'Person_studyAt_University', + 'Person_workAt_Company', 'Post_hasTag_Tag', 'Comment_hasTag_Tag') +STATIC = ('Place', 'Organisation', 'Tag', 'TagClass') +EDGE_KEYS = { + 'Person_knows_Person': ('Person1Id', 'Person2Id'), + 'Forum_hasMember_Person': ('ForumId', 'PersonId'), + 'Forum_hasTag_Tag': ('ForumId', 'TagId'), + 'Person_hasInterest_Tag': ('PersonId', 'TagId'), + 'Person_likes_Comment': ('PersonId', 'CommentId'), + 'Person_likes_Post': ('PersonId', 'PostId'), + 'Person_studyAt_University': ('PersonId', 'UniversityId'), + 'Person_workAt_Company': ('PersonId', 'CompanyId'), + 'Post_hasTag_Tag': ('PostId', 'TagId'), + 'Comment_hasTag_Tag': ('CommentId', 'TagId'), +} + + +def key(entity, row): + if entity in EDGE_KEYS: + fields = tuple(int(row[f]) for f in EDGE_KEYS[entity]) + return tuple(sorted(fields)) if entity == 'Person_knows_Person' else fields + return int(row['id']) + + +def number(row, name): + return int(row.get(name) or -1) + + +def date(text): + return datetime.fromisoformat(text.replace('Z', '+00:00')).astimezone(timezone.utc) + + +def millis(text): + delta = date(text) - datetime(1970, 1, 1, tzinfo=timezone.utc) + return delta.days * 86400000 + delta.seconds * 1000 + delta.microseconds // 1000 + + +def load(snapshot): + """Read an existing initial_snapshot directory. Never download data.""" + tables = {entity: {} for entity in (*ENTITIES, *STATIC)} + for entity in tables: + directory = snapshot / ("static" if entity in STATIC else "dynamic") / entity + paths = sorted([*directory.glob("*.csv"), *directory.glob("*.csv.gz")]) + if not paths: + raise FileNotFoundError(f"missing {entity} CSV in {directory}") + for path in paths: + opener = gzip.open if path.suffix == ".gz" else open + with opener(path, "rt", newline="", encoding="utf-8") as source: + for row in csv.DictReader(source, delimiter="|"): + identity = key(entity, row) + if identity in tables[entity]: + raise ValueError(f"duplicate {entity} {identity}") + tables[entity][identity] = row + return project(tables) + + +def project(tables): + result = {name: set() for name in SCHEMA} + def add(name, *fields): + assert len(fields) == len(SCHEMA[name]), name + result[name].add(tuple(fields)) + for pid,r in tables['Person'].items(): + birthday = date(r['birthday']+'T00:00:00+00:00') + created = date(r['creationDate']) + add('person', pid,r['firstName'],r['lastName'],r['gender'],millis(birthday.isoformat()),millis(r['creationDate']), + r['locationIP'],r['browserUsed'],number(r,'LocationCityId'),birthday.month,birthday.day,created.year*12+created.month) + for source,target in (('email','email'),('language','language')): + for value in r[source].split(';'): + if value: + add(target,pid,value) + for tag,kind in enumerate(('Post','Comment')): + for mid,r in tables[kind].items(): + created = date(r['creationDate']) + add('message',mid,tag,millis(r['creationDate']),number(r,'CreatorPersonId'),number(r,'LocationCountryId'), + r['content'],r.get('imageFile',''),number(r,'length'),r.get('language',''),number(r,'ContainerForumId'), + number(r,'ParentPostId') if r.get('ParentPostId') else number(r,'ParentCommentId'), + created.year,created.year*12+created.month,millis(created.replace(hour=0,minute=0,second=0,microsecond=0).isoformat())) + for fid,r in tables['Forum'].items(): + add('forum',fid,r['title'],millis(r['creationDate']),number(r,'ModeratorPersonId')) + for (a,b),r in tables['Person_knows_Person'].items(): + add('knows',a,b,millis(r['creationDate'])) + for _,r in tables['Forum_hasMember_Person'].items(): + add('member',number(r,'ForumId'),number(r,'PersonId'),millis(r['creationDate'])) + for kind, relation, fields in ( + ('Tag','tag',('id','name','TypeTagClassId')), + ('TagClass','tagclass',('id','name','SubclassOfTagClassId')), + ('Place','place',('id','name','type','PartOfPlaceId')), + ('Organisation','org',('id','name','type','LocationPlaceId')), + ('Person_hasInterest_Tag','interest',('PersonId','TagId')), + ('Post_hasTag_Tag','mtag',('PostId','TagId')), + ('Comment_hasTag_Tag','mtag',('CommentId','TagId')), + ('Forum_hasTag_Tag','ftag',('ForumId','TagId')), + ('Person_likes_Post','likes',('PersonId','PostId','creationDate')), + ('Person_likes_Comment','likes',('PersonId','CommentId','creationDate')), + ('Person_studyAt_University','study',('PersonId','UniversityId','classYear')), + ('Person_workAt_Company','work',('PersonId','CompanyId','workFrom')), + ): + for r in tables[kind].values(): + add(relation,*(millis(r[f]) if f=='creationDate' else r[f] if f in ('name','type') else number(r,f) for f in fields)) + ids = [m[0] for m in result['message']] + assert len(ids) == len(set(ids)), 'Post and Comment IDs overlap' + return result diff --git a/interactive/server/bench/ldbc/snb/parameters.py b/interactive/server/bench/ldbc/snb/parameters.py new file mode 100644 index 000000000..4b9f6106e --- /dev/null +++ b/interactive/server/bench/ldbc/snb/parameters.py @@ -0,0 +1,118 @@ +"""Deterministic smoke bindings and snapshot checks, outside engine timings.""" +from collections import Counter +from datetime import datetime, timezone +import itertools +import struct + +from .data import millis +from .rel import evaluate + + +def parameters(graph, query=None): + people=sorted(graph['person']) + messages=sorted(graph['message']) + authors=Counter(m[3] for m in messages if m[1]==1) + pid=authors.most_common(1)[0][0] + knows=sorted(graph['knows']) + p1,p2,_=next((edge for edge in knows if pid in edge[:2]),knows[0]) + tags={r[0]:r for r in graph['tag']} + places={r[0]:r for r in graph['place']} + classes={r[0]:r for r in graph['tagclass']} + tag=Counter(r[1] for r in graph['mtag']).most_common(1)[0][0] + classid=tags[tag][2] + tagged_ids={r[0] for r in graph['mtag'] if r[1]==tag} + tagday=Counter(m[13] for m in messages if m[0] in tagged_ids).most_common(1)[0][0] + countries=Counter(places[p[8]][3] for p in people) + countries=[places[k][1] for k,_ in countries.most_common()] + parent=Counter(m[10] for m in messages if m[1]==1).most_common(1)[0][0] + result = dict(rid=1,pid=pid,p1=p1,p2=p2,firstname=people[0][1],mid=parent, + maxdate=millis('2014-01-01T00:00:00Z'),start=millis('2010-01-01T00:00:00Z'), + end=millis('2014-01-01T00:00:00Z'),date=millis('2010-01-01T00:00:00Z'),days=1500, + mindate=millis('2010-01-01T00:00:00Z'),month=people[0][9],tagname=tags[tag][1], + classname=classes[classid][1],countryname=countries[0],countryx=countries[0],countryy=countries[1], + country1=countries[0],country2=countries[1],workyear=2100,minimum=1,maximum=4, + length=500,language='en',enabled=1,delta=12,maxknows=100,city1=people[0][8],city2=people[1][8], + company=next(r[1] for r in graph['org'] if r[2].lower()=='company'), + taga=tags[tag][1],tagb=tags[tag][1],datea=tagday,dateb=tagday,endmonth=2014*12+1) + # Deterministic smoke parameters, not LDBC's official parameter generator. + # Select witnesses from input facts; query answers are never fed to DDIR. + homes={p[0]:places[p[8]][3] for p in people} + if query=='bi2': result['date']=tagday + if query=='ic3': + visits={p[0]:set() for p in people} + for m in messages: + if m[3] in homes and m[4]!=homes[m[3]]: visits[m[3]].add(m[4]) + for a,b,_ in knows: + author,start=(a,b) if len(visits[a])>=2 else (b,a) + foreign=sorted(visits[author]-{homes[start]}) + if len(foreign)>=2: + result.update(pid=start,countryx=places[foreign[0]][1],countryy=places[foreign[1]][1]) + break + if query=='bi14': + for a,b,_ in knows: + if homes[a]!=homes[b]: + result.update(country1=places[homes[a]][1],country2=places[homes[b]][1]) + break + if query=='bi20': + studies={p[0]:set() for p in people} + for person,org,_ in graph['study']: studies[person].add(org) + employers={p[0]:[] for p in people} + orgs={r[0]:r for r in graph['org']} + for person,org,_ in sorted(graph['work']): employers[person].append(orgs[org][1]) + for a,b,_ in knows: + if studies[a]&studies[b] and (employers[a] or employers[b]): + employee,start=(a,b) if employers[a] else (b,a) + result.update(p2=start,company=employers[employee][0]) + break + return result + + +def json_value(value): + if isinstance(value,float): + bits=struct.unpack('>Q',struct.pack('>d',value))[0] + ordered=(~bits & ((1<<64)-1)) if bits>>63 else bits^(1<<63) + payload=ordered^(1<<63) + return dict(tag=0,payload=payload if payload < 1<<63 else payload-(1<<64)) + if isinstance(value,tuple): return [json_value(v) for v in value] + if isinstance(value,bool): return int(value) + return value + + +def reference(plan, schemas, data): + encoded={name:[tuple(tuple(v.encode()) if kind=='bytes' else v for v,kind in zip(row,schema.values())) for row in data[name]] for name,schema in schemas} + rows=evaluate(plan,encoded) + values=[n for n in plan.fields if n not in ('rid','rank')] + return sorted([[[r['rid'],r['rank']],[json_value(r[n]) for n in values],1] for r in rows], key=lambda row: row[0]) + + +def requests(name, schema, params, rid): + """One binding, or BI12's language-set relation, with explicit identity.""" + params = dict(params, rid=rid) + if 'endmonth' in schema: + end = datetime.fromtimestamp(params['end']/1000, timezone.utc) + params['endmonth'] = end.year*12 + end.month + if name == 'bi12' and params['language'] == []: + params.update(language='', enabled=0) + choices = [] + for field, kind in schema.items(): + value = params[field] + values = value if isinstance(value, list) else [value] + if isinstance(value, list) and (name, field) != ('bi12', 'language'): + raise ValueError('only BI12 language takes a set of values') + if any(type(v) is not (int if kind == 'int' else str) for v in values): + raise ValueError(f'{name}: wrong type for {field}') + choices.append(values) + return set(itertools.product(*choices)) + + +def alternate(graph, params): + """Vary actual bindings, not just request IDs; not an official parameter mix.""" + result = dict(params) + for field, relation, index in (('pid', 'person', 0), ('p1', 'person', 0), + ('p2', 'person', 0), ('mid', 'message', 0), + ('tagname', 'tag', 1), ('firstname', 'person', 1)): + choices = sorted({row[index] for row in graph[relation]}) + value = params[field] + result[field] = next((v for v in choices if v > value), choices[0]) + result['maxdate'] -= 86400000 + return result diff --git a/interactive/server/bench/ldbc/snb/queries.py b/interactive/server/bench/ldbc/snb/queries.py new file mode 100644 index 000000000..abf8f3d64 --- /dev/null +++ b/interactive/server/bench/ldbc/snb/queries.py @@ -0,0 +1,515 @@ +"""Current SNB read queries, expressed using named fields and relational steps. + +Specification: ldbc_snb_docs b2269610f433da72e7c97041f01680aae369a903. +Interactive 14 uses v2 (one cheapest interaction path), not v1's all-shortest +paths. The definitions compile to ordinary DDP; there are no Python query +callbacks in measured engine execution. +""" +from functools import cached_property + +from .data import SCHEMA +from .rel import R, c, col, expr, choose, absolute, call, empty + +DAY = 86400000 +QUERIES = {} + + +def query(name, params, title): + def register(fn): + QUERIES[name] = (fn,dict(field.split(':') for field in ('rid:int '+params).split()),title) + return fn + return register + + +def used(e): + e = expr(e) + if e.op == 'column': return {e.args[0]} + return set().union(*(used(a) for a in e.args if hasattr(a,'op'))) + + +def finish(rows, outputs, order, limit=None): + needed = {'rid'} | set().union(*(used(e) for e in (*outputs.values(),*order))) + rows = rows.select(*(n for n in rows.fields if n in needed)) + return rows.rank(order,limit).select('rid','rank',**outputs) + + +def body(content, image): return choose(call('len',content)>0,content,image) +def sumof(e): return ('sum',expr(e)) +def count(): return ('count',expr(1)) + + +class Context: + def __init__(self): + self.tables = {name:R.source(name,schema) for name,schema in SCHEMA.items()} + def __getattr__(self, name): return self.tables[name] + + @cached_property + def edges(self): + k = self.knows + return k.union(k.select(src=c.dst,dst=c.src,created=c.created)).distinct().semi(self.person,{'src':'id'}).semi(self.person,{'dst':'id'}) + + @cached_property + def residents(self): + return self.person.join(self.place,{'city':'id'},'city_').join(self.place,{'city_parent':'id'},'country_') + + @cached_property + def threads(self): + roots = self.message.where(c.kind==0).select(mid=c.id,root=c.id,forum=c.forum,author=c.creator,language=c.language) + links = self.message.where(c.kind==1).select('parent',child=c.id) + return roots.closure(links,{'mid':'parent'}, + dict(mid=c.next_child,root=c.root,forum=c.forum,author=c.author,language=c.language), + keys=('mid',),best=('root','forum','author','language')) + + @cached_property + def ancestry(self): + roots = self.tagclass.select(descendant=c.id,ancestor=c.id) + links = self.tagclass.where(c.parent>=0).select('id','parent') + return roots.closure(links,{'ancestor':'id'},dict(descendant=c.descendant,ancestor=c.next_parent), + keys=('descendant','ancestor'),best=()) + + @cached_property + def tagged(self): + return self.mtag.join(self.tag,{'tag':'id'},'tag_') + + @cached_property + def replies(self): + return self.message.where(c.kind==1).join(self.message,{'parent':'id'},'parent_') + + @cached_property + def interactions(self): + a = self.replies.select(comment=c.id,src=c.creator,dst=c.parent_creator,score2=choose(c.parent_kind==0,2,1)).where(c.src!=c.dst) + return a.union(a.select('comment',src=c.dst,dst=c.src,score2=c.score2)) + + @cached_property + def weighted(self): + pairs = self.interactions.group(('src','dst'),n=count()).semi(self.edges,{'src':'src','dst':'dst'}) + # Exact integer thresholds for max(round(40-sqrt(n)),1), no float + # approximation and no lookup precomputation outside the dataflow. + weight = 1+sum(choose(c.n < k*(k-1)+1,1,0) for k in range(1,40)) + return pairs.select('src','dst',weight=weight) + + def friends(self, q, hops): + q = q.semi(self.person,{'pid':'id'}) + frontier = q.select(*q.fields,node=c.pid,distance=0) + result = None + for _ in range(hops): + frontier = frontier.join(self.edges,{'node':'src'},'edge_').select(*q.fields,node=c.edge_dst,distance=c.distance+1) + result = frontier if result is None else result.union(frontier) + return result.where(c.node!=c.pid).group(('rid','node'),distance=('min',c.distance)).join(q,{'rid':'rid'}) + + def shortest(self, q, edges, start='p1', path=False, floating=False): + # Hop-indexed Bellman-Ford. Bounding by |V|-1 avoids count-to-infinity + # when deletions disconnect a cyclic component from the source. + bound = self.person.group((),bound=count()) + seed = q.join(bound).select('rid',node=col(start),hops=0,cost=call('float',0) if floating else 0,bound=c.bound-1, + **({'path':call('list',col(start))} if path else {})) + projection = dict(rid=c.rid,node=c.next_dst,hops=c.hops+1,cost=call('fadd',c.cost,c.next_weight) if floating else c.cost+c.next_weight,bound=c.bound) + if path: projection['path'] = call('append',c.path,call('list',c.next_dst)) + reached = seed.closure(edges,{'node':'src',**({'rid':'rid'} if 'rid' in edges.fields else {})},projection, + keys=('rid','node','hops'),best=('cost','bound',*(['path'] if path else [])),condition=c.hops=c.start)&(c.m_created0)&(c.ny>0)) + return finish(r,dict(person=c.node,first=c.p_first,last=c.p_last,xCount=c.nx,yCount=c.ny,total=c.nx+c.ny),[-(c.nx+c.ny),c.node],20) + + +@query('ic4','pid:int start:int days:int','New topics') +def ic4(x,q): + r=x.friends(q,1).join(x.message.where(c.kind==0),{'node':'creator'},'m_').join(x.tagged,{'m_id':'message'},'t_') + old=r.where(c.m_created=c.start)&(c.m_createdc.mindate).select('rid','node','forum').distinct() + posts=members.join(x.message.where(c.kind==0),{'node':'creator','forum':'forum'},'m_').group(('rid','forum'),n=count()) + r=members.select('rid','forum').distinct().left(posts,{'rid':'rid','forum':'forum'},dict(n=0)).join(x.forum,{'forum':'id'},'f_') + return finish(r,dict(title=c.f_title,count=c.n),[-c.n,c.forum],20) + + +@query('ic6','pid:int tagname:bytes','Tag co-occurrence') +def ic6(x,q): + r=x.friends(q,2).join(x.message.where(c.kind==0),{'node':'creator'},'m_').join(x.tagged,{'m_id':'message'},'t_').where(c.t_tag_name==c.tagname) + r=r.select('rid','m_id','tagname').distinct().join(x.tagged,{'m_id':'message'},'other_').where(c.other_tag_name!=c.tagname) + r=r.group(('rid','other_tag_name'),n=count()) + return finish(r,dict(tag=c.other_tag_name,count=c.n),[-c.n,c.other_tag_name],10) + + +@query('ic7','pid:int','Recent likers') +def ic7(x,q): + r=q.join(x.message,{'pid':'creator'},'m_').join(x.likes,{'m_id':'message'},'l_') + r=r.group(('rid','pid','l_person'),best=('min',call('tuple',-c.l_created,c.m_id))) + r=r.select('rid','pid','l_person',when=-c.best.at(0),mid=c.best.at(1)).join(x.message,{'mid':'id'},'m_').join(x.person,{'l_person':'id'},'p_') + flags=x.edges.select('src','dst',known=1) + r=r.left(flags,{'pid':'src','l_person':'dst'},dict(src=c.pid,dst=c.l_person,known=0)) + return finish(r,dict(person=c.l_person,first=c.p_first,last=c.p_last,created=c.when,message=c.mid, + content=body(c.m_content,c.m_image),latency=call('idiv',c.when-c.m_created,60000),isNew=1-c.known),[-c.when,c.l_person],20) + + +@query('ic8','pid:int','Recent replies') +def ic8(x,q): + r=q.join(x.replies,{'pid':'parent_creator'}).join(x.person,{'creator':'id'},'p_') + return finish(r,dict(author=c.creator,first=c.p_first,last=c.p_last,created=c.created,comment=c.id,content=c.content),[-c.created,c.id],20) + + +@query('ic9','pid:int maxdate:int','Recent messages by friends or friends of friends') +def ic9(x,q): return recent(x,q,2) + + +@query('ic10','pid:int month:int','Friend recommendation') +def ic10(x,q): + candidates=x.friends(q,2).where(c.distance==2).join(x.residents,{'node':'id'},'p_') + nextmonth=choose(c.month==12,1,c.month+1) + candidates=candidates.where(((c.p_birthmonth==c.month)&(c.p_birthdaynum>=21))|((c.p_birthmonth==nextmonth)&(c.p_birthdaynum<22))) + posts=candidates.select('rid','pid','node').join(x.message.where(c.kind==0),{'node':'creator'},'m_') + common=posts.join(x.mtag,{'m_id':'message'},'t_').semi(x.interest,{'pid':'person','t_tag':'tag'}).select('rid','node','m_id').distinct() + positive=common.select('rid','node',score=1) + negative=posts.anti(common,{'rid':'rid','node':'node','m_id':'m_id'}).select('rid','node',score=-1) + scores=positive.union(negative).group(('rid','node'),score=sumof(c.score)) + r=candidates.left(scores,{'rid':'rid','node':'node'},dict(score=0)) + return finish(r,dict(person=c.node,first=c.p_first,last=c.p_last,score=c.score,gender=c.p_gender,city=c.p_city_name),[-c.score,c.node],10) + + +@query('ic11','pid:int countryname:bytes workyear:int','Job referral') +def ic11(x,q): + r=x.friends(q,2).join(x.work,{'node':'person'}).where(c.year=c.date)&(c.m_created=c.date+100*DAY,1,0))) + return finish(r,dict(tag=c.t_tag_name,window1=c.n1,window2=c.n2,diff=absolute(c.n1-c.n2)),[-absolute(c.n1-c.n2),c.t_tag_name],100) + + +@query('bi3','classname:bytes countryname:bytes','Popular topics in a country') +def bi3(x,q): + r=q.join(x.tagclass,{'classname':'name'},'class_').join(x.tagged,{'class_id':'tag_class'},'t_').select('rid','countryname','t_message').distinct() + r=r.join(x.threads,{'t_message':'mid'}).join(x.forum,{'forum':'id'},'f_').join(x.residents,{'f_moderator':'id'},'p_').where(c.p_country_name==c.countryname) + r=r.group(('rid','forum','f_title','f_created','f_moderator'),n=count()) + return finish(r,dict(forum=c.forum,title=c.f_title,created=c.f_created,moderator=c.f_moderator,count=c.n),[-c.n,c.forum],20) + + +@query('bi4','date:int','Top message creators by country') +def bi4(x,q): + forums=q.join(x.forum).where(c.created>c.date).join(x.member,{'id':'forum'},'member_').join(x.residents,{'member_person':'id'},'p_') + popularity=forums.group(('rid','id','p_country_id'),n=count()).group(('rid','id'),negative=('min',-c.n)) + top=popularity.rank([c.negative,c.id],100).select('rid',forum=c.id) + members=top.join(x.member,{'forum':'forum'}).select('rid','person').distinct() + messages=top.join(x.threads,{'forum':'forum'}).join(x.message,{'mid':'id'},'m_') + counts=messages.group(('rid','m_creator'),n=count()).rename(m_creator='person') + r=members.left(counts,{'rid':'rid','person':'person'},dict(n=0)).join(x.person,{'person':'id'},'p_') + return finish(r,dict(person=c.person,first=c.p_first,last=c.p_last,created=c.p_created,count=c.n),[-c.n,c.person],100) + + +@query('bi5','tagname:bytes','Most active posters of a topic') +def bi5(x,q): + messages=tagged_messages(x,q).select('rid','m_id','m_creator').distinct() + counts=messages.group(('rid','m_creator'),messages=count()) + likes=messages.join(x.likes,{'m_id':'message'}).group(('rid','m_creator'),likes=count()) + replies=messages.join(x.message.where(c.kind==1),{'m_id':'parent'},'r_').group(('rid','m_creator'),replies=count()) + r=counts.left(likes,{'rid':'rid','m_creator':'m_creator'},dict(likes=0)).left(replies,{'rid':'rid','m_creator':'m_creator'},dict(replies=0)) + score=c.messages+2*c.replies+10*c.likes + return finish(r,dict(person=c.m_creator,replies=c.replies,likes=c.likes,messages=c.messages,score=score),[-score,c.m_creator],100) + + +@query('bi6','tagname:bytes','Most authoritative users on a topic') +def bi6(x,q): + messages=tagged_messages(x,q).select('rid','m_id','m_creator').distinct() + authors=messages.select('rid','m_creator').distinct() + pairs=messages.join(x.likes,{'m_id':'message'}).select('rid','m_creator','person').distinct() + popularity=x.message.join(x.likes,{'id':'message'},'l_').group(('creator',),popularity=count()) + scores=pairs.join(popularity,{'person':'creator'}).group(('rid','m_creator'),score=sumof(c.popularity)) + r=authors.left(scores,{'rid':'rid','m_creator':'m_creator'},dict(score=0)) + return finish(r,dict(person=c.m_creator,score=c.score),[-c.score,c.m_creator],100) + + +@query('bi7','tagname:bytes','Related topics') +def bi7(x,q): + messages=tagged_messages(x,q).select('rid','m_id').distinct() + replies=messages.join(x.message.where(c.kind==1),{'m_id':'parent'},'r_').anti(messages,{'rid':'rid','r_id':'m_id'}) + r=replies.join(x.tagged,{'r_id':'message'},'t_').group(('rid','t_tag_name'),n=count()) + return finish(r,dict(tag=c.t_tag_name,count=c.n),[-c.n,c.t_tag_name],100) + + +@query('bi8','tagname:bytes start:int end:int','Central person for a tag') +def bi8(x,q): + interests=q.join(x.tag,{'tagname':'name'},'t_').join(x.interest,{'t_id':'tag'}).select('rid','person',score=100) + messages=tagged_messages(x,q).where((c.m_created>c.start)&(c.m_created=c.start)&(c.created<=c.end)) + threads=posts.select('rid','creator','id').group(('rid','creator'),threads=count()) + messages=posts.join(x.threads,{'id':'root'},'t_').join(x.message,{'t_mid':'id'},'m_').where((c.m_created>=c.start)&(c.m_created<=c.end)) + counts=messages.group(('rid','creator'),messages=count()) + r=threads.join(counts,{'rid':'rid','creator':'creator'}).join(x.person,{'creator':'id'},'p_') + return finish(r,dict(person=c.creator,first=c.p_first,last=c.p_last,threads=c.threads,messages=c.messages),[-c.messages,c.creator],100) + + +@query('bi10','pid:int countryname:bytes classname:bytes minimum:int maximum:int','Experts in social circle') +def bi10(x,q): + # The parameter maximum is not a compile-time unrolling bound. + paths=x.shortest(q,x.edges.select('src','dst',weight=1),start='pid').join(q,{'rid':'rid'}) + paths=paths.where((c.cost>=c.minimum)&(c.cost<=c.maximum)&(c.node!=c.pid)).join(x.residents,{'node':'id'},'p_').where(c.p_country_name==c.countryname) + qualified=paths.join(x.message,{'node':'creator'},'m_').join(x.tagged,{'m_id':'message'},'t_').join(x.tagclass,{'t_tag_class':'id'},'class_').where(c.class_name==c.classname).select('rid','node','m_id').distinct() + r=qualified.join(x.tagged,{'m_id':'message'},'tag_').group(('rid','node','tag_tag_name'),n=count()) + return finish(r,dict(person=c.node,tag=c.tag_tag_name,count=c.n),[-c.n,c.tag_tag_name,c.node],100) + + +@query('bi11','countryname:bytes start:int end:int','Friend triangles') +def bi11(x,q): + residents=q.join(x.residents,{'countryname':'country_name'}).select('rid','start','end',person=c.id) + edges=residents.join(x.edges,{'person':'src'}).where((c.created>=c.start)&(c.created<=c.end)).semi(residents,{'rid':'rid','dst':'person'}).select('rid','src','dst') + triangles=edges.join(edges,{'rid':'rid','dst':'src'},'e_').where((c.src0)&(c.created>c.start)&(c.length=c.created)&(c.m_created<=c.end)).group(('rid','id'),n=count()) + zombies=candidates.left(counts,{'rid':'rid','id':'id'},dict(n=0)).where(c.n0,call('fdiv',call('float',c.zombies),call('float',c.total)),call('float',0))) + return finish(r,dict(person=c.zombie,zombieLikes=c.zombies,totalLikes=c.total,score=c.score),[call('fneg',c.score),c.zombie],100) + + +@query('bi15','p1:int p2:int start:int end:int','Trusted connection paths through forums') +def bi15(x,q): + interactions=q.join(x.interactions).join(x.threads,{'comment':'mid'},'t_').join(x.forum,{'t_forum':'id'},'f_') + interactions=interactions.where((c.f_created>=c.start)&(c.f_created<=c.end)).group(('rid','src','dst'),score2=sumof(c.score2)) + pairs=q.select('rid').join(x.edges.select('src','dst')).left(interactions,{'rid':'rid','src':'src','dst':'dst'},dict(score2=0)) + edges=pairs.select('rid','src','dst',weight=call('fdiv',call('float',2),call('float',c.score2+2))) + paths=x.shortest(q,edges,floating=True).join(q,{'rid':'rid'}).where(c.node==c.p2).select('rid','cost') + r=q.left(paths,{'rid':'rid'},dict(cost=call('float',-1))) + return finish(r,dict(weight=c.cost),[c.rid]) diff --git a/interactive/server/bench/ldbc/snb/rel.py b/interactive/server/bench/ldbc/snb/rel.py new file mode 100644 index 000000000..f06d81d46 --- /dev/null +++ b/interactive/server/bench/ldbc/snb/rel.py @@ -0,0 +1,310 @@ +"""Small named relational notation with DDP emission and Python evaluation. + +This is benchmark authoring infrastructure, not another execution backend. +The Python evaluator checks DDP lowering and incremental execution against +snapshot evaluation of the same logical plan. It is not an independent query +specification oracle; query semantics also need hand-checked witnesses. +""" +from collections import Counter, defaultdict +from dataclasses import dataclass +import operator + + +@dataclass(eq=False) +class E: + op: str + args: tuple + + def __add__(self, x): return E('+', (self, expr(x))) + def __radd__(self, x): return expr(x)+self + def __sub__(self, x): return E('-', (self, expr(x))) + def __rsub__(self, x): return expr(x)-self + def __mul__(self, x): return E('*', (self, expr(x))) + def __rmul__(self, x): return expr(x)*self + def __neg__(self): return 0-self + def __eq__(self, x): return E('==', (self, expr(x))) + def __ne__(self, x): return E('!=', (self, expr(x))) + def __lt__(self, x): return E('<', (self, expr(x))) + def __le__(self, x): return E('<=', (self, expr(x))) + def __gt__(self, x): return E('>', (self, expr(x))) + def __ge__(self, x): return E('>=', (self, expr(x))) + def __and__(self, x): return E('&&', (self, expr(x))) + def __or__(self, x): return E('or', (self, expr(x))) + def __bool__(self): raise TypeError('use & / | for relational expressions') + def at(self, index): return E('at', (self, index)) + + +def expr(x): return x if isinstance(x,E) else E('literal',(x,)) +def col(name): return E('column',(name,)) +def choose(cond, yes, no): return E('if',(expr(cond),expr(yes),expr(no))) +def absolute(x): return choose(x < 0,-x,x) +def call(name, *args): return E(name,tuple(expr(x) for x in args)) +def empty(kind): return E('empty',(kind,)) + + +class Columns: + def __getattr__(self, name): return col(name) + def __getitem__(self, name): return col(name) +c = Columns() + + +def value(e, row): + op,args = e.op,e.args + if op == 'column': return row[args[0]] + if op == 'literal': + x = args[0] + return tuple(x.encode()) if isinstance(x,str) else x + if op == 'empty': return () + if op == 'at': return value(args[0],row)[args[1]] + if op == 'if': return value(args[1] if value(args[0],row) else args[2],row) + vs = [value(a,row) for a in args] + funcs = {'+':operator.add,'-':operator.sub,'*':operator.mul,'==':operator.eq, + '!=':operator.ne,'<':operator.lt,'<=':operator.le,'>':operator.gt,'>=':operator.ge, + '&&':lambda a,b:int(bool(a) and bool(b)),'or':lambda a,b:int(bool(a) or bool(b)), + 'len':len,'tuple':lambda *xs:tuple(xs),'list':lambda *xs:tuple(xs),'append':operator.add, + 'float':float,'fneg':operator.neg,'fadd':operator.add,'fsub':operator.sub,'fmul':operator.mul,'fdiv':operator.truediv, + 'idiv':lambda a,b: (abs(a)//abs(b))*(-1 if (a<0)!=(b<0) else 1) if b else 0} + return funcs[op](*vs) + + +def term(e, fields): + op,args = e.op,e.args + if op == 'column': return fields[args[0]] + if op == 'literal': + x = args[0] + if isinstance(x,str): + return 'list('+','.join(str(b) for b in x.encode())+')' if x else term(empty('bytes'),fields) + return str(int(x)) + if op == 'empty': + name = {'bytes':'EmptyBytes','strings':'EmptyStrings','triples':'EmptyTriples','ints':'EmptyBytes'}[args[0]] + return f'case {name}(list()) {{ {name}(x) => x }}' + if op == 'at': return f'({term(args[0],fields)})[{args[1]}]' + ts = [term(a,fields) for a in args] + if op in ('+','-','*','==','!=','<','<=','>','>=','&&'): + return '('+f' {op} '.join(ts)+')' + return op+'('+', '.join(ts)+')' + + +class R: + def __init__(self, op, fields, *args): + self.op,self.fields,self.args = op,tuple(fields),args + if len(set(self.fields)) != len(self.fields): + raise ValueError(f'duplicate fields {fields}') + + @staticmethod + def source(name, schema): return R('source',schema,name,dict(schema)) + def select(self, *names, **computed): + pairs = {name:col(name) for name in names} + pairs.update({name:expr(e) for name,e in computed.items()}) + return R('select',pairs,self,pairs) + def rename(self, **mapping): return self.select(**{mapping.get(n,n):col(n) for n in self.fields}) + def where(self, pred): return R('where',self.fields,self,expr(pred)) + def distinct(self): return R('distinct',self.fields,self) + def union(self, other): + assert self.fields == other.fields,(self.fields,other.fields) + return R('union',self.fields,self,other) + def join(self, other, on=None, prefix=''): + on = on or {} + right = {n:prefix+n for n in other.fields if not (n in on.values() and not prefix and n in self.fields)} + return R('join',(*self.fields,*right.values()),self,other,dict(on),right) + def semi(self, other, on): + return R('semi',self.fields,self,other,dict(on)) + def anti(self, other, on): + return R('anti',self.fields,self,other,dict(on)) + def left(self, other, on, defaults, prefix=''): + joined = self.join(other,on,prefix) + missing = self.anti(other,on).select(*self.fields,**defaults) + assert joined.fields == missing.fields,(joined.fields,missing.fields) + return joined.union(missing) + def group(self, keys=(), **aggregates): + # aggregate := ('sum'|'count'|'min'|'collect', scalar expression) + return R('group',(*keys,*aggregates),self,tuple(keys),aggregates) + def rank(self, order, limit=None, groups=('rid',)): + return R('rank',(*self.fields,'rank'),self,tuple(expr(e) for e in order),limit,tuple(groups)) + def closure(self, next_relation, on, select, keys, best, condition=1): + """Least fixpoint of seed UNION recursive equijoin, min per key. + + The step sees left fields and right fields under `next_` prefix. + `best` is the lexicographically ordered value field list. + """ + assert set(keys)|set(best) == set(self.fields) + return R('closure',self.fields,self,next_relation,dict(on),{k:expr(v) for k,v in select.items()},tuple(keys),tuple(best),expr(condition)) + + +def evaluate(root, data, cache=None): + # A caller evaluating several exports against one snapshot can share this + # cache. It must start a fresh cache when the input snapshot changes. + cache = {} if cache is None else cache + def ev(r): + if id(r) in cache: return cache[id(r)] + op,args = r.op,r.args + if op == 'source': + rows = [dict(zip(r.fields,row)) for row in data[args[0]]] + elif op == 'select': rows = [{n:value(e,row) for n,e in args[1].items()} for row in ev(args[0])] + elif op == 'where': rows = [row for row in ev(args[0]) if value(args[1],row)] + elif op == 'distinct': rows = [dict(zip(r.fields,row)) for row in set(tuple(row[n] for n in r.fields) for row in ev(args[0]))] + elif op == 'union': rows = ev(args[0])+ev(args[1]) + elif op in ('join','semi','anti'): + left,right,on = args[:3] + index = defaultdict(list) + for row in ev(right): index[tuple(row[n] for n in on.values())].append(row) + rows = [] + for row in ev(left): + matches = index[tuple(row[n] for n in on)] + if op == 'join': rows.extend({**row,**{out:other[n] for n,out in args[3].items()}} for other in matches) + elif bool(matches) == (op=='semi'): rows.append(row) + elif op == 'group': + source,keys,aggs = args + groups = defaultdict(list) + for row in ev(source): groups[tuple(row[n] for n in keys)].append(row) + rows = [] + for key,members in groups.items(): + row = dict(zip(keys,key)) + for n,(kind,e) in aggs.items(): + vs = [value(expr(e),member) for member in members] + row[n] = {'sum':sum,'count':len,'min':min,'collect':lambda xs:tuple(sorted(xs))}[kind](vs) + rows.append(row) + elif op == 'rank': + source,order,limit,keys = args + groups = defaultdict(list) + for row in ev(source): groups[tuple(row[n] for n in keys)].append(row) + rows = [{**row,'rank':i} for members in groups.values() + for i,row in enumerate(sorted(members,key=lambda row:(tuple(value(e,row) for e in order),tuple(row[n] for n in source.fields)))[:limit])] + elif op == 'closure': + seed,nextrel,on,projection,keys,best,condition = args + state = {} + index = defaultdict(list) + for row in ev(nextrel): index[tuple(row[n] for n in on.values())].append(row) + pending = ev(seed) + while pending: + fresh = {} + for row in pending: + key = tuple(row[n] for n in keys) + if key not in state or tuple(row[n] for n in best) < tuple(state[key][n] for n in best): + state[key] = row + fresh[key] = row + pending = [{n:value(e,{**row,**{'next_'+k:v for k,v in other.items()}}) for n,e in projection.items()} + for row in fresh.values() for other in index[tuple(row[n] for n in on)] + if value(condition,{**row,**{'next_'+k:v for k,v in other.items()}})] + rows = list(state.values()) + else: raise ValueError(op) + cache[id(r)] = rows + return rows + return ev(root) + + +def source_shape(schema): + return '(' + ', '.join({'int': 'int', 'bytes': 'List(int)'}[kind] for kind in schema.values()) + ')' + + +class Compiler: + def __init__(self, imports=False): + self.lines = ['type ByteString = EmptyBytes List(int);', + 'type StringSet = EmptyStrings List(List(int));', + 'type TripleSet = EmptyTriples List((List(int), int, List(int)));'] + self.names = {} + self.inputs = [] + self.imports = imports + + def emit(self, r): + if id(r) in self.names: return self.names[id(r)] + op,args = r.op,r.args + name = 'r'+str(len(self.names)) + self.names[id(r)] = name + def positions(fields, slot=0): return {n:f'${slot}[{i}]' for i,n in enumerate(fields)} + def projection(pairs, refs): return ', '.join(term(expr(e),refs) for e in pairs) + if op == 'source': + self.inputs.append((args[0],args[1])) + if self.imports: + code = 'input 0' if args[0] == 'request' else f'import "ldbc.{args[0]}"' + else: + code = f'input {len(self.inputs)-1}' + code += f' : ({source_shape(args[1])} ; ())' + elif op == 'select': + source,pairs = args + code = f'{self.emit(source)} | map({projection(pairs.values(),positions(source.fields))} ;)' + elif op == 'where': + source,pred = args + code = f'{self.emit(source)} | filter({term(pred,positions(source.fields))})' + elif op == 'distinct': code = f'{self.emit(args[0])} | distinct' + elif op == 'union': code = f'({self.emit(args[0])}) + ({self.emit(args[1])})' + elif op in ('join','semi','anti'): + left,right,on = args[:3] + l,rhs = self.emit(left),self.emit(right) + lkey = projection([col(n) for n in on],positions(left.fields)) + rkey = projection([col(n) for n in on.values()],positions(right.fields)) + keyed_left = f'({l} | key({lkey} ; $0))' + if op == 'join': + keyed_right = f'({rhs} | key({rkey} ; $0))' + out = [f'$1[{i}]' for i in range(len(left.fields))]+[positions(right.fields,2)[n] for n in args[3]] + else: + keyed_right = f'({rhs} | key({rkey} ;) | distinct)' + out = [f'$1[{i}]' for i in range(len(left.fields))] + matched = f'{keyed_left} | join({keyed_right}, ({", ".join(out)} ;))' + code = f'{l} - ({matched})' if op=='anti' else matched + elif op == 'group': + source,keys,aggs = args + source_name = self.emit(source) + refs = positions(source.fields) + if len(aggs)==1 and next(iter(aggs.values()))[0]=='min': + _,e = next(iter(aggs.values())) + code = f'{source_name} | key({projection([col(n) for n in keys],refs)} ; {term(expr(e),refs)}) | min | map($0, $1[0] ;)' + self.lines.append(f'-- {name}: ({", ".join(r.fields)})\nlet {name} = {code};') + return name + vals = projection([e for _,e in aggs.values()],refs) + outs = [f'$0[{i}]' for i in range(len(keys))] + for i,(kind,e) in enumerate(aggs.values()): + if kind=='count': out = 'len($1)' + elif kind=='sum': out = f'fold($1, 0, ^1 + ^0[{i}])' + elif kind=='min': out = f'fold($1, $1[0][{i}], if(^0[{i}] < ^1, ^0[{i}], ^1))' + elif kind=='collect': + # A dedicated value-only collect avoids lists of singleton + # tuples and is useful for nested IC1/IC12 results. + assert len(aggs)==1 + code = f'{source_name} | key({projection([col(n) for n in keys],refs)} ; {term(expr(e),refs)}) | value($1[0]) | collect | map($0, ($1) ;)' + self.lines.append(f'-- {name}: ({", ".join(r.fields)})\nlet {name} = {code};') + return name + else: raise ValueError(kind) + outs.append(out) + code = f'{source_name} | key({projection([col(n) for n in keys],refs)} ; {vals}) | collect | map({", ".join(outs)} ;)' + elif op == 'rank': + source,order,limit,keys = args + src = self.emit(source) + refs = positions(source.fields) + sort = projection(order,refs) + packed = 'tuple('+','.join(refs[n] for n in source.fields)+')' + code = f'{src} | key({projection([col(n) for n in keys],refs)} ; {sort}, {packed}) | collect | flatmap($1)' + if limit is not None: code += f' | filter($1[0] < {limit})' + code += ' | map('+', '.join(f'$1[1][{len(order)}][{i}]' for i in range(len(source.fields)))+', $1[0] ;)' + elif op == 'closure': + seed,nextrel,on,select,keys,best,condition = args + s,n = self.emit(seed),self.emit(nextrel) + refs = positions(seed.fields,1) + refs.update({'next_'+k:v for k,v in positions(nextrel.fields,2).items()}) + fields = positions(seed.fields) + k = projection([col(x) for x in keys],fields) + b = projection([col(x) for x in best],fields) + restore = {**{n:f'$0[{i}]' for i,n in enumerate(keys)},**{n:f'$1[{i}]' for i,n in enumerate(best)}} + # Keep a flat-row recursive variable, with min in its definition. + # Evaluate the condition on the joined environment before reshaping. + combined = list(seed.fields)+['next_'+x for x in nextrel.fields] + joined = ', '.join(refs[x] for x in combined) + step_refs = positions(combined) + self.lines.append(f'{name}_loop: {{\n let proposals = state | key({projection([col(x) for x in on],fields)} ; $0) | join(({n} | key({projection([col(x) for x in on.values()],positions(nextrel.fields))} ; $0)), ({joined} ;)) | filter({term(condition,step_refs)}) | map({projection(select.values(),step_refs)} ;);\n var state = ({s} + proposals) | key({k} ; {b}) | min | map({", ".join(restore[x] for x in seed.fields)} ;);\n}}') + code = f'{name}_loop::state' + else: raise ValueError(op) + self.lines.append(f'-- {name}: ({", ".join(r.fields)})\nlet {name} = {code};') + return name + + def compile(self, root): + return self.compile_many({'result':root}) + + def compile_many(self, roots): + exports=[] + for export,root in roots.items(): + if not export.replace('_','').replace('.','').isalnum(): raise ValueError('invalid export name') + name = self.emit(root) + refs = {n:f'$0[{i}]' for i,n in enumerate(root.fields)} + values = [refs[n] for n in root.fields if n not in ('rid','rank')] + exports.append(f'export "{export}" = {name} | map({refs["rid"]}, {refs["rank"]} ; {", ".join(values)}) | arrange;') + return '\n\n'.join([*self.lines,*exports])+'\n' diff --git a/interactive/server/bench/ldbc/snb/witness.py b/interactive/server/bench/ldbc/snb/witness.py new file mode 100644 index 000000000..be7faa5d8 --- /dev/null +++ b/interactive/server/bench/ldbc/snb/witness.py @@ -0,0 +1,89 @@ +"""Small hand-built graph with non-empty witnesses for every SNB read query. + +This complements the generated fixture; it is not a scale-factor dataset. +""" +from .data import SCHEMA +from .data import millis + + +def graph(): + g={name:set() for name in SCHEMA} + def add(relation,**row): g[relation].add(tuple(row[n] for n in SCHEMA[relation])) + t=millis('2012-01-01T00:00:00Z') + born=millis('1980-01-22T00:00:00Z') + created=millis('2011-01-01T00:00:00Z') + for base,name in ((100,'Alpha'),(200,'Beta'),(300,'Gamma')): + add('place',id=base,name=name,type='Country',parent=-1) + add('place',id=base+1,name=name+' City',type='City',parent=base) + cities={1:301,2:101,3:201,4:301,5:101,6:201,7:101,8:201} + for pid,city in cities.items(): + add('person',id=pid,first='Alex',last='A' if pid==2 else 'Alphabet' if pid==3 else f'Person{pid}',gender='female' if pid%2 else 'male', + birthday=born,created=created,ip='127.0.0.1',browser='Browser',city=city,birthmonth=1,birthdaynum=22,creationmonth=2011*12+1) + add('email',person=pid,value=f'{pid}@example.invalid') + add('language',person=pid,value='en') + add('interest',person=pid,tag=10) + add('study',person=pid,org=20,year=2000+pid%3) + add('org',id=20,name='University',type='University',place=101) + add('org',id=21,name='Company',type='Company',place=100) + add('work',person=7,org=21,year=2005) + add('work',person=2,org=21,year=2008) + for a,b in ((1,2),(1,3),(2,4),(3,4),(2,5),(5,7),(2,7),(4,5),(3,6),(6,8)): + add('knows',src=a,dst=b,created=created+1000) + for fid,moderator in ((1001,1),(1002,2),(1003,3),(1004,4)): + add('forum',id=fid,title=f'Forum {fid}',created=created,moderator=moderator) + add('ftag',forum=fid,tag=10) + for fid,people in ((1001,(2,3,4)),(1002,(2,3,5)),(1003,(1,4,5,6,7,8)),(1004,(4,5))): + for pid in people: add('member',forum=fid,person=pid,created=created+2000) + for cid,name,parent in ((500,'Root',-1),(501,'TopicClass',500),(502,'SubTopicClass',501)): + add('tagclass',id=cid,name=name,parent=parent) + for tag,name,cls in ((10,'Topic',501),(11,'Other',501),(12,'Z',502),(13,'Alphabet',501)): + add('tag',id=tag,name=name,**{'class':cls}) + def message(mid,author,when,parent=-1,forum=1001,country=None,tags=(10,11),content='text',language='en',image=''): + add('message',id=mid,kind=int(parent>=0),created=when,creator=author,country=country or cities[author]-1, + content=content,image=image,length=len(content),language=language if parent<0 else '',forum=forum if parent<0 else -1, + parent=parent,year=2012,month=2012*12+1,day=when-when%86400000) + for tag in tags: add('mtag',message=mid,tag=tag) + # Explicit propagation witness: 1 posts in F1; 2 posts a day later in F2; + # 3 replies, both 2/3 belong to F1, and 1 is not a member of F2. + message(10001,1,t) + message(10002,2,t+86400000,forum=1002,country=200) + message(10003,3,t+86401000,parent=10002) + message(10004,2,t+86402000,parent=10001) + message(10005,5,t+86403000,parent=10002) + message(10006,4,t+86404000,parent=10003,tags=(11,)) + message(10007,7,t+86405000,forum=1003) + message(10008,2,t+86406000,country=300) + message(10009,4,t+86407000,country=100,forum=1003) + message(10010,4,t+86408000,country=200,forum=1003) + message(10011,1,t+86409000,tags=(11,13)) + # Same timestamps, enough candidates to exercise replacement at rank 20; + # empty text/image, Unicode, and different-length tag strings are present. + for i in range(24): + message(10100+i,2 if i%2==0 else 3,t+2*86400000,forum=1002, + tags=(10,13) if i%2 else (10,11),content='' if i==0 else 'café',image='photo.png' if i==0 else '') + for person,mid,offset in ((2,10001,1),(4,10001,2),(4,10011,2),(1,10001,3),(5,10007,4),(8,10002,5)): + add('likes',person=person,message=mid,created=t+3*86400000+offset) + return g + + +def params(name): + base=dict(pid=1,p1=1,p2=5,firstname='Alex',mid=10001,tagname='Topic',classname='TopicClass', + countryname='Alpha',countryx='Alpha',countryy='Beta',country1='Alpha',country2='Gamma', + start=millis('2011-01-01T00:00:00Z'),end=millis('2014-01-01T00:00:00Z'),endmonth=2014*12+1, + date=millis('2011-11-01T00:00:00Z'),mindate=millis('2010-01-01T00:00:00Z'), + month=1,minimum=1,maximum=4,city1=301,city2=101,company='Company',language='en', + taga='Topic',tagb='Other',datea=millis('2012-01-03T00:00:00Z'),dateb=millis('2012-01-03T00:00:00Z')) + if name=='bi4': base['date']=millis('2010-01-01T00:00:00Z') + if name=='ic12': base['classname']='Root' + if name=='bi20': base['p2']=1 + return base + + +def changed(original): + """A valid projected update: remove a friendship and a post, rename a person.""" + g={name:set(rows) for name,rows in original.items()} + g['knows']={row for row in g['knows'] if row[:2]!=(1,2)} + g['message']={row for row in g['message'] if row[0]!=10100} + g['mtag']={row for row in g['mtag'] if row[0]!=10100} + g['person']={tuple('Renamed' if i==1 and row[0]==2 else v for i,v in enumerate(row)) for row in g['person']} + return g diff --git a/interactive/server/bench/ldbc/suite.py b/interactive/server/bench/ldbc/suite.py new file mode 100644 index 000000000..62ebdf5dc --- /dev/null +++ b/interactive/server/bench/ldbc/suite.py @@ -0,0 +1,278 @@ +#!/usr/bin/env python3 +"""Run the complete SNB read catalogue through a private, real ddir_server.""" +import argparse +from collections import defaultdict, deque +import hashlib +import json +import os +from pathlib import Path +import platform +import shutil +import statistics +import subprocess +import sys +import tempfile +import time + +from client import Server, decode +from run import checked, digest, positive +from snb import data, witness +from snb.parameters import alternate, parameters, reference, requests +from snb.queries import Context, QUERIES +from snb.rel import Compiler, R, source_shape + +HERE = Path(__file__).resolve().parent +SPEC = 'b2269610f433da72e7c97041f01680aae369a903' +CATALOGUE = [f'{family}{i}' for family, end in (('is', 8), ('ic', 15), ('bi', 21)) for i in range(1, end)] +assert set(CATALOGUE) == set(QUERIES) + + +def compile_queries(names): + result = {} + for name in names: + fn, schema, title = QUERIES[name] + plan = fn(Context(), R.source('request', schema)) + compiler = Compiler(imports=True) + program = compiler.compile_many({name + '.answer': plan}) + result[name] = dict(plan=plan, sources=compiler.inputs, program=program, + parameters=schema, title=title) + return result + + +def graph_program(): + return '\n'.join(f'export "ldbc.{name}" = input {i} : ({source_shape(schema)} ; ()) | arrange;' + for i, (name, schema) in enumerate(data.SCHEMA.items())) + '\n' + + +def fingerprint(graph): + h = hashlib.sha256() + for name, rows in sorted(graph.items()): + h.update(name.encode()) + for row in sorted(rows): + h.update((json.dumps(row, ensure_ascii=True) + '\n').encode()) + return h.hexdigest() + + +def run_server(args, backend, names, graph, changed, bank, plans, report): + suffix = '-'.join(names) if args.isolated else 'all' + record = dict(backend=backend, workers=args.workers, queries=names, events=[], bindings=[]) + report['runs'].append(record) + standing = names if args.mode == 'maintained' else [n for n in names if n.startswith('bi')] + dynamic = [n for n in names if n not in standing] + active = {name: set() for name in names} + with Server(args.server, args.output / f'{backend}-{suffix}.log', backend, args.workers, + args.timeout, args.max_rss_gib) as server: + context = dict(round=-1, warmup=True, state='setup') + + def command(phase, make_commands): + start = time.perf_counter() + commands = make_commands() + prepared = time.perf_counter() + metrics, replies = server.commands(commands) + metrics.update(prepare_ms=1000*(prepared-start), client_ms=1000*(time.perf_counter()-start)) + record['events'].append(dict(context, phase=phase, **metrics)) + return replies + + def read(current, selected=names): + for name in selected: + # Same logical plan, evaluated from scratch. Outside all timers; + # checks lowering/maintenance, not independent spec conformance. + want = reference(plans[name]['plan'], plans[name]['sources'], + {**current, 'request': active[name]}) if active[name] else [] + phase = 'empty' if not active[name] else 'maintained' if name in standing else 'read' + lines, = command(phase + ':' + name, lambda: [f'peek {name}.answer']) + start = time.perf_counter() + actual = decode(lines) + event = record['events'][-1] + event.update(decode_ms=1000*(time.perf_counter()-start), rows=len(actual)) + event['client_ms'] += event['decode_ms'] + checked(actual, want, f'{backend}/{context}/{name}') + event['answer_sha256'] = hashlib.sha256(json.dumps(actual).encode()).hexdigest() + + def binding(name, cycle): + schema = plans[name]['parameters'] + # Two equal bindings under distinct IDs, plus varying values as the + # batch grows. Reusing IDs on the next round must give fresh answers. + return set().union(*(requests(name, schema, bank[name][(cycle+i//2) % len(bank[name])], i+1) + for i in range(args.batch_size))) + + def update(before, after, phase): + def commands(): + result = [] + for i, table in enumerate(data.SCHEMA): + for rows, diff in ((before[table]-after[table], -1), (after[table]-before[table], 1)): + if rows: + result.append(Server.feed('graph', i, sorted(rows), diff)) + return [*result, 'tick'] + command(phase, commands) + + for name, program in [('graph', graph_program()), *[(n, plans[n]['program']) for n in names]]: + command('install:' + name, lambda: [f'load {name} begin\n{program}\n{{rid}} end-load', 'tick']) + for i, table in enumerate(data.SCHEMA): + rows = sorted(graph[table]) + for offset in range(0, len(rows), 1000): + command('load:' + table, lambda: [Server.feed('graph', i, rows[offset:offset+1000], 1)]) + for name in standing: + active[name] = requests(name, plans[name]['parameters'], bank[name][0], 0) + command('standing:' + name, lambda: [Server.feed(name, 0, sorted(active[name]), 1)]) + command('initial_tick', lambda: ['tick']) + read(graph) + + for cycle in range(args.warmup + args.rounds): + context.update(round=cycle-args.warmup, warmup=cycle < args.warmup) + for state, current in (('initial', graph), ('changed', changed)): + context['state'] = state + if state == 'changed': + update(graph, changed, 'update') + read(current) + if dynamic: + for name in dynamic: + active[name] = binding(name, 2*cycle + int(state == 'changed')) + record['bindings'].append(dict(context, rows={n: sorted(active[n]) for n in dynamic})) + first = len(record['events']) + command('bind', lambda: [Server.feed(n, 0, sorted(active[n]), 1) for n in dynamic] + ['tick']) + read(current, dynamic) + elapsed = sum(e['client_ms'] for e in record['events'][first:]) + record['events'].append(dict(context, phase='batch_answers', client_ms=elapsed, + derived=True, requests=len(dynamic)*args.batch_size)) + command('release', lambda: [Server.feed(n, 0, sorted(active[n]), -1) for n in dynamic] + ['tick']) + for name in dynamic: + active[name] = set() + read(current, dynamic) + context['state'] = 'restored' + update(changed, graph, 'restore') + read(graph) + print(f'{backend}/{suffix}: round {cycle-args.warmup+1}/{args.rounds}, {len(names)} queries checked', flush=True) + context.update(state='retired', warmup=True) + if standing: + command('retire', lambda: [Server.feed(n, 0, sorted(active[n]), -1) for n in standing] + ['tick']) + for name in standing: + active[name] = set() + read(graph) + record['peak_server_rss_bytes_sampled'] = server.peak_rss + buckets = defaultdict(list) + for event in record['events']: + if not event['warmup']: + buckets[(event['state'], event['phase'])].append(event['client_ms']) + record['summary'] = {f'{state}/{phase}': dict(samples=len(xs), median_ms=statistics.median(xs), + min_ms=min(xs), max_ms=max(xs)) + for (state, phase), xs in sorted(buckets.items())} + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--server', type=Path, help='existing ddir_server (release for timing)') + parser.add_argument('--snapshot', type=Path, help='existing BI composite-merged-fk initial_snapshot directory') + parser.add_argument('--parameters', type=Path, help='JSON object: query name -> list of parameter objects') + parser.add_argument('--backend', choices=('vec', 'corgi', 'both'), default='both') + parser.add_argument('--workers', type=positive, default=1) + parser.add_argument('--queries', nargs='+', default=['all'], help='all, is, ic, bi, or individual names') + parser.add_argument('--mode', choices=('mixed', 'maintained'), default='mixed', + help='mixed: standing BI, transient IS/IC; maintained: standing bindings for every read') + parser.add_argument('--isolated', action='store_true', help='one fresh server per query, for attribution') + parser.add_argument('--rounds', type=positive, default=3) + parser.add_argument('--warmup', type=int, default=1) + parser.add_argument('--batch-size', type=positive, default=3) + parser.add_argument('--changes', type=positive, default=1, help='snapshot mode: retractions per edge table') + parser.add_argument('--timeout', type=positive, default=120) + parser.add_argument('--max-rss-gib', type=positive, default=6) + parser.add_argument('--output', type=Path, help='new artifact directory') + parser.add_argument('--emit', action='store_true', help='emit all selected DDP plans without starting a server') + parser.add_argument('--list', action='store_true', help='list query names, titles, and parameter fields') + args = parser.parse_args() + if args.warmup < 0: + parser.error('warmup must be non-negative') + names = [] + for selector in args.queries: + selected = CATALOGUE if selector == 'all' else [n for n in CATALOGUE if n.startswith(selector)] if selector in ('is', 'ic', 'bi') else [selector] + for name in selected: + if name not in QUERIES: + parser.error(f'unknown query {name}') + if name not in names: + names.append(name) + if args.list: + for name in names: + print(f'{name}: {QUERIES[name][2]} ({", ".join(QUERIES[name][1])})') + return + if not args.emit and args.server is None: + parser.error('--server is required unless --emit or --list is used') + if args.server: + args.server = args.server.resolve(strict=True) + if args.snapshot: + args.snapshot = args.snapshot.resolve(strict=True) + args.output = args.output.resolve() if args.output else Path(tempfile.mkdtemp(prefix='ddir-snb-')) + if not args.output.exists(): + args.output.mkdir(parents=True) + elif any(args.output.iterdir()): + parser.error('--output must be empty or new') + print(f'Artifacts: {args.output}', flush=True) + plans = compile_queries(names) + for name, plan in plans.items(): + (args.output / f'{name}.ddp').write_text(plan['program']) + (args.output / 'graph.ddp').write_text(graph_program()) + catalogue = {n: dict(title=plans[n]['title'], parameters=plans[n]['parameters'], + outputs=list(plans[n]['plan'].fields[2:])) for n in names} + (args.output / 'catalogue.json').write_text(json.dumps(catalogue, indent=2) + '\n') + if args.emit: + return + repo = HERE.parents[3] + def git(*command): + return subprocess.run(['git', '-C', str(repo), *command], capture_output=True, text=True, check=False).stdout.strip() + report = dict(format_version='snb-suite-1', status='running', spec_commit=SPEC, catalogue=catalogue, + config={k: str(v) if isinstance(v, Path) else v for k, v in vars(args).items()}, + platform=platform.platform(), python=platform.python_version(), logical_cpus=os.cpu_count(), + revision=git('rev-parse', 'HEAD'), worktree=git('status', '--porcelain'), + binary_sha256=digest(args.server), + plans_sha256={p.name: digest(p) for p in sorted(args.output.glob('*.ddp'))}, + sources_sha256={str(p.relative_to(HERE)): digest(p) for p in sorted(HERE.rglob('*.py'))}, runs=[]) + lock = repo / 'Cargo.lock' + if lock.exists(): + shutil.copyfile(lock, args.output / 'Cargo.lock') + report['cargo_lock_sha256'] = digest(lock) + try: + graph = data.load(args.snapshot) if args.snapshot else witness.graph() + if args.snapshot: + changed = {n: set(rows) for n, rows in graph.items()} + for table in ('knows', 'member', 'likes'): + changed[table].difference_update(sorted(graph[table])[:args.changes]) + else: + changed = witness.changed(graph) + overrides = json.loads(args.parameters.read_text()) if args.parameters else {} + if set(overrides) - set(QUERIES): + raise ValueError('unknown queries in --parameters') + bank = {} + for name in names: + base = parameters(graph, name) + if not args.snapshot: + base.update(witness.params(name)) + bank[name] = [base, alternate(graph, base)] + if name in overrides: + if not isinstance(overrides[name], list) or not overrides[name]: + raise ValueError(f'{name}: parameters must be a nonempty list of objects') + bank[name] = [{**base, **p} for p in overrides[name]] + for binding in bank[name]: + requests(name, plans[name]['parameters'], binding, 0) + report.update(data=dict(rows={n: len(rows) for n, rows in graph.items()}, sha256=fingerprint(graph)), + parameter_bank=bank, changed_sha256=fingerprint(changed), + changes={n: dict(removed=sorted(graph[n]-changed[n]), added=sorted(changed[n]-graph[n])) + for n in graph if graph[n] != changed[n]}) + print(f'{len(names)} queries; {sum(map(len, graph.values()))} projected rows', flush=True) + for backend in ('vec', 'corgi') if args.backend == 'both' else (args.backend,): + for selected in [[n] for n in names] if args.isolated else [names]: + run_server(args, backend, selected, graph, changed, bank, plans, report) + report['status'] = 'passed' + except BaseException as error: + report.update(status='failed', error=f'{type(error).__name__}: {error}') + for log in args.output.glob('*.log'): + with log.open(errors='replace') as source: + tail = ''.join(deque(source, maxlen=15))[-6000:] + if tail: + print(f'{log.name}:\n{tail}', file=sys.stderr) + raise + finally: + (args.output / 'report.json').write_text(json.dumps(report, indent=2) + '\n') + print(f'Report: {args.output / "report.json"}', flush=True) + + +if __name__ == '__main__': + main() diff --git a/interactive/server/bench/ldbc/test_snb.py b/interactive/server/bench/ldbc/test_snb.py new file mode 100644 index 000000000..8fc806fb8 --- /dev/null +++ b/interactive/server/bench/ldbc/test_snb.py @@ -0,0 +1,73 @@ +"""Read-suite witnesses, hand-checked answers, and both engine backends.""" +import struct +import unittest + +from snb.queries import Context,QUERIES +from snb.rel import R,Compiler +from snb.parameters import parameters,reference +from snb import witness + + +def decode_float(encoded): + ordered=(encoded['payload'] & ((1<<64)-1))^(1<<63) + bits=ordered^(1<<63) if ordered>>63 else ~ordered & ((1<<64)-1) + return struct.unpack('>d',struct.pack('>Q',bits))[0] + + +class SuiteTests(unittest.TestCase): + @classmethod + def setUpClass(cls): cls.graph=witness.graph() + + def rows(self,name,graph=None,**overrides): + graph=self.graph if graph is None else graph + fn,schema,_=QUERIES[name] + params=parameters(graph);params.update(witness.params(name));params.update(overrides) + plan=fn(Context(),R.source('request',schema));cc=Compiler();cc.compile(plan) + rows=reference(plan,cc.inputs,{**graph,'request':{tuple(params[n] for n in schema)}}) + return [r[1] for r in sorted(rows,key=lambda r:r[0])] + + def test_complete_catalog_and_positive_witnesses(self): + self.assertEqual(set(QUERIES),{f'is{i}' for i in range(1,8)}|{f'ic{i}' for i in range(1,15)}|{f'bi{i}' for i in range(1,21)}) + for name in QUERIES: + with self.subTest(query=name): self.assertTrue(self.rows(name)) + + def test_hand_checked_graph_answers(self): + self.assertEqual(self.rows('bi11'),[[1]]) + self.assertEqual(self.rows('bi11',countryname='Gamma'),[[0]]) + self.assertEqual(self.rows('bi17'),[[1,1]]) + self.assertEqual(self.rows('bi20'),[[2,2]]) + self.assertEqual(self.rows('ic13'),[[2]]) + self.assertEqual(self.rows('ic13',p2=1),[[0]]) + self.assertEqual(self.rows('ic14'),[[[1,2,5],78]]) + self.assertEqual(decode_float(self.rows('bi15')[0][0]),1.0) + changed=witness.changed(self.graph) + self.assertEqual(self.rows('ic13',changed),[[3]]) + self.assertEqual(self.rows('ic14',changed),[]) + self.assertAlmostEqual(decode_float(self.rows('bi15',changed)[0][0]),8/3) + + def test_hand_checked_ranking_and_optional_matches(self): + self.assertEqual([r[3] for r in self.rows('ic2')],list(range(10100,10120))) + self.assertEqual(bytes(self.rows('ic2')[0][4]),b'photo.png') + self.assertEqual([r[0] for r in self.rows('ic7')],[1,4,2]) + self.assertEqual([r[-1] for r in self.rows('ic7')],[1,1,0]) + self.assertEqual([r[4] for r in self.rows('ic7')],[10001,10001,10001]) + self.assertEqual([(bytes(r[0]).decode(),r[1]) for r in self.rows('ic4')],[('Topic',26),('Other',14),('Alphabet',12)]) + self.assertEqual(self.rows('bi12'),[[1,2],[0,2],[14,1],[13,1],[3,1],[2,1]]) + self.assertEqual(self.rows('bi12',language='',enabled=0),[[0,8]]) + # Empty nested collections for people without work history are real + # typed empty lists, not missing rows or singleton placeholder tuples. + rows=self.rows('ic1') + self.assertEqual(len(rows),7) + self.assertTrue(any(not r[-1] for r in rows)) + + def test_hand_checked_floating_aggregates(self): + rows=self.rows('bi1') + self.assertEqual([(r[0],r[1],r[2],r[3],r[5]) for r in rows],[(2012,0,0,31,120),(2012,1,0,4,16)]) + self.assertAlmostEqual(decode_float(rows[0][4]),120/31) + self.assertAlmostEqual(decode_float(rows[0][6]),3100/35) + self.assertEqual([(r[0],r[1],r[2],decode_float(r[3])) for r in self.rows('bi13')],[(7,1,1,1.0),(2,0,1,0.0),(5,0,0,0.0)]) + + + +if __name__ == "__main__": + unittest.main() From 1df4f4dfad02e4234cc47a4c52595adf7476e45d Mon Sep 17 00:00:00 2001 From: Frank McSherry Date: Mon, 7 Sep 2026 08:44:07 -0400 Subject: [PATCH 5/8] Document compressed-memory limits of benchmark RSS monitoring --- interactive/server/bench/ldbc/README.md | 10 +++++++--- interactive/server/bench/ldbc/SNB.md | 5 +++++ 2 files changed, 12 insertions(+), 3 deletions(-) diff --git a/interactive/server/bench/ldbc/README.md b/interactive/server/bench/ldbc/README.md index 63a02a4ec..87a642594 100644 --- a/interactive/server/bench/ldbc/README.md +++ b/interactive/server/bench/ldbc/README.md @@ -55,9 +55,13 @@ measures single bindings **per interactive query**; in the default panel that still means two simultaneous requests per tick. Increase batch size separately from data size to study dispatch amortization. Defaults enforce a 120-second command deadline and a sampled 6-GiB **server** RSS ceiling. The latter is not -a hard memory limit and excludes the Python loader/reference process. -SF1 can exceed the default ceiling during loading; use `--max-rss-gib 8` -if the host has sufficient memory for both processes. +a hard memory limit and excludes the Python loader/reference process. RSS can +fall as pages are compressed or swapped, even while the total memory burden +grows. Do not treat the ceiling as protection against exhausting the host. +Start with small snapshots and selected queries; for larger runs use external +resource limits or monitoring that covers the server and Python descendants, +compressed memory, swap and host headroom. Increasing `--max-rss-gib` alone +does not establish that a workload fits. ## Queries and lifecycle diff --git a/interactive/server/bench/ldbc/SNB.md b/interactive/server/bench/ldbc/SNB.md index 6ebcedf89..96f869def 100644 --- a/interactive/server/bench/ldbc/SNB.md +++ b/interactive/server/bench/ldbc/SNB.md @@ -187,6 +187,11 @@ the hop-indexed path plans are intended to make all-query SF1 runs cheap. The default 120-second command deadline and sampled 6-GiB server RSS ceiling exclude Python and are not a hard OS memory limit. Small debug runs establish correctness, not competitive performance; use release builds and a quiet machine for timing. +RSS can fall under compression or swapping while memory pressure grows. For +larger runs, apply external limits or monitoring to the whole process group, +including the Python loader/reference process, and retain host headroom. The +built-in RSS ceiling alone cannot make an all-query SF1 run safe on a +memory-constrained machine. Useful initial attribution groups (not official choke-point classifications): IC1/IC12 for optional/nested output; IC2/IC6 for selective joins and ranking; From 6ff8a788a1e0c135629a373ca67b781320aab4ca Mon Sep 17 00:00:00 2001 From: Frank McSherry Date: Mon, 7 Sep 2026 11:10:44 -0400 Subject: [PATCH 6/8] Document fatal shape-contract violations in the SNB guide --- interactive/server/bench/ldbc/SNB.md | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/interactive/server/bench/ldbc/SNB.md b/interactive/server/bench/ldbc/SNB.md index 96f869def..4fa1c34d0 100644 --- a/interactive/server/bench/ldbc/SNB.md +++ b/interactive/server/bench/ldbc/SNB.md @@ -75,7 +75,10 @@ leaves are 64-bit integers, not byte-packed or dictionary-encoded strings. Post and Comment share a tuple with a kind field; they are **not** an enum-layout optimization experiment. Source shape ascriptions supply the element encoding of empty lists and inactive sum lanes. They are execution-time contracts, not -transactional validation at `feed` admission. +transactional validation at `feed` admission. A mismatched row can panic a +dataflow worker and take down the shared server on either backend, disconnecting +other clients. There is no per-program failure isolation; use trusted programs +and shape-correct data. The benchmark starts private server processes. Default bindings are deterministic smoke parameters selected from input facts, not LDBC's parameter generator. The fixture has nonempty witnesses for all 41 From 1b37122ffda715530d8f7cca9d19ad70b274cfed Mon Sep 17 00:00:00 2001 From: Frank McSherry Date: Mon, 7 Sep 2026 12:29:59 -0400 Subject: [PATCH 7/8] ddir: unify snapshot loading and separate benchmark timing components Use one Spark BI CSV reader, compute signed graph deltas once at setup, and summarize client preparation, encoding, request/response, decoding and total timings separately. Version the reports and make comparisons select an explicit component, defaulting to request/response time. --- .github/workflows/test.yml | 1 + interactive/server/bench/ldbc/README.md | 24 +++++- interactive/server/bench/ldbc/SNB.md | 18 ++++- interactive/server/bench/ldbc/compare.py | 20 +++-- interactive/server/bench/ldbc/measure.py | 44 +++++++++++ interactive/server/bench/ldbc/run.py | 38 +++------- interactive/server/bench/ldbc/snapshot.py | 62 +++++++++++++++ interactive/server/bench/ldbc/snb/data.py | 21 ++--- interactive/server/bench/ldbc/suite.py | 60 +++++++-------- interactive/server/bench/ldbc/test_harness.py | 76 +++++++++++++++++++ interactive/server/bench/ldbc/workload.py | 46 +++++------ 11 files changed, 300 insertions(+), 110 deletions(-) create mode 100644 interactive/server/bench/ldbc/measure.py create mode 100644 interactive/server/bench/ldbc/snapshot.py create mode 100644 interactive/server/bench/ldbc/test_harness.py diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index fbc72599c..5aed7e6bf 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -51,6 +51,7 @@ jobs: - name: Complete SNB read catalogue if: matrix.os == 'ubuntu' && matrix.toolchain == 'stable' run: | + python3 interactive/server/bench/ldbc/test_harness.py python3 interactive/server/bench/ldbc/test_snb.py for workers in 1 4; do python3 interactive/server/bench/ldbc/suite.py --server target/debug/ddir_server \ diff --git a/interactive/server/bench/ldbc/README.md b/interactive/server/bench/ldbc/README.md index 87a642594..f4672a529 100644 --- a/interactive/server/bench/ldbc/README.md +++ b/interactive/server/bench/ldbc/README.md @@ -48,6 +48,15 @@ and large scale factors. Missing table directories fail rather than silently becoming empty tables. Only seven relations/needed columns are loaded; the reported projected row counts and data hash define the actual input. +Both runners share [snapshot.py](snapshot.py), supporting the default Spark BI +export dialect: UTF-8, pipe separators, double-quoted fields and backslash +escapes inside quoted fields. Literal backslashes in unquoted fields survive. +This follows the [BI snapshot writer](https://github.com/ldbc/ldbc_snb_datagen_spark/blob/b3dc986898efac7c1676abba865a30865334922e/src/main/scala/ldbc/snb/datagen/io/graphs.scala#L49) +and [Spark CSV options](https://spark.apache.org/docs/3.5.6/sql-data-sources-csv.html#data-source-option), +not the raw generator's unquoted format. Custom export dialects are unsupported. +Plain `.csv` and `.csv.gz` partitions are accepted; when both copies exist under +the same partition name, the plain file takes precedence and is read once. + `--queries ic6` isolates one query over the same common graph substrate; `--queries bi5 bi11` measures BI maintenance without interactive consumers. Compare those controls with the default concurrent panel. `--batch-size 1` @@ -129,7 +138,14 @@ LDBC conformance suite. separate; subtracting them does not remove the maintenance cost. Events split command preparation, encoding, wire wait/receipt and Value decoding. -Wire time includes server work and TCP overhead; it is not a kernel profile. +Every phase's `summary` contains separate `prepare_ms`, `encode_ms`, `wire_ms`, +`decode_ms` and `client_ms` statistics. `wire_ms` is the default printed/comparison +metric: time from sending the encoded command group through receipt of its final +acknowledgement, including transport and Python protocol handling. It excludes +command construction, request encoding and answer Value decoding, but is **not +server CPU time**. `client_ms` retains the timed phase's total including those +client costs; it excludes parameter selection and oracle/validation work. +Derived `batch_answers` sums each metric over bind/read phases separately. Medians/min/max exclude warmup; raw samples, result bytes, reference answers, sampled server RSS, machine metadata, binary/source/data hashes, repository revision/status and the available Cargo lockfile accompany the report. Keep the @@ -144,6 +160,12 @@ python3 interactive/server/bench/ldbc/compare.py \ /tmp/ldbc-baseline/report.json /tmp/ldbc-candidate/report.json ``` +Use `--metric client_ms` (or another timing field) to compare that component +instead. Reports now use format version 2 (`snb-suite-2` for the full suite). +The comparison rejects old reports: their summaries used client totals, and +the old full-suite update/restore totals included full-graph set differences. +Establish a fresh baseline after this harness correction. + The comparison rejects failed runs, changed data/parameters/schedules/answers, different worker counts or environments, and changed measurement code. Changed DDP plans are allowed and called out: plan improvements are benchmark subjects diff --git a/interactive/server/bench/ldbc/SNB.md b/interactive/server/bench/ldbc/SNB.md index 4fa1c34d0..6e7f25ab0 100644 --- a/interactive/server/bench/ldbc/SNB.md +++ b/interactive/server/bench/ldbc/SNB.md @@ -61,7 +61,10 @@ python3 interactive/server/bench/ldbc/suite.py --server target/release/ddir_serv ``` The adapter reads all 18 required tables in SNB BI CSV `composite-merged-fk` -layout, including `.csv.gz` partitions. It projects 17 relations: people, +layout, including `.csv.gz` partitions, through the same +[snapshot dialect reader](README.md#four-query-control-panel) as the small panel +(double-quoted fields, backslash escaping within quotes; not raw-generator CSV). +A plain partition takes precedence over its `.csv.gz` copy. It projects 17 relations: people, posts/comments, forums, friendships, memberships, tags/classes, places, organisations, interests, message/forum tags, likes, education, employment, email, and languages. Missing files fail explicitly. All reads have their @@ -135,8 +138,17 @@ Standing-result reads are `maintained:*`; cleanup checks are `empty:*`, so empty reads do not dilute the latency summary for bound requests. Per-event command preparation, encoding, wire wait/receipt, and decoding are -recorded. Wire time includes server work and TCP overhead; it is not a kernel -profile. Shared update/bind time cannot be attributed to one query by inspecting +recorded and summarized separately as `prepare_ms`, `encode_ms`, `wire_ms`, +`decode_ms` and `client_ms`. Comparisons default to client-observed request/response +time (`wire_ms`), which includes server work, transport and Python protocol +handling; it is not server CPU time. `client_ms` includes command preparation, +encoding and answer decoding, but excludes parameter selection and oracle work. +`batch_answers` sums each component over bind/read phases separately. +The snapshot deltas are computed once during setup (`delta_prepare_ms`) and +reused for every update/restore; these phases do work proportional to the delta, +not repeated full-graph differences in the client. Version `snb-suite-2` reports +require a fresh baseline; the comparison tool rejects the old summary format. +Shared update/bind time cannot be attributed to one query by inspecting its later `peek`. Use `--isolated` and selected concurrent panels for attribution. Parsing/planning is setup cost here: **this is not an ad-hoc re-planning test**. Do not omit release or standing maintenance when assessing sustained work. diff --git a/interactive/server/bench/ldbc/compare.py b/interactive/server/bench/ldbc/compare.py index 7f79d77b7..83e15aba2 100644 --- a/interactive/server/bench/ldbc/compare.py +++ b/interactive/server/bench/ldbc/compare.py @@ -4,17 +4,21 @@ import json from pathlib import Path +from measure import TIMINGS + def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument('baseline', type=Path) parser.add_argument('candidate', type=Path) + parser.add_argument('--metric', choices=TIMINGS, default='wire_ms', + help='wire_ms is client-observed request/response time, not server CPU time') args = parser.parse_args() before, after = [json.loads(path.read_text()) for path in (args.baseline, args.candidate)] - suite = before['format_version'] == 'snb-suite-1' + suite = before['format_version'] == 'snb-suite-2' for report in (before, after): - if report['status'] != 'passed' or report['format_version'] != ('snb-suite-1' if suite else 1): - parser.error('both reports must be successful runs of the same supported format') + if report['status'] != 'passed' or report['format_version'] != ('snb-suite-2' if suite else 2): + parser.error('both reports must be successful runs of the same v2 timing format; establish a fresh baseline') fields = ('catalogue', 'spec_commit', 'changed_sha256', 'changes') if suite else ('standing', 'retracted_rows', 'reference_answers') for key in ('data', 'parameter_bank', 'platform', 'python', 'logical_cpus', *fields): if before[key] != after[key]: @@ -22,7 +26,8 @@ def main(): for key in ('workers', 'queries', 'rounds', 'warmup', 'batch_size', 'changes', *(('mode', 'isolated') if suite else ())): if before['config'][key] != after['config'][key]: parser.error(f'incomparable configuration: {key}') - sources = ('client.py', 'run.py', 'suite.py', 'snb/data.py', 'snb/parameters.py', 'snb/witness.py') if suite else ('client.py', 'workload.py', 'run.py') + sources = ('client.py', 'measure.py', 'snapshot.py', 'run.py', *( + ('suite.py', 'snb/data.py', 'snb/parameters.py', 'snb/witness.py') if suite else ('workload.py',))) for source in sources: if before['sources_sha256'][source] != after['sources_sha256'][source]: parser.error(f'measurement/reference code changed: {source}; establish a fresh baseline') @@ -49,10 +54,11 @@ def answers(run): parser.error(f'{backend}: answers changed') if a[backend]['summary'].keys() != b[backend]['summary'].keys(): parser.error(f'{backend}: measured phases changed') - print(f'\n{backend}: milliseconds, baseline / candidate; >1x is faster') + print(f'\n{backend}: {args.metric}, milliseconds, baseline / candidate; >1x is faster') print(f'{"phase":36} {"baseline":>10} {"candidate":>10} {"ratio":>8}') - for phase, old in a[backend]['summary'].items(): - new = b[backend]['summary'][phase] + for phase, metrics in a[backend]['summary'].items(): + old = metrics[args.metric] + new = b[backend]['summary'][phase][args.metric] ratio = old['median_ms'] / new['median_ms'] if new['median_ms'] else float('inf') print(f'{phase:36} {old["median_ms"]:10.3f} {new["median_ms"]:10.3f} {ratio:7.2f}x') diff --git a/interactive/server/bench/ldbc/measure.py b/interactive/server/bench/ldbc/measure.py new file mode 100644 index 000000000..b6bbafbf3 --- /dev/null +++ b/interactive/server/bench/ldbc/measure.py @@ -0,0 +1,44 @@ +"""Client-observed phase timings; none of these measures server CPU time.""" +from collections import defaultdict +import statistics +import time + +from client import decode + +TIMINGS = ('prepare_ms', 'encode_ms', 'wire_ms', 'decode_ms', 'client_ms') + + +def commands(server, make_commands): + start = time.perf_counter() + prepared_commands = make_commands() + prepared = time.perf_counter() + metrics, replies = server.commands(prepared_commands) + metrics.update(prepare_ms=1000*(prepared-start), decode_ms=0, + client_ms=1000*(time.perf_counter()-start)) + return metrics, replies + + +def decode_answer(lines, event): + start = time.perf_counter() + rows = decode(lines) + event['decode_ms'] = 1000*(time.perf_counter()-start) + event['client_ms'] += event['decode_ms'] + event['rows'] = len(rows) + return rows + + +def total(events): + """Sum measured phases, excluding intervening oracle/validation work.""" + return {metric: sum(event[metric] for event in events) for metric in TIMINGS} + + +def summarize(events): + buckets = defaultdict(list) + for event in events: + if not event['warmup']: + buckets[event['state'] + '/' + event['phase']].append(event) + def stats(values): + return dict(samples=len(values), median_ms=statistics.median(values), + min_ms=min(values), max_ms=max(values)) + return {phase: {metric: stats([event[metric] for event in rows]) for metric in TIMINGS} + for phase, rows in sorted(buckets.items())} diff --git a/interactive/server/bench/ldbc/run.py b/interactive/server/bench/ldbc/run.py index 8c2503583..e6e91e562 100644 --- a/interactive/server/bench/ldbc/run.py +++ b/interactive/server/bench/ldbc/run.py @@ -1,20 +1,20 @@ #!/usr/bin/env python3 """Reproducible maintained/parameterized LDBC-derived server workload (stdlib only).""" import argparse -from collections import defaultdict, deque +from collections import deque import hashlib import json import os from pathlib import Path import platform import shutil -import statistics import subprocess import sys import tempfile import time -from client import Server, decode +from client import Server +import measure from workload import HERE, TABLES, QUERIES, Reference, changes, expected, fingerprint, load, parameters @@ -51,23 +51,14 @@ def run_server(args, backend, graph, delta, bank, standing, answers, report): context = dict(round=-1, warmup=True, state='setup') def command(phase, make_commands): - start = time.perf_counter() - commands = make_commands() - prepared = time.perf_counter() - metrics, replies = server.commands(commands) - metrics['prepare_ms'] = 1000*(prepared-start) - metrics['client_ms'] = 1000*(time.perf_counter()-start) + metrics, replies = measure.commands(server, make_commands) record['events'].append(dict(context, phase=phase, **metrics)) return replies def read(name, want, phase='read'): lines, = command(f'{phase}:{name}', lambda: [f'peek {name}.answer']) - start = time.perf_counter() - actual = decode(lines) event = record['events'][-1] - event['decode_ms'] = 1000*(time.perf_counter()-start) - event['client_ms'] += event['decode_ms'] - event['rows'] = len(actual) + actual = measure.decode_answer(lines, event) # Validation/reference work is deliberately outside all timings. checked(actual, want, f'{backend}/{context}/{phase}/{name}') @@ -109,8 +100,8 @@ def read(name, want, phase='read'): want.append([[rid, key[1]], value, diff]) read(name, want) # Sum measured phases, excluding the intervening answer checks. - elapsed = sum(e['client_ms'] for e in record['events'][first_event:]) - record['events'].append(dict(context, phase='batch_answers', client_ms=elapsed, + totals = measure.total(record['events'][first_event:]) + record['events'].append(dict(context, phase='batch_answers', **totals, derived=True, requests=len(interactive)*args.batch_size)) command('release', lambda: [Server.feed(name, 0, bindings[name], -1) for name in interactive] + ['tick']) for name in interactive: @@ -122,16 +113,11 @@ def read(name, want, phase='read'): read(name, answers['initial'][name][standing[name]], 'maintained') print(f'{backend}: round {cycle-args.warmup + 1}/{args.rounds}, answers verified', flush=True) record['peak_server_rss_bytes_sampled'] = server.peak_rss - buckets = defaultdict(list) - for event in record['events']: - if not event['warmup']: - buckets[(event['state'], event['phase'])].append(event['client_ms']) - record['summary'] = {f'{state}/{phase}': dict(samples=len(values), median_ms=statistics.median(values), - min_ms=min(values), max_ms=max(values)) - for (state, phase), values in sorted(buckets.items())} - for label, summary in record['summary'].items(): + record['summary'] = measure.summarize(record['events']) + for label, metrics in record['summary'].items(): if not label.split('/')[-1].startswith('empty'): - print(f' {label}: {summary["median_ms"]:.3f} ms median ({summary["samples"]} samples)') + summary = metrics['wire_ms'] + print(f' {label}: {summary["median_ms"]:.3f} ms request/response median ({summary["samples"]} samples)') def main(): @@ -164,7 +150,7 @@ def main(): def git(*command): return subprocess.run(['git', '-C', str(repo), *command], capture_output=True, text=True, check=False).stdout.strip() - report = dict(format_version=1, status='running', config={k: str(v) if isinstance(v, Path) else v for k, v in vars(args).items()}, + report = dict(format_version=2, status='running', config={k: str(v) if isinstance(v, Path) else v for k, v in vars(args).items()}, platform=platform.platform(), python=platform.python_version(), logical_cpus=os.cpu_count(), revision=git('rev-parse', 'HEAD'), worktree=git('status', '--porcelain'), binary_sha256=digest(args.server), diff --git a/interactive/server/bench/ldbc/snapshot.py b/interactive/server/bench/ldbc/snapshot.py new file mode 100644 index 000000000..ad91773a2 --- /dev/null +++ b/interactive/server/bench/ldbc/snapshot.py @@ -0,0 +1,62 @@ +"""Shared reader for Spark's default SNB BI pipe-delimited CSV export. + +Double quotes delimit fields; backslashes escape quotes/backslashes *inside* +quoted fields. Raw-generator CSV and custom export dialects are not supported. +""" +import csv +import gzip + + +def csv_rows(source): + # Normalize Spark's quoted-field escapes to the doubled quotes understood + # by csv.reader. Setting escapechar='\\' on csv.reader would also consume + # literal backslashes in UNQUOTED fields, unlike Spark's writer/reader. + def normalized(): + quoted, field_start = False, True + for line in source: + if not quoted and '"' not in line: + yield line + field_start = True + continue + out, i = [], 0 + while i < len(line): + ch = line[i] + if quoted: + if ch == '\\' and i + 1 < len(line) and line[i+1] in ('\\', '"'): + out.append('""' if line[i+1] == '"' else '\\') + i += 2 + continue + if ch == '"': + quoted = False + else: + if ch == '"' and field_start: + quoted = True + field_start = ch in '|\r\n' + out.append(ch) + i += 1 + yield ''.join(out) + + reader = csv.DictReader(normalized(), delimiter='|', strict=True) + fields = reader.fieldnames + if not fields or len(set(fields)) != len(fields): + raise ValueError('missing or duplicate CSV header') + for row in reader: + if None in row or any(value is None for value in row.values()): + raise ValueError(f'CSV row ending at line {reader.line_num} has the wrong field count') + yield row + + +def read_table(snapshot, category, entity): + directory = snapshot / category / entity + # A decompressed partition and its archive are the same logical input. + paths = sorted([*directory.glob('*.csv'), + *(p for p in directory.glob('*.csv.gz') if not p.with_suffix('').exists())]) + if not paths: + raise FileNotFoundError(f'missing {entity} CSV in {directory}') + for path in paths: + opener = gzip.open if path.suffix == '.gz' else open + with opener(path, 'rt', newline='', encoding='utf-8') as source: + try: + yield from csv_rows(source) + except (csv.Error, ValueError) as error: + raise ValueError(f'{path}: {error}') from error diff --git a/interactive/server/bench/ldbc/snb/data.py b/interactive/server/bench/ldbc/snb/data.py index 3ce5f106f..c97bdb5d0 100644 --- a/interactive/server/bench/ldbc/snb/data.py +++ b/interactive/server/bench/ldbc/snb/data.py @@ -1,7 +1,7 @@ """Project the 18 SNB BI composite-merged-fk snapshot tables; no query results.""" -import csv from datetime import datetime, timezone -import gzip + +from snapshot import read_table # Field names are shared with the named relational query definitions. SCHEMA = { @@ -67,18 +67,11 @@ def load(snapshot): """Read an existing initial_snapshot directory. Never download data.""" tables = {entity: {} for entity in (*ENTITIES, *STATIC)} for entity in tables: - directory = snapshot / ("static" if entity in STATIC else "dynamic") / entity - paths = sorted([*directory.glob("*.csv"), *directory.glob("*.csv.gz")]) - if not paths: - raise FileNotFoundError(f"missing {entity} CSV in {directory}") - for path in paths: - opener = gzip.open if path.suffix == ".gz" else open - with opener(path, "rt", newline="", encoding="utf-8") as source: - for row in csv.DictReader(source, delimiter="|"): - identity = key(entity, row) - if identity in tables[entity]: - raise ValueError(f"duplicate {entity} {identity}") - tables[entity][identity] = row + for row in read_table(snapshot, "static" if entity in STATIC else "dynamic", entity): + identity = key(entity, row) + if identity in tables[entity]: + raise ValueError(f"duplicate {entity} {identity}") + tables[entity][identity] = row return project(tables) diff --git a/interactive/server/bench/ldbc/suite.py b/interactive/server/bench/ldbc/suite.py index 62ebdf5dc..2d7332447 100644 --- a/interactive/server/bench/ldbc/suite.py +++ b/interactive/server/bench/ldbc/suite.py @@ -1,20 +1,20 @@ #!/usr/bin/env python3 """Run the complete SNB read catalogue through a private, real ddir_server.""" import argparse -from collections import defaultdict, deque +from collections import deque import hashlib import json import os from pathlib import Path import platform import shutil -import statistics import subprocess import sys import tempfile import time -from client import Server, decode +from client import Server +import measure from run import checked, digest, positive from snb import data, witness from snb.parameters import alternate, parameters, reference, requests @@ -53,7 +53,7 @@ def fingerprint(graph): return h.hexdigest() -def run_server(args, backend, names, graph, changed, bank, plans, report): +def run_server(args, backend, names, graph, changed, delta, bank, plans, report): suffix = '-'.join(names) if args.isolated else 'all' record = dict(backend=backend, workers=args.workers, queries=names, events=[], bindings=[]) report['runs'].append(record) @@ -65,11 +65,7 @@ def run_server(args, backend, names, graph, changed, bank, plans, report): context = dict(round=-1, warmup=True, state='setup') def command(phase, make_commands): - start = time.perf_counter() - commands = make_commands() - prepared = time.perf_counter() - metrics, replies = server.commands(commands) - metrics.update(prepare_ms=1000*(prepared-start), client_ms=1000*(time.perf_counter()-start)) + metrics, replies = measure.commands(server, make_commands) record['events'].append(dict(context, phase=phase, **metrics)) return replies @@ -81,11 +77,8 @@ def read(current, selected=names): {**current, 'request': active[name]}) if active[name] else [] phase = 'empty' if not active[name] else 'maintained' if name in standing else 'read' lines, = command(phase + ':' + name, lambda: [f'peek {name}.answer']) - start = time.perf_counter() - actual = decode(lines) event = record['events'][-1] - event.update(decode_ms=1000*(time.perf_counter()-start), rows=len(actual)) - event['client_ms'] += event['decode_ms'] + actual = measure.decode_answer(lines, event) checked(actual, want, f'{backend}/{context}/{name}') event['answer_sha256'] = hashlib.sha256(json.dumps(actual).encode()).hexdigest() @@ -96,13 +89,15 @@ def binding(name, cycle): return set().union(*(requests(name, schema, bank[name][(cycle+i//2) % len(bank[name])], i+1) for i in range(args.batch_size))) - def update(before, after, phase): + def update(phase, reverse=False): def commands(): result = [] + removed, added = ('added', 'removed') if reverse else ('removed', 'added') for i, table in enumerate(data.SCHEMA): - for rows, diff in ((before[table]-after[table], -1), (after[table]-before[table], 1)): - if rows: - result.append(Server.feed('graph', i, sorted(rows), diff)) + if table in delta: + for side, diff in ((removed, -1), (added, 1)): + if delta[table][side]: + result.append(Server.feed('graph', i, delta[table][side], diff)) return [*result, 'tick'] command(phase, commands) @@ -123,7 +118,7 @@ def commands(): for state, current in (('initial', graph), ('changed', changed)): context['state'] = state if state == 'changed': - update(graph, changed, 'update') + update('update') read(current) if dynamic: for name in dynamic: @@ -132,15 +127,15 @@ def commands(): first = len(record['events']) command('bind', lambda: [Server.feed(n, 0, sorted(active[n]), 1) for n in dynamic] + ['tick']) read(current, dynamic) - elapsed = sum(e['client_ms'] for e in record['events'][first:]) - record['events'].append(dict(context, phase='batch_answers', client_ms=elapsed, + totals = measure.total(record['events'][first:]) + record['events'].append(dict(context, phase='batch_answers', **totals, derived=True, requests=len(dynamic)*args.batch_size)) command('release', lambda: [Server.feed(n, 0, sorted(active[n]), -1) for n in dynamic] + ['tick']) for name in dynamic: active[name] = set() read(current, dynamic) context['state'] = 'restored' - update(changed, graph, 'restore') + update('restore', reverse=True) read(graph) print(f'{backend}/{suffix}: round {cycle-args.warmup+1}/{args.rounds}, {len(names)} queries checked', flush=True) context.update(state='retired', warmup=True) @@ -150,13 +145,7 @@ def commands(): active[name] = set() read(graph) record['peak_server_rss_bytes_sampled'] = server.peak_rss - buckets = defaultdict(list) - for event in record['events']: - if not event['warmup']: - buckets[(event['state'], event['phase'])].append(event['client_ms']) - record['summary'] = {f'{state}/{phase}': dict(samples=len(xs), median_ms=statistics.median(xs), - min_ms=min(xs), max_ms=max(xs)) - for (state, phase), xs in sorted(buckets.items())} + record['summary'] = measure.summarize(record['events']) def main(): @@ -218,7 +207,7 @@ def main(): repo = HERE.parents[3] def git(*command): return subprocess.run(['git', '-C', str(repo), *command], capture_output=True, text=True, check=False).stdout.strip() - report = dict(format_version='snb-suite-1', status='running', spec_commit=SPEC, catalogue=catalogue, + report = dict(format_version='snb-suite-2', status='running', spec_commit=SPEC, catalogue=catalogue, config={k: str(v) if isinstance(v, Path) else v for k, v in vars(args).items()}, platform=platform.platform(), python=platform.python_version(), logical_cpus=os.cpu_count(), revision=git('rev-parse', 'HEAD'), worktree=git('status', '--porcelain'), @@ -252,14 +241,17 @@ def git(*command): bank[name] = [{**base, **p} for p in overrides[name]] for binding in bank[name]: requests(name, plans[name]['parameters'], binding, 0) - report.update(data=dict(rows={n: len(rows) for n, rows in graph.items()}, sha256=fingerprint(graph)), - parameter_bank=bank, changed_sha256=fingerprint(changed), - changes={n: dict(removed=sorted(graph[n]-changed[n]), added=sorted(changed[n]-graph[n])) - for n in graph if graph[n] != changed[n]}) + # Compute full-snapshot differences once, outside every server phase. + started = time.perf_counter() + delta = {n: dict(removed=sorted(graph[n]-changed[n]), added=sorted(changed[n]-graph[n])) + for n in graph if graph[n] != changed[n]} + report.update(delta_prepare_ms=1000*(time.perf_counter()-started), + data=dict(rows={n: len(rows) for n, rows in graph.items()}, sha256=fingerprint(graph)), + parameter_bank=bank, changed_sha256=fingerprint(changed), changes=delta) print(f'{len(names)} queries; {sum(map(len, graph.values()))} projected rows', flush=True) for backend in ('vec', 'corgi') if args.backend == 'both' else (args.backend,): for selected in [[n] for n in names] if args.isolated else [names]: - run_server(args, backend, selected, graph, changed, bank, plans, report) + run_server(args, backend, selected, graph, changed, delta, bank, plans, report) report['status'] = 'passed' except BaseException as error: report.update(status='failed', error=f'{type(error).__name__}: {error}') diff --git a/interactive/server/bench/ldbc/test_harness.py b/interactive/server/bench/ldbc/test_harness.py new file mode 100644 index 000000000..e3e83e554 --- /dev/null +++ b/interactive/server/bench/ldbc/test_harness.py @@ -0,0 +1,76 @@ +"""Snapshot dialect and timing boundaries, independent of a running server.""" +import gzip +import io +from pathlib import Path +import tempfile +import unittest +from unittest.mock import patch + +import measure +from snapshot import csv_rows +from snb import data +import workload + + +class HarnessTests(unittest.TestCase): + def test_snapshot_readers_share_the_spark_dialect(self): + text = '\n'.join(( + 'id|name|TypeTagClassId', + '1|"a|b"|2', + r'2|"say \"hi\" and \\ end"|2', + r'3|unquoted\literal|2', + '4|"é\nsecond line"|2', + '5|after multiline|2', + '', + )) + want = {(1, 'a|b'), (2, 'say "hi" and \\ end'), (3, r'unquoted\literal'), + (4, 'é\nsecond line'), (5, 'after multiline')} + with tempfile.TemporaryDirectory() as folder: + snapshot = Path(folder) + for entity in (*data.ENTITIES, *data.STATIC): + directory = snapshot / ('static' if entity in data.STATIC else 'dynamic') / entity + directory.mkdir(parents=True) + (directory / 'part.csv').write_text('id\n', encoding='utf-8') + tag = snapshot / 'static' / 'Tag' / 'part.csv' + compressed = tag.with_suffix('.csv.gz') + # Plain, compressed, and a partition plus its decompressed copy + # must all represent the same rows, exactly once. + for layout in ('plain', 'both', 'gzip'): + with self.subTest(layout=layout): + if layout == 'plain': + tag.write_text(text, encoding='utf-8') + elif layout == 'both': + with gzip.open(compressed, 'wt', encoding='utf-8', newline='') as out: + out.write(text) + else: + tag.unlink() + self.assertEqual(workload.load(snapshot)['tag'], want) + self.assertEqual({row[:2] for row in data.load(snapshot)['tag']}, want) + self.assertEqual(list(csv_rows(io.StringIO('id|name\n1|""\n2|\n'))), + [{'id': '1', 'name': ''}, {'id': '2', 'name': ''}]) + with self.assertRaisesRegex(ValueError, 'field count'): + list(csv_rows(io.StringIO('id|name\n1|extra|field\n'))) + + def test_phase_metrics_remain_separate(self): + class FakeServer: + def commands(self, commands): + self.commands_seen = commands + return dict(encode_ms=3000, wire_ms=5000), [[]] + server = FakeServer() + # Deliberately distinct durations. No wall-clock thresholds or sleeps. + with patch('measure.time.perf_counter', side_effect=[0, 2, 10, 20, 27]): + metrics, replies = measure.commands(server, lambda: ['peek answer']) + self.assertEqual(measure.decode_answer(replies[0], metrics), []) + self.assertEqual(server.commands_seen, ['peek answer']) + self.assertEqual({k: metrics[k] for k in measure.TIMINGS}, + dict(prepare_ms=2000, encode_ms=3000, wire_ms=5000, + decode_ms=7000, client_ms=17000)) + event = dict(metrics, warmup=False, state='initial', phase='read') + summary = measure.summarize([event, dict(event, warmup=True, wire_ms=999999)]) + self.assertEqual(summary['initial/read']['wire_ms']['median_ms'], 5000) + self.assertEqual(summary['initial/read']['client_ms']['median_ms'], 17000) + self.assertEqual(measure.total([event, event]), {k: 2*metrics[k] for k in measure.TIMINGS}) + + +if __name__ == '__main__': + unittest.main() diff --git a/interactive/server/bench/ldbc/workload.py b/interactive/server/bench/ldbc/workload.py index 80d07e0db..485700f5f 100644 --- a/interactive/server/bench/ldbc/workload.py +++ b/interactive/server/bench/ldbc/workload.py @@ -1,11 +1,12 @@ """Narrow SNB snapshot adapter and independent, untimed reference queries.""" from collections import Counter, defaultdict -import csv from datetime import datetime import hashlib import json from pathlib import Path +from snapshot import read_table + HERE = Path(__file__).resolve().parent TABLES = ('person', 'knows', 'message', 'tag', 'message_tag', 'likes', 'place') QUERIES = ('is3', 'ic6', 'bi5', 'bi11') @@ -29,30 +30,25 @@ def load(snapshot): } for entity, table in entities.items(): category = 'static' if entity in ('Tag', 'Place') else 'dynamic' - paths = sorted((snapshot / category / entity).glob('*.csv')) - if not paths: - raise ValueError(f'missing CSV partition(s): {category}/{entity}') - for path in paths: - with path.open(newline='', encoding='utf-8') as source: - for r in csv.DictReader(source, delimiter='|', quoting=csv.QUOTE_NONE): - if table == 'person': - row = (int(r['id']), r['firstName'], r['lastName'], int(r['LocationCityId'])) - elif table == 'knows': - a, b = sorted((int(r['Person1Id']), int(r['Person2Id']))) - row = (a, b, millis(r['creationDate'])) - elif table == 'message': - parent = r.get('ParentPostId') or r.get('ParentCommentId') or '-1' - row = (int(r['id']), int(entity == 'Comment'), int(r['CreatorPersonId']), int(parent)) - elif table == 'tag': - row = (int(r['id']), r['name']) - elif table == 'place': - row = (int(r['id']), r['name'], int(r.get('PartOfPlaceId') or '-1')) - else: - mid = int(r['PostId'] if 'PostId' in r else r['CommentId']) - row = (mid, int(r['TagId'])) if table == 'message_tag' else (int(r['PersonId']), mid) - if any(isinstance(v, str) and not v for v in row): - raise ValueError(f'{path}: empty strings require typed server inputs; not padded here') - result[table].add(row) + for r in read_table(snapshot, category, entity): + if table == 'person': + row = (int(r['id']), r['firstName'], r['lastName'], int(r['LocationCityId'])) + elif table == 'knows': + a, b = sorted((int(r['Person1Id']), int(r['Person2Id']))) + row = (a, b, millis(r['creationDate'])) + elif table == 'message': + parent = r.get('ParentPostId') or r.get('ParentCommentId') or '-1' + row = (int(r['id']), int(entity == 'Comment'), int(r['CreatorPersonId']), int(parent)) + elif table == 'tag': + row = (int(r['id']), r['name']) + elif table == 'place': + row = (int(r['id']), r['name'], int(r.get('PartOfPlaceId') or '-1')) + else: + mid = int(r['PostId'] if 'PostId' in r else r['CommentId']) + row = (mid, int(r['TagId'])) if table == 'message_tag' else (int(r['PersonId']), mid) + if any(isinstance(v, str) and not v for v in row): + raise ValueError(f'{entity}: empty strings require typed server inputs; not padded here') + result[table].add(row) return result From b07b86ed10053092082af9cdb96afb114be344ca Mon Sep 17 00:00:00 2001 From: Frank McSherry Date: Mon, 7 Sep 2026 12:30:48 -0400 Subject: [PATCH 8/8] ddir: strengthen IC3 witnesses while preserving baseline query plans Add a hand-checked IC3 request and retraction witness and vary the tiny server bank to an out-of-neighborhood requester. Keep BI19's unused rank as an optimizer target and leave BI13's endpoint policy unchanged, documenting both. --- interactive/server/bench/ldbc/SNB.md | 18 ++++++++++++++---- interactive/server/bench/ldbc/snb/queries.py | 8 ++++---- interactive/server/bench/ldbc/snb/witness.py | 6 ++++++ interactive/server/bench/ldbc/suite.py | 2 ++ interactive/server/bench/ldbc/test_snb.py | 12 ++++++++++++ 5 files changed, 38 insertions(+), 8 deletions(-) diff --git a/interactive/server/bench/ldbc/SNB.md b/interactive/server/bench/ldbc/SNB.md index 6e7f25ab0..5ee15c6e6 100644 --- a/interactive/server/bench/ldbc/SNB.md +++ b/interactive/server/bench/ldbc/SNB.md @@ -62,7 +62,7 @@ python3 interactive/server/bench/ldbc/suite.py --server target/release/ddir_serv The adapter reads all 18 required tables in SNB BI CSV `composite-merged-fk` layout, including `.csv.gz` partitions, through the same -[snapshot dialect reader](README.md#four-query-control-panel) as the small panel +[snapshot dialect reader](README.md#run) as the small panel (double-quoted fields, backslash escaping within quotes; not raw-generator CSV). A plain partition takes precedence over its `.csv.gz` copy. It projects 17 relations: people, posts/comments, forums, friendships, memberships, tags/classes, places, @@ -89,6 +89,8 @@ reads; generated datasets may legitimately give empty answers. Actual values (people, messages, tags, dates) vary between interactive request batches, not just request IDs. Equal bindings under distinct IDs are also exercised at the default batch size. The full bank and exact bindings are recorded. +The tiny IC3 bank includes a requester outside the qualifying person's two-hop +neighborhood, so varying bindings changes the answer, not just request IDs. Override bindings with `--parameters /path/to/bindings.json`. Its format is a query-name object containing a nonempty list of parameter objects; omitted @@ -171,9 +173,11 @@ fresh baseline. Four-query and full-suite reports are not interchangeable. Every returned row, rank, multiplicity, and empty result is checked against fresh Python evaluation of the **same logical plan**, outside the timers. This checks DDP lowering and incremental execution; it is not an independent query -specification oracle. [test_snb.py](test_snb.py) adds hand-counted witnesses for -paths, ranking/ties, optional nested collections, triangles, propagation, -recruitment, language sets, and floating aggregates. The older four-query panel +specification oracle. All 41 have nonempty default witnesses; **14 queries** +also have independent hand-checked expectations in [test_snb.py](test_snb.py), +not necessarily a complete oracle for each query. IC3 checks two-hop requester +sensitivity and loss of a required-country message. The other 27 have no +independent semantic oracle here. The older four-query panel retains independent traversal/counting oracles. Official conformance validation is still separate work. @@ -216,3 +220,9 @@ BI17 for broad temporal joins. Keep optimizer and kernel changes separate from these definitions so each improvement can be measured against an unchanged workload. In particular, the IC6 baseline has not been hand-rewritten to push the tag filter ahead of friendship/message expansion. +BI19 retains an unbounded rank whose result is discarded by its next projection; +eliminating that unused ranking operation is an optimizer opportunity, not a +query-semantics correction. BI13's current message-count interval includes the +end date while profiles use a strict upper bound. The specification's endpoint +wording remains an interpretation question; this harness correction does not +change that policy or its inclusive calendar-month count. diff --git a/interactive/server/bench/ldbc/snb/queries.py b/interactive/server/bench/ldbc/snb/queries.py index abf8f3d64..faf862a37 100644 --- a/interactive/server/bench/ldbc/snb/queries.py +++ b/interactive/server/bench/ldbc/snb/queries.py @@ -450,12 +450,12 @@ def bi18(x,q): @query('bi19','city1:int city2:int','Interaction path between cities') def bi19(x,q): - # One start per person, carried in the internal request key. Expand request - # identity to (original rid, source) through a deterministic id relation. + # One start per person. The unbounded rank is discarded by the seeds + # projection: retain this baseline form to exercise optimizer elimination + # of an unused ranking operation, rather than hand-optimizing the query. starts=q.join(x.person,{'city1':'city'},'p_').select('rid','city2',source=c.p_id) indexed=starts.rank([c.source],groups=('rid',)).rename(rank='startindex') - # Source IDs are globally unique. Use them as internal request IDs; preserve - # the outer request in a separate field by running paths per (rid,source). + # Carry both outer request identity and source through each path search. # The composite key is encoded as a tuple, not a collision-prone hash. seeds=indexed.select(rid=call('tuple',c.rid,c.source),p1=c.source) paths=x.shortest(seeds,x.weighted).select(internal=c.rid,node=c.node,cost=c.cost) diff --git a/interactive/server/bench/ldbc/snb/witness.py b/interactive/server/bench/ldbc/snb/witness.py index be7faa5d8..eec7167ae 100644 --- a/interactive/server/bench/ldbc/snb/witness.py +++ b/interactive/server/bench/ldbc/snb/witness.py @@ -79,6 +79,12 @@ def params(name): return base +def alternate_params(name): + # Person 4 is beyond two hops from 8; unlike 1 and 2, this requester has + # no IC3 answer. Exercise parameter sensitivity through the real server. + return {'pid': 8} if name == 'ic3' else {} + + def changed(original): """A valid projected update: remove a friendship and a post, rename a person.""" g={name:set(rows) for name,rows in original.items()} diff --git a/interactive/server/bench/ldbc/suite.py b/interactive/server/bench/ldbc/suite.py index 2d7332447..4c05fd599 100644 --- a/interactive/server/bench/ldbc/suite.py +++ b/interactive/server/bench/ldbc/suite.py @@ -235,6 +235,8 @@ def git(*command): if not args.snapshot: base.update(witness.params(name)) bank[name] = [base, alternate(graph, base)] + if not args.snapshot: + bank[name][1].update(witness.alternate_params(name)) if name in overrides: if not isinstance(overrides[name], list) or not overrides[name]: raise ValueError(f'{name}: parameters must be a nonempty list of objects') diff --git a/interactive/server/bench/ldbc/test_snb.py b/interactive/server/bench/ldbc/test_snb.py index 8fc806fb8..45173ef76 100644 --- a/interactive/server/bench/ldbc/test_snb.py +++ b/interactive/server/bench/ldbc/test_snb.py @@ -60,6 +60,18 @@ def test_hand_checked_ranking_and_optional_matches(self): self.assertEqual(len(rows),7) self.assertTrue(any(not r[-1] for r in rows)) + def test_ic3_depends_on_request_and_foreign_messages(self): + # Person 4 lives in Gamma, is two hops from 1, and has one message + # in each of Alpha and Beta. From 8, person 4 is beyond two hops. + self.assertEqual(self.rows('ic3'), [[4, list(b'Alex'), list(b'Person4'), 1, 1, 2]]) + self.assertEqual(self.rows('ic3', **witness.alternate_params('ic3')), []) + # Retract the only Alpha message (and its tags). One visited country + # is insufficient even though the friendship paths still exist. + changed = {name: set(rows) for name, rows in self.graph.items()} + changed['message'] = {row for row in changed['message'] if row[0] != 10009} + changed['mtag'] = {row for row in changed['mtag'] if row[0] != 10009} + self.assertEqual(self.rows('ic3', changed), []) + def test_hand_checked_floating_aggregates(self): rows=self.rows('bi1') self.assertEqual([(r[0],r[1],r[2],r[3],r[5]) for r in rows],[(2012,0,0,31,120),(2012,1,0,4,16)])