diff --git a/publications/comnet/experiments/analysis/aggregate.py b/publications/comnet/experiments/analysis/aggregate.py index dcbd3c24..4e3a4951 100755 --- a/publications/comnet/experiments/analysis/aggregate.py +++ b/publications/comnet/experiments/analysis/aggregate.py @@ -6,6 +6,8 @@ import sys from pathlib import Path +from scipy import stats + def load_results(results_dir: Path) -> dict[str, list]: experiments: dict[str, list] = {} @@ -49,7 +51,8 @@ def stdev(values: list[float]) -> float: def ci95(values: list[float]) -> float: if len(values) < 2: return 0.0 - return 1.96 * stdev(values) / math.sqrt(len(values)) + n = len(values) + return float(stats.t.ppf(0.975, n - 1)) * stdev(values) / math.sqrt(n) def extract_metric(data: dict) -> dict[str, float]: diff --git a/publications/comnet/experiments/analysis/buildcheck.py b/publications/comnet/experiments/analysis/buildcheck.py new file mode 100644 index 00000000..d0064464 --- /dev/null +++ b/publications/comnet/experiments/analysis/buildcheck.py @@ -0,0 +1,72 @@ +import csv +import json +import statistics +import sys +from pathlib import Path + +CONFIGS = ["tcp", "tls", "quic-main"] +BUILDS = ["bcfleet", "bclater"] +FANOUT = 8 + + +def collect(base): + cells = {} + excluded = [] + for directory in sorted(base.glob("03h_buildcheck_g*")): + accepted = {row["label"]: row for row in csv.DictReader(open(directory / "acceptance.csv"))} + for row in csv.DictReader(open(directory / "manifest.csv")): + gate = accepted.get(row["label"]) + if gate is None or gate["pass"] != "True": + excluded.append((row["label"], gate["failures"] if gate else "not evaluated")) + continue + sub = json.load(open(directory / f"{row['label']}.json")) + pub = json.load(open(directory / f"{row['label']}_pub.json")) + cells.setdefault((row["phase"], row["config"]), []).append({ + "group": directory.name[-2:], + "unique": sub["results"]["throughput_avg"] / FANOUT, + "offered": pub["results"]["offered_rate"], + "sha": row["binary_sha"], + }) + return cells, excluded + + +def group_means(samples, field): + per_group = {} + for sample in samples: + per_group.setdefault(sample["group"], []).append(sample[field]) + return {group: statistics.mean(values) for group, values in per_group.items()} + + +def within_group_ratio(numerator, denominator): + left, right = group_means(numerator, "unique"), group_means(denominator, "unique") + return [left[g] / right[g] for g in sorted(set(left) & set(right))] + + +def main(results_dir): + cells, excluded = collect(Path(results_dir)) + print(f"excluded runs: {len(excluded)}") + for label, reason in excluded: + print(f" {label}: {reason}") + print("\n=== 10% loss, router arm, unpaced. unique msg/s pooled mean (per-group means); offered msg/s") + for config in CONFIGS: + for build in BUILDS: + samples = cells.get((build, config), []) + if not samples: + continue + groups = " ".join(f"{g}:{v:.0f}" for g, v in sorted(group_means(samples, "unique").items())) + shas = {s["sha"][:8] for s in samples} + print(f" {config:10s} {build:8s} n={len(samples)} unique {statistics.mean(s['unique'] for s in samples):7.1f} ({groups}) " + f"offered {statistics.mean(s['offered'] for s in samples):9.0f} sha {','.join(sorted(shas))}") + ratios = within_group_ratio(cells.get(("bclater", config), []), cells.get(("bcfleet", config), [])) + if ratios: + print(f" {config:10s} later/fleet {statistics.mean(ratios):.3f} (range {min(ratios):.3f}-{max(ratios):.3f}, {len(ratios)} groups)") + for build in BUILDS: + for other in ("tls", "tcp"): + ratios = within_group_ratio(cells.get((build, "quic-main"), []), cells.get((build, other), [])) + if ratios: + print(f" QUIC/{other} {build:8s} {statistics.mean(ratios):.3f} (range {min(ratios):.3f}-{max(ratios):.3f}, {len(ratios)} groups)") + + +if __name__ == "__main__": + default = Path(__file__).resolve().parent.parent / "results-v5" + main(sys.argv[1] if len(sys.argv) > 1 else default) diff --git a/publications/comnet/experiments/analysis/capped_ppub.py b/publications/comnet/experiments/analysis/capped_ppub.py new file mode 100644 index 00000000..7fc0ed34 --- /dev/null +++ b/publications/comnet/experiments/analysis/capped_ppub.py @@ -0,0 +1,158 @@ +import csv +import json +import statistics +import sys +from pathlib import Path + +CONFIGS = ["quic-ppub", "quic-main-ppub", "quic-ctl"] +LABELS = { + "quic-ppub": "per-publish delivery (client per-publish)", + "quic-main-ppub": "per-topic delivery (client per-publish)", + "quic-ctl": "control-only delivery (client control-only)", +} +RATES = [1250, 2500, 5000, 10000, 20000, 0] +CAP_PROBE_RATES = {10000, 20000} +FANOUT = 8 +PACING_TOLERANCE = 0.02 +SUB_SATURATION_PREFIX = "P3: sub busy" +METRICS = ["offered", "delivered", "ratio", "cpu", "cpu_us_per_delivered", "cpu_us_per_offered", "pkts_per_delivered"] + + +def load_json(path): + try: + return json.load(open(path)) + except (OSError, json.JSONDecodeError): + return None + + +def active_cpu(path): + try: + values = [float(row["cpu_percent"]) for row in csv.DictReader(open(path)) if row.get("cpu_percent")] + except (OSError, ValueError): + return None + if not values: + return None + return statistics.mean(value for value in values if value >= 0.5 * max(values)) + + +def acceptance(directory): + path = directory / "acceptance.csv" + if not path.exists(): + raise SystemExit(f"{path} missing: run single_segment_accept.py {directory} first") + return {row["label"]: row for row in csv.DictReader(open(path))} + + +def collect(base): + cells = {} + excluded = [] + for directory in sorted(base.glob("03f_capped_g*")): + accepted = acceptance(directory) + rows = {row["label"]: row for row in csv.DictReader(open(directory / "manifest.csv"))}.values() + for row in rows: + label = row["label"] + gate = accepted.get(label) + if gate is None: + excluded.append((label, "not evaluated by acceptance")) + continue + reasons = [reason for reason in gate["failures"].split("; ") if reason] + if any(not reason.startswith(SUB_SATURATION_PREFIX) for reason in reasons): + excluded.append((label, gate["failures"])) + continue + saturated = bool(reasons) + pub = load_json(directory / f"{label}_pub.json") + sub = load_json(directory / f"{label}.json") + if not pub or not sub: + excluded.append((label, "missing results")) + continue + target = int(row["rate"]) + configured = int(pub.get("config", {}).get("rate", 0) or 0) + offered = pub["results"].get("offered_rate") or 0.0 + if configured != target: + excluded.append((label, f"bench rate {configured} != planned {target}")) + continue + if target and abs(offered / target - 1) > PACING_TOLERANCE: + excluded.append((label, f"paced at {offered:.0f}/s, target {target}/s")) + continue + delivered = sub["results"]["throughput_avg"] + received = sub["results"].get("received") or 0 + cpu = None if saturated else active_cpu(directory / f"{label}_broker_resources.csv") + skbs = float(gate["skbs"]) if gate.get("skbs") else None + cells.setdefault((row["config"], target, int(row["loss"])), []).append({ + "group": directory.name[-2:], + "saturated": saturated, + "offered": offered, + "delivered": delivered, + "ratio": delivered / (FANOUT * offered) if offered else None, + "cpu": cpu, + "cpu_us_per_delivered": cpu / 100 / delivered * 1e6 if cpu and delivered else None, + "cpu_us_per_offered": cpu / 100 / offered * 1e6 if cpu and offered else None, + "pkts_per_delivered": skbs / received if skbs and received else None, + }) + return cells, excluded + + +def summary(samples, field): + values = [s[field] for s in samples if s[field] is not None] + if not values: + return "-" + per_group = {} + for sample in samples: + if sample[field] is not None: + per_group.setdefault(sample["group"], []).append(sample[field]) + groups = " ".join(f"{g}:{statistics.mean(v):.3g}" for g, v in sorted(per_group.items())) + return f"{statistics.mean(values):.3g} ({groups})" + + +def paired_differences(cells, loss, field, left, right): + out = [] + for rate in RATES: + a, b = cells.get((left, rate, loss)), cells.get((right, rate, loss)) + if not a or not b: + continue + diffs = [] + for group in sorted({s["group"] for s in a} & {s["group"] for s in b}): + va = [s[field] for s in a if s["group"] == group and s[field] is not None] + vb = [s[field] for s in b if s["group"] == group and s[field] is not None] + if va and vb: + diffs.append(statistics.mean(va) / statistics.mean(vb)) + if diffs: + out.append((rate, statistics.mean(diffs), min(diffs), max(diffs), len(diffs))) + return out + + +def main(results_dir): + cells, excluded = collect(Path(results_dir)) + print(f"excluded runs: {len(excluded)}") + for label, reason in excluded: + print(f" {label}: {reason}") + for loss in (0, 1): + print(f"\n=== loss {loss}% (router arm), 8 subscribers. Values: pooled mean (per-group means g1..g3)") + for config in CONFIGS: + print(f" {LABELS[config]}") + for rate in RATES: + samples = cells.get((config, rate, loss)) + if not samples: + continue + name = f"rate {rate}" if rate else "uncapped" + tags = [] + if rate in CAP_PROBE_RATES and config == "quic-ppub": + tags.append("cap probe") + saturated = sum(s["saturated"] for s in samples) + if saturated: + tags.append(f"{saturated} sub-saturated (cpu omitted)") + print(f" {name:11s} n={len(samples):2d} {'[' + ', '.join(tags) + ']' if tags else ''}") + for field in METRICS: + print(f" {field:22s} {summary(samples, field)}") + for field in ("cpu_us_per_delivered", "pkts_per_delivered"): + for other in ("quic-main-ppub", "quic-ctl"): + rows = paired_differences(cells, loss, field, "quic-ppub", other) + if rows: + print(f" within-group ratio {field}: per-publish / {other}") + for rate, mean, low, high, n in rows: + name = f"rate {rate}" if rate else "uncapped" + print(f" {name:11s} {mean:5.2f} (range {low:.2f}-{high:.2f}, {n} groups)") + + +if __name__ == "__main__": + default = Path(__file__).resolve().parent.parent / "results-v5" + main(sys.argv[1] if len(sys.argv) > 1 else default) diff --git a/publications/comnet/experiments/analysis/e1_fairness.py b/publications/comnet/experiments/analysis/e1_fairness.py new file mode 100644 index 00000000..9281d0d5 --- /dev/null +++ b/publications/comnet/experiments/analysis/e1_fairness.py @@ -0,0 +1,114 @@ +import csv +import glob +import json +import statistics +import sys +from pathlib import Path + + +def broker_tx_mbit(broker_csv: Path): + rows = [r for r in csv.DictReader(open(broker_csv)) if r["net_tx_bytes"].isdigit()] + if len(rows) < 4: + return None + rows = rows[1:-1] + dt = float(rows[-1]["timestamp"]) - float(rows[0]["timestamp"]) + db = int(rows[-1]["net_tx_bytes"]) - int(rows[0]["net_tx_bytes"]) + return db * 8 / 1e6 / dt if dt > 0 else None + + +def iperf_mbit(iperf_json: Path): + end = json.load(open(iperf_json)).get("end", {}) + if "sum_received" not in end: + return None + return end["sum_received"]["bits_per_second"] / 1e6 + + +def arm_fairness(results_dir: Path, arm: str, loss: int): + prefix = f"{arm}_rate*_loss{loss}pct" + shares, iperfs, jains = [], [], [] + for bc in sorted(glob.glob(str(results_dir / f"{prefix}_run*_broker_resources.csv"))): + run = Path(bc).name.replace("_broker_resources.csv", "") + ic = results_dir / f"{run}_iperf.json" + mc = results_dir / f"{run}_messages.csv" + tx = broker_tx_mbit(Path(bc)) + ip = iperf_mbit(ic) if ic.exists() else None + if tx and ip and tx > ip: + shares.append(100 * (tx - ip) / tx) + iperfs.append(ip) + if mc.exists(): + gp = per_connection_goodput(mc) + if len(gp) > 1: + jains.append(jain_index(list(gp.values()))) + med = lambda v: round(statistics.median(v), 1) if v else None + return {"n": len(shares), "mqtt_wire_share_pct": med(shares), + "iperf_mbit": med(iperfs), "jain": (round(statistics.median(jains), 3) if jains else None)} + + +def analyze_dir(results_dir: Path): + arms = ["tcp-1conn", "tcp-Nconn", "quic-control", "quic-pertopic"] + print(f"E1 fairness (MQTT share of a bottleneck shared with one greedy TCP flow)") + print(f"{'arm':>14} {'loss':>5} | {'MQTT share%':>11} {'iperf Mbit':>11} {'intra-Jain':>11} {'n':>3}") + for arm in arms: + for loss in [0, 1]: + r = arm_fairness(results_dir, arm, loss) + if r["n"]: + print(f"{arm:>14} {loss:>4}% | {str(r['mqtt_wire_share_pct']):>11} " + f"{str(r['iperf_mbit']):>11} {str(r['jain']):>11} {r['n']:>3}") + + +def per_connection_goodput(messages_csv: Path, trim: float = 0.1): + receive_ns = {} + for row in csv.DictReader(open(messages_csv)): + conn = int(row["conn_idx"]) + receive_ns.setdefault(conn, []).append(int(row["receive_ns"])) + + all_ns = [ns for series in receive_ns.values() for ns in series] + if not all_ns: + return {} + lo, hi = min(all_ns), max(all_ns) + span = hi - lo + start = lo + int(span * trim) + end = hi - int(span * trim) + window_s = max((end - start) / 1e9, 1e-9) + + goodput = {} + for conn, series in receive_ns.items(): + in_window = sum(1 for ns in series if start <= ns <= end) + goodput[conn] = in_window / window_s + return goodput + + +def jain_index(values): + if not values: + return 0.0 + n = len(values) + total = sum(values) + total_sq = sum(v * v for v in values) + if total_sq == 0: + return 0.0 + return (total * total) / (n * total_sq) + + +def main(messages_csv: Path): + goodput = per_connection_goodput(messages_csv) + if not goodput: + print(f"no data in {messages_csv}") + return + flows = [goodput[c] for c in sorted(goodput)] + print(f"file: {messages_csv}") + print(f"connections: {len(flows)}") + for conn in sorted(goodput): + print(f" conn {conn}: {goodput[conn]:.1f} msg/s") + print(f"aggregate: {sum(flows):.1f} msg/s") + print(f"Jain fairness index: {jain_index(flows):.4f} (1.0 = perfectly fair)") + + +if __name__ == "__main__": + if len(sys.argv) < 2: + print(f"usage: {sys.argv[0]} ") + sys.exit(1) + target = Path(sys.argv[1]) + if target.is_dir(): + analyze_dir(target) + else: + main(target) diff --git a/publications/comnet/experiments/analysis/figures/fig01_spike_iso_vs_loss.py b/publications/comnet/experiments/analysis/figures/fig01_spike_iso_vs_loss.py index 442f954a..a7ce54b4 100644 --- a/publications/comnet/experiments/analysis/figures/fig01_spike_iso_vs_loss.py +++ b/publications/comnet/experiments/analysis/figures/fig01_spike_iso_vs_loss.py @@ -16,8 +16,8 @@ save_figure, ) -LOSS_RATES = [0, 1, 2, 5] -LOSS_LABELS = ["0%", "1%", "2%", "5%"] +LOSS_RATES = [1, 2, 5] +LOSS_LABELS = ["1%", "2%", "5%"] RUNS = range(1, 16) @@ -64,13 +64,6 @@ def main(results_dir: Path, output_dir: Path): fig, ax = plt.subplots(figsize=(7, 4.5)) - x_offsets = { - "tcp": -0.15, - "quic-control": -0.05, - "quic-pertopic": 0.05, - "quic-perpub": 0.15, - } - group_positions = np.arange(len(LOSS_RATES)) for transport in TRANSPORT_ORDER: @@ -82,7 +75,7 @@ def main(results_dir: Path, output_dir: Path): mean, ci_half = compute_ci(data[transport][loss]) means.append(mean) ci_halves.append(ci_half) - positions.append(group_positions[loss_idx] + x_offsets[transport]) + positions.append(group_positions[loss_idx]) ax.errorbar( positions, diff --git a/publications/comnet/experiments/analysis/figures/fig03_spike_iso_vs_topics.py b/publications/comnet/experiments/analysis/figures/fig03_spike_iso_vs_topics.py index edc7f271..2efe8372 100644 --- a/publications/comnet/experiments/analysis/figures/fig03_spike_iso_vs_topics.py +++ b/publications/comnet/experiments/analysis/figures/fig03_spike_iso_vs_topics.py @@ -30,9 +30,13 @@ def load_topic_scaling_data(results_dir: Path): for run in RUNS: filepath = exp02_dir / f"{transport}_loss1pct_run{run}.json" if filepath.exists(): + try: + with open(filepath) as f: + result = json.load(f) + except json.JSONDecodeError: + print(f" warning: skipping invalid JSON: {filepath.name}") + continue sources.setdefault(transport, {}).setdefault(8, []) - with open(filepath) as f: - result = json.load(f) sources[transport][8].append( result["results"]["windowed_correlation"] ) @@ -46,11 +50,15 @@ def load_topic_scaling_data(results_dir: Path): / f"{transport}_{topic_count}topics_run{run}.json" ) if filepath.exists(): + try: + with open(filepath) as f: + result = json.load(f) + except json.JSONDecodeError: + print(f" warning: skipping invalid JSON: {filepath.name}") + continue sources.setdefault(transport, {}).setdefault( topic_count, [] ) - with open(filepath) as f: - result = json.load(f) sources[transport][topic_count].append( result["results"]["windowed_correlation"] ) diff --git a/publications/comnet/experiments/analysis/figures/fig04_wcorr_vs_spike_iso.py b/publications/comnet/experiments/analysis/figures/fig04_wcorr_vs_spike_iso.py index 35d84cf8..0a5c40bc 100644 --- a/publications/comnet/experiments/analysis/figures/fig04_wcorr_vs_spike_iso.py +++ b/publications/comnet/experiments/analysis/figures/fig04_wcorr_vs_spike_iso.py @@ -16,8 +16,8 @@ save_figure, ) -LOSS_RATES = [0, 1, 2, 5] -LOSS_LABELS = ["0%", "1%", "2%", "5%"] +LOSS_RATES = [1, 2, 5] +LOSS_LABELS = ["1%", "2%", "5%"] RUNS = range(1, 16) @@ -64,13 +64,6 @@ def main(results_dir: Path, output_dir: Path): fig, ax = plt.subplots(figsize=(7, 4.5)) - x_offsets = { - "tcp": -0.15, - "quic-control": -0.05, - "quic-pertopic": 0.05, - "quic-perpub": 0.15, - } - group_positions = np.arange(len(LOSS_RATES)) for transport in TRANSPORT_ORDER: @@ -82,7 +75,7 @@ def main(results_dir: Path, output_dir: Path): mean, ci_half = compute_ci(data[transport][loss]) means.append(mean) ci_halves.append(ci_half) - positions.append(group_positions[loss_idx] + x_offsets[transport]) + positions.append(group_positions[loss_idx]) ax.errorbar( positions, @@ -105,9 +98,9 @@ def main(results_dir: Path, output_dir: Path): ax.set_ylabel("Spike Isolation Ratio") ax.set_xticks(group_positions) ax.set_xticklabels(LOSS_LABELS) - ax.set_ylim(-0.05, 1.15) + ax.set_ylim(0.5, 1.05) ax.axhline(y=1.0, color="gray", linewidth=0.5, linestyle="--", zorder=1) - ax.legend(loc="center right", framealpha=0.9) + ax.legend(loc="lower right", framealpha=0.9) fig.tight_layout() save_figure(fig, output_dir, "fig04_wcorr_vs_spike_iso") diff --git a/publications/comnet/experiments/analysis/figures/fig08_connection_latency.py b/publications/comnet/experiments/analysis/figures/fig08_connection_latency.py index 51f60727..e03349a4 100644 --- a/publications/comnet/experiments/analysis/figures/fig08_connection_latency.py +++ b/publications/comnet/experiments/analysis/figures/fig08_connection_latency.py @@ -3,6 +3,7 @@ from pathlib import Path import matplotlib.pyplot as plt +from matplotlib.ticker import FixedLocator, FuncFormatter, NullLocator import numpy as np from scipy import stats @@ -18,6 +19,8 @@ DELAYS_MS = [0, 25, 50, 100, 200] DELAY_LABELS = ["0ms", "25ms", "50ms", "100ms", "200ms"] RUNS = range(1, 16) +US_PER_MS = 1000 +Y_TICKS_MS = [0.5, 1, 2, 5, 10, 20, 50, 100, 200, 500] def load_connection_latency(results_dir: Path): @@ -38,8 +41,8 @@ def load_connection_latency(results_dir: Path): continue with open(filepath) as f: result = json.load(f) - p50_values.append(result["results"]["p50_connect_us"]) - p95_values.append(result["results"]["p95_connect_us"]) + p50_values.append(result["results"]["p50_connect_us"] / US_PER_MS) + p95_values.append(result["results"]["p95_connect_us"] / US_PER_MS) if p50_values: data[transport][delay] = {"p50": p50_values, "p95": p95_values} return data @@ -96,9 +99,13 @@ def main(results_dir: Path, output_dir: Path): linewidth=0.5, ) - ax.set_xlabel("One-Way Delay (RTT = 2x)") - ax.set_ylabel("Connect Latency (us)") + ax.set_xlabel("Emulated Path Delay (ms)") + ax.set_ylabel("Connect Latency (ms)") ax.set_yscale("log") + ax.yaxis.set_major_locator(FixedLocator(Y_TICKS_MS)) + ax.yaxis.set_minor_locator(NullLocator()) + ax.yaxis.set_major_formatter(FuncFormatter(lambda value, _: f"{value:g}")) + ax.set_ylim(Y_TICKS_MS[0], 800) ax.set_title("Connection Setup Latency vs. Network Delay") ax.set_xticks(group_positions) ax.set_xticklabels(DELAY_LABELS) diff --git a/publications/comnet/experiments/analysis/figures/fig09_throughput_vs_loss.py b/publications/comnet/experiments/analysis/figures/fig09_throughput_vs_loss.py index add938e8..66ab3553 100644 --- a/publications/comnet/experiments/analysis/figures/fig09_throughput_vs_loss.py +++ b/publications/comnet/experiments/analysis/figures/fig09_throughput_vs_loss.py @@ -1,8 +1,10 @@ +import csv import json import sys from pathlib import Path import matplotlib.pyplot as plt +from matplotlib.ticker import FixedLocator, FuncFormatter, NullLocator import numpy as np from scipy import stats @@ -18,96 +20,111 @@ LOSS_RATES = [0, 1, 2, 5, 10] LOSS_LABELS = ["0%", "1%", "2%", "5%", "10%"] -QOS_LEVELS = [0, 1] RUNS = range(1, 16) - - -def load_throughput_data(results_dir: Path): +Y_TICKS = [500, 1000, 2000, 5000, 10000, 20000, 50000] +GROUPS = (1, 2, 3) +PER_PACKET_MODE = "router" +PER_PACKET_PHASE = "main" +PER_PACKET_CONFIGS = { + "tcp": "tcp", + "tls": "tls", + "quic-control-only": "quic-main", + "quic-per-topic": "quic-main-ptopic", + "quic-per-publish": "quic-main-ppub", +} + + +def load_per_buffer(results_dir: Path): exp_dir = results_dir / "03_throughput_under_loss" - if not exp_dir.exists(): - print(f" WARNING: {exp_dir} not found, skipping fig09") - return None - data = {} for strategy in THROUGHPUT_ORDER: - data[strategy] = {} - for qos in QOS_LEVELS: - data[strategy][qos] = {} - for loss in LOSS_RATES: - values = [] - for run in RUNS: - filepath = exp_dir / f"{strategy}_qos{qos}_loss{loss}pct_run{run}.json" - if not filepath.exists(): - continue - with open(filepath) as f: - result = json.load(f) - values.append(result["results"]["throughput_avg"]) - if values: - data[strategy][qos][loss] = values + for loss in LOSS_RATES: + values = [] + for run in RUNS: + filepath = exp_dir / f"{strategy}_qos0_loss{loss}pct_run{run}.json" + if not filepath.exists(): + continue + result = json.load(open(filepath)) + subscribers = result["config"].get("subscribers") or 1 + values.append(result["results"]["throughput_avg"] / subscribers) + if values: + data[(strategy, loss)] = values + return data + + +def load_per_packet(results_dir: Path): + strategy_of = {config: strategy for strategy, config in PER_PACKET_CONFIGS.items()} + data = {} + for group in GROUPS: + exp_dir = results_dir / f"03e_single_segment_g{group}" + accepted = {row["label"] for row in csv.DictReader(open(exp_dir / "acceptance.csv")) if row["pass"] == "True"} + for row in csv.DictReader(open(exp_dir / "manifest.csv")): + if row["phase"] != PER_PACKET_PHASE or row["mode"] != PER_PACKET_MODE or row["broker_probe"] == "1": + continue + strategy = strategy_of.get(row["config"]) + if strategy is None or row["label"] not in accepted: + continue + result = json.load(open(exp_dir / f"{row['label']}.json")) + subscribers = result["config"].get("subscribers") or 1 + data.setdefault((strategy, int(row["loss"])), []).append(result["results"]["throughput_avg"] / subscribers) return data def compute_ci(values, confidence=0.95): - n = len(values) - if n < 2: - return np.mean(values), 0.0 - m = np.mean(values) - sem = stats.sem(values) - t_crit = stats.t.ppf((1 + confidence) / 2, df=n - 1) - return m, t_crit * sem + mean = np.mean(values) + if len(values) < 2: + return mean, 0.0 + return mean, stats.t.ppf((1 + confidence) / 2, df=len(values) - 1) * stats.sem(values) + + +def series(data, strategy): + points = [(loss, *compute_ci(data[(strategy, loss)])) for loss in LOSS_RATES if (strategy, loss) in data] + return [np.array(column) for column in zip(*points)] if points else None def main(results_dir: Path, output_dir: Path): apply_style() - data = load_throughput_data(results_dir) - if data is None: - return - - fig, axes = plt.subplots(2, 1, figsize=(7, 8), sharex=True) - - for qos_idx, qos in enumerate(QOS_LEVELS): - ax = axes[qos_idx] - for strategy in THROUGHPUT_ORDER: - x_vals = [] - y_means = [] - y_lo = [] - y_hi = [] - for loss in LOSS_RATES: - if loss in data[strategy][qos]: - m, ci_half = compute_ci(data[strategy][qos][loss]) - x_vals.append(loss) - y_means.append(m) - y_lo.append(m - ci_half) - y_hi.append(m + ci_half) - - if not x_vals: - continue + per_buffer = load_per_buffer(results_dir) + per_packet = load_per_packet(results_dir) - ax.plot( - x_vals, - y_means, - marker=THROUGHPUT_MARKERS[strategy], - color=THROUGHPUT_COLORS[strategy], - label=THROUGHPUT_LABELS[strategy], - linewidth=1.5, - markersize=5, - ) - ax.fill_between( - x_vals, - y_lo, - y_hi, - alpha=0.15, - color=THROUGHPUT_COLORS[strategy], - ) - - ax.set_ylabel("Throughput (msgs/sec)") - ax.set_title(f"QoS {qos}") - ax.legend(loc="best", framealpha=0.9) - - axes[1].set_xlabel("Packet Loss Rate (%)") - axes[1].set_xticks(LOSS_RATES) - axes[1].set_xticklabels(LOSS_LABELS) - fig.suptitle("Throughput vs. Packet Loss Rate", fontsize=12, y=0.98) + fig, ax = plt.subplots(figsize=(7, 4)) + for strategy in THROUGHPUT_ORDER: + color = THROUGHPUT_COLORS[strategy] + reference = series(per_buffer, strategy) + if reference is not None: + x, mean, _ = reference + ax.plot(x, mean, linestyle="--", color=color, alpha=0.35, linewidth=1.0) + current = series(per_packet, strategy) + if current is None: + continue + x, mean, half = current + ax.errorbar( + x, + mean, + yerr=half, + marker=THROUGHPUT_MARKERS[strategy], + color=color, + label=THROUGHPUT_LABELS[strategy], + linewidth=1.5, + markersize=5, + capsize=3, + ) + for loss in x: + values = per_packet[(strategy, int(loss))] + print(f"{strategy} {int(loss)}%: n={len(values)} mean={np.mean(values):.0f}") + + ax.plot([], [], linestyle="--", color="grey", alpha=0.5, label="Per-buffer loss on broker egress") + ax.set_yscale("log") + ax.yaxis.set_major_locator(FixedLocator(Y_TICKS)) + ax.yaxis.set_minor_locator(NullLocator()) + ax.yaxis.set_major_formatter(FuncFormatter(lambda value, _: f"{value / 1000:g}K")) + ax.set_ylim(Y_TICKS[0], 80000) + ax.set_ylabel("Delivered throughput (unique msg/s)") + ax.set_xlabel("Packet Loss Rate (%)") + ax.set_xticks(LOSS_RATES) + ax.set_xticklabels(LOSS_LABELS) + ax.legend(loc="best", framealpha=0.9) + ax.set_title("Delivered Throughput vs. Packet Loss Rate (QoS 0)") fig.tight_layout() save_figure(fig, output_dir, "fig09_throughput_vs_loss") diff --git a/publications/comnet/experiments/analysis/figures/fig14_strategy_comparison.py b/publications/comnet/experiments/analysis/figures/fig14_strategy_comparison.py index 03802fab..65c4a80d 100644 --- a/publications/comnet/experiments/analysis/figures/fig14_strategy_comparison.py +++ b/publications/comnet/experiments/analysis/figures/fig14_strategy_comparison.py @@ -42,11 +42,17 @@ def load_data(results_dir: Path): for topics in TOPIC_COUNTS: tp_values = [] for run in RUNS: - tp_path = exp_dir / f"{strategy}_{topics}topics_throughput_run{run}.json" + tp_path = exp_dir / f"{strategy}_{topics}topics_throughput_run{run}_pub.json" if tp_path.exists(): - with open(tp_path) as f: - d = json.load(f) - tp_values.append(d["results"]["throughput_avg"]) + try: + with open(tp_path) as f: + d = json.load(f) + except json.JSONDecodeError: + print(f" warning: skipping invalid JSON: {tp_path.name}") + continue + elapsed = d["results"].get("elapsed_secs", 0) + if elapsed: + tp_values.append(d["results"]["published"] / elapsed) if tp_values: data[strategy]["throughput"][topics] = tp_values @@ -85,10 +91,13 @@ def main(results_dir: Path, output_dir: Path): ) ax_tp.set_xlabel("Topic Count") - ax_tp.set_ylabel("Throughput (K msgs/sec)") - ax_tp.set_title("Throughput vs. Topic Count") + ax_tp.set_ylabel("Publish rate (K msg/s)") + ax_tp.set_title("Single-Connection Publish Rate vs. Topic Count") ax_tp.set_xticks(TOPIC_COUNTS) - ax_tp.legend(loc="best", fontsize=8) + ax_tp.set_yticks([0, 10, 20, 30, 40, 50]) + ax_tp.set_ylim(0, 57) + ax_tp.grid(True, which="major", axis="y", linewidth=0.4, alpha=0.4) + ax_tp.legend(loc="center", bbox_to_anchor=(0.5, 0.36), fontsize=8) fig.tight_layout() save_figure(fig, output_dir, "fig14_strategy_comparison") diff --git a/publications/comnet/experiments/analysis/figures/fig_hol_excess.py b/publications/comnet/experiments/analysis/figures/fig_hol_excess.py new file mode 100644 index 00000000..8a0e6be7 --- /dev/null +++ b/publications/comnet/experiments/analysis/figures/fig_hol_excess.py @@ -0,0 +1,85 @@ +import json +import sys +from pathlib import Path + +import matplotlib.pyplot as plt + +sys.path.insert(0, str(Path(__file__).parent)) +from style import ( + TRANSPORT_COLORS, + TRANSPORT_LABELS, + TRANSPORT_MARKERS, + TRANSPORT_ORDER, + apply_style, + save_figure, +) + +TOPICS = [2, 4, 8, 16, 32] +RATES = [125, 250, 500, 1000, 2000] + + +def main(decomp_path: Path, output_dir: Path): + apply_style() + with open(decomp_path) as f: + d = json.load(f)["1"] + + fig, (ax1, ax2) = plt.subplots(2, 1, figsize=(5, 6.4)) + + for tr in TRANSPORT_ORDER: + cells = d["topics"].get(tr, {}) + xs, ys, es = [], [], [] + for t in TOPICS: + c = cells.get(str(t)) + if c: + xs.append(t) + ys.append(c["excess"][0]) + es.append(c["excess"][1]) + ax1.errorbar( + range(len(xs)), ys, yerr=es, + fmt=TRANSPORT_MARKERS[tr] + "-", color=TRANSPORT_COLORS[tr], + label=TRANSPORT_LABELS[tr], markersize=7, capsize=3, linewidth=1.5, + markeredgecolor="white", markeredgewidth=0.7, zorder=3, + ) + ax1.set_xlabel("Topics (streams)") + ax1.set_ylabel("Excess co-occurrence (real coupling)") + ax1.set_xticks(range(len(TOPICS))) + ax1.set_xticklabels([str(t) for t in TOPICS]) + ax1.set_ylim(0, 1.0) + ax1.text(0.03, 0.06, "more isolated", transform=ax1.transAxes, fontsize=8, style="italic", color="0.4") + ax1.legend(loc="lower right", framealpha=0.9, fontsize=8) + ax1.set_title("(a) Isolation vs. topic count (rate 500)", fontsize=9) + + for tr in TRANSPORT_ORDER: + cells = d["rate"].get(tr, {}) + xs, ys, es = [], [], [] + for r in RATES: + c = cells.get(str(r)) + if c: + xs.append(r) + ys.append(c["null"][0]) + es.append(c["null"][1]) + ax2.errorbar( + range(len(xs)), ys, yerr=es, + fmt=TRANSPORT_MARKERS[tr] + "-", color=TRANSPORT_COLORS[tr], + label=TRANSPORT_LABELS[tr], markersize=7, capsize=3, linewidth=1.5, + markeredgecolor="white", markeredgewidth=0.7, zorder=3, + ) + ax2.set_xlabel("Offered rate (msg/s, 8 topics)") + ax2.set_ylabel("Null co-occurrence (density artifact)") + ax2.set_xticks(range(len(RATES))) + ax2.set_xticklabels([str(r) for r in RATES]) + ax2.set_ylim(0, 0.6) + ax2.set_title("(b) Density confound vs. rate", fontsize=9) + + fig.tight_layout() + save_figure(fig, output_dir, "fig_hol_excess") + + +if __name__ == "__main__": + script_dir = Path(__file__).resolve().parent + default_decomp = script_dir.parent.parent / "results-v5" / "02c_topic_rate_sweep" / "decomposition.json" + default_output = script_dir / "output" + decomp = Path(sys.argv[1]) if len(sys.argv) > 1 else default_decomp + output_dir = Path(sys.argv[2]) if len(sys.argv) > 2 else default_output + output_dir.mkdir(parents=True, exist_ok=True) + main(decomp, output_dir) diff --git a/publications/comnet/experiments/analysis/figures/fig_hol_tail_vs_loss.py b/publications/comnet/experiments/analysis/figures/fig_hol_tail_vs_loss.py new file mode 100644 index 00000000..16b317af --- /dev/null +++ b/publications/comnet/experiments/analysis/figures/fig_hol_tail_vs_loss.py @@ -0,0 +1,67 @@ +import glob +import json +import sys +from pathlib import Path + +import matplotlib.pyplot as plt +import numpy as np + +sys.path.insert(0, str(Path(__file__).parent)) +from style import ( + TRANSPORT_COLORS, + TRANSPORT_LABELS, + TRANSPORT_MARKERS, + TRANSPORT_ORDER, + apply_style, + save_figure, +) + +LOSSES = [0, 1, 2, 5] +BASE = Path(__file__).resolve().parent.parent.parent / "results-v5" / "02_hol_blocking" + + +def run_avgp99(transport, loss): + vals = [] + for f in glob.glob(str(BASE / f"{transport}_loss{loss}pct_run*.json")): + tp = json.load(open(f))["results"].get("topics", []) + p = [x.get("p99_us", 0) / 1000 for x in tp if x.get("p99_us")] + if p: + vals.append(sum(p) / len(p)) + return vals + + +def main(output_dir: Path): + apply_style() + fig, ax = plt.subplots(figsize=(7, 4.3)) + x = range(len(LOSSES)) + for tr in TRANSPORT_ORDER: + med, lo, hi = [], [], [] + for loss in LOSSES: + v = run_avgp99(tr, loss) + if v: + med.append(float(np.median(v))) + lo.append(float(np.percentile(v, 10))) + hi.append(float(np.percentile(v, 90))) + else: + med.append(np.nan) + lo.append(np.nan) + hi.append(np.nan) + ax.plot(x, med, TRANSPORT_MARKERS[tr] + "-", color=TRANSPORT_COLORS[tr], + label=TRANSPORT_LABELS[tr], markersize=7, linewidth=1.6, + markeredgecolor="white", markeredgewidth=0.7, zorder=3) + ax.fill_between(x, lo, hi, color=TRANSPORT_COLORS[tr], alpha=0.13, zorder=1) + + ax.set_xlabel("Packet loss rate") + ax.set_ylabel("Mean per-topic p99 latency (ms)") + ax.set_xticks(list(x)) + ax.set_xticklabels([f"{l}%" for l in LOSSES]) + ax.set_ylim(bottom=0) + ax.legend(loc="upper left", framealpha=0.9) + fig.tight_layout() + save_figure(fig, output_dir, "fig_hol_tail_vs_loss") + + +if __name__ == "__main__": + out = Path(sys.argv[1]) if len(sys.argv) > 1 else Path(__file__).resolve().parent / "output" + out.mkdir(parents=True, exist_ok=True) + main(out) diff --git a/publications/comnet/experiments/analysis/figures/fig_hol_taillatency.py b/publications/comnet/experiments/analysis/figures/fig_hol_taillatency.py new file mode 100644 index 00000000..00338c1e --- /dev/null +++ b/publications/comnet/experiments/analysis/figures/fig_hol_taillatency.py @@ -0,0 +1,78 @@ +import glob +import json +import statistics +import sys +from pathlib import Path + +import matplotlib.pyplot as plt + +sys.path.insert(0, str(Path(__file__).parent)) +from style import ( + TRANSPORT_COLORS, + TRANSPORT_LABELS, + TRANSPORT_MARKERS, + TRANSPORT_ORDER, + apply_style, + save_figure, +) + +TOPICS = [2, 4, 8, 16, 32] +LOSS = 5 +RESULTS = Path(__file__).resolve().parent.parent.parent / "results-v5" + + +def cell_files(transport, topics): + if topics == 8: + return glob.glob(str(RESULTS / "02_hol_blocking" / f"{transport}_loss{LOSS}pct_run*.json")) + return glob.glob(str(RESULTS / "02c_topic_rate_sweep" / f"{transport}_t{topics}_r500_loss{LOSS}pct_run*.json")) + + +def worst_best(transport, topics): + worst, best = [], [] + for f in cell_files(transport, topics): + tp = json.load(open(f))["results"].get("topics", []) + p99 = [x.get("p99_us", 0) / 1000 for x in tp if x.get("p99_us")] + if p99: + worst.append(max(p99)) + best.append(min(p99)) + if not worst: + return None + return statistics.median(worst), statistics.median(best) + + +def main(output_dir: Path): + apply_style() + fig, ax = plt.subplots(figsize=(7, 4.3)) + x = range(len(TOPICS)) + for tr in TRANSPORT_ORDER: + w, b = [], [] + for t in TOPICS: + r = worst_best(tr, t) + w.append(r[0] if r else None) + b.append(r[1] if r else None) + ax.plot( + x, w, TRANSPORT_MARKERS[tr] + "-", color=TRANSPORT_COLORS[tr], + label=TRANSPORT_LABELS[tr], markersize=7, linewidth=1.6, + markeredgecolor="white", markeredgewidth=0.7, zorder=3, + ) + if tr == "quic-pertopic": + ax.plot(x, b, TRANSPORT_MARKERS[tr] + "--", color=TRANSPORT_COLORS[tr], + markersize=6, linewidth=1.2, alpha=0.7, zorder=3) + ax.fill_between(x, b, w, color=TRANSPORT_COLORS[tr], alpha=0.12, zorder=1) + + ax.set_xlabel("Topics") + ax.set_ylabel("Per-topic p99 latency (ms)") + ax.set_xticks(list(x)) + ax.set_xticklabels([str(t) for t in TOPICS]) + ax.set_ylim(bottom=0) + ax.legend(loc="center left", framealpha=0.9) + ax.text(0.98, 0.05, "solid: worst topic | dashed: best (per-topic)", + transform=ax.transAxes, fontsize=7.5, style="italic", color="0.4", ha="right") + fig.tight_layout() + save_figure(fig, output_dir, "fig_hol_taillatency") + + +if __name__ == "__main__": + out = Path(sys.argv[1]) if len(sys.argv) > 1 else Path(__file__).resolve().parent / "output" + out.mkdir(parents=True, exist_ok=True) + main(out) diff --git a/publications/comnet/experiments/analysis/figures/fig_transport_limits.py b/publications/comnet/experiments/analysis/figures/fig_transport_limits.py new file mode 100644 index 00000000..d435759e --- /dev/null +++ b/publications/comnet/experiments/analysis/figures/fig_transport_limits.py @@ -0,0 +1,217 @@ +import csv +import glob +import json +import statistics +import sys +from pathlib import Path + +import matplotlib.pyplot as plt +import numpy as np +from scipy import stats as st + +sys.path.insert(0, str(Path(__file__).parent)) +from style import STRATEGY_COLORS, STRATEGY_LABELS, STRATEGY_MARKERS, apply_style, save_figure + +TOPIC_COUNTS = [1, 2, 4, 8, 16] +STREAM_LIMITS = [25, 100, 250, 1000] +STREAM_WINDOWS = [131_072, 262_144, 1_048_576] +BYTES_PER_MESSAGE = 312.8 + + +def publish_rates(base, label): + rates = [] + for path in glob.glob(str(base / f"{label}_run*_pub.json")): + try: + results = json.load(open(path))["results"] + except (json.JSONDecodeError, KeyError, OSError): + continue + if results.get("elapsed_secs") and results.get("published"): + rates.append(results["published"] / results["elapsed_secs"]) + return rates + + +def mean_ci(values): + if not values: + return None, None + mean = statistics.mean(values) + if len(values) < 2: + return mean, 0.0 + return mean, st.t.ppf(0.975, len(values) - 1) * st.sem(values) + + +def median_rtt(base, label): + samples = [] + for path in glob.glob(str(base / f"{label}_run*_broker_quic_*.csv")): + rows = list(csv.DictReader(open(path))) + if rows: + samples.append(statistics.median([int(r["rtt_us"]) for r in rows if r.get("rtt_us")])) + return statistics.median(samples) / 1e6 if samples else None + + +def series(base, labels): + means, errs = [], [] + for label in labels: + mean, err = mean_ci(publish_rates(base, label)) + means.append(np.nan if mean is None else mean / 1000.0) + errs.append(0.0 if err is None else err / 1000.0) + return means, errs + + +def topic_sweep(base, output_dir): + fig, ax = plt.subplots(1, 1, figsize=(4.5, 3.2)) + for strategy in ["control-only", "per-topic", "per-publish"]: + labels = [f"{strategy}_t{t}_sdef_wdef_d25_l2_tput" for t in TOPIC_COUNTS] + means, errs = series(base, labels) + ax.errorbar( + TOPIC_COUNTS, means, yerr=errs, + marker=STRATEGY_MARKERS[strategy], color=STRATEGY_COLORS[strategy], + label=STRATEGY_LABELS[strategy], linewidth=2, markersize=8, capsize=3, + ) + ax.set_xscale("log", base=2) + ax.set_xticks(TOPIC_COUNTS) + ax.set_xticklabels([str(t) for t in TOPIC_COUNTS]) + ax.set_xlabel("Topics on the connection") + ax.set_ylabel("Publish rate (K msg/s)") + ax.set_ylim(0, 62) + ax.set_yticks([0, 10, 20, 30, 40, 50, 60]) + ax.grid(True, axis="y", linewidth=0.4, alpha=0.4) + ax.legend(loc="center", bbox_to_anchor=(0.5, 0.30), fontsize=8) + fig.tight_layout() + save_figure(fig, output_dir, "fig_topic_sweep") + + +def limit_sweep(base, output_dir): + fig, (ax_credit, ax_window) = plt.subplots(2, 1, figsize=(4.5, 5.6)) + + labels = [f"per-publish_t8_s{limit}_wdef_d25_l2_tput" for limit in STREAM_LIMITS] + means, errs = series(base, labels) + rtt = median_rtt(base, labels[1]) or 0.0252 + model = [limit / rtt / 1000.0 for limit in STREAM_LIMITS] + ax_credit.plot(STREAM_LIMITS, model, linestyle="--", color="0.45", linewidth=1.5, + label="credit $\\div$ RTT") + ax_credit.errorbar(STREAM_LIMITS, means, yerr=errs, marker=STRATEGY_MARKERS["per-publish"], + color=STRATEGY_COLORS["per-publish"], linewidth=2, markersize=8, capsize=3, + label=STRATEGY_LABELS["per-publish"]) + for strategy in ["control-only", "per-topic"]: + flat, flat_errs = series(base, [f"{strategy}_t8_s{limit}_wdef_d25_l2_tput" for limit in [100, 1000]]) + ax_credit.errorbar([100, 1000], flat, yerr=flat_errs, marker=STRATEGY_MARKERS[strategy], + color=STRATEGY_COLORS[strategy], linewidth=2, markersize=8, capsize=3, + label=STRATEGY_LABELS[strategy]) + ax_credit.set_xscale("log") + ax_credit.set_yscale("log") + ax_credit.set_xticks(STREAM_LIMITS) + ax_credit.set_xticklabels([str(limit) for limit in STREAM_LIMITS]) + ax_credit.set_yticks([1, 3, 10, 30, 60]) + ax_credit.set_yticklabels(["1", "3", "10", "30", "60"]) + ax_credit.set_xlabel("(a) Concurrent stream limit") + ax_credit.set_ylabel("Publish rate (K msg/s)") + ax_credit.grid(True, axis="y", linewidth=0.4, alpha=0.4) + ax_credit.legend(loc="lower right", fontsize=7) + + window_series = [ + ("control-only", 8, STRATEGY_COLORS["control-only"], STRATEGY_MARKERS["control-only"], "-", "Control-only"), + ("per-topic", 1, STRATEGY_COLORS["per-topic"], STRATEGY_MARKERS["per-topic"], "--", "Per-topic, 1 topic"), + ("per-topic", 8, STRATEGY_COLORS["per-topic"], "o", "-", "Per-topic, 8 topics"), + ] + x_kb = [w / 1024 for w in STREAM_WINDOWS] + model_w = [w / rtt / BYTES_PER_MESSAGE / 1000.0 for w in STREAM_WINDOWS] + ax_window.plot(x_kb, model_w, linestyle=":", color="0.45", linewidth=1.5, + label="window $\\div$ RTT") + for strategy, topics, color, marker, linestyle, label in window_series: + labels = [f"{strategy}_t{topics}_sdef_w{w}_d25_l2_tput" for w in STREAM_WINDOWS] + means, errs = series(base, labels) + ax_window.errorbar(x_kb, means, yerr=errs, marker=marker, color=color, linestyle=linestyle, + linewidth=2, markersize=8, capsize=3, label=label) + ax_window.set_xscale("log", base=2) + ax_window.set_xticks(x_kb) + ax_window.set_xticklabels(["128", "256", "1024"]) + ax_window.set_ylim(0, 92) + ax_window.set_yticks([0, 20, 40, 60, 80]) + ax_window.set_xlabel("(b) Per-stream receive window (KB)") + ax_window.set_ylabel("Publish rate (K msg/s)") + ax_window.grid(True, axis="y", linewidth=0.4, alpha=0.4) + ax_window.legend(loc="upper left", fontsize=7) + + fig.tight_layout() + save_figure(fig, output_dir, "fig_transport_limits") + + +import re + +RATE_CEILING = 54_000.0 +DEFAULT_WINDOW = 262_144 +DEFAULT_STREAM_LIMIT = 100 + +CELL_PATTERN = re.compile( + r"(?P[a-z-]+)_t(?P\d+)_s(?Pdef|\d+)" + r"_w(?Pdef|\d+)_d(?P\d+)_l(?P\d+)_tput" +) + + +def predicted_rate(strategy, topics, stream_limit, window, rtt): + if strategy == "per-publish": + return min(stream_limit / rtt, RATE_CEILING) + streams = 1 if strategy == "control-only" else topics + return min(streams * window / rtt / BYTES_PER_MESSAGE, RATE_CEILING) + + +def model_collapse(base, output_dir): + points = {} + for pub in glob.glob(str(base / "*_tput_run1_pub.json")): + label = Path(pub).name.replace("_run1_pub.json", "") + match = CELL_PATTERN.match(label) + if not match: + continue + cell = match.groupdict() + rates = publish_rates(base, label) + rtt = median_rtt(base, label) + if not rates or not rtt: + continue + window = DEFAULT_WINDOW if cell["window"] == "def" else int(cell["window"]) + limit = DEFAULT_STREAM_LIMIT if cell["streams"] == "def" else int(cell["streams"]) + predicted = predicted_rate(cell["strategy"], int(cell["topics"]), limit, window, rtt) + points.setdefault(cell["strategy"], []).append((predicted / 1000.0, statistics.mean(rates) / 1000.0)) + + fig, ax = plt.subplots(1, 1, figsize=(4.5, 3.6)) + diagonal = np.array([0.7, 80.0]) + ax.fill_between(diagonal, diagonal * 0.9, diagonal * 1.1, color="0.85", zorder=0, + label="$\\pm$10%") + ax.plot(diagonal, diagonal, color="0.35", linewidth=1.2, zorder=1) + for strategy in ["control-only", "per-topic", "per-publish"]: + if strategy not in points: + continue + xs, ys = zip(*points[strategy]) + ax.scatter(xs, ys, s=46, marker=STRATEGY_MARKERS[strategy], color=STRATEGY_COLORS[strategy], + edgecolor="white", linewidth=0.6, zorder=3, label=STRATEGY_LABELS[strategy]) + ax.set_xscale("log") + ax.set_yscale("log") + ax.set_xlim(0.7, 80) + ax.set_ylim(0.7, 80) + ticks = [1, 3, 10, 30, 60] + ax.set_xticks(ticks); ax.set_xticklabels([str(t) for t in ticks]) + ax.set_yticks(ticks); ax.set_yticklabels([str(t) for t in ticks]) + ax.set_xlabel("Rate predicted by the binding limit (K msg/s)") + ax.set_ylabel("Measured rate (K msg/s)") + ax.grid(True, linewidth=0.4, alpha=0.4) + ax.legend(loc="upper left", fontsize=8) + fig.tight_layout() + save_figure(fig, output_dir, "fig_model_collapse") + + +def main(results_dir, output_dir): + apply_style() + base = Path(results_dir) / "04_transport_limits" + if not base.exists(): + print(f" WARNING: {base} not found") + return + topic_sweep(base, output_dir) + limit_sweep(base, output_dir) + model_collapse(base, output_dir) + + +if __name__ == "__main__": + script_dir = Path(__file__).resolve().parent + results = Path(sys.argv[1]) if len(sys.argv) > 1 else script_dir.parent.parent / "results-v5" + output = Path(sys.argv[2]) if len(sys.argv) > 2 else script_dir / "output" + output.mkdir(parents=True, exist_ok=True) + main(results, output) diff --git a/publications/comnet/experiments/analysis/figures/style.py b/publications/comnet/experiments/analysis/figures/style.py index 64a04bef..963d7c7d 100644 --- a/publications/comnet/experiments/analysis/figures/style.py +++ b/publications/comnet/experiments/analysis/figures/style.py @@ -4,10 +4,10 @@ import matplotlib.pyplot as plt TRANSPORT_COLORS = { - "tcp": "#1f77b4", - "quic-control": "#ff7f0e", - "quic-pertopic": "#2ca02c", - "quic-perpub": "#d62728", + "tcp": "#0072B2", + "quic-control": "#E69F00", + "quic-pertopic": "#009E73", + "quic-perpub": "#D55E00", } TRANSPORT_LABELS = { @@ -27,37 +27,40 @@ TRANSPORT_ORDER = ["tcp", "quic-control", "quic-pertopic", "quic-perpub"] CONN_TRANSPORT_ORDER = ["tcp", "tls", "quic"] -CONN_TRANSPORT_COLORS = {"tcp": "#1f77b4", "tls": "#9467bd", "quic": "#2ca02c"} -CONN_TRANSPORT_LABELS = {"tcp": "TCP", "tls": "TLS 1.3", "quic": "QUIC"} +CONN_TRANSPORT_COLORS = {"tcp": "#0072B2", "tls": "#762A83", "quic": "#009E73"} +CONN_TRANSPORT_LABELS = {"tcp": "TCP", "tls": "TCP+TLS 1.3", "quic": "QUIC"} CONN_TRANSPORT_MARKERS = {"tcp": "o", "tls": "P", "quic": "^"} -THROUGHPUT_ORDER = ["tcp", "quic-control-only", "quic-per-topic", "quic-per-publish"] +THROUGHPUT_ORDER = ["tcp", "tls", "quic-control-only", "quic-per-topic", "quic-per-publish"] THROUGHPUT_COLORS = { - "tcp": "#1f77b4", - "quic-control-only": "#ff7f0e", - "quic-per-topic": "#2ca02c", - "quic-per-publish": "#d62728", + "tcp": "#0072B2", + "tls": "#762A83", + "quic-control-only": "#E69F00", + "quic-per-topic": "#009E73", + "quic-per-publish": "#D55E00", } THROUGHPUT_LABELS = { "tcp": "TCP", + "tls": "TCP+TLS 1.3", "quic-control-only": "QUIC control", "quic-per-topic": "QUIC per-topic", "quic-per-publish": "QUIC per-publish", } THROUGHPUT_MARKERS = { "tcp": "o", + "tls": "v", "quic-control-only": "s", "quic-per-topic": "^", "quic-per-publish": "D", } DATAGRAM_ORDER = ["quic-stream", "quic-datagram"] -DATAGRAM_COLORS = {"quic-stream": "#2ca02c", "quic-datagram": "#e377c2"} +DATAGRAM_COLORS = {"quic-stream": "#009E73", "quic-datagram": "#CC79A7"} DATAGRAM_LABELS = {"quic-stream": "QUIC Stream", "quic-datagram": "QUIC Datagram"} DATAGRAM_MARKERS = {"quic-stream": "^", "quic-datagram": "X"} STRATEGY_ORDER = ["control-only", "per-publish", "per-topic"] -STRATEGY_COLORS = {"control-only": "#ff7f0e", "per-publish": "#d62728", "per-topic": "#2ca02c"} +STRATEGY_COLORS = {"control-only": "#E69F00", "per-publish": "#D55E00", "per-topic": "#009E73"} STRATEGY_LABELS = {"control-only": "Control-only", "per-publish": "Per-publish", "per-topic": "Per-topic"} STRATEGY_MARKERS = {"control-only": "s", "per-publish": "D", "per-topic": "^"} diff --git a/publications/comnet/experiments/analysis/hol_excess_decomposition.py b/publications/comnet/experiments/analysis/hol_excess_decomposition.py new file mode 100644 index 00000000..d4670979 --- /dev/null +++ b/publications/comnet/experiments/analysis/hol_excess_decomposition.py @@ -0,0 +1,210 @@ +import csv +import glob +import math +import random +from pathlib import Path +from typing import NamedTuple + +import numpy as np +from scipy import stats + + +class Decomp(NamedTuple): + n: int + obs: tuple[float, float] + null: tuple[float, float] + excess: tuple[float, float] + +SW = 50 +TH = 2.0 +COW = 10_000_000 +N_SHIFTS = 100 +NETEM_ONE_WAY_DELAY_US = 25_000 + +D02 = Path("results-v5/02_hol_blocking") +DSW = Path("results-v5/02c_topic_rate_sweep") + + +def load_topics(path, nt): + cols = [[] for _ in range(nt)] + with open(path) as f: + for row in csv.DictReader(f): + ti = int(row["topic_idx"]) + if 0 <= ti < nt: + cols[ti].append((int(row["receive_ns"]), float(row["latency_us"]))) + out = [] + for d in cols: + d.sort(key=lambda x: x[0]) + recv = np.array([r for r, _ in d], dtype=np.int64) + lat = np.array([l for _, l in d], dtype=np.float64) + out.append((recv, lat)) + return out + + +def spikes_of(topic_arrays): + per_topic = [] + for recv, lat in topic_arrays: + L = len(lat) + if L <= SW: + per_topic.append(np.empty(0, dtype=np.int64)) + continue + W = np.lib.stride_tricks.sliding_window_view(lat, SW) + med = np.partition(W, SW // 2, axis=1)[:, SW // 2] + idx = np.arange(SW, L) + m = med[idx - SW] + mask = (m > 0) & (lat[idx] > TH * m) + per_topic.append(recv[idx][mask]) + return per_topic + + +def co_ratio(times, topics): + n = len(times) + if n == 0: + return 0.0 + co = 0 + j0 = 0 + for i in range(n): + ti = times[i] + while times[j0] < ti - COW: + j0 += 1 + j = j0 + found = False + while j < n and times[j] <= ti + COW: + if j != i and topics[j] != topics[i]: + found = True + break + j += 1 + if found: + co += 1 + return co / n + + +def flatten(per_topic): + times = [] + topics = [] + for ti, arr in enumerate(per_topic): + for v in arr: + times.append(v) + topics.append(ti) + order = np.argsort(times, kind="stable") + return np.array(times)[order], np.array(topics)[order] + + +def observed(per_topic): + t, tp = flatten(per_topic) + return co_ratio(t, tp) + + +def null_ratio(per_topic, rng): + allt = [v for arr in per_topic for v in arr] + if not allt: + return 0.0 + lo, hi = min(allt), max(allt) + span = max(hi - lo, 1) + vals = [] + for _ in range(N_SHIFTS): + pt = [((arr - lo + rng.randrange(span)) % span).astype(np.int64) for arr in per_topic] + t, tp = flatten(pt) + vals.append(co_ratio(t, tp)) + return sum(vals) / len(vals) + + +def ci(vals): + n = len(vals) + if n < 2: + return (vals[0] if vals else 0.0), 0.0 + m = sum(vals) / n + sd = math.sqrt(sum((x - m) ** 2 for x in vals) / (n - 1)) + return m, float(stats.t.ppf(0.975, n - 1)) * sd / math.sqrt(n) + + +def cell_files(key, topics, rate, loss): + if topics == 8 and rate == 500: + return sorted(glob.glob(str(D02 / f"{key}_loss{loss}pct_run*_messages.csv"))) + return sorted(glob.glob(str(DSW / f"{key}_t{topics}_r{rate}_loss{loss}pct_run*_messages.csv"))) + + +def emulation_applied(topic_arrays): + latencies = np.concatenate([lat for _, lat in topic_arrays if len(lat)]) + return len(latencies) > 0 and float(np.median(latencies)) >= NETEM_ONE_WAY_DELAY_US + + +def decompose(key, topics, rate, loss): + files = cell_files(key, topics, rate, loss) + if not files: + return None + rng = random.Random(42) + obs, nul, exc = [], [], [] + kept = 0 + for fp in files: + pt = load_topics(fp, topics) + if not emulation_applied(pt): + print(f" excluded (median latency below the emulated delay, netem not applied): {fp}") + continue + kept += 1 + sp = spikes_of(pt) + o = observed(sp) + nv = null_ratio(sp, rng) + obs.append(o) + nul.append(nv) + exc.append(o - nv) + if kept == 0: + return None + return Decomp(n=kept, obs=ci(obs), null=ci(nul), excess=ci(exc)) + + +def main(): + import json + + transports = ["tcp", "quic-control", "quic-pertopic", "quic-perpub"] + dump: dict = {} + for loss in (1, 5): + dump[loss] = {"topics": {}, "rate": {}} + for key in transports: + dump[loss]["topics"][key] = {} + for t in (2, 4, 8, 16, 32): + r = decompose(key, t, 500, loss) + if r: + dump[loss]["topics"][key][t] = r._asdict() + dump[loss]["rate"][key] = {} + for rate in (125, 250, 500, 1000, 2000): + r = decompose(key, 8, rate, loss) + if r: + dump[loss]["rate"][key][rate] = r._asdict() + out = DSW / "decomposition.json" + with open(out, "w") as f: + json.dump(dump, f, indent=1) + print(f"wrote {out}") + + for loss in (1, 5): + print(f"\n########## LOSS {loss}% ##########") + print("=== AXIS 1: topics @ rate 500 [obs / null / excess (mean±95%CI over runs)] ===") + for key in transports: + print(f" {key}") + for t in (2, 4, 8, 16, 32): + r = decompose(key, t, 500, loss) + if not r: + print(f" t={t:2d}: (missing)") + continue + print( + f" t={t:2d} n={r.n:2d}: obs={r.obs[0]:.3f}±{r.obs[1]:.3f}" + f" null={r.null[0]:.3f}±{r.null[1]:.3f}" + f" excess={r.excess[0]:+.3f}±{r.excess[1]:.3f}" + ) + print("=== AXIS 2: rate @ 8 topics ===") + for key in transports: + print(f" {key}") + for rate in (125, 250, 500, 1000, 2000): + r = decompose(key, 8, rate, loss) + if not r: + print(f" r={rate:4d}: (missing)") + continue + print( + f" r={rate:4d} n={r.n:2d}: obs={r.obs[0]:.3f}±{r.obs[1]:.3f}" + f" null={r.null[0]:.3f}±{r.null[1]:.3f}" + f" excess={r.excess[0]:+.3f}±{r.excess[1]:.3f}" + ) + + +if __name__ == "__main__": + main() diff --git a/publications/comnet/experiments/analysis/offload_ablation.py b/publications/comnet/experiments/analysis/offload_ablation.py new file mode 100644 index 00000000..97cbc5c0 --- /dev/null +++ b/publications/comnet/experiments/analysis/offload_ablation.py @@ -0,0 +1,124 @@ +import csv +import glob +import json +import re +import statistics +import sys +from pathlib import Path + +ARMS = {"tls": "TCP+TLS 1.3", "quic-control": "QUIC control-only"} +LOSSES = [0, 1, 2, 5, 10] +OFFLOAD = ["on", "off"] + + +def rates(base, arm, offload, loss): + out = [] + for g in (1, 2, 3): + pat = base / f"03c_offload_ablation_g{g}" / f"{arm}_offload-{offload}_loss{loss}pct_run*.json" + for f in glob.glob(str(pat)): + if f.endswith("_pub.json"): + continue + try: + r = json.load(open(f))["results"] + except (json.JSONDecodeError, KeyError, OSError): + continue + if r.get("throughput_avg"): + out.append(r["throughput_avg"]) + return out + + +def broker_cpu(base, arm, offload, loss): + vals = [] + for g in (1, 2, 3): + pat = base / f"03c_offload_ablation_g{g}" / f"{arm}_offload-{offload}_loss{loss}pct_run*_broker_resources.csv" + for f in glob.glob(str(pat)): + rows = list(csv.DictReader(open(f))) + v = [float(r["cpu_percent"]) for r in rows if r.get("cpu_percent")] + if v: + vals.append(statistics.median(v)) + return statistics.median(vals) if vals else None + + +def qdisc_bpp(base, arm, offload, loss): + ratios = [] + for g in (1, 2, 3): + d = base / f"03c_offload_ablation_g{g}" + for after in glob.glob(str(d / f"{arm}_offload-{offload}_loss{loss}pct_run*_qdisc_after.txt")): + before = after.replace("_after.txt", "_before.txt") + try: + b = _sent(open(before).read()) + a = _sent(open(after).read()) + except OSError: + continue + if a and b and a[1] > b[1]: + dbytes, dpkts = a[0] - b[0], a[1] - b[1] + if dpkts > 0: + ratios.append(dbytes / dpkts) + return statistics.median(ratios) if ratios else None + + +def _sent(text): + m = re.search(r"Sent (\d+) bytes (\d+) pkt", text) + return (int(m.group(1)), int(m.group(2))) if m else None + + +def main(results_dir): + base = Path(results_dir) + print("Delivered throughput (K msg/s aggregate), n runs pooled over 3 groups\n") + header = f"{'arm':20s} {'offload':>7s} " + " ".join(f"{'loss' + str(l):>13s}" for l in LOSSES) + print(header) + table = {} + for arm in ARMS: + for off in OFFLOAD: + cells = [] + for loss in LOSSES: + r = rates(base, arm, off, loss) + table[(arm, off, loss)] = r + cells.append(f"{statistics.mean(r) / 1000:7.1f}(n={len(r)})" if r else f"{'-':>13s}") + print(f"{ARMS[arm]:20s} {off:>7s} " + " ".join(f"{c:>13s}" for c in cells)) + + print("\nFold-degradation ratio rate(0%) / rate(10%)") + folds = {} + for arm in ARMS: + for off in OFFLOAD: + r0, r10 = table[(arm, off, 0)], table[(arm, off, 10)] + if r0 and r10: + fold = statistics.mean(r0) / statistics.mean(r10) + folds[(arm, off)] = fold + print(f" {ARMS[arm]:20s} offload {off:3s}: {fold:6.1f}x") + + print("\nDiscriminator G = fold(TCP+TLS) / fold(QUIC-control)") + for off in OFFLOAD: + if ("tls", off) in folds and ("quic-control", off) in folds: + g = folds[("tls", off)] / folds[("quic-control", off)] + print(f" offload {off:3s}: G = {g:5.2f}") + if all(("tls", o) in folds and ("quic-control", o) in folds for o in OFFLOAD): + g_on = folds[("tls", "on")] / folds[("quic-control", "on")] + g_off = folds[("tls", "off")] / folds[("quic-control", "off")] + print(f"\n G(on)={g_on:.2f} G(off)={g_off:.2f}") + print(" G(off)~=G(on) and >~2 => QUIC advantage is a real transport property (artifact refuted)") + print(" G(off)->1 => advantage was a segmentation-offload artifact (confirmed)") + + print("\nOffload sanity: median bytes-per-packet through the broker netem qdisc (want OFF ~= MTU 1500)") + for arm in ARMS: + for off in OFFLOAD: + row = [] + for loss in LOSSES: + bpp = qdisc_bpp(base, arm, off, loss) + row.append(f"{bpp:6.0f}" if bpp else f"{'-':>6s}") + print(f" {ARMS[arm]:20s} {off:>3s}: " + " ".join(row)) + + print("\nBroker CPU (median % of 400); flag > 320 (80% of 4 cores) => CPU-bound, fold suspect") + for arm in ARMS: + for off in OFFLOAD: + row = [] + for loss in LOSSES: + c = broker_cpu(base, arm, off, loss) + mark = "*" if c and c > 320 else " " + row.append(f"{c:5.0f}{mark}" if c else f"{'-':>6s}") + print(f" {ARMS[arm]:20s} {off:>3s}: " + " ".join(row)) + + +if __name__ == "__main__": + default = Path(__file__).resolve().parent.parent / "results-v5" + main(sys.argv[1] if len(sys.argv) > 1 else default) diff --git a/publications/comnet/experiments/analysis/paced_offered_load.py b/publications/comnet/experiments/analysis/paced_offered_load.py new file mode 100644 index 00000000..638aadd2 --- /dev/null +++ b/publications/comnet/experiments/analysis/paced_offered_load.py @@ -0,0 +1,135 @@ +import csv +import json +import statistics +import sys +from pathlib import Path + +CONFIGS = ["tcp", "tls", "quic-main"] +RATES = [2500, 20000, 0] +LOSSES = [1, 10] +FANOUT = 8 +PACING_TOLERANCE = 0.02 +SUB_SATURATION_PREFIX = "P3: sub busy" + + +def load_json(path): + try: + return json.load(open(path)) + except (OSError, json.JSONDecodeError): + return None + + +def active_cpu(path): + try: + values = [float(row["cpu_percent"]) for row in csv.DictReader(open(path)) if row.get("cpu_percent")] + except (OSError, ValueError): + return None + if not values: + return None + return statistics.mean(value for value in values if value >= 0.5 * max(values)) + + +def acceptance(directory): + path = directory / "acceptance.csv" + if not path.exists(): + raise SystemExit(f"{path} missing: run single_segment_accept.py {directory} first") + return {row["label"]: row for row in csv.DictReader(open(path))} + + +def collect(base): + cells = {} + excluded = [] + for directory in sorted(base.glob("03g_paced_g*")): + accepted = acceptance(directory) + for row in {row["label"]: row for row in csv.DictReader(open(directory / "manifest.csv"))}.values(): + label = row["label"] + gate = accepted.get(label) + if gate is None: + excluded.append((label, "not evaluated by acceptance")) + continue + reasons = [reason for reason in gate["failures"].split("; ") if reason] + if any(not reason.startswith(SUB_SATURATION_PREFIX) for reason in reasons): + excluded.append((label, gate["failures"])) + continue + pub = load_json(directory / f"{label}_pub.json") + sub = load_json(directory / f"{label}.json") + if not pub or not sub: + excluded.append((label, "missing results")) + continue + target = int(row["rate"]) + configured = int(pub.get("config", {}).get("rate", 0) or 0) + offered = pub["results"].get("offered_rate") or 0.0 + if configured != target: + excluded.append((label, f"bench rate {configured} != planned {target}")) + continue + if target and abs(offered / target - 1) > PACING_TOLERANCE: + excluded.append((label, f"paced at {offered:.0f}/s, target {target}/s")) + continue + cells.setdefault((row["config"], int(row["loss"]), target), []).append({ + "group": directory.name[-2:], + "saturated": bool(reasons), + "offered": offered, + "unique": sub["results"]["throughput_avg"] / FANOUT, + "cpu": active_cpu(directory / f"{label}_broker_resources.csv"), + }) + return cells, excluded + + +def group_means(samples, field): + per_group = {} + for sample in samples: + if sample[field] is not None: + per_group.setdefault(sample["group"], []).append(sample[field]) + return {group: statistics.mean(values) for group, values in per_group.items()} + + +def describe(samples, field): + means = group_means(samples, field) + if not means: + return "-" + pooled = statistics.mean(s[field] for s in samples if s[field] is not None) + groups = " ".join(f"{g}:{v:.4g}" for g, v in sorted(means.items())) + return f"{pooled:.4g} ({groups})" + + +def ratio_by_group(cells, numerator, denominator): + left, right = group_means(cells.get(numerator, []), "unique"), group_means(cells.get(denominator, []), "unique") + shared = sorted(set(left) & set(right)) + return [(group, left[group] / right[group]) for group in shared] + + +def main(results_dir): + cells, excluded = collect(Path(results_dir)) + print(f"excluded runs: {len(excluded)}") + for label, reason in excluded: + print(f" {label}: {reason}") + for loss in LOSSES: + print(f"\n=== loss {loss}% (router arm). Values: pooled mean (per-group means)") + for config in CONFIGS: + print(f" {config}") + for rate in RATES: + samples = cells.get((config, loss, rate)) + if not samples: + continue + name = f"paced {rate}" if rate else "unpaced" + saturated = sum(s["saturated"] for s in samples) + tag = f" [{saturated} sub-saturated]" if saturated else "" + print(f" {name:12s} n={len(samples)}{tag}") + for field in ("offered", "unique", "cpu"): + print(f" {field:8s} {describe(samples, field)}") + paced = ratio_by_group(cells, (config, loss, 20000), (config, loss, 0)) + if paced: + values = [v for _, v in paced] + print(f" paced 20000 / unpaced delivered: {statistics.mean(values):.3f} (range {min(values):.3f}-{max(values):.3f}, {len(values)} groups)") + for rate in RATES: + for other in ("tls", "tcp"): + rows = ratio_by_group(cells, ("quic-main", loss, rate), (other, loss, rate)) + if rows: + values = [v for _, v in rows] + name = f"paced {rate}" if rate else "unpaced" + print(f" QUIC/{other} {name:12s} {statistics.mean(values):.3f} (range {min(values):.3f}-{max(values):.3f}, {len(values)} groups)") + + +if __name__ == "__main__": + default = Path(__file__).resolve().parent.parent / "results-v5" + main(sys.argv[1] if len(sys.argv) > 1 else default) diff --git a/publications/comnet/experiments/analysis/single_segment_accept.py b/publications/comnet/experiments/analysis/single_segment_accept.py new file mode 100644 index 00000000..2d916016 --- /dev/null +++ b/publications/comnet/experiments/analysis/single_segment_accept.py @@ -0,0 +1,464 @@ +import csv +import glob +import json +import math +import re +import statistics +import sys +from pathlib import Path + +WIRE_FRAME_MAX = 1474 +ACCOUNTING_TOLERANCE = 0.001 +Z_999 = 3.29 +NETEM_LIMIT = 100_000 +BACKLOG_CEILING = NETEM_LIMIT // 10 +ROUTER_CORE_P95_MAX = 60.0 +CLIENT_BUSY_P95_MAX = 95.0 +CLEAN_RETRANS_MAX = 0.0005 +CLEAN_RTO_MAX = 0.00001 +MAIN_CAMPAIGN_PREFIX = {"tcp": "tcp", "tls": "tls", "quic-main": "quic-control-only"} +MAIN_CAMPAIGN_GROUP = {"tcp": 2, "tls": 1, "quic-main": 2} +LIMITED_MODES = ("single", "legacy", "tbf") +LEGACY_REFERENCE_03C = {"tls": 0.0895, "quic-main": 0.0768} + + +def sections(path): + out = {} + current = None + try: + text = Path(path).read_text() + except OSError: + return out + for line in text.splitlines(): + match = re.match(r"^=== (.+) ===$", line) + if match: + current = match.group(1) + out[current] = [] + elif current: + out[current].append(line) + return out + + +def qdisc_counters(lines, handle): + for index, line in enumerate(lines): + if re.match(rf"^qdisc \S+ {re.escape(handle)} ", line): + block = " ".join(lines[index + 1:index + 4]) + sent = re.search(r"Sent (\d+) bytes (\d+) pkt \(dropped (\d+)", block) + backlog = re.search(r"backlog \S+ (\d+)p", block) + if sent: + return { + "bytes": int(sent.group(1)), "pkts": int(sent.group(2)), "dropped": int(sent.group(3)), + "backlog": int(backlog.group(1)) if backlog else 0, + } + return None + + +def table_value(lines, table, field): + rows = [line.split() for line in lines if line.startswith(f"{table}:")] + for header, values in zip(rows[0::2], rows[1::2]): + if field in header: + return int(values[header.index(field)]) + return None + + +def table_delta(before, after, section, table, field): + start = table_value(before.get(section, []), table, field) + end = table_value(after.get(section, []), table, field) + if start is None or end is None: + return None + return end - start + + +def dev_counters(lines, iface="ens4"): + for line in lines: + name, _, rest = line.partition(":") + if name.strip() == iface: + fields = rest.split() + return {"rx_pkts": int(fields[1]), "rx_drop": int(fields[3]), "tx_pkts": int(fields[9]), "tx_drop": int(fields[11])} + return None + + +def softnet_column(lines, column): + total = 0 + for line in lines: + fields = line.split() + if len(fields) > column: + total += int(fields[column], 16) + return total + + +def ethtool_drops(lines): + matched = [int(value) for name, _, value in (line.partition(":") for line in lines) + if re.search(r"drop|discard|err", name, re.I) and value.strip().isdigit()] + return sum(matched) if matched else None + + +def single_number(lines): + for line in lines: + if line.strip().isdigit(): + return int(line.strip()) + return None + + +def probe(path): + try: + text = Path(path).read_text() + except OSError: + return None + enq = re.search(r"^@enq: (\d+)", text, re.M) + if not enq: + return None + values = {"enq": int(enq.group(1))} + for key in ("gso", "maxlen"): + match = re.search(rf"^@{key}: (\d+)", text, re.M) + values[key] = int(match.group(1)) if match else 0 + return values + + +def csv_rows(path): + try: + return list(csv.DictReader(open(path))) + except OSError: + return [] + + +def p95(values): + ordered = sorted(values) + return ordered[min(len(ordered) - 1, int(math.ceil(0.95 * len(ordered))) - 1)] + + +def active_cpu(rows): + values = [] + for row in rows: + try: + values.append(float(row["cpu_percent"])) + except (KeyError, TypeError, ValueError): + continue + if not values: + return None + return statistics.mean(value for value in values if value >= 0.5 * max(values)) + + +def throughput(path): + try: + results = json.load(open(path))["results"] + return results.get("throughput_avg", results.get("measured_rate")) + except (json.JSONDecodeError, KeyError, OSError, TypeError): + return None + + +def binomial_bound(loss, skbs): + return Z_999 * math.sqrt(loss * (1 - loss) / skbs) if skbs and loss > 0 else 0.0 + + +def monitor_integrity(base, failures): + for name in ("broker", "pub", "sub"): + rows = csv_rows(f"{base}_{name}_resources.csv") + stamps = [row["timestamp"] for row in rows] + if not rows: + failures.append(f"P5: {name} monitor empty") + elif abs(len(stamps) - len(set(stamps))) > 1: + failures.append(f"P5: {name} monitor has {len(stamps)} rows for {len(set(stamps))} timestamps") + router = csv_rows(f"{base}_router_resources.csv") + if not router: + failures.append("P5: router monitor empty") + else: + cores = len({row["cpu"] for row in router}) + stamps = len({row["timestamp"] for row in router}) + if abs(len(router) - cores * stamps) > cores: + failures.append(f"P5: router monitor has {len(router)} rows for {stamps}x{cores}") + + +def headroom(base, result, failures): + router = csv_rows(f"{base}_router_resources.csv") + if router: + per_core = {} + for row in router: + per_core.setdefault(row["cpu"], []).append(float(row["sys_pct"]) + float(row["softirq_pct"])) + worst = max(p95(values) for values in per_core.values()) + result["router_core_p95"] = worst + if worst >= ROUTER_CORE_P95_MAX: + failures.append(f"P3: router core sys+softirq p95 {worst:.1f}%") + for name in ("pub", "sub"): + busy = [float(row["cpu_busy"]) for row in csv_rows(f"{base}_{name}_resources.csv") if row.get("cpu_busy")] + if busy: + busy_p95 = p95(busy) + result[f"{name}_busy_p95"] = busy_p95 + if busy_p95 >= CLIENT_BUSY_P95_MAX: + failures.append(f"P3: {name} busy p95 {busy_p95:.1f}%") + broker_rows = csv_rows(f"{base}_broker_resources.csv") + load = active_cpu(broker_rows) + if load is not None: + result["broker_cpu_mean"] = round(load, 1) + for column in ("host_softirq", "host_steal"): + values = [float(row[column]) for row in broker_rows if row.get(column)] + if values: + result[f"broker_{column}_p95"] = p95(values) + + +def router_gates(base, loss, entry, result, failures): + guarded = entry.get("router_guard") == "1" + before = sections(f"{base}_router_before.txt") + after = sections(f"{base}_router_after.txt") + rb = qdisc_counters(before.get("tc", []), "10:") + ra = qdisc_counters(after.get("tc", []), "10:") + if not (rb and ra): + failures.append("router netem counters missing") + return + sent, dropped = ra["pkts"] - rb["pkts"], ra["dropped"] - rb["dropped"] + skbs = sent + dropped + fraction = dropped / skbs if skbs else 0.0 + bound = binomial_bound(loss, skbs) + result.update(skbs=skbs, drop_fraction=round(fraction, 6), bound=round(bound, 6), + bytes_per_pkt=round((ra["bytes"] - rb["bytes"]) / sent, 1) if sent else None) + if not skbs: + failures.append("a: no broker traffic crossed the router netem") + if loss > 0 and abs(fraction - loss) > bound: + failures.append(f"a: drop fraction {fraction:.5f} outside {loss}±{bound:.5f}") + if loss == 0 and dropped: + failures.append(f"a: {dropped} router drops at 0% loss") + if sent and (ra["bytes"] - rb["bytes"]) / sent > WIRE_FRAME_MAX: + failures.append("d: bytes per packet above one wire frame") + + probed = probe(f"{base}_router_probe.txt") + if entry.get("router_probe") == "1" and probed is None: + failures.append("b/c: router probe output missing") + if probed: + expected = skbs + (ra["backlog"] - rb["backlog"]) + result.update(probe_enq=probed["enq"], probe_gso=probed["gso"], probe_maxlen=probed["maxlen"]) + if expected and abs(probed["enq"] - expected) / expected > ACCOUNTING_TOLERANCE: + failures.append(f"b: probe enq {probed['enq']} vs counters {expected}") + if probed["gso"] or probed["maxlen"] > WIRE_FRAME_MAX: + failures.append(f"c: gso skbs {probed['gso']}, max len {probed['maxlen']}") + + ingress_before, ingress_after = single_number(before.get("ingress", [])), single_number(after.get("ingress", [])) + if ingress_before is None or ingress_after is None: + failures.append("g: router ingress counter missing") + elif skbs: + ingress = ingress_after - ingress_before + result["ingress_over_skbs"] = round(ingress / skbs, 5) + if not guarded and abs(ingress - skbs) / skbs > ACCOUNTING_TOLERANCE: + failures.append(f"g: router ingress {ingress} vs netem skbs {skbs}") + + dev_b, dev_a = dev_counters(before.get("dev", [])), dev_counters(after.get("dev", [])) + if not (dev_b and dev_a): + failures.append("f: router interface counters missing") + elif dev_a["rx_drop"] - dev_b["rx_drop"] or dev_a["tx_drop"] - dev_b["tx_drop"]: + failures.append("f: router interface drops") + if not before.get("softnet") or not after.get("softnet"): + failures.append("f: router softnet counters missing") + else: + dropped_softnet = softnet_column(after["softnet"], 1) - softnet_column(before["softnet"], 1) + result["router_softnet_squeezed"] = softnet_column(after["softnet"], 2) - softnet_column(before["softnet"], 2) + if dropped_softnet: + failures.append(f"f: router softnet drops {dropped_softnet}") + nic_before, nic_after = ethtool_drops(before.get("ethtool", [])), ethtool_drops(after.get("ethtool", [])) + if nic_before is None or nic_after is None: + failures.append("f: router NIC drop counters missing") + elif nic_after - nic_before: + failures.append(f"f: router NIC drop/error counters +{nic_after - nic_before}") + for section, table, field in (("snmp", "Ip", "InDiscards"), ("snmp", "Ip", "OutDiscards"), + ("netstat", "IpExt", "InNoRoutes"), ("snmp", "Ip", "OutNoRoutes")): + change = table_delta(before, after, section, table, field) + if change is None: + failures.append(f"f: router {table} {field} missing") + elif change: + failures.append(f"f: router {table} {field} +{change}") + + +def broker_gates(base, loss, entry, result, failures): + before = sections(f"{base}_broker_before.txt") + after = sections(f"{base}_broker_after.txt") + bb = qdisc_counters(before.get("tc", []), "10:") + ba = qdisc_counters(after.get("tc", []), "10:") + if not (bb and ba): + if entry["mode"] != "router": + failures.append("e: broker netem counters missing") + else: + sent, dropped = ba["pkts"] - bb["pkts"], ba["dropped"] - bb["dropped"] + result["broker_netem_dropped"] = dropped + if entry["mode"] == "single" and dropped: + failures.append(f"e: broker delay qdisc dropped {dropped}") + if entry["mode"] in ("legacy", "tbf") and sent + dropped: + result["broker_drop_fraction"] = round(dropped / (sent + dropped), 6) + if entry["mode"] in LIMITED_MODES: + backlog = [int(row["backlog_pkts"]) for row in csv_rows(f"{base}_broker_backlog.csv") if row.get("backlog_pkts")] + if not backlog: + failures.append("e: broker backlog samples missing") + else: + result["broker_backlog_max"] = max(backlog) + if max(backlog) >= BACKLOG_CEILING: + failures.append(f"e: broker netem backlog reached {max(backlog)} packets") + + retrans = table_delta(before, after, "snmp", "Tcp", "RetransSegs") + segments = table_delta(before, after, "snmp", "Tcp", "OutSegs") + if entry["config"] in ("tcp", "tls"): + if retrans is None or not segments: + failures.append("P4: broker TCP counters missing") + else: + ratio = retrans / segments + lost = table_delta(before, after, "netstat", "TcpExt", "TCPLostRetransmit") + timeouts = table_delta(before, after, "netstat", "TcpExt", "TCPTimeouts") + result["tcp_retrans_ratio"] = round(ratio, 6) + result["tcp_timeouts"] = timeouts + result["tcp_lost_retransmit"] = lost + if lost is None or timeouts is None: + failures.append("P4: broker TcpExt counters missing") + if loss == 0 and ratio > CLEAN_RETRANS_MAX: + failures.append(f"P4: TCP retransmits {ratio:.5f} of segments on a clean path") + if loss == 0 and lost is not None and lost / segments > CLEAN_RTO_MAX: + failures.append(f"P4: {lost} lost retransmissions on a clean path") + if loss == 0 and timeouts is not None and timeouts / segments > CLEAN_RTO_MAX: + failures.append(f"P4: {timeouts} retransmission timeouts on a clean path") + + if entry["phase"] == "accuracy" and entry["mode"] == "legacy": + probed = probe(f"{base}_broker_probe.txt") + fraction = result.get("broker_drop_fraction") + clustered = fraction is not None and bb and ba and \ + loss - fraction > binomial_bound(loss, (ba["pkts"] - bb["pkts"]) + (ba["dropped"] - bb["dropped"])) + result["positive_control_gso"] = probed["gso"] if probed else None + result["positive_control_reference_03c"] = LEGACY_REFERENCE_03C.get(entry["config"]) + if probed and bb and ba: + counted = (ba["pkts"] - bb["pkts"]) + (ba["dropped"] - bb["dropped"]) + (ba["backlog"] - bb["backlog"]) + result["positive_control_skbs_over_counted"] = round(probed["enq"] / counted, 4) if counted else None + result["positive_control_detected"] = bool(probed and probed["gso"] > 0 and clustered) + + +def evaluate(directory, entry): + label = entry["label"] + base = directory / label + loss = float(entry["loss"]) / 100.0 + result = {"label": label, "phase": entry["phase"], "mode": entry["mode"], "config": entry["config"], "loss": entry["loss"]} + failures = [] + + value = throughput(directory / f"{label}.json") + result["throughput"] = value + if not value: + failures.append("throughput missing or zero") + for key in ("ilb_health", "ilb_health_after"): + health = entry.get(key, "") + if not health or any(state != "HEALTHY" for state in health.split()): + failures.append(f"{key} '{health}'") + base_delay = 25_000 if entry.get("workload") == "hol" else 10_000 + if int(entry.get("delay_us", 0)) + int(entry.get("router_hop_us", 0)) != base_delay: + failures.append(f"delay {entry.get('delay_us')}us + hop {entry.get('router_hop_us')}us != {base_delay}us") + if entry["phase"] == "calib-direct" and entry.get("router_hop_us") != "0": + failures.append("calib-direct run compensated for a router hop that is not in the path") + expected_path = "off" if entry["phase"] == "calib-direct" else "on" + if entry.get("path_state") != expected_path: + failures.append(f"router path {entry.get('path_state')} (expected {expected_path})") + + if entry["phase"] != "calib-direct": + router_gates(base, loss if entry["mode"] in ("single", "router") else 0.0, entry, result, failures) + broker_gates(base, loss, entry, result, failures) + headroom(base, result, failures) + monitor_integrity(base, failures) + + result["pass"] = not failures + result["failures"] = "; ".join(failures) + return result + + +def manifest_entries(directory): + return list({row["label"]: row for row in csv.DictReader(open(directory / "manifest.csv"))}.values()) + + +def main_campaign_cpu(main_dir, prefix): + loads = [active_cpu(csv_rows(path)) for path in glob.glob(str(main_dir / f"{prefix}_qos0_loss0pct_run*_broker_resources.csv"))] + loads = [load for load in loads if load is not None] + return statistics.mean(loads) if loads else None + + +def calibration_report(directory, results, group): + main_dir = directory.parent / "03_throughput_under_loss" + failures = 0 + if not any(result["phase"] in ("calib", "calib-direct") for result in results): + return failures + print(" P1 calibration at 0% loss (X = max(5%, 2*CV of the main-campaign cell))") + for config, prefix in MAIN_CAMPAIGN_PREFIX.items(): + main_values = [v for v in (throughput(p) for p in glob.glob(str(main_dir / f"{prefix}_qos0_loss0pct_run*.json"))) if v] + arms = {} + loads = {} + for result in results: + if result["config"] == config and result["loss"] == "0" and result["phase"] in ("calib", "calib-direct") and result["throughput"]: + arms.setdefault((result["phase"], result["mode"]), []).append(result["throughput"]) + if "broker_cpu_mean" in result: + loads.setdefault((result["phase"], result["mode"]), []).append(result["broker_cpu_mean"]) + if not main_values: + print(f" {config:10s} FAIL: no main-campaign 0% data") + failures += 1 + continue + if not arms: + print(f" {config:10s} FAIL: no calibration runs") + failures += 1 + continue + main_mean = statistics.mean(main_values) + tolerance = max(0.05, 2 * statistics.stdev(main_values) / main_mean) + legacy = arms.get(("calib", "legacy-calib")) + via_router = arms.get(("calib", "single")) + direct = arms.get(("calib-direct", "single")) + matched = MAIN_CAMPAIGN_GROUP[config] == group + if legacy: + drift = statistics.mean(legacy) / main_mean - 1 + passed = abs(drift) <= tolerance + if matched and not passed: + failures += 1 + print(f" {config:10s} (i) legacy vs main campaign g{MAIN_CAMPAIGN_GROUP[config]}: {drift:+.1%} " + f"(tol {tolerance:.1%}) {'PASS' if passed else 'FAIL'}{'' if matched else ' [cross-group, not gated]'}") + main_load = main_campaign_cpu(main_dir, prefix) + legacy_load = loads.get(("calib", "legacy-calib")) + if not (main_load and legacy_load): + print(f" {config:10s} (i) broker CPU comparison missing data {'FAIL' if matched else '[cross-group, not gated]'}") + failures += 1 if matched else 0 + else: + cpu_drift = statistics.mean(legacy_load) / main_load - 1 + cpu_passed = abs(cpu_drift) <= 0.05 + if matched and not cpu_passed: + failures += 1 + print(f" {config:10s} (i) broker CPU vs main campaign: {cpu_drift:+.1%} (tol 5.0%) " + f"{'PASS' if cpu_passed else 'FAIL'}{'' if matched else ' [cross-group, not gated]'}") + if not (via_router and direct): + print(f" {config:10s} (ii)/(iii) FAIL: missing {'router' if not via_router else 'direct'} calibration arm") + failures += 1 + else: + gap = statistics.mean(via_router) / statistics.mean(direct) - 1 + passed = abs(gap) <= tolerance + failures += 0 if passed else 1 + print(f" {config:10s} (ii) direct vs (iii) router: {gap:+.1%} (tol {tolerance:.1%}) {'PASS' if passed else 'FAIL'}") + if legacy and via_router: + print(f" {config:10s} tail-drop fix effect (iii)/(i): {statistics.mean(via_router) / statistics.mean(legacy) - 1:+.1%}") + return failures + + +def main(directory): + directory = Path(directory) + if not (directory / "manifest.csv").exists(): + print(f"missing {directory / 'manifest.csv'}") + return 1 + match = re.search(r"_g(\d+)$", directory.name) + if not match: + print(f"cannot read group from {directory.name}") + return 1 + group = int(match.group(1)) + results = [evaluate(directory, entry) for entry in manifest_entries(directory)] + fields = sorted({key for result in results for key in result}, key=lambda k: (k != "label", k)) + out = directory / "acceptance.csv" + with open(out, "w", newline="") as handle: + writer = csv.DictWriter(handle, fieldnames=fields, lineterminator="\n") + writer.writeheader() + writer.writerows(results) + failed = [r for r in results if not r["pass"]] + print(f"{directory.name}: {len(results)} runs, {len(failed)} failing gates -> {out}") + for result in failed: + print(f" FAIL {result['label']}: {result['failures']}") + for control in (r for r in results if "positive_control_detected" in r): + print(f" positive control {control['label']}: detected={control['positive_control_detected']} " + f"gso={control['positive_control_gso']} drop_fraction={control.get('broker_drop_fraction')} " + f"(03c reference {control.get('positive_control_reference_03c')}) " + f"probe skbs / counter segments={control.get('positive_control_skbs_over_counted')}") + return len(failed) + calibration_report(directory, results, group) + + +if __name__ == "__main__": + targets = sys.argv[1:] or sorted(glob.glob(str(Path(__file__).resolve().parent.parent / "results-v5" / "03e_single_segment_g*"))) + sys.exit(1 if sum(main(target) for target in targets) else 0) diff --git a/publications/comnet/experiments/analysis/single_segment_fold.py b/publications/comnet/experiments/analysis/single_segment_fold.py new file mode 100644 index 00000000..27d75d70 --- /dev/null +++ b/publications/comnet/experiments/analysis/single_segment_fold.py @@ -0,0 +1,319 @@ +import csv +import glob +import json +import math +import random +import statistics +import sys +from pathlib import Path + +LOSSES = [0, 1, 2, 5, 10] +QUIC_CONFIGS = ["quic-main", "quic-main-ppub", "quic-main-ptopic", "quic-ctl", "quic-ppub"] +BOOTSTRAP = 5000 +MAIN_CAMPAIGN_FILES = { + "tcp": "tcp", "tls": "tls", "quic-main": "quic-control-only", + "quic-main-ppub": "quic-per-publish", "quic-main-ptopic": "quic-per-topic", +} +TCP_MSS = 1408 +PAYLOAD_BYTES = 256 +RTT_S = 0.0101 +SOFTIRQ = {} +CPU_BOUND = 380.0 + + +def throughput(path): + try: + return json.load(open(path))["results"]["throughput_avg"] + except (json.JSONDecodeError, KeyError, OSError, TypeError): + return None + + +def broker_cpu(path): + try: + values = [float(row["cpu_percent"]) for row in csv.DictReader(open(path)) if row.get("cpu_percent")] + except (OSError, ValueError, KeyError): + return None + if not values: + return None + active = [value for value in values if value >= 0.5 * max(values)] + return statistics.mean(active) + + +def broker_softirq(path): + try: + rows = [row for row in csv.DictReader(open(path)) if row.get("cpu_percent") and row.get("host_softirq")] + peak = max(float(row["cpu_percent"]) for row in rows) if rows else 0 + values = [float(row["host_softirq"]) for row in rows if float(row["cpu_percent"]) >= 0.5 * peak] + except (OSError, ValueError, KeyError): + return None + return statistics.mean(values) if values else None + + +def failing_labels(directory): + path = directory / "acceptance.csv" + if not path.exists(): + raise SystemExit(f"{path} missing: run single_segment_accept.py first") + return {row["label"] for row in csv.DictReader(open(path)) if row["pass"] != "True"} + + +def per_buffer_drop_fractions(directories): + fractions = {} + for directory in directories: + for row in csv.DictReader(open(directory / "acceptance.csv")): + if row["phase"] == "main" and row["mode"] == "legacy" and row["pass"] == "True" and row.get("broker_drop_fraction"): + fractions.setdefault((row["config"], int(row["loss"])), []).append(float(row["broker_drop_fraction"])) + return {key: statistics.mean(values) for key, values in fractions.items()} + + +def clustering_prediction(directories, low=1, high=10): + fractions = per_buffer_drop_fractions(directories) + keys = [("tls", low), ("tls", high), ("quic-main", low), ("quic-main", high)] + if any(key not in fractions or fractions[key] <= 0 for key in keys): + return None + tls_span = fractions[("tls", high)] / fractions[("tls", low)] + quic_span = fractions[("quic-main", high)] / fractions[("quic-main", low)] + return 1 / math.sqrt(tls_span / quic_span) + + +def load_rerun(directories): + cells = {} + cpu = {} + excluded = [] + for directory in directories: + manifest = directory / "manifest.csv" + if not manifest.exists(): + continue + failing = failing_labels(directory) + entries = {row["label"]: row for row in csv.DictReader(open(manifest))}.values() + for entry in entries: + if entry["phase"] not in ("main", "crosscheck") or entry.get("broker_probe") == "1": + continue + value = throughput(directory / f"{entry['label']}.json") + if entry["label"] in failing or not value or value <= 0: + excluded.append(entry["label"]) + continue + key = (entry["mode"], entry["config"], int(entry["loss"])) + cells.setdefault(key, []).append((directory.name, value)) + load = broker_cpu(directory / f"{entry['label']}_broker_resources.csv") + if load is not None: + cpu.setdefault(key, []).append(load) + softirq = broker_softirq(directory / f"{entry['label']}_broker_resources.csv") + if softirq is not None: + SOFTIRQ.setdefault(key, []).append(softirq) + return cells, cpu, excluded + + +def load_main_campaign(base): + cells = {} + cpu = {} + for config, prefix in MAIN_CAMPAIGN_FILES.items(): + for loss in LOSSES: + key = ("main-campaign", config, loss) + for path in glob.glob(str(base / f"{prefix}_qos0_loss{loss}pct_run*.json")): + value = throughput(path) + if value and value > 0: + cells.setdefault(key, []).append(("main", value)) + load = broker_cpu(path.replace(".json", "_broker_resources.csv")) + if load is not None: + cpu.setdefault(key, []).append(load) + return cells, cpu + + +def resample(samples, rng): + by_group = {} + for group, value in samples: + by_group.setdefault(group, []).append(value) + drawn = [rng.choice(values) for values in by_group.values() for _ in values] + return statistics.mean(drawn) + + +def interval(statistic, rng): + draws = sorted(statistic(rng) for _ in range(BOOTSTRAP)) + return draws[int(0.025 * BOOTSTRAP)], draws[int(0.975 * BOOTSTRAP) - 1] + + +def mean_of(samples): + return statistics.mean(value for _, value in samples) + + +def fold_ratio(cells, mode, reference, quic, low, high): + needed = [(mode, reference, low), (mode, reference, high), (mode, quic, low), (mode, quic, high)] + if not all(key in cells for key in needed): + return None + + def compute(draw): + ref_low, ref_high, quic_low, quic_high = (draw(cells[key]) for key in needed) + return (ref_low / ref_high) / (quic_low / quic_high) + + rng = random.Random(f"{mode}-{reference}-{quic}-{low}-{high}") + return compute(mean_of), interval(lambda r: compute(lambda s: resample(s, r)), rng) + + +def same_loss_ratio(cells, mode, reference, quic, loss): + if (mode, reference, loss) not in cells or (mode, quic, loss) not in cells: + return None + rng = random.Random(f"R-{mode}-{reference}-{quic}-{loss}") + point = mean_of(cells[(mode, quic, loss)]) / mean_of(cells[(mode, reference, loss)]) + ci = interval(lambda r: resample(cells[(mode, quic, loss)], r) / resample(cells[(mode, reference, loss)], r), rng) + return point, ci + + +def loss_slope(cells, mode, config): + points = [(math.log(loss / 100), math.log(mean_of(cells[(mode, config, loss)]))) + for loss in (1, 2, 5, 10) if (mode, config, loss) in cells] + if len(points) < 3: + return None + mean_x = statistics.mean(x for x, _ in points) + mean_y = statistics.mean(y for _, y in points) + return -sum((x - mean_x) * (y - mean_y) for x, y in points) / sum((x - mean_x) ** 2 for x, _ in points) + + +def verdict(ci, main_value): + low, high = ci + if high < 1: + return "reversed" + if low <= 1: + return "not supported: headline gap was an artifact" + if low <= main_value <= high: + return "robust: gap and magnitude hold" + return "direction robust, magnitude was a netem artifact" + + +def cross_ratio(cells, numerator, denominator): + if numerator not in cells or denominator not in cells: + return None + rng = random.Random(f"X-{numerator}-{denominator}") + point = mean_of(cells[numerator]) / mean_of(cells[denominator]) + return point, interval(lambda r: resample(cells[numerator], r) / resample(cells[denominator], r), rng) + + +def g_shift(cells, arm): + keys = [(mode, config, loss) for mode in (arm, "main-campaign") for config in ("tls", "quic-main") for loss in (1, 10)] + if not all(key in cells for key in keys): + return None + + def compute(draw): + def g(mode): + return (draw(cells[(mode, "tls", 1)]) / draw(cells[(mode, "tls", 10)])) / \ + (draw(cells[(mode, "quic-main", 1)]) / draw(cells[(mode, "quic-main", 10)])) + return g(arm) / g("main-campaign") + + rng = random.Random(f"g-shift-{arm}") + return compute(mean_of), interval(lambda r: compute(lambda samples: resample(samples, r)), rng) + + +def cpu_label(cpu, key): + if key not in cpu: + return "-" + load = statistics.mean(cpu[key]) + return f"{load:.0f}{'*' if load >= CPU_BOUND else ''}" + + +def fmt(result): + if not result: + return "n/a" + point, (low, high) = result + return f"{point:6.2f} [{low:5.2f}, {high:5.2f}]" + + +def main(results_dir): + base = Path(results_dir) + directories = sorted(base.glob("03e_single_segment_g*")) + cells, cpu, excluded = load_rerun(directories) + main_cells, main_cpu = load_main_campaign(base / "03_throughput_under_loss") + cells.update(main_cells) + cpu.update(main_cpu) + print(f"groups: {[d.name for d in directories]}; runs excluded (failed gates or no throughput): {len(excluded)}") + for label in excluded: + print(f" excluded {label}") + print("main-campaign QUIC cells ran with the broker's default per-topic delivery") + + print("\nDelivered throughput, K msg/s (n) / broker CPU % of 400 (* = CPU-bound, >= 380)") + modes = ["main-campaign", "legacy", "single", "router", "tbf"] + for mode in modes: + for config in ["tcp", "tls"] + QUIC_CONFIGS: + row = [f"{mean_of(cells[(mode, config, loss)]) / 1000:7.2f}({len(cells[(mode, config, loss)]):2d})" + if (mode, config, loss) in cells else f"{'-':>11s}" for loss in LOSSES] + if any(cell.strip() != "-" for cell in row): + print(f" {mode:14s} {config:16s} " + " ".join(row)) + loads = [cpu_label(cpu, (mode, config, loss)) for loss in LOSSES] + print(f" {'':14s} {'broker cpu':16s} " + " ".join(f"{load:>11s}" for load in loads)) + if any((mode, config, loss) in SOFTIRQ for loss in LOSSES): + irq = [f"{statistics.mean(SOFTIRQ[(mode, config, loss)]):.1f}" if (mode, config, loss) in SOFTIRQ else "-" for loss in LOSSES] + print(f" {'':14s} {'host softirq %':16s} " + " ".join(f"{value:>11s}" for value in irq)) + + print("\nLoss-bound fold ratio G' = [X_ref(1)/X_ref(10)] / [X_quic(1)/X_quic(10)], 95% CI blocked by group") + main_g = fold_ratio(cells, "main-campaign", "tls", "quic-main", 1, 10) + for mode in modes: + for reference in ("tls", "tcp"): + for quic in QUIC_CONFIGS: + result = fold_ratio(cells, mode, reference, quic, 1, 10) + if result: + loads = ", ".join(cpu_label(cpu, (mode, cfg, loss)) for cfg in (reference, quic) for loss in (1, 10)) + print(f" {mode:14s} {reference:4s} vs {quic:16s} G'={fmt(result)} cpu[{loads}]") + + print("\nHeadline fold G0 = [X(0)/X(10)] ratio (0% level is broker-CPU bound, reported only)") + for mode in modes: + result = fold_ratio(cells, mode, "tls", "quic-main", 0, 10) + if result: + loads = ", ".join(cpu_label(cpu, (mode, cfg, loss)) for cfg in ("tls", "quic-main") for loss in (0, 10)) + print(f" {mode:14s} tls vs quic-main G0={fmt(result)} cpu[{loads}]") + + print("\nSame-loss ratio R(p) = X_quic(p) / X_ref(p), broker CPU of quic/ref cells") + for mode in modes: + for reference in ("tls", "tcp"): + row = [fmt(same_loss_ratio(cells, mode, reference, "quic-main", loss)) for loss in (1, 2, 5, 10)] + if any(cell != "n/a" for cell in row): + print(f" {mode:14s} quic-main/{reference:4s} " + " | ".join(row)) + loads = [f"{cpu_label(cpu, (mode, 'quic-main', loss))}/{cpu_label(cpu, (mode, reference, loss))}" for loss in (1, 2, 5, 10)] + print(f" {'':14s} {'cpu':14s} " + " | ".join(f"{load:>20s}" for load in loads)) + + print("\nRerun over main campaign, per cell X_single(p) / X_main(p)") + for config in ("tcp", "tls", "quic-main"): + row = [fmt(cross_ratio(cells, ("single", config, loss), ("main-campaign", config, loss))) for loss in (1, 2, 5, 10)] + if any(cell != "n/a" for cell in row): + print(f" {config:10s} " + " | ".join(row)) + + print("\nLoss exponent b in X ~ p^-b (Mathis predicts 0.5) and G' implied by the slopes, 10^(b_ref - b_quic)") + for mode in modes: + slopes = {config: loss_slope(cells, mode, config) for config in ["tcp", "tls"] + QUIC_CONFIGS} + for config, slope in slopes.items(): + if slope is not None: + print(f" {mode:14s} {config:16s} b={slope:5.2f}") + for reference in ("tls", "tcp"): + for quic in QUIC_CONFIGS: + if slopes.get(reference) is not None and slopes.get(quic) is not None: + print(f" {mode:14s} slope-implied G' {reference} vs {quic}: {10 ** (slopes[reference] - slopes[quic]):.2f}") + + print(f"\nMathis constant C = X_bytes * RTT * sqrt(p) / MSS per subscriber connection (Reno ~1.22), " + f"lower bound using {PAYLOAD_BYTES} B payload, MSS {TCP_MSS}, RTT {RTT_S * 1000:.0f} ms") + for mode in modes: + for config in ("tcp", "tls"): + row = [] + for loss in (1, 2, 5, 10): + if (mode, config, loss) in cells: + per_connection = mean_of(cells[(mode, config, loss)]) / 8 + row.append(f"{per_connection * PAYLOAD_BYTES * RTT_S * math.sqrt(loss / 100) / TCP_MSS:5.2f}") + else: + row.append(f"{'-':>5s}") + if any(cell.strip() != "-" for cell in row): + print(f" {mode:14s} {config:4s} " + " ".join(row)) + + for arm in ("router", "single"): + result = fold_ratio(cells, arm, "tls", "quic-main", 1, 10) + if result and main_g: + print(f"\n{arm} arm vs main campaign, G' tls vs quic-main: {verdict(result[1], main_g[0])}; rerun G'={fmt(result)}") + rerun = fold_ratio(cells, "single", "tls", "quic-main", 1, 10) + if rerun and main_g: + print(f"\nDECISION (pre-registered, G' single-segment tls vs quic-main): {verdict(rerun[1], main_g[0])}") + print(f" main-campaign G'={main_g[0]:.2f}; rerun G'={fmt(rerun)}") + prediction = clustering_prediction(sorted(Path(results_dir).glob("03e_single_segment_g*"))) + if prediction: + print(f" clustering-only prediction (X ~ 1/sqrt(p_eff), measured per-buffer drop fractions) for rerun G' / main-campaign G': {prediction:.2f}") + for arm in ("single", "router", "tbf"): + print(f" {arm:6s} G' / main-campaign G' = {fmt(g_shift(cells, arm))}") + + +if __name__ == "__main__": + default = Path(__file__).resolve().parent.parent / "results-v5" + main(sys.argv[1] if len(sys.argv) > 1 else default) diff --git a/publications/comnet/experiments/analysis/stream_limit_sweep.py b/publications/comnet/experiments/analysis/stream_limit_sweep.py new file mode 100644 index 00000000..eef9b48d --- /dev/null +++ b/publications/comnet/experiments/analysis/stream_limit_sweep.py @@ -0,0 +1,72 @@ +import glob +import json +import statistics +import sys +from pathlib import Path + +STRATEGIES = ["control-only", "per-topic", "per-publish"] +LIMITS = [100, 250, 1000] +RTT_S = 0.025 + + +def load(pattern, field): + values = [] + for path in glob.glob(pattern): + try: + results = json.load(open(path))["results"] + except (json.JSONDecodeError, KeyError, OSError): + continue + value = results.get(field) + if value: + values.append(value) + return values + + +def summarise(values, scale=1.0): + if not values: + return None + mean = statistics.mean(values) / scale + spread = (statistics.stdev(values) / statistics.mean(values) * 100) if len(values) > 1 else 0.0 + return mean, spread, len(values) + + +def main(results_dir): + base = Path(results_dir) / "04b_stream_limit_sweep" + if not base.exists(): + print(f"missing {base}") + return + + print("Single-connection publish rate (K msg/s), 8 topics, 25 ms, 2% loss") + print(f"{'strategy':14s} " + " ".join(f"{'limit ' + str(l):>18s}" for l in LIMITS)) + for strategy in STRATEGIES: + cells = [] + for limit in LIMITS: + rates = load(str(base / f"{strategy}_limit{limit}_throughput_run*_pub.json"), "offered_rate") + stat = summarise(rates, 1000.0) + cells.append(f"{stat[0]:8.2f}K ±{stat[1]:4.1f}% n={stat[2]}" if stat else f"{'-':>18s}") + print(f"{strategy:14s} " + " ".join(f"{c:>18s}" for c in cells)) + + print() + print("per-publish only: concurrent streams in flight (rate x 25 ms) against the cap") + cells = [] + for limit in LIMITS: + rates = load(str(base / f"per-publish_limit{limit}_throughput_run*_pub.json"), "offered_rate") + stat = summarise(rates) + cells.append(f"{stat[0] * RTT_S:5.0f}/{limit}" if stat else f"{'-':>10s}") + print(" " + " ".join(f"{c:>10s}" for c in cells)) + print(" (only per-publish opens one stream per message, so this ratio is meaningful only there)") + + print() + print("Experiment 2 cell: delivered rate at 2000 msg/s offered, 8 topics, 5% loss") + for strategy in STRATEGIES: + cells = [] + for limit in LIMITS: + rates = load(str(base / f"{strategy}_limit{limit}_hol_r2000_loss5pct_run*.json"), "measured_rate") + stat = summarise(rates) + cells.append(f"{stat[0]:8.1f} ±{stat[1]:4.1f}% n={stat[2]}" if stat else f"{'-':>18s}") + print(f" {strategy:14s} " + " ".join(f"{c:>18s}" for c in cells)) + + +if __name__ == "__main__": + default = Path(__file__).resolve().parent.parent / "results-v5" + main(sys.argv[1] if len(sys.argv) > 1 else default) diff --git a/publications/comnet/experiments/analysis/transport_limits.py b/publications/comnet/experiments/analysis/transport_limits.py new file mode 100644 index 00000000..9186c015 --- /dev/null +++ b/publications/comnet/experiments/analysis/transport_limits.py @@ -0,0 +1,169 @@ +import csv +import glob +import json +import statistics +import sys +from pathlib import Path + +from scipy import stats as st + +STRATEGIES = ["control-only", "per-topic", "per-publish"] +DEFAULT_STREAM_WINDOW = 262_144 +DEFAULT_MAX_STREAMS = 100 + + +def cell_paths(base, label): + return sorted(p for p in glob.glob(str(base / f"{label}_run*_pub.json"))) + + +def publish_rates(base, label): + rates = [] + for path in cell_paths(base, label): + try: + results = json.load(open(path))["results"] + except (json.JSONDecodeError, KeyError, OSError): + continue + elapsed = results.get("elapsed_secs") + published = results.get("published") + if elapsed and published: + rates.append(published / elapsed) + return rates + + +def hol_rates(base, label): + rates = [] + for path in sorted(glob.glob(str(base / f"{label}_run*.json"))): + if path.endswith("_pub.json"): + continue + try: + results = json.load(open(path))["results"] + except (json.JSONDecodeError, KeyError, OSError): + continue + if results.get("measured_rate"): + rates.append(results["measured_rate"]) + return rates + + +def summarise(values): + if not values: + return None + mean = statistics.mean(values) + if len(values) < 2: + return mean, 0.0, 1 + half = st.t.ppf(0.975, len(values) - 1) * st.sem(values) + return mean, half, len(values) + + +def fmt(stat, scale=1000.0): + if stat is None: + return f"{'-':>20s}" + return f"{stat[0] / scale:8.2f} ±{stat[1] / scale:5.2f} n={stat[2]}" + + +def broker_stats(base, label): + rtts, blocked_stream, blocked_conn, blocked_streams_uni = [], [], [], [] + for path in sorted(glob.glob(str(base / f"{label}_run*_broker_quic_*.csv"))): + rows = list(csv.DictReader(open(path))) + if not rows: + continue + last = rows[-1] + rtt_values = [int(r["rtt_us"]) for r in rows if r.get("rtt_us")] + if rtt_values: + rtts.append(statistics.median(rtt_values)) + blocked_stream.append(int(last.get("stream_data_blocked", 0))) + blocked_conn.append(int(last.get("data_blocked", 0))) + blocked_streams_uni.append(int(last.get("streams_blocked_uni", 0))) + if not rtts: + return None + return { + "rtt_ms": statistics.median(rtts) / 1000.0, + "stream_data_blocked": max(blocked_stream), + "data_blocked": max(blocked_conn), + "streams_blocked_uni": max(blocked_streams_uni), + } + + +def in_flight(base, label): + rates = publish_rates(base, label) + info = broker_stats(base, label) + if not rates or not info: + return None + return statistics.mean(rates) * info["rtt_ms"] / 1000.0 + + +def part_a(base): + print("PART A publish rate (K msg/s) by topic count, default transport config, 25 ms, 2% loss") + topics = [1, 2, 4, 8, 16] + print(f"{'strategy':14s}" + "".join(f"{'t=' + str(t):>21s}" for t in topics)) + for strategy in STRATEGIES: + cells = [fmt(summarise(publish_rates(base, f"{strategy}_t{t}_sdef_wdef_d25_l2_tput"))) for t in topics] + print(f"{strategy:14s}" + "".join(f"{c:>21s}" for c in cells)) + + +def part_b(base): + print("\nPART B stream credit, 8 topics, 25 ms, 2% loss") + print(f"{'cell':34s} {'K msg/s':>20s} {'rtt ms':>7s} {'streams in flight':>12s} {'of cap':>10s}") + for limit in [25, 100, 250, 1000]: + label = f"per-publish_t8_s{limit}_wdef_d25_l2_tput" + stat = summarise(publish_rates(base, label)) + info = broker_stats(base, label) + flight = in_flight(base, label) + rtt = f"{info['rtt_ms']:7.1f}" if info else f"{'-':>7s}" + pct = 100.0 * flight / limit if flight else float("nan") + print(f"{'per-publish limit ' + str(limit):34s} {fmt(stat):>20s} {rtt} {flight:12.1f} {pct:9.1f}%") + for strategy in ["control-only", "per-topic"]: + for limit in [100, 1000]: + label = f"{strategy}_t8_s{limit}_wdef_d25_l2_tput" + print(f"{strategy + ' limit ' + str(limit):34s} {fmt(summarise(publish_rates(base, label))):>20s}") + + print("\nPART B delay sweep (a credit ceiling scales with 1/RTT)") + for strategy in ["per-publish", "control-only"]: + for delay in [10, 25, 50]: + label = f"{strategy}_t8_s100_wdef_d{delay}_l2_tput" + info = broker_stats(base, label) + rtt = f"{info['rtt_ms']:7.1f}" if info else f"{'-':>7s}" + print(f"{strategy + ' delay ' + str(delay) + 'ms':34s} {fmt(summarise(publish_rates(base, label))):>20s} {rtt}") + + print("\nPART B Experiment 2 cell: delivered rate at 2000 msg/s offered, 8 topics, 5% loss") + for limit in [100, 1000]: + label = f"per-publish_t8_s{limit}_wdef_d25_l5_hol_r2000" + print(f"{'per-publish limit ' + str(limit):34s} {fmt(summarise(hol_rates(base, label)), 1.0):>20s} msg/s") + + +def part_c(base): + print("\nPART C per-stream receive window, 25 ms, 2% loss") + print(f"{'cell':34s} {'K msg/s':>20s} {'rtt ms':>7s} {'implied B/msg':>15s}") + print(" (implied B/msg is constant only where the per-stream window is the binding limit)") + for window in [131_072, 262_144, 1_048_576]: + for strategy, topics in [("control-only", 8), ("per-topic", 1), ("per-topic", 8)]: + label = f"{strategy}_t{topics}_sdef_w{window}_d25_l2_tput" + stat = summarise(publish_rates(base, label)) + info = broker_stats(base, label) + streams = topics if strategy == "per-topic" else 1 + flight = in_flight(base, label) + rtt = f"{info['rtt_ms']:7.1f}" if info else f"{'-':>7s}" + implied = window * streams / flight if flight else float("nan") + name = f"{strategy} t{topics} w{window // 1024}K" + print(f"{name:34s} {fmt(stat):>20s} {rtt} {implied:15.1f}") + + print("\nPART C 0% loss (separates congestion response from a fixed ceiling)") + for strategy, window in [("control-only", 262_144), ("per-topic", 262_144), ("control-only", 1_048_576)]: + label = f"{strategy}_t8_sdef_w{window}_d25_l0_tput" + info = broker_stats(base, label) + rtt = f"{info['rtt_ms']:7.1f}" if info else f"{'-':>7s}" + print(f"{strategy + ' w' + str(window // 1024) + 'K loss 0%':34s} {fmt(summarise(publish_rates(base, label))):>20s} {rtt}") + + +def main(results_dir): + base = Path(results_dir) / "04_transport_limits" + if not base.exists(): + print(f"missing {base}") + return + part_a(base) + part_b(base) + part_c(base) + + +if __name__ == "__main__": + default = Path(__file__).resolve().parent.parent / "results-v5" + main(sys.argv[1] if len(sys.argv) > 1 else default) diff --git a/publications/comnet/experiments/analysis/verify_reruns.py b/publications/comnet/experiments/analysis/verify_reruns.py new file mode 100644 index 00000000..c132861e --- /dev/null +++ b/publications/comnet/experiments/analysis/verify_reruns.py @@ -0,0 +1,87 @@ +import json +import statistics +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent / "results-v5" +D3 = ROOT / "03_throughput_under_loss" +D4 = ROOT / "04_stream_strategies" +LOSSES = [0, 1, 2, 5, 10] +TOPICS = [1, 4, 8, 16] +RUNS = range(1, 16) +ARMS3 = ["tcp", "tls", "quic-control-only", "quic-per-topic", "quic-per-publish"] +STRATS4 = ["control-only", "per-topic", "per-publish"] + + +def load(path: Path): + try: + if path.stat().st_size < 50: + return None + return json.load(open(path)) + except (OSError, ValueError): + return None + + +def median_or_none(values): + return round(statistics.median(values), 2) if values else None + + +def exp3(qos: int): + print(f"=== Exp3 QoS{qos}: unique delivered (throughput_avg/subs) | fan-out recv/pub | ingest pub/s ===") + print(f" loss: {LOSSES}") + for arm in ARMS3: + uniq, fan, ing = [], [], [] + for loss in LOSSES: + u, f, i = [], [], [] + for run in RUNS: + sub = load(D3 / f"{arm}_qos{qos}_loss{loss}pct_run{run}.json") + pub = load(D3 / f"{arm}_qos{qos}_loss{loss}pct_run{run}_pub.json") + if sub: + subs = sub["config"].get("subscribers") or 1 + u.append(sub["results"]["throughput_avg"] / subs) + if sub and pub and pub["results"].get("published"): + f.append(sub["results"]["received"] / pub["results"]["published"]) + if pub and pub["results"].get("elapsed_secs"): + i.append(pub["results"]["published"] / pub["results"]["elapsed_secs"]) + uniq.append(median_or_none(u)) + fan.append(median_or_none(f)) + ing.append(median_or_none(i)) + n = sum(1 for run in RUNS if load(D3 / f"{arm}_qos{qos}_loss0pct_run{run}.json")) + print(f" {arm:18s} n={n:2d} uniq={uniq}") + print(f" {'':18s} fanout={fan}") + print(f" {'':18s} ingest={ing}") + + +def exp4(): + print("=== Exp4: publisher send rate (published/elapsed) | subscriber delivered (throughput_avg) ===") + print(f" topics: {TOPICS}") + for strat in STRATS4: + send, deliv = [], [] + for topics in TOPICS: + s, d = [], [] + for run in RUNS: + pub = load(D4 / f"{strat}_{topics}topics_throughput_run{run}_pub.json") + sub = load(D4 / f"{strat}_{topics}topics_throughput_run{run}.json") + if pub and pub["results"].get("elapsed_secs"): + s.append(pub["results"]["published"] / pub["results"]["elapsed_secs"]) + if sub: + d.append(sub["results"]["throughput_avg"]) + send.append(median_or_none(s)) + deliv.append(median_or_none(d)) + print(f" {strat:12s} send={send}") + print(f" {'':12s} deliv={deliv}") + + +def main(): + which = sys.argv[1] if len(sys.argv) > 1 else "all" + if which in ("all", "exp3"): + exp3(1) + print() + if which in ("all", "exp4"): + exp4() + if which == "exp3qos0": + exp3(0) + + +if __name__ == "__main__": + main() diff --git a/publications/comnet/experiments/monitor/client_monitor.sh b/publications/comnet/experiments/monitor/client_monitor.sh index 00164953..9e575b0c 100755 --- a/publications/comnet/experiments/monitor/client_monitor.sh +++ b/publications/comnet/experiments/monitor/client_monitor.sh @@ -3,60 +3,50 @@ set -euo pipefail INTERVAL="${1:-1}" +exec 9>/tmp/client_monitor.lock +if ! flock -w 5 9; then + echo "client_monitor already running" >&2 + exit 1 +fi + IFACE=$(ip route show default 2>/dev/null | awk '{print $5; exit}') : "${IFACE:=eth0}" read_cpu() { - awk '/^cpu / {print $2, $4, $5}' /proc/stat + awk '/^cpu / {print $2, $3, $4, $5, $6, $7, $8, $9}' /proc/stat } read_net_counters() { awk -v iface="${IFACE}:" '$1 == iface {print $2, $3, $10, $11}' /proc/net/dev } -echo "timestamp,cpu_user,cpu_sys,cpu_idle,net_rx_bytes,net_tx_bytes,net_rx_packets,net_tx_packets" +echo "timestamp,cpu_user,cpu_sys,cpu_idle,net_rx_bytes,net_tx_bytes,net_rx_packets,net_tx_packets,cpu_nice,cpu_iowait,cpu_irq,cpu_softirq,cpu_steal,cpu_busy" -prev_cpu=$(read_cpu) -prev_user=$(echo "$prev_cpu" | awk '{print $1}') -prev_sys=$(echo "$prev_cpu" | awk '{print $2}') -prev_idle=$(echo "$prev_cpu" | awk '{print $3}') +read -r p_user p_nice p_sys p_idle p_iowait p_irq p_softirq p_steal <<< "$(read_cpu)" -sleep "$INTERVAL" +sleep "$INTERVAL" 9>&- while true; do ts=$(date +%s) - - cur_cpu=$(read_cpu) - cur_user=$(echo "$cur_cpu" | awk '{print $1}') - cur_sys=$(echo "$cur_cpu" | awk '{print $2}') - cur_idle=$(echo "$cur_cpu" | awk '{print $3}') - - d_user=$((cur_user - prev_user)) - d_sys=$((cur_sys - prev_sys)) - d_idle=$((cur_idle - prev_idle)) - d_total=$((d_user + d_sys + d_idle)) - - if [ "$d_total" -gt 0 ]; then - cpu_user=$(awk "BEGIN {printf \"%.1f\", 100*${d_user}/${d_total}}") - cpu_sys=$(awk "BEGIN {printf \"%.1f\", 100*${d_sys}/${d_total}}") - cpu_idle=$(awk "BEGIN {printf \"%.1f\", 100*${d_idle}/${d_total}}") - else - cpu_user="0.0" - cpu_sys="0.0" - cpu_idle="100.0" - fi - - prev_user=$cur_user - prev_sys=$cur_sys - prev_idle=$cur_idle - - net=$(read_net_counters) - rx_bytes=$(echo "$net" | awk '{print $1}') - rx_packets=$(echo "$net" | awk '{print $2}') - tx_bytes=$(echo "$net" | awk '{print $3}') - tx_packets=$(echo "$net" | awk '{print $4}') + read -r c_user c_nice c_sys c_idle c_iowait c_irq c_softirq c_steal <<< "$(read_cpu)" + read -r rx_bytes rx_packets tx_bytes tx_packets <<< "$(read_net_counters)" : "${rx_bytes:=0}" "${rx_packets:=0}" "${tx_bytes:=0}" "${tx_packets:=0}" - echo "${ts},${cpu_user},${cpu_sys},${cpu_idle},${rx_bytes},${tx_bytes},${rx_packets},${tx_packets}" - sleep "$INTERVAL" + awk -v ts="$ts" \ + -v du=$((c_user - p_user)) -v dn=$((c_nice - p_nice)) -v ds=$((c_sys - p_sys)) \ + -v di=$((c_idle - p_idle)) -v dw=$((c_iowait - p_iowait)) -v dq=$((c_irq - p_irq)) \ + -v dsq=$((c_softirq - p_softirq)) -v dst=$((c_steal - p_steal)) \ + -v rxb="$rx_bytes" -v txb="$tx_bytes" -v rxp="$rx_packets" -v txp="$tx_packets" \ + 'BEGIN { + total = du + dn + ds + di + dw + dq + dsq + dst + if (total <= 0) { total = 1; di = 1 } + printf "%s,%.1f,%.1f,%.1f,%s,%s,%s,%s,%.1f,%.1f,%.1f,%.1f,%.1f,%.1f\n", ts, + 100*du/total, 100*ds/total, 100*di/total, rxb, txb, rxp, txp, + 100*dn/total, 100*dw/total, 100*dq/total, 100*dsq/total, 100*dst/total, + 100*(total - di - dw)/total + }' + + p_user=$c_user; p_nice=$c_nice; p_sys=$c_sys; p_idle=$c_idle + p_iowait=$c_iowait; p_irq=$c_irq; p_softirq=$c_softirq; p_steal=$c_steal + sleep "$INTERVAL" 9>&- done diff --git a/publications/comnet/experiments/monitor/resource_monitor.sh b/publications/comnet/experiments/monitor/resource_monitor.sh index 5cf926c9..9e5ab021 100755 --- a/publications/comnet/experiments/monitor/resource_monitor.sh +++ b/publications/comnet/experiments/monitor/resource_monitor.sh @@ -4,6 +4,12 @@ set -euo pipefail PID="${1:?usage: $0 }" INTERVAL="${2:-1}" +exec 9>/tmp/resource_monitor.lock +if ! flock -w 5 9; then + echo "resource_monitor already running" >&2 + exit 1 +fi + IFACE=$(ip route show default 2>/dev/null | awk '{print $5; exit}') : "${IFACE:=eth0}" @@ -18,13 +24,18 @@ read_cpu_jiffies() { "/proc/${PID}/stat" 2>/dev/null || echo 0 } -echo "timestamp,rss_kb,cpu_percent,threads,net_rx_bytes,net_tx_bytes,net_rx_packets,net_tx_packets" +read_host_cpu() { + awk '/^cpu / {print $2, $3, $4, $5, $6, $7, $8, $9}' /proc/stat +} + +echo "timestamp,rss_kb,cpu_percent,threads,net_rx_bytes,net_tx_bytes,net_rx_packets,net_tx_packets,host_user,host_nice,host_sys,host_idle,host_iowait,host_irq,host_softirq,host_steal,host_busy" prev_jiffies=$(read_cpu_jiffies) prev_time=$(date +%s.%N) +read -r p_user p_nice p_sys p_idle p_iowait p_irq p_softirq p_steal <<< "$(read_host_cpu)" while kill -0 "$PID" 2>/dev/null; do - sleep "$INTERVAL" + sleep "$INTERVAL" 9>&- kill -0 "$PID" 2>/dev/null || break now=$(date +%s.%N) ts="${now%.*}" @@ -35,11 +46,19 @@ while kill -0 "$PID" 2>/dev/null; do prev_time=$now rss=$(awk '/^VmRSS:/ {print $2}' "/proc/${PID}/status" 2>/dev/null || echo 0) threads=$(awk '/^Threads:/ {print $2}' "/proc/${PID}/status" 2>/dev/null || echo 0) - net=$(read_net_counters) - rx_bytes=$(echo "$net" | awk '{print $1}') - rx_packets=$(echo "$net" | awk '{print $2}') - tx_bytes=$(echo "$net" | awk '{print $3}') - tx_packets=$(echo "$net" | awk '{print $4}') + read -r rx_bytes rx_packets tx_bytes tx_packets <<< "$(read_net_counters)" : "${rx_bytes:=0}" "${rx_packets:=0}" "${tx_bytes:=0}" "${tx_packets:=0}" - echo "${ts},${rss},${cpu},${threads},${rx_bytes},${tx_bytes},${rx_packets},${tx_packets}" + read -r c_user c_nice c_sys c_idle c_iowait c_irq c_softirq c_steal <<< "$(read_host_cpu)" + host=$(awk -v du=$((c_user - p_user)) -v dn=$((c_nice - p_nice)) -v ds=$((c_sys - p_sys)) \ + -v di=$((c_idle - p_idle)) -v dw=$((c_iowait - p_iowait)) -v dq=$((c_irq - p_irq)) \ + -v dsq=$((c_softirq - p_softirq)) -v dst=$((c_steal - p_steal)) \ + 'BEGIN { + total = du + dn + ds + di + dw + dq + dsq + dst + if (total <= 0) { total = 1; di = 1 } + printf "%.1f,%.1f,%.1f,%.1f,%.1f,%.1f,%.1f,%.1f,%.1f", 100*du/total, 100*dn/total, 100*ds/total, + 100*di/total, 100*dw/total, 100*dq/total, 100*dsq/total, 100*dst/total, 100*(total - di - dw)/total + }') + p_user=$c_user; p_nice=$c_nice; p_sys=$c_sys; p_idle=$c_idle + p_iowait=$c_iowait; p_irq=$c_irq; p_softirq=$c_softirq; p_steal=$c_steal + echo "${ts},${rss},${cpu},${threads},${rx_bytes},${tx_bytes},${rx_packets},${tx_packets},${host}" done diff --git a/publications/comnet/experiments/monitor/router_monitor.sh b/publications/comnet/experiments/monitor/router_monitor.sh new file mode 100755 index 00000000..83c974fc --- /dev/null +++ b/publications/comnet/experiments/monitor/router_monitor.sh @@ -0,0 +1,34 @@ +#!/usr/bin/env bash +set -euo pipefail + +INTERVAL="${1:-1}" + +exec 9>/tmp/router_monitor.lock +if ! flock -w 5 9; then + echo "router_monitor already running" >&2 + exit 1 +fi + +snapshot() { + awk '/^cpu[0-9]+ / { + busy = $2 + $3 + $4 + $7 + $8 + $9 + print $1, busy, busy + $5 + $6, $4, $8 + }' /proc/stat +} + +echo "timestamp,cpu,busy_pct,sys_pct,softirq_pct" + +prev=$(snapshot) +sleep "$INTERVAL" 9>&- + +while true; do + ts=$(date +%s) + cur=$(snapshot) + paste -d' ' <(echo "$prev") <(echo "$cur") | awk -v ts="$ts" '{ + total = $8 - $3 + if (total <= 0) total = 1 + printf "%s,%s,%.1f,%.1f,%.1f\n", ts, $1, 100*($7 - $2)/total, 100*($9 - $4)/total, 100*($10 - $5)/total + }' + prev=$cur + sleep "$INTERVAL" 9>&- +done diff --git a/publications/comnet/experiments/netem/apply_bottleneck.sh b/publications/comnet/experiments/netem/apply_bottleneck.sh new file mode 100755 index 00000000..99662f32 --- /dev/null +++ b/publications/comnet/experiments/netem/apply_bottleneck.sh @@ -0,0 +1,11 @@ +#!/usr/bin/env bash +set -euo pipefail + +DELAY_MS="${1:?usage: $0 [limit_pkts] [loss_pct]}" +RATE_MBIT="${2:?usage: $0 [limit_pkts] [loss_pct]}" +LIMIT_PKTS="${3:-1000}" +LOSS_PCT="${4:-0}" + +DEV=$(ip route show default | awk '{print $5}' | head -1) +tc qdisc replace dev "$DEV" root netem delay "${DELAY_MS}ms" loss "${LOSS_PCT}%" rate "${RATE_MBIT}mbit" limit "${LIMIT_PKTS}" +echo "netem-bottleneck: dev=${DEV} delay=${DELAY_MS}ms rate=${RATE_MBIT}mbit limit=${LIMIT_PKTS}pkts loss=${LOSS_PCT}%" diff --git a/publications/comnet/experiments/netem/broker_backlog.sh b/publications/comnet/experiments/netem/broker_backlog.sh new file mode 100755 index 00000000..dad0fc4d --- /dev/null +++ b/publications/comnet/experiments/netem/broker_backlog.sh @@ -0,0 +1,16 @@ +#!/usr/bin/env bash +set -euo pipefail + +INTERVAL="${1:-1}" + +DEV=$(ip route show default | awk '{print $5}' | head -1) +echo "timestamp,backlog_bytes,backlog_pkts,dropped" +while true; do + tc -s -j qdisc show dev "$DEV" | python3 -c ' +import json, sys, time +for qdisc in json.load(sys.stdin): + if qdisc.get("kind") == "netem" and qdisc.get("handle") == "10:": + print(f"{int(time.time())},{qdisc.get("backlog", 0)},{qdisc.get("qlen", 0)},{qdisc.get("drops", 0)}") +' + sleep "$INTERVAL" +done diff --git a/publications/comnet/experiments/netem/broker_netem.sh b/publications/comnet/experiments/netem/broker_netem.sh new file mode 100755 index 00000000..6d9a3ec4 --- /dev/null +++ b/publications/comnet/experiments/netem/broker_netem.sh @@ -0,0 +1,11 @@ +#!/usr/bin/env bash +set -euo pipefail + +DELAY_US="${1:?usage: $0 [limit]}" +LOSS_PCT="${2:-0}" +LIMIT="${3:-}" + +DEV=$(ip route show default | awk '{print $5}' | head -1) +tc qdisc del dev "$DEV" root 2>/dev/null || true +tc qdisc add dev "$DEV" root handle 10: netem delay "${DELAY_US}us" loss "${LOSS_PCT}%" ${LIMIT:+limit "$LIMIT"} +echo "broker netem: dev=${DEV} delay=${DELAY_US}us loss=${LOSS_PCT}% limit=${LIMIT:-default}" diff --git a/publications/comnet/experiments/netem/broker_tbf.sh b/publications/comnet/experiments/netem/broker_tbf.sh new file mode 100755 index 00000000..fba6ca58 --- /dev/null +++ b/publications/comnet/experiments/netem/broker_tbf.sh @@ -0,0 +1,13 @@ +#!/usr/bin/env bash +set -euo pipefail + +DELAY_US="${1:?usage: $0 }" +LOSS_PCT="${2:-0}" + +DEV=$(ip route show default | awk '{print $5}' | head -1) +MTU=$(cat "/sys/class/net/${DEV}/mtu") +BURST=$((MTU + 14)) +tc qdisc del dev "$DEV" root 2>/dev/null || true +tc qdisc add dev "$DEV" root handle 1: tbf rate 100gbit burst "$BURST" latency 100ms +tc qdisc add dev "$DEV" parent 1:1 handle 10: netem delay "${DELAY_US}us" loss "${LOSS_PCT}%" limit 100000 +echo "broker tbf+netem: dev=${DEV} burst=${BURST} delay=${DELAY_US}us loss=${LOSS_PCT}%" diff --git a/publications/comnet/experiments/netem/netem_probe.bt b/publications/comnet/experiments/netem/netem_probe.bt new file mode 100644 index 00000000..d369a8dd --- /dev/null +++ b/publications/comnet/experiments/netem/netem_probe.bt @@ -0,0 +1,11 @@ +kprobe:netem_enqueue +{ + $skb = (struct sk_buff *)arg0; + $shinfo = (struct skb_shared_info *)($skb->head + $skb->end); + @enq = count(); + @maxlen = max($skb->len); + if ($shinfo->gso_size > 0 || $shinfo->gso_segs > 1) { + @gso = count(); + @gso_segs = hist($shinfo->gso_segs); + } +} diff --git a/publications/comnet/experiments/netem/router_clear.sh b/publications/comnet/experiments/netem/router_clear.sh new file mode 100755 index 00000000..e4cbb4cf --- /dev/null +++ b/publications/comnet/experiments/netem/router_clear.sh @@ -0,0 +1,8 @@ +#!/usr/bin/env bash +set -euo pipefail + +DEV=$(ip route show default | awk '{print $5}' | head -1) +tc qdisc del dev "$DEV" root 2>/dev/null || true +tc qdisc del dev "$DEV" clsact 2>/dev/null || true +rm -f /tmp/router_netem_parent +echo "router: cleared dev=${DEV}" diff --git a/publications/comnet/experiments/netem/router_ingress_count.sh b/publications/comnet/experiments/netem/router_ingress_count.sh new file mode 100755 index 00000000..10bfaa2b --- /dev/null +++ b/publications/comnet/experiments/netem/router_ingress_count.sh @@ -0,0 +1,5 @@ +#!/usr/bin/env bash +set -euo pipefail + +DEV=$(ip route show default | awk '{print $5}' | head -1) +tc -s filter show dev "$DEV" ingress | awk '/Sent [0-9]+ bytes [0-9]+ pkt/ { for (i = 1; i <= NF; i++) if ($i == "pkt") total += $(i - 1) } END { print total + 0 }' diff --git a/publications/comnet/experiments/netem/router_loss.sh b/publications/comnet/experiments/netem/router_loss.sh new file mode 100755 index 00000000..fc6b08b3 --- /dev/null +++ b/publications/comnet/experiments/netem/router_loss.sh @@ -0,0 +1,10 @@ +#!/usr/bin/env bash +set -euo pipefail + +LOSS_PCT="${1:?usage: $0 [delay_us]}" +DELAY_US="${2:-0}" + +DEV=$(ip route show default | awk '{print $5}' | head -1) +PARENT=$(cat /tmp/router_netem_parent) +tc qdisc change dev "$DEV" parent "$PARENT" handle 10: netem delay "${DELAY_US}us" loss "${LOSS_PCT}%" limit 100000 +echo "router netem: dev=${DEV} parent=${PARENT} delay=${DELAY_US}us loss=${LOSS_PCT}%" diff --git a/publications/comnet/experiments/netem/router_setup.sh b/publications/comnet/experiments/netem/router_setup.sh new file mode 100755 index 00000000..84701d22 --- /dev/null +++ b/publications/comnet/experiments/netem/router_setup.sh @@ -0,0 +1,47 @@ +#!/usr/bin/env bash +set -euo pipefail + +BROKER_IP="${1:?usage: $0 [guard 0|1]}" +PUB_IP="${2:?usage: $0 [guard 0|1]}" +SUB_IP="${3:?usage: $0 [guard 0|1]}" +GUARD="${4:-0}" + +DEV=$(ip route show default | awk '{print $5}' | head -1) +MTU=$(cat "/sys/class/net/${DEV}/mtu") +BURST=$((MTU + 14)) + +sysctl -qw net.ipv4.ip_forward=1 +sysctl -qw net.ipv4.conf.all.send_redirects=0 +sysctl -qw "net.ipv4.conf.${DEV}.send_redirects=0" +for feature in gro rx-gro-hw lro rx-gro-list rx-udp-gro-forwarding; do + ethtool -K "$DEV" "$feature" off 2>/dev/null || true +done + +ethtool -k "$DEV" | grep -E '^(generic-receive-offload|rx-gro-hw|large-receive-offload|rx-gro-list|rx-udp-gro-forwarding):' +still_on=$(ethtool -k "$DEV" | grep -E '^(generic-receive-offload|rx-gro-hw|large-receive-offload|rx-gro-list):' | grep -c ': on' || true) + +if [ "$still_on" -gt 0 ] && [ "$GUARD" != "1" ]; then + echo "ERROR: receive coalescing still on for ${DEV}; rerun with guard=1" >&2 + exit 3 +fi + +tc qdisc del dev "$DEV" root 2>/dev/null || true +tc qdisc del dev "$DEV" clsact 2>/dev/null || true + +tc qdisc add dev "$DEV" clsact +tc filter add dev "$DEV" ingress protocol ip prio 1 u32 \ + match ip src "${BROKER_IP}/32" match ip dst "${PUB_IP}/32" action pass +tc filter add dev "$DEV" ingress protocol ip prio 2 u32 \ + match ip src "${BROKER_IP}/32" match ip dst "${SUB_IP}/32" action pass + +tc qdisc add dev "$DEV" root handle 1: prio bands 3 priomap 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 +if [ "$GUARD" = "1" ]; then + tc qdisc add dev "$DEV" parent 1:1 handle 5: tbf rate 100gbit burst "$BURST" latency 50ms + tc qdisc add dev "$DEV" parent 5:1 handle 10: netem loss 0% limit 100000 + echo "5:1" > /tmp/router_netem_parent +else + tc qdisc add dev "$DEV" parent 1:1 handle 10: netem loss 0% limit 100000 + echo "1:1" > /tmp/router_netem_parent +fi +tc filter add dev "$DEV" parent 1: protocol ip prio 1 u32 match ip src "${BROKER_IP}/32" flowid 1:1 +echo "router: dev=${DEV} broker=${BROKER_IP} pub=${PUB_IP} sub=${SUB_IP} guard=${GUARD} coalescing_on=${still_on}" diff --git a/publications/comnet/experiments/netem/tbf_probe.bt b/publications/comnet/experiments/netem/tbf_probe.bt new file mode 100644 index 00000000..f64196c4 --- /dev/null +++ b/publications/comnet/experiments/netem/tbf_probe.bt @@ -0,0 +1,4 @@ +kretprobe:tbf_enqueue +{ + @ret[retval] = count(); +} diff --git a/publications/comnet/experiments/netem/tcp_cwr_probe.bt b/publications/comnet/experiments/netem/tcp_cwr_probe.bt new file mode 100644 index 00000000..4a86df6c --- /dev/null +++ b/publications/comnet/experiments/netem/tcp_cwr_probe.bt @@ -0,0 +1,10 @@ +kprobe:tcp_enter_cwr +{ + @cwr = count(); +} + +kretprobe:ip_queue_xmit +/retval != 0/ +{ + @xmit_err[retval] = count(); +} diff --git a/publications/comnet/experiments/parallel/02c_topic_rate_sweep_v5.sh b/publications/comnet/experiments/parallel/02c_topic_rate_sweep_v5.sh new file mode 100644 index 00000000..d8c8d01f --- /dev/null +++ b/publications/comnet/experiments/parallel/02c_topic_rate_sweep_v5.sh @@ -0,0 +1,144 @@ +#!/usr/bin/env bash +set -euo pipefail +source "$(dirname "${BASH_SOURCE[0]}")/common_parallel.sh" + +EXPERIMENT="02c_topic_rate_sweep" +LOSSES=(1 5) +DELAY=25 +RUNS_PER_DATAPOINT=10 + +RESULTS_DIR="${ROOT_DIR}/results-v5" +OUTPUT_DIR="${RESULTS_DIR}/${EXPERIMENT}" +mkdir -p "$OUTPUT_DIR" + +BROKER_TLS="--tls-cert /opt/mqtt-certs/server.pem --tls-key /opt/mqtt-certs/server.key" +BROKER_QUIC="--quic-host 0.0.0.0:14567" +CA="--ca-cert /opt/mqtt-certs/ca.pem" + +declare -A TRANSPORT_URLS +TRANSPORT_URLS[tcp]="mqtt://${BROKER_IP}:1883" +TRANSPORT_URLS[quic-control]="quic://${BROKER_IP}:14567" +TRANSPORT_URLS[quic-pertopic]="quic://${BROKER_IP}:14567" +TRANSPORT_URLS[quic-perpub]="quic://${BROKER_IP}:14567" + +declare -A TRANSPORT_FLAGS +TRANSPORT_FLAGS[tcp]="" +TRANSPORT_FLAGS[quic-control]="--quic-stream-strategy control-only ${CA}" +TRANSPORT_FLAGS[quic-pertopic]="--quic-stream-strategy per-topic ${CA}" +TRANSPORT_FLAGS[quic-perpub]="--quic-stream-strategy per-publish ${CA}" + +declare -A BROKER_DELIVERY +BROKER_DELIVERY[tcp]="" +BROKER_DELIVERY[quic-control]="--quic-delivery-strategy control-only" +BROKER_DELIVERY[quic-pertopic]="--quic-delivery-strategy per-topic" +BROKER_DELIVERY[quic-perpub]="--quic-delivery-strategy per-publish" + +: "${V3_TRANSPORTS:=tcp quic-control quic-pertopic quic-perpub}" +read -ra TRANSPORTS <<< "$V3_TRANSPORTS" + +CELLS=( + "2 500" "4 500" "16 500" "32 500" + "8 125" "8 250" "8 1000" "8 2000" +) + +start_monitors() { + BROKER_MONITOR_PID=$(ssh_broker "nohup bash /opt/mqtt-lib/experiments/monitor/resource_monitor.sh ${BROKER_PID} \ + > /tmp/monitor.csv 2>&1 & echo \$!") || BROKER_MONITOR_PID="" + PUB_MONITOR_PID=$(ssh_pub "nohup bash /opt/mqtt-lib/experiments/monitor/client_monitor.sh \ + > /tmp/client_monitor.csv 2>&1 & echo \$!") || PUB_MONITOR_PID="" +} + +stop_monitors() { + local output_dir="$1" + local run_label="$2" + ssh_broker "kill ${BROKER_MONITOR_PID}" 2>/dev/null || true + ssh_pub "kill ${PUB_MONITOR_PID}" 2>/dev/null || true + scp -i "$SSH_KEY_PATH" $SSH_OPTS "${SSH_USER}@${BROKER_SSH_IP}:/tmp/monitor.csv" \ + "${output_dir}/${run_label}_broker_resources.csv" 2>/dev/null || true + scp -i "$SSH_KEY_PATH" $SSH_OPTS "${SSH_USER}@${PUB_IP}:/tmp/client_monitor.csv" \ + "${output_dir}/${run_label}_pub_resources.csv" 2>/dev/null || true + BROKER_MONITOR_PID="" + PUB_MONITOR_PID="" +} + +collect_traces() { + local run_label="$1" + local remote_dir="$2" + for csv in messages.csv quinn_stats.csv; do + scp -i "$SSH_KEY_PATH" "${SSH_USER}@${PUB_IP}:${remote_dir}/${csv}" \ + "${OUTPUT_DIR}/${run_label}_${csv}" 2>/dev/null || true + done + ssh_pub "rm -rf ${remote_dir}" 2>/dev/null || true +} + +run_hol_colocated() { + local label="$1" + shift + local bench_args="$*" + echo " running (co-located): ${label}" + ssh_pub "ulimit -n 65536; mqttv5 bench ${bench_args}" \ + > "${OUTPUT_DIR}/${label}.json" 2>/dev/null || true + warn_if_empty "${OUTPUT_DIR}/${label}.json" + echo " saved: ${OUTPUT_DIR}/${label}.json" +} + +for tname in "${TRANSPORTS[@]}"; do + url="${TRANSPORT_URLS[$tname]}" + flags="${TRANSPORT_FLAGS[$tname]}" + delivery="${BROKER_DELIVERY[$tname]}" + + stop_broker 2>/dev/null || true + if ! start_broker "${BROKER_TLS} ${BROKER_QUIC} ${delivery}"; then + echo "WARN: broker start failed for ${tname}, skipping transport" >&2 + continue + fi + + for cell in "${CELLS[@]}"; do + read -r topics rate <<< "$cell" + for loss in "${LOSSES[@]}"; do + label="${tname}_t${topics}_r${rate}_loss${loss}pct" + + done_runs=0 + for run in $(seq 1 "$RUNS_PER_DATAPOINT"); do + f="${OUTPUT_DIR}/${label}_run${run}.json" + if [ -s "$f" ] && [ "$(wc -c < "$f")" -gt 500 ]; then + done_runs=$((done_runs + 1)) + fi + done + if [ "$done_runs" = "$RUNS_PER_DATAPOINT" ]; then + echo "[${EXPERIMENT}] ${label} complete (${done_runs}/${RUNS_PER_DATAPOINT}), skipping" + continue + fi + + apply_netem "$DELAY" "$loss" + echo "[${EXPERIMENT}] ${label} (${done_runs}/${RUNS_PER_DATAPOINT} done)" + + bench_args="--url ${url} ${flags} --mode hol-blocking --topics ${topics} --duration 60 --warmup 5 --payload-size 256 --rate ${rate} --trace-dir /tmp/hol-traces" + + for run in $(seq 1 "$RUNS_PER_DATAPOINT"); do + run_label="${label}_run${run}" + existing="${OUTPUT_DIR}/${run_label}.json" + if [ -s "$existing" ] && [ "$(wc -c < "$existing")" -gt 500 ]; then + echo " skip (exists): ${run_label}" + continue + fi + if [ "$BROKER_FRESH" = "1" ]; then + BROKER_FRESH=0 + elif ! restart_broker; then + echo "WARN: broker restart failed, skipping ${run_label}" >&2 + continue + fi + start_monitors + run_hol_colocated "$run_label" "$bench_args" + stop_monitors "$OUTPUT_DIR" "$run_label" + collect_traces "$run_label" "/tmp/hol-traces" + sleep 5 + done + + clear_netem + done + done +done + +stop_broker +echo "experiment ${EXPERIMENT} complete (group ${GROUP})" diff --git a/publications/comnet/experiments/parallel/03_watched.sh b/publications/comnet/experiments/parallel/03_watched.sh new file mode 100644 index 00000000..00456ac3 --- /dev/null +++ b/publications/comnet/experiments/parallel/03_watched.sh @@ -0,0 +1,40 @@ +#!/usr/bin/env bash +set -euo pipefail +source "$(dirname "${BASH_SOURCE[0]}")/common_parallel.sh" + +EXPERIMENT="03_throughput_under_loss" +LOSSES=(0 5) +DELAY=10 +QOS_LEVELS=(0 1) +STRATEGIES=("per-topic") +RUNS_PER_DATAPOINT=2 + +RESULTS_DIR="${ROOT_DIR}/results-watched" +mkdir -p "$RESULTS_DIR" + +BENCH_COMMON="--mode throughput --duration 30 --warmup 5 --payload-size 256 --publishers 16 --subscribers 8 --inflight 64" + +start_broker "--tls-cert /opt/mqtt-certs/server.pem --tls-key /opt/mqtt-certs/server.key --quic-host 0.0.0.0:14567" + +for qos in "${QOS_LEVELS[@]}"; do + for loss in "${LOSSES[@]}"; do + apply_netem "$DELAY" "$loss" + + label="tcp_qos${qos}_loss${loss}pct" + echo "[${EXPERIMENT}] ${label}" + run_monitored_split "$EXPERIMENT" "$label" \ + "--url mqtt://${BROKER_IP}:1883 ${BENCH_COMMON} --qos ${qos}" + + for strategy in "${STRATEGIES[@]}"; do + label="quic-${strategy}_qos${qos}_loss${loss}pct" + echo "[${EXPERIMENT}] ${label}" + run_monitored_split "$EXPERIMENT" "$label" \ + "--url quic://${BROKER_IP}:14567 --ca-cert /opt/mqtt-certs/ca.pem --quic-stream-strategy ${strategy} ${BENCH_COMMON} --qos ${qos}" + done + + clear_netem + done +done + +stop_broker +echo "watched slice complete (group ${GROUP})" diff --git a/publications/comnet/experiments/parallel/03b_tls_throughput_v5.sh b/publications/comnet/experiments/parallel/03b_tls_throughput_v5.sh new file mode 100644 index 00000000..489ba634 --- /dev/null +++ b/publications/comnet/experiments/parallel/03b_tls_throughput_v5.sh @@ -0,0 +1,52 @@ +#!/usr/bin/env bash +set -euo pipefail +source "$(dirname "${BASH_SOURCE[0]}")/common_parallel.sh" + +EXPERIMENT="03_throughput_under_loss" +LOSSES=(${LOSSES_OVERRIDE:-0 1 2 5 10}) +DELAY=10 +QOS_LEVELS=(${QOS_OVERRIDE:-0 1}) +RUNS_PER_DATAPOINT=15 + +RESULTS_DIR="${ROOT_DIR}/results-v5" +OUTPUT_DIR="${RESULTS_DIR}/${EXPERIMENT}" +mkdir -p "$OUTPUT_DIR" + +run_tls_cell() { + local label="$1" + shift + local bench_args="$*" + for run in $(seq 1 "$RUNS_PER_DATAPOINT"); do + local run_label="${label}_run${run}" + local f="${OUTPUT_DIR}/${run_label}.json" + if [ -f "$f" ] && [ "$(wc -c < "$f")" -gt 300 ]; then + continue + fi + if [ "$BROKER_FRESH" = "1" ]; then + BROKER_FRESH=0 + elif ! restart_broker; then + echo "WARN: broker restart failed, skipping ${run_label}" >&2 + continue + fi + start_monitors + run_bench_split "$EXPERIMENT" "$run_label" "$bench_args" + stop_monitors "$OUTPUT_DIR" "$run_label" + sleep 5 + done +} + +start_broker "--tls-cert /opt/mqtt-certs/server.pem --tls-key /opt/mqtt-certs/server.key --quic-host 0.0.0.0:14567" + +for qos in "${QOS_LEVELS[@]}"; do + for loss in "${LOSSES[@]}"; do + apply_netem "$DELAY" "$loss" + label="tls_qos${qos}_loss${loss}pct" + echo "[${EXPERIMENT}] ${label}" + run_tls_cell "$label" \ + "--url mqtts://${BROKER_IP}:8883 --ca-cert /opt/mqtt-certs/ca.pem --mode throughput --duration 60 --warmup 5 --payload-size 256 --qos ${qos} --publishers 16 --subscribers 8 --inflight 64" + clear_netem + done +done + +stop_broker +echo "experiment ${EXPERIMENT} TLS arm complete (group ${GROUP})" diff --git a/publications/comnet/experiments/parallel/03c_offload_ablation.sh b/publications/comnet/experiments/parallel/03c_offload_ablation.sh new file mode 100644 index 00000000..6e583eba --- /dev/null +++ b/publications/comnet/experiments/parallel/03c_offload_ablation.sh @@ -0,0 +1,79 @@ +#!/usr/bin/env bash +set -euo pipefail +source "$(dirname "${BASH_SOURCE[0]}")/common_parallel.sh" + +EXPERIMENT="03c_offload_ablation_g${GROUP}" +LOSSES=(${LOSSES_OVERRIDE:-0 1 2 5 10}) +DELAY=10 +RUNS_PER_DATAPOINT="${RUNS_OVERRIDE:-5}" + +RESULTS_DIR="${ROOT_DIR}/results-v5" +OUTPUT_DIR="${RESULTS_DIR}/${EXPERIMENT}" +mkdir -p "$OUTPUT_DIR" + +DEV_CMD='DEV=$(ip route show default | awk "{print \$5}" | head -1)' + +set_offload() { + local mode="$1" + if [ "$mode" = "off" ]; then + ssh_broker "${DEV_CMD}; for f in gso tso gro lro tx-udp-segmentation rx-gro-list; do sudo ethtool -K \$DEV \$f off 2>/dev/null || true; done; echo '--- ethtool -k after off ---'; ethtool -k \$DEV | grep -E 'tcp-segmentation|generic-segmentation|generic-receive|large-receive|udp-segmentation'" + else + ssh_broker "${DEV_CMD}; for f in gso tso gro; do sudo ethtool -K \$DEV \$f on 2>/dev/null || true; done; echo '--- ethtool -k after on ---'; ethtool -k \$DEV | grep -E 'tcp-segmentation|generic-segmentation|generic-receive|large-receive|udp-segmentation'" + fi +} + +qdisc_snapshot() { ssh_broker "${DEV_CMD}; tc -s qdisc show dev \$DEV"; } + +run_ablation_cell() { + local label="$1" + shift + local bench_args="$*" + for run in $(seq 1 "$RUNS_PER_DATAPOINT"); do + local rl="${label}_run${run}" + local f="${OUTPUT_DIR}/${rl}.json" + if [ -s "$f" ] && [ "$(wc -c < "$f")" -gt 300 ]; then + continue + fi + if [ "$BROKER_FRESH" = "1" ]; then + BROKER_FRESH=0 + elif ! restart_broker; then + echo "WARN: broker restart failed, skipping ${rl}" >&2 + continue + fi + start_monitors + qdisc_snapshot > "${OUTPUT_DIR}/${rl}_qdisc_before.txt" 2>/dev/null || true + run_bench_split "$EXPERIMENT" "$rl" "$bench_args" + qdisc_snapshot > "${OUTPUT_DIR}/${rl}_qdisc_after.txt" 2>/dev/null || true + stop_monitors "$OUTPUT_DIR" "$rl" + sleep 5 + done +} + +BENCH_COMMON="--mode throughput --duration 60 --warmup 5 --payload-size 256 --qos 0 --publishers 16 --subscribers 8 --inflight 64" + +for offload in ${OFFLOAD_OVERRIDE:-on off}; do + broker_flags="--tls-cert /opt/mqtt-certs/server.pem --tls-key /opt/mqtt-certs/server.key --quic-host 0.0.0.0:14567" + if [ "$offload" = "off" ]; then + broker_flags="${broker_flags} --quic-disable-offload" + fi + stop_broker 2>/dev/null || true + echo "[${EXPERIMENT}] setting NIC offload ${offload} on broker" + set_offload "$offload" | tee "${OUTPUT_DIR}/ethtool_${offload}.txt" + if ! start_broker "$broker_flags"; then + echo "ERROR: broker failed to start (offload ${offload})" >&2 + continue + fi + for loss in "${LOSSES[@]}"; do + apply_netem "$DELAY" "$loss" + echo "[${EXPERIMENT}] tls offload=${offload} loss=${loss}%" + run_ablation_cell "tls_offload-${offload}_loss${loss}pct" \ + "--url mqtts://${BROKER_IP}:8883 --ca-cert /opt/mqtt-certs/ca.pem ${BENCH_COMMON}" + echo "[${EXPERIMENT}] quic-control offload=${offload} loss=${loss}%" + run_ablation_cell "quic-control_offload-${offload}_loss${loss}pct" \ + "--url quic://${BROKER_IP}:14567 --ca-cert /opt/mqtt-certs/ca.pem --quic-stream-strategy control-only ${BENCH_COMMON}" + clear_netem + done +done + +stop_broker +echo "experiment ${EXPERIMENT} complete (group ${GROUP})" diff --git a/publications/comnet/experiments/parallel/03d_offload_lowload.sh b/publications/comnet/experiments/parallel/03d_offload_lowload.sh new file mode 100644 index 00000000..e09df480 --- /dev/null +++ b/publications/comnet/experiments/parallel/03d_offload_lowload.sh @@ -0,0 +1,79 @@ +#!/usr/bin/env bash +set -euo pipefail +source "$(dirname "${BASH_SOURCE[0]}")/common_parallel.sh" + +EXPERIMENT="03d_offload_lowload_g${GROUP}" +LOSSES=(${LOSSES_OVERRIDE:-0 1 2 5 10}) +DELAY=10 +RUNS_PER_DATAPOINT="${RUNS_OVERRIDE:-5}" + +RESULTS_DIR="${ROOT_DIR}/results-v5" +OUTPUT_DIR="${RESULTS_DIR}/${EXPERIMENT}" +mkdir -p "$OUTPUT_DIR" + +DEV_CMD='DEV=$(ip route show default | awk "{print \$5}" | head -1)' + +set_offload() { + local mode="$1" + if [ "$mode" = "off" ]; then + ssh_broker "${DEV_CMD}; for f in gso tso gro lro tx-udp-segmentation rx-gro-list; do sudo ethtool -K \$DEV \$f off 2>/dev/null || true; done; echo '--- ethtool -k after off ---'; ethtool -k \$DEV | grep -E 'tcp-segmentation|generic-segmentation|generic-receive|large-receive|udp-segmentation'" + else + ssh_broker "${DEV_CMD}; for f in gso tso gro; do sudo ethtool -K \$DEV \$f on 2>/dev/null || true; done; echo '--- ethtool -k after on ---'; ethtool -k \$DEV | grep -E 'tcp-segmentation|generic-segmentation|generic-receive|large-receive|udp-segmentation'" + fi +} + +qdisc_snapshot() { ssh_broker "${DEV_CMD}; tc -s qdisc show dev \$DEV"; } + +run_ablation_cell() { + local label="$1" + shift + local bench_args="$*" + for run in $(seq 1 "$RUNS_PER_DATAPOINT"); do + local rl="${label}_run${run}" + local f="${OUTPUT_DIR}/${rl}.json" + if [ -s "$f" ] && [ "$(wc -c < "$f")" -gt 300 ]; then + continue + fi + if [ "$BROKER_FRESH" = "1" ]; then + BROKER_FRESH=0 + elif ! restart_broker; then + echo "WARN: broker restart failed, skipping ${rl}" >&2 + continue + fi + start_monitors + qdisc_snapshot > "${OUTPUT_DIR}/${rl}_qdisc_before.txt" 2>/dev/null || true + run_bench_split "$EXPERIMENT" "$rl" "$bench_args" + qdisc_snapshot > "${OUTPUT_DIR}/${rl}_qdisc_after.txt" 2>/dev/null || true + stop_monitors "$OUTPUT_DIR" "$rl" + sleep 5 + done +} + +BENCH_COMMON="--mode throughput --duration 60 --warmup 5 --payload-size 256 --qos 0 --publishers 1 --subscribers 1 --inflight 64" + +for offload in ${OFFLOAD_OVERRIDE:-on off}; do + broker_flags="--tls-cert /opt/mqtt-certs/server.pem --tls-key /opt/mqtt-certs/server.key --quic-host 0.0.0.0:14567" + if [ "$offload" = "off" ]; then + broker_flags="${broker_flags} --quic-disable-offload" + fi + stop_broker 2>/dev/null || true + echo "[${EXPERIMENT}] setting NIC offload ${offload} on broker" + set_offload "$offload" | tee "${OUTPUT_DIR}/ethtool_${offload}.txt" + if ! start_broker "$broker_flags"; then + echo "ERROR: broker failed to start (offload ${offload})" >&2 + continue + fi + for loss in "${LOSSES[@]}"; do + apply_netem "$DELAY" "$loss" + echo "[${EXPERIMENT}] tls offload=${offload} loss=${loss}%" + run_ablation_cell "tls_offload-${offload}_loss${loss}pct" \ + "--url mqtts://${BROKER_IP}:8883 --ca-cert /opt/mqtt-certs/ca.pem ${BENCH_COMMON}" + echo "[${EXPERIMENT}] quic-control offload=${offload} loss=${loss}%" + run_ablation_cell "quic-control_offload-${offload}_loss${loss}pct" \ + "--url quic://${BROKER_IP}:14567 --ca-cert /opt/mqtt-certs/ca.pem --quic-stream-strategy control-only ${BENCH_COMMON}" + clear_netem + done +done + +stop_broker +echo "experiment ${EXPERIMENT} complete (group ${GROUP})" diff --git a/publications/comnet/experiments/parallel/03e_infra.sh b/publications/comnet/experiments/parallel/03e_infra.sh new file mode 100755 index 00000000..ccab3cce --- /dev/null +++ b/publications/comnet/experiments/parallel/03e_infra.sh @@ -0,0 +1,226 @@ +#!/usr/bin/env bash +set -euo pipefail +source "$(dirname "${BASH_SOURCE[0]}")/common_parallel.sh" + +ACTION="${1:?usage: GROUP=G $0 up|tools|guest|health|smoke|hop|route-off|route-on|down|down-shared}" + +ZONE="us-west1-b" +NETWORK="default" +ROUTER="exp3-router-${GROUP}" +BROKER_VM="mqoq-broker-${GROUP}" +BROKER_TAG="exp3-brk-${GROUP}" +IG="exp3-ig-${GROUP}" +BACKEND="exp3-bs-${GROUP}" +RULE="exp3-fr-${GROUP}" +HEALTH="exp3-hc" +FIREWALL="exp3-allow-hc" +BPFTRACE_URL="https://github.com/bpftrace/bpftrace/releases/download/v0.27.0/bpftrace-x86_64" +: "${PUB_INTERNAL_IP:?Set PUB_INTERNAL_IP in group${GROUP}.env}" +: "${SUB_INTERNAL_IP:?Set SUB_INTERNAL_IP in group${GROUP}.env}" + +exists() { gc "$@" --format='value(name)' >/dev/null 2>&1; } + +pbr_name() { echo "exp3-g${GROUP}-${1//./-}"; } + +router_internal_ip() { + gc compute instances describe "$ROUTER" --zone="$ZONE" --format='value(networkInterfaces[0].networkIP)' +} + +clear_broker_qdisc() { + ssh_broker "sudo bash ${REMOTE_EXPERIMENTS_DIR}/netem/clear.sh" +} + +up() { + local image + image=$(gc compute disks describe "$(gc compute instances describe "$BROKER_VM" --zone="$ZONE" \ + --format='value(disks[0].source.basename())')" --zone="$ZONE" --format='value(sourceImage)') + echo "router image: ${image}" + + exists compute instances describe "$ROUTER" --zone="$ZONE" || \ + gc compute instances create "$ROUTER" --zone="$ZONE" --machine-type=n2-standard-4 --can-ip-forward \ + --no-address --image="$image" --subnet="$NETWORK" --tags=exp3-router + exists compute firewall-rules describe "$FIREWALL" || \ + gc compute firewall-rules create "$FIREWALL" --network="$NETWORK" --allow=tcp:22 \ + --source-ranges=35.191.0.0/16,130.211.0.0/22 --target-tags=exp3-router + exists compute health-checks describe "$HEALTH" --region="$GCP_REGION" || \ + gc compute health-checks create tcp "$HEALTH" --region="$GCP_REGION" --port=22 + exists compute instance-groups unmanaged describe "$IG" --zone="$ZONE" || \ + gc compute instance-groups unmanaged create "$IG" --zone="$ZONE" + gc compute instance-groups unmanaged list-instances "$IG" --zone="$ZONE" \ + --format='value(instance.basename())' | grep -qx "$ROUTER" || \ + gc compute instance-groups unmanaged add-instances "$IG" --zone="$ZONE" --instances="$ROUTER" + exists compute backend-services describe "$BACKEND" --region="$GCP_REGION" || \ + gc compute backend-services create "$BACKEND" --region="$GCP_REGION" --load-balancing-scheme=INTERNAL \ + --protocol=UNSPECIFIED --health-checks="$HEALTH" --health-checks-region="$GCP_REGION" + gc compute backend-services describe "$BACKEND" --region="$GCP_REGION" \ + --format='value(backends[].group)' | grep -q "instanceGroups/${IG}" || \ + gc compute backend-services add-backend "$BACKEND" --region="$GCP_REGION" \ + --instance-group="$IG" --instance-group-zone="$ZONE" + exists compute forwarding-rules describe "$RULE" --region="$GCP_REGION" || \ + gc compute forwarding-rules create "$RULE" --region="$GCP_REGION" --load-balancing-scheme=INTERNAL \ + --network="$NETWORK" --subnet="$NETWORK" --ip-protocol=L3_DEFAULT --ports=ALL --backend-service="$BACKEND" + + local ilb dst + ilb=$(gc compute forwarding-rules describe "$RULE" --region="$GCP_REGION" --format='value(IPAddress)') + echo "ilb next hop: ${ilb}" + for dst in "$PUB_INTERNAL_IP" "$SUB_INTERNAL_IP"; do + exists network-connectivity policy-based-routes describe "$(pbr_name "$dst")" || \ + gc network-connectivity policy-based-routes create "$(pbr_name "$dst")" \ + --network="projects/${GCP_PROJECT}/global/networks/${NETWORK}" \ + --source-range="${BROKER_IP}/32" --destination-range="${dst}/32" \ + --ip-protocol=ALL --protocol-version=IPV4 --next-hop-ilb-ip="$ilb" \ + --tags="$BROKER_TAG" --priority=100 + done + echo "router internal ip: $(router_internal_ip) (set ROUTER_IP in group${GROUP}.env, then run tools, guest, route-on)" +} + +extract_bpftrace() { + echo "test -x /opt/bpftrace/AppRun || { cd /tmp && chmod 755 bpftrace.appimage && rm -rf squashfs-root && ./bpftrace.appimage --appimage-extract >/dev/null && sudo rm -rf /opt/bpftrace && sudo mv /tmp/squashfs-root /opt/bpftrace; }" +} + +tools() { + local install='sudo tee /usr/local/bin/bpftrace >/dev/null <<< $'"'"'#!/bin/sh\nexec /opt/bpftrace/AppRun "$@"'"'"' && sudo chmod 755 /usr/local/bin/bpftrace && sudo modprobe sch_netem && sudo modprobe sch_tbf' + ssh_broker "test -s /tmp/bpftrace.appimage || curl -fsSL -o /tmp/bpftrace.appimage ${BPFTRACE_URL}" + ssh_broker "$(extract_bpftrace) && ${install}" + if ! ssh_router "test -x /opt/bpftrace/AppRun"; then + ssh -A -i "$SSH_KEY_PATH" $SSH_OPTS "${SSH_USER}@${BROKER_SSH_IP}" \ + "scp -q -o StrictHostKeyChecking=no /tmp/bpftrace.appimage ${SSH_USER}@${ROUTER_IP}:/tmp/bpftrace.appimage" + ssh_router "$(extract_bpftrace)" + fi + ssh_router "$install" + ssh_broker "sudo bpftrace --version; uname -r" + ssh_router "sudo bpftrace --version; uname -r" +} + +guest() { + COPYFILE_DISABLE=1 tar --no-xattrs -C "$ROOT_DIR" -czf - netem monitor | \ + ssh_router "sudo mkdir -p ${REMOTE_EXPERIMENTS_DIR} && sudo chown \$(id -u):\$(id -g) ${REMOTE_EXPERIMENTS_DIR} && tar -xzmf - -C ${REMOTE_EXPERIMENTS_DIR}" + ssh_router "sudo bash ${REMOTE_EXPERIMENTS_DIR}/netem/router_setup.sh ${BROKER_IP} ${PUB_INTERNAL_IP} ${SUB_INTERNAL_IP} ${ROUTER_GUARD:-0}" +} + +health() { + local state + state=$(ilb_health) + echo "ilb backend health: ${state}" + [ "$state" = "HEALTHY" ] +} + +ping_loss() { + ssh_broker "ping -q -c ${2:-500} -i 0.01 $1" | sed -n 's/.* \([0-9.]*\)% packet loss.*/\1/p' +} + +smoke() { + local path dst loss + health + clear_broker_qdisc + ssh_router "sudo bash ${REMOTE_EXPERIMENTS_DIR}/netem/router_loss.sh 0" + path=$(router_path_state) + echo "router path: ${path}" + [ "$path" = "on" ] || { echo "SMOKE FAIL: broker traffic is not crossing the router" >&2; return 1; } + for dst in "$SUB_INTERNAL_IP" "$PUB_INTERNAL_IP"; do + loss=$(ping_loss "$dst") || loss="" + echo "broker -> ${dst} through router: ${loss:-no reply}% loss" + [ "$loss" = "0" ] || { echo "SMOKE FAIL: packets lost or not delivered to ${dst}" >&2; return 1; } + done + echo "SMOKE PASS (ICMP; loaded TCP/UDP accounting is gated per run by the calib phase)" +} + +wait_for_path() { + local want="$1" attempt state + for attempt in $(seq 1 30); do + state=$(router_path_state) + if [ "$state" = "$want" ]; then + echo "router path ${want} (attempt ${attempt})" + return 0 + fi + sleep 10 + done + echo "router path did not become ${want} (last: ${state})" >&2 + return 1 +} + +route_off() { + gc compute instances remove-tags "$BROKER_VM" --zone="$ZONE" --tags="$BROKER_TAG" + wait_for_path off +} + +route_on() { + gc compute instances add-tags "$BROKER_VM" --zone="$ZONE" --tags="$BROKER_TAG" + wait_for_path on +} + +median_rtt_ms() { + ssh_broker "ping -c 1000 -i 0.005 $1" | sed -n 's/.*time=\([0-9.]*\) ms/\1/p' | sort -n | \ + awk '{a[NR] = $1} END { if (NR == 0) exit 1; print a[int((NR + 1) / 2)] }' +} + +hop() { + local with without + clear_broker_qdisc + ssh_router "sudo bash ${REMOTE_EXPERIMENTS_DIR}/netem/router_loss.sh 0" + wait_for_path on + with=$(median_rtt_ms "$SUB_INTERNAL_IP") + trap 'route_on' EXIT + route_off + without=$(median_rtt_ms "$SUB_INTERNAL_IP") + trap - EXIT + route_on + if ! awk -v w="$with" -v o="$without" 'BEGIN { exit !(w > o && o > 0) }'; then + echo "HOP FAIL: via router ${with} ms is not above direct ${without} ms" >&2 + return 1 + fi + awk -v w="$with" -v o="$without" -v g="$GROUP" \ + 'BEGIN { printf "group %s median rtt via router %.3f ms, direct %.3f ms, ROUTER_HOP_US=%d\n", g, w, o, (w - o) * 1000 + 0.5 }' +} + +down() { + local dst leftover="" + clear_broker_qdisc || true + gc compute instances remove-tags "$BROKER_VM" --zone="$ZONE" --tags="$BROKER_TAG" || true + for dst in "$PUB_INTERNAL_IP" "$SUB_INTERNAL_IP"; do + gc network-connectivity policy-based-routes delete "$(pbr_name "$dst")" || true + done + gc compute forwarding-rules delete "$RULE" --region="$GCP_REGION" || true + gc compute backend-services delete "$BACKEND" --region="$GCP_REGION" || true + gc compute instance-groups unmanaged delete "$IG" --zone="$ZONE" || true + gc compute instances delete "$ROUTER" --zone="$ZONE" || true + for dst in "$PUB_INTERNAL_IP" "$SUB_INTERNAL_IP"; do + exists network-connectivity policy-based-routes describe "$(pbr_name "$dst")" && leftover+=" $(pbr_name "$dst")" + done + exists compute forwarding-rules describe "$RULE" --region="$GCP_REGION" && leftover+=" ${RULE}" + exists compute backend-services describe "$BACKEND" --region="$GCP_REGION" && leftover+=" ${BACKEND}" + exists compute instance-groups unmanaged describe "$IG" --zone="$ZONE" && leftover+=" ${IG}" + exists compute instances describe "$ROUTER" --zone="$ZONE" && leftover+=" ${ROUTER}" + if [ -n "$leftover" ]; then + echo "TEARDOWN INCOMPLETE:${leftover}" >&2 + return 1 + fi + echo "group ${GROUP} torn down; shared ${HEALTH} and ${FIREWALL} remain until: GROUP=${GROUP} $0 down-shared" +} + +down_shared() { + gc compute health-checks delete "$HEALTH" --region="$GCP_REGION" || true + gc compute firewall-rules delete "$FIREWALL" || true + local leftover="" + exists compute health-checks describe "$HEALTH" --region="$GCP_REGION" && leftover+=" ${HEALTH}" + exists compute firewall-rules describe "$FIREWALL" && leftover+=" ${FIREWALL}" + if [ -n "$leftover" ]; then + echo "TEARDOWN INCOMPLETE:${leftover}" >&2 + return 1 + fi + echo "shared resources removed" +} + +case "$ACTION" in + up) up ;; + tools) tools ;; + guest) guest ;; + health) health ;; + smoke) smoke ;; + hop) hop ;; + route-off) route_off ;; + route-on) route_on ;; + down) down ;; + down-shared) down_shared ;; + *) echo "unknown action ${ACTION}" >&2; exit 1 ;; +esac diff --git a/publications/comnet/experiments/parallel/03e_plan.py b/publications/comnet/experiments/parallel/03e_plan.py new file mode 100644 index 00000000..f06b16e0 --- /dev/null +++ b/publications/comnet/experiments/parallel/03e_plan.py @@ -0,0 +1,146 @@ +import argparse +import csv +import random +import sys +from pathlib import Path + +LOSSES = [0, 1, 2, 5, 10] +SWEEP_CONFIGS = ["tcp", "tls", "quic-main", "quic-main-ppub", "quic-main-ptopic", "quic-ctl", "quic-ppub"] +LEGACY_CONFIGS = ["tcp", "tls", "quic-main"] +HOL_CONFIGS = ["tcp", "quic-ctl", "quic-ptopic", "quic-ppub"] +FIELDS = ["order", "phase", "mode", "config", "loss", "run", "workload", "router_probe", "broker_probe", "rate"] +CAPPED_CONFIGS = ["quic-ppub", "quic-main-ppub", "quic-ctl"] +CAPPED_RATES = [1250, 2500, 5000, 10000, 20000] +PACED_RATES = [2500, 20000, 0] + + +def row(phase, mode, config, loss, run, workload="tput", router_probe=0, broker_probe=0, rate=0): + return { + "phase": phase, "mode": mode, "config": config, "loss": loss, "run": run, "workload": workload, + "router_probe": router_probe, "broker_probe": broker_probe, "rate": rate, + } + + +def calibration(phase): + mode_runs = {"calib": ["legacy-calib", "single"], "calib-direct": ["single"]}[phase] + return [row(phase, mode, config, 0, run) for run in (1, 2, 3) for mode in mode_runs for config in LEGACY_CONFIGS] + + +def accuracy(rng): + rows = [row("accuracy", "single", config, loss, 1, router_probe=1) + for loss in (1, 5, 10) for config in SWEEP_CONFIGS] + rows += [row("accuracy", "legacy", config, 10, 1, broker_probe=1) for config in LEGACY_CONFIGS] + rng.shuffle(rows) + return rows + + +def main_sweep(rng): + rows = [] + for run in range(1, 6): + losses = LOSSES[:] + rng.shuffle(losses) + for loss in losses: + block = [row("main", mode, config, loss, run, router_probe=int(run == 1)) + for mode in ("single", "router") for config in SWEEP_CONFIGS] + if run <= 3 and loss > 0: + block += [row("main", "legacy", config, loss, run) for config in LEGACY_CONFIGS] + rng.shuffle(block) + rows += block + return rows + + +def router_check(rng): + rows = [row("routercheck", "router", config, loss, 1, router_probe=1) + for loss in (0, 1) for config in LEGACY_CONFIGS] + rng.shuffle(rows) + return rows + + +def capped(rng): + rows = [] + for run in (1, 2, 3): + block = [row("capped", "router", config, loss, run, rate=rate) + for config in CAPPED_CONFIGS for loss in (0, 1) for rate in CAPPED_RATES] + block += [row("capped", "router", config, loss, run, rate=0) for config in CAPPED_CONFIGS for loss in (0, 1)] + rng.shuffle(block) + rows += block + return rows + + +def paced(rng): + rows = [] + for run in (1, 2): + block = [row("paced", "router", config, loss, run, rate=rate) + for config in LEGACY_CONFIGS for loss in (1, 10) for rate in PACED_RATES] + rng.shuffle(block) + rows += block + return rows + + +def buildcheck(phase, block, rng): + rows = [row(phase, "router", config, 10, block, rate=0) for config in LEGACY_CONFIGS] + rng.shuffle(rows) + return rows + + +def crosscheck(rng): + rows = [] + for run in (1, 2, 3): + block = [row("crosscheck", "tbf", config, loss, run, broker_probe=int(run == 3)) + for loss in (0, 1, 5, 10) for config in LEGACY_CONFIGS] + rng.shuffle(block) + rows += block + return rows + + +def hol(rng): + rows = [] + for run in range(1, 6): + block = [row("hol", "single", config, loss, run, workload="hol") for loss in (0, 5) for config in HOL_CONFIGS] + rng.shuffle(block) + rows += block + return rows + + +def build(phase, group, block): + rng = random.Random(f"03e-{phase}-g{group}-b{block}") + builders = { + "calib": lambda: calibration("calib"), + "calib-direct": lambda: calibration("calib-direct"), + "accuracy": lambda: accuracy(rng), + "main": lambda: main_sweep(rng), + "crosscheck": lambda: crosscheck(rng), + "routercheck": lambda: router_check(rng), + "capped": lambda: capped(rng), + "paced": lambda: paced(rng), + "hol": lambda: hol(rng), + "bcfleet": lambda: buildcheck("bcfleet", block, rng), + "bclater": lambda: buildcheck("bclater", block, rng), + } + rows = builders[phase]() + for index, entry in enumerate(rows, start=1): + entry["order"] = index + return rows + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--group", type=int, required=True) + parser.add_argument("--phase", required=True, + choices=["calib", "calib-direct", "accuracy", "routercheck", "main", "crosscheck", "hol", "capped", "paced", "bcfleet", "bclater"]) + parser.add_argument("--block", type=int, default=1) + parser.add_argument("--out", type=Path, required=True) + args = parser.parse_args() + if args.out.exists(): + print(f"plan exists, keeping it: {args.out}", file=sys.stderr) + return + args.out.parent.mkdir(parents=True, exist_ok=True) + with open(args.out, "w", newline="") as handle: + writer = csv.DictWriter(handle, fieldnames=FIELDS, lineterminator="\n") + writer.writeheader() + writer.writerows(build(args.phase, args.group, args.block)) + print(f"wrote {args.out}") + + +if __name__ == "__main__": + main() diff --git a/publications/comnet/experiments/parallel/03e_single_segment.sh b/publications/comnet/experiments/parallel/03e_single_segment.sh new file mode 100755 index 00000000..ecc57095 --- /dev/null +++ b/publications/comnet/experiments/parallel/03e_single_segment.sh @@ -0,0 +1,424 @@ +#!/usr/bin/env bash +set -euo pipefail +source "$(dirname "${BASH_SOURCE[0]}")/common_parallel.sh" + +: "${PHASE:?Set PHASE=calib|calib-direct|accuracy|routercheck|main|crosscheck|hol|capped|paced|bcfleet|bclater}" +: "${ROUTER_IP:?Set ROUTER_IP in group${GROUP}.env}" +: "${ROUTER_HOP_US:?Set ROUTER_HOP_US (measured router one-way delay, microseconds) in group${GROUP}.env}" +: "${PUB_INTERNAL_IP:?Set PUB_INTERNAL_IP in group${GROUP}.env}" +: "${SUB_INTERNAL_IP:?Set SUB_INTERNAL_IP in group${GROUP}.env}" +ROUTER_GUARD="${ROUTER_GUARD:-0}" +if [ "$ROUTER_HOP_US" -lt 0 ] || [ "$ROUTER_HOP_US" -ge 10000 ]; then + echo "ROUTER_HOP_US=${ROUTER_HOP_US} is outside [0, 10000)" >&2 + exit 1 +fi + +if [ "$PHASE" = "capped" ]; then + EXPERIMENT="03f_capped_g${GROUP}" +elif [ "$PHASE" = "paced" ]; then + EXPERIMENT="03g_paced_g${GROUP}" +elif [ "$PHASE" = "bcfleet" ] || [ "$PHASE" = "bclater" ]; then + EXPERIMENT="03h_buildcheck_g${GROUP}" +else + EXPERIMENT="03e_single_segment_g${GROUP}" +fi +RESULTS_DIR="${ROOT_DIR}/results-v5" +OUTPUT_DIR="${RESULTS_DIR}/${EXPERIMENT}" +PLAN="${OUTPUT_DIR}/plan_${PHASE}${BLOCK:+_b${BLOCK}}.csv" +MANIFEST="${OUTPUT_DIR}/manifest.csv" +REMOTE_DIR="$REMOTE_EXPERIMENTS_DIR" +mkdir -p "$OUTPUT_DIR" + +if [ "$PHASE" = "calib-direct" ]; then + EXPECTED_PATH="off" + HOP_US=0 +else + EXPECTED_PATH="on" + HOP_US="$ROUTER_HOP_US" +fi + +CA="--ca-cert /opt/mqtt-certs/ca.pem" +BROKER_BASE="--tls-cert /opt/mqtt-certs/server.pem --tls-key /opt/mqtt-certs/server.key --quic-host 0.0.0.0:14567" +TPUT_ARGS="--mode throughput --duration 60 --warmup 5 --payload-size 256 --qos 0 --publishers 16 --subscribers 8 --inflight 64" +HOL_ARGS="--mode hol-blocking --topics 8 --duration 60 --warmup 5 --payload-size 256 --rate 500 --trace-dir /tmp/hol-traces" +QUIC_URL="--url quic://${BROKER_IP}:14567 ${CA}" + +declare -A CLIENT_ARGS=( + [tcp]="--url mqtt://${BROKER_IP}:1883" + [tls]="--url mqtts://${BROKER_IP}:8883 ${CA}" + [quic-main]="${QUIC_URL} --quic-stream-strategy control-only" + [quic-main-ppub]="${QUIC_URL} --quic-stream-strategy per-publish" + [quic-main-ptopic]="${QUIC_URL} --quic-stream-strategy per-topic" + [quic-ctl]="${QUIC_URL} --quic-stream-strategy control-only" + [quic-ptopic]="${QUIC_URL} --quic-stream-strategy per-topic" + [quic-ppub]="${QUIC_URL} --quic-stream-strategy per-publish" +) +declare -A BROKER_EXTRA=( + [tcp]="" + [tls]="" + [quic-main]="" + [quic-main-ppub]="" + [quic-main-ptopic]="" + [quic-ctl]="--quic-delivery-strategy control-only" + [quic-ptopic]="--quic-delivery-strategy per-topic" + [quic-ppub]="--quic-delivery-strategy per-publish" +) +declare -A BASE_DELAY_US=([tput]=10000 [hol]=25000) +RERUN_SHA="7c65659d5e768d8a38f251bfbdb96c0ee9bdcaa79e2fd5eda36ac4f644377a07" +FOLLOWUP_SHA="f7f2afe469b3328b0b4929dfc14031c47d889b655c9c04f9956bc86912f734d7" +rated_phase() { + case "$1" in + capped|paced|bcfleet|bclater) return 0 ;; + *) return 1 ;; + esac +} +if [ "$PHASE" = "bcfleet" ]; then + EXPECTED_SHA="$RERUN_SHA" +else + EXPECTED_SHA="$FOLLOWUP_SHA" +fi +if rated_phase "$PHASE"; then + MANIFEST_HEADER="label,order,phase,mode,config,loss,run,workload,router_probe,broker_probe,rate,delay_us,router_hop_us,router_guard,path_state,ilb_health,ilb_health_after,started,finished,binary_sha,broker_flags,bench_args" +else + EXPECTED_SHA="$RERUN_SHA" + MANIFEST_HEADER="label,order,phase,mode,config,loss,run,workload,router_probe,broker_probe,delay_us,router_hop_us,router_guard,path_state,ilb_health,ilb_health_after,started,finished,broker_flags,bench_args" +fi +SKIPPED=0 + +scp_router() { + scp -q -i "$SSH_KEY_PATH" -o StrictHostKeyChecking=no -o ProxyJump="${SSH_USER}@${BROKER_SSH_IP}" \ + "${SSH_USER}@${ROUTER_IP}:$1" "$2" +} + +scp_broker() { + scp -q -i "$SSH_KEY_PATH" -o StrictHostKeyChecking=no "${SSH_USER}@${BROKER_SSH_IP}:$1" "$2" +} + +deploy_scripts() { + local host + for host in ssh_broker ssh_pub ssh_sub ssh_router; do + COPYFILE_DISABLE=1 tar --no-xattrs -C "$ROOT_DIR" -czf - netem monitor | \ + "$host" "sudo mkdir -p ${REMOTE_DIR} && sudo chown \$(id -u):\$(id -g) ${REMOTE_DIR} && tar -xzmf - -C ${REMOTE_DIR}" + done +} + +stop_leftover_probes() { + ssh_broker "sudo pkill -INT -x bpftrace; for _ in \$(seq 1 20); do pgrep -x bpftrace >/dev/null || break; sleep 0.5; done; ! pgrep -x bpftrace" || return 1 + ssh_router "sudo pkill -INT -x bpftrace; for _ in \$(seq 1 20); do pgrep -x bpftrace >/dev/null || break; sleep 0.5; done; ! pgrep -x bpftrace" || return 1 +} + +reset_impairment() { + ssh_broker "sudo bash ${REMOTE_DIR}/netem/clear.sh" >/dev/null || return 1 + ssh_router "sudo bash ${REMOTE_DIR}/netem/router_loss.sh 0" >/dev/null || return 1 +} + +ilb_healthy() { + local state + state=$(ilb_health) + echo "$state" + [ "$state" = "HEALTHY" ] +} + +preflight_step() { + local report="$1" name="$2" + shift 2 + echo "=== ${name} ===" >> "$report" + if ! "$@" >> "$report" 2>&1; then + echo "PREFLIGHT FAILED at '${name}', see ${report}" >&2 + tail -n 20 "$report" >&2 + exit 1 + fi +} + +host_report() { + "$1" "uname -r; sha256sum ${REMOTE_DIR}/netem/* ${REMOTE_DIR}/monitor/*; ethtool -k ens4; mqttv5 --version; sha256sum \$(readlink -f \$(command -v mqttv5))" +} + +binary_check() { + local host sha + for host in ssh_broker ssh_pub ssh_sub; do + sha=$("$host" "sha256sum \$(readlink -f \$(command -v mqttv5)) | cut -d' ' -f1" < /dev/null) || return 1 + echo "${host} ${sha}" + if [ "$sha" != "$EXPECTED_SHA" ]; then + echo "binary mismatch on ${host}: ${sha} (expected ${EXPECTED_SHA})" >&2 + return 1 + fi + done +} + +manifest_check() { + if [ -f "$MANIFEST" ] && [ "$(head -1 "$MANIFEST")" != "$MANIFEST_HEADER" ]; then + echo "manifest header mismatch in ${MANIFEST}" >&2 + return 1 + fi +} + +preflight() { + local report path + report="${OUTPUT_DIR}/preflight_${PHASE}_$(date +%Y%m%dT%H%M%S).txt" + preflight_step "$report" "local git" git -C "$ROOT_DIR" log -1 --format=%H + preflight_step "$report" "local changes" git -C "$ROOT_DIR" status --short -- netem monitor parallel analysis + preflight_step "$report" "manifest header" manifest_check + preflight_step "$report" "binary" binary_check + preflight_step "$report" "leftover probes" stop_leftover_probes + preflight_step "$report" "broker" host_report ssh_broker + preflight_step "$report" "pub" host_report ssh_pub + preflight_step "$report" "sub" host_report ssh_sub + preflight_step "$report" "router" ssh_router "uname -r; sha256sum ${REMOTE_DIR}/netem/* ${REMOTE_DIR}/monitor/*; ethtool -k ens4; command -v bpftrace; cat /sys/class/net/ens4/device/features" + preflight_step "$report" "router setup" ssh_router "sudo bash ${REMOTE_DIR}/netem/router_setup.sh ${BROKER_IP} ${PUB_INTERNAL_IP} ${SUB_INTERNAL_IP} ${ROUTER_GUARD}" + preflight_step "$report" "impairment reset" reset_impairment + preflight_step "$report" "ilb health" ilb_healthy + path=$(router_path_state) + echo "=== router path: ${path} (expected ${EXPECTED_PATH}) ===" >> "$report" + if [ "$path" != "$EXPECTED_PATH" ]; then + echo "PREFLIGHT FAILED: router path is ${path}, phase ${PHASE} needs ${EXPECTED_PATH} (03e_infra.sh route-on/route-off)" >&2 + exit 1 + fi + echo "preflight ok: ${report}" +} + +set_impairment() { + local mode="$1" loss="$2" delay_us="$3" + case "$mode" in + single) + ssh_broker "sudo bash ${REMOTE_DIR}/netem/broker_netem.sh ${delay_us} 0 100000" + ssh_router "sudo bash ${REMOTE_DIR}/netem/router_loss.sh ${loss}" + ;; + legacy) + ssh_broker "sudo bash ${REMOTE_DIR}/netem/broker_netem.sh ${delay_us} ${loss} 100000" + ssh_router "sudo bash ${REMOTE_DIR}/netem/router_loss.sh 0" + ;; + legacy-calib) + ssh_broker "sudo bash ${REMOTE_DIR}/netem/broker_netem.sh ${delay_us} ${loss}" + ssh_router "sudo bash ${REMOTE_DIR}/netem/router_loss.sh 0" + ;; + router) + ssh_broker "sudo bash ${REMOTE_DIR}/netem/clear.sh" + ssh_router "sudo bash ${REMOTE_DIR}/netem/router_loss.sh ${loss} ${delay_us}" + ;; + tbf) + ssh_broker "sudo bash ${REMOTE_DIR}/netem/broker_tbf.sh ${delay_us} ${loss}" + ssh_router "sudo bash ${REMOTE_DIR}/netem/router_loss.sh 0" + ;; + *) + echo "unknown mode ${mode}" >&2 + return 1 + ;; + esac +} + +snapshot() { + local label="$1" when="$2" + ssh_broker "echo '=== tc ==='; tc -s -d qdisc show dev ens4; echo '=== snmp ==='; cat /proc/net/snmp; \ + echo '=== netstat ==='; cat /proc/net/netstat; echo '=== dev ==='; cat /proc/net/dev" \ + > "${OUTPUT_DIR}/${label}_broker_${when}.txt" 2>&1 || true + ssh_router "echo '=== tc ==='; tc -s -d qdisc show dev ens4; echo '=== ingress ==='; sudo bash ${REMOTE_DIR}/netem/router_ingress_count.sh; \ + echo '=== dev ==='; cat /proc/net/dev; echo '=== softnet ==='; cat /proc/net/softnet_stat; \ + echo '=== snmp ==='; cat /proc/net/snmp; echo '=== netstat ==='; cat /proc/net/netstat; echo '=== ethtool ==='; ethtool -S ens4" \ + > "${OUTPUT_DIR}/${label}_router_${when}.txt" 2>&1 || true +} + +ROUTER_PROBE_PID="" +BROKER_PROBE_PID="" +ROUTER_MONITOR_PID="" +BACKLOG_PID="" + +start_probe() { + local host="$1" script="$2" + "$host" "sudo nohup bpftrace ${REMOTE_DIR}/netem/${script} > /tmp/probe.txt 2>&1 & echo \$!" +} + +stop_probe() { + local host="$1" pid="$2" + "$host" "sudo kill -INT ${pid}; for _ in \$(seq 1 20); do sudo kill -0 ${pid} 2>/dev/null || break; sleep 0.5; done" || true +} + +start_side_monitors() { + ssh_router "pkill -f '[r]outer_monitor.sh'" 2>/dev/null || true + ssh_broker "pkill -f '[b]roker_backlog.sh'" 2>/dev/null || true + ROUTER_MONITOR_PID=$(ssh_router "nohup bash ${REMOTE_DIR}/monitor/router_monitor.sh > /tmp/router_monitor.csv 2>&1 & echo \$!") || ROUTER_MONITOR_PID="" + BACKLOG_PID=$(ssh_broker "nohup bash ${REMOTE_DIR}/netem/broker_backlog.sh > /tmp/broker_backlog.csv 2>&1 & echo \$!") || BACKLOG_PID="" +} + +stop_side_monitors() { + local label="$1" + if [ -n "$ROUTER_MONITOR_PID" ]; then + ssh_router "kill ${ROUTER_MONITOR_PID}" 2>/dev/null || true + fi + if [ -n "$BACKLOG_PID" ]; then + ssh_broker "kill ${BACKLOG_PID}" 2>/dev/null || true + fi + if [ -n "$label" ]; then + scp_router /tmp/router_monitor.csv "${OUTPUT_DIR}/${label}_router_resources.csv" || true + scp_broker /tmp/broker_backlog.csv "${OUTPUT_DIR}/${label}_broker_backlog.csv" || true + fi + ROUTER_MONITOR_PID="" + BACKLOG_PID="" +} + +stop_run_probes() { + local label="$1" + if [ -n "$ROUTER_PROBE_PID" ]; then + stop_probe ssh_router "$ROUTER_PROBE_PID" + [ -n "$label" ] && { scp_router /tmp/probe.txt "${OUTPUT_DIR}/${label}_router_probe.txt" || echo "WARN: router probe copy failed ${label}" >&2; } + ROUTER_PROBE_PID="" + fi + if [ -n "$BROKER_PROBE_PID" ]; then + stop_probe ssh_broker "$BROKER_PROBE_PID" + [ -n "$label" ] && { scp_broker /tmp/probe.txt "${OUTPUT_DIR}/${label}_broker_probe.txt" || echo "WARN: broker probe copy failed ${label}" >&2; } + BROKER_PROBE_PID="" + fi +} + +cleanup() { + trap - EXIT INT TERM + stop_run_probes "" + stop_side_monitors "" + stop_stale_monitors + stop_broker || true + ssh_broker "sudo bash ${REMOTE_DIR}/netem/clear.sh" >/dev/null 2>&1 || true + ssh_router "sudo bash ${REMOTE_DIR}/netem/router_loss.sh 0" >/dev/null 2>&1 || true +} + +run_hol() { + local label="$1" bench_args="$2" + ssh_pub "rm -rf /tmp/hol-traces; ulimit -n 65536; mqttv5 bench ${bench_args}" \ + > "${OUTPUT_DIR}/${label}.json" 2>/dev/null || true + warn_if_empty "${OUTPUT_DIR}/${label}.json" + local csv + for csv in messages.csv quinn_stats.csv; do + scp -q -i "$SSH_KEY_PATH" "${SSH_USER}@${PUB_IP}:/tmp/hol-traces/${csv}" \ + "${OUTPUT_DIR}/${label}_${csv}" 2>/dev/null || true + done +} + +json_complete() { + local f="${OUTPUT_DIR}/$1.json" + [ -f "$f" ] && [ "$(wc -c < "$f")" -gt 300 ] && python3 - "$f" <<'PY' +import json +import sys +results = json.load(open(sys.argv[1])).get("results", {}) +sys.exit(0 if (results.get("throughput_avg") or results.get("measured_rate") or 0) > 0 else 1) +PY +} + +run_recorded() { + [ -f "$MANIFEST" ] && grep -q "^$1," "$MANIFEST" && json_complete "$1" +} + +run_line() { + local order="$1" phase="$2" mode="$3" config="$4" loss="$5" run="$6" workload="$7" router_probe="$8" broker_probe="$9" rate="${10:-0}" + local label="${phase}_${mode}_${config}_loss${loss}pct_run${run}" + if rated_phase "$phase"; then + label="${phase}_${mode}_${config}_r${rate}_loss${loss}pct_run${run}" + fi + if run_recorded "$label"; then + echo " skip (complete): ${label}" + return 0 + fi + local delay_us=$((BASE_DELAY_US[$workload] - HOP_US)) + local broker_flags="${BROKER_BASE} ${BROKER_EXTRA[$config]}" + local bench_args + if [ "$workload" = "hol" ]; then + bench_args="${CLIENT_ARGS[$config]} ${HOL_ARGS}" + else + bench_args="${CLIENT_ARGS[$config]} ${TPUT_ARGS}" + fi + if [ "$rate" -gt 0 ]; then + bench_args="${bench_args} --rate ${rate}" + fi + echo "[${EXPERIMENT}] #${order} ${label} delay=${delay_us}us" + + rm -f "${OUTPUT_DIR}/${label}.json" "${OUTPUT_DIR}/${label}_pub.json" + stop_broker + reset_impairment + local path health health_after + path=$(router_path_state) + health=$(ilb_health) + if [ "$path" != "$EXPECTED_PATH" ] || [ "$health" != "HEALTHY" ]; then + echo "WARN: router path ${path} (expected ${EXPECTED_PATH}), ilb '${health}'; skipping ${label}" >&2 + SKIPPED=$((SKIPPED + 1)) + return 0 + fi + set_impairment "$mode" "$loss" "$delay_us" + if ! start_broker "$broker_flags"; then + echo "WARN: broker start failed, skipping ${label}" >&2 + SKIPPED=$((SKIPPED + 1)) + return 0 + fi + local started + started=$(date +%s) + + snapshot "$label" before + if [ "$router_probe" = "1" ]; then + ROUTER_PROBE_PID=$(start_probe ssh_router netem_probe.bt) || { ROUTER_PROBE_PID=""; echo "WARN: router probe start failed ${label}" >&2; } + fi + if [ "$broker_probe" = "1" ]; then + local broker_script="netem_probe.bt" + [ "$mode" = "tbf" ] && broker_script="tbf_probe.bt" + [ "$mode" = "router" ] || [ "$mode" = "single" ] || [ "$mode" = "legacy-calib" ] && broker_script="tcp_cwr_probe.bt" + BROKER_PROBE_PID=$(start_probe ssh_broker "$broker_script") || { BROKER_PROBE_PID=""; echo "WARN: broker probe start failed ${label}" >&2; } + fi + if [ -n "$ROUTER_PROBE_PID$BROKER_PROBE_PID" ]; then + sleep 3 + fi + start_monitors + start_side_monitors + + if [ "$workload" = "hol" ]; then + run_hol "$label" "$bench_args" + else + run_bench_split "$EXPERIMENT" "$label" "$bench_args" + fi + + stop_side_monitors "$label" + stop_monitors "$OUTPUT_DIR" "$label" + stop_run_probes "$label" + snapshot "$label" after + health_after=$(ilb_health) + + if ! json_complete "$label"; then + echo "WARN: no usable result for ${label}; not recorded, will retry on resume" >&2 + SKIPPED=$((SKIPPED + 1)) + sleep 5 + return 0 + fi + [ -f "$MANIFEST" ] || echo "$MANIFEST_HEADER" > "$MANIFEST" + if rated_phase "$phase"; then + echo "${label},${order},${phase},${mode},${config},${loss},${run},${workload},${router_probe},${broker_probe},${rate},${delay_us},${HOP_US},${ROUTER_GUARD},${path},\"${health}\",\"${health_after}\",${started},$(date +%s),${EXPECTED_SHA},\"${broker_flags}\",\"${bench_args}\"" >> "$MANIFEST" + else + echo "${label},${order},${phase},${mode},${config},${loss},${run},${workload},${router_probe},${broker_probe},${delay_us},${HOP_US},${ROUTER_GUARD},${path},\"${health}\",\"${health_after}\",${started},$(date +%s),\"${broker_flags}\",\"${bench_args}\"" >> "$MANIFEST" + fi + sleep 5 +} + +trap cleanup EXIT +trap 'cleanup; exit 130' INT +trap 'cleanup; exit 143' TERM + +case "$PHASE" in + main|crosscheck|hol) + if ! python3 "${ROOT_DIR}/analysis/single_segment_accept.py" "$OUTPUT_DIR"; then + if [ -z "${ACCEPT_OVERRIDE:-}" ]; then + echo "pilot acceptance has not passed for ${EXPERIMENT}; refusing PHASE=${PHASE} (set ACCEPT_OVERRIDE= to proceed)" >&2 + exit 1 + fi + echo "$(date -u +%FT%TZ) PHASE=${PHASE} proceeding despite failed pilot acceptance: ${ACCEPT_OVERRIDE}" | tee -a "${OUTPUT_DIR}/acceptance_overrides.log" >&2 + fi + ;; +esac + +python3 "${SCRIPT_DIR}/03e_plan.py" --group "$GROUP" --phase "$PHASE" ${BLOCK:+--block "$BLOCK"} --out "$PLAN" +deploy_scripts +preflight + +while IFS=, read -r order phase mode config loss run workload router_probe broker_probe rate; do + [ "$order" = "order" ] && continue + run_line "$order" "$phase" "$mode" "$config" "$loss" "$run" "$workload" "$router_probe" "$broker_probe" "${rate%$'\r'}" < /dev/null +done < "$PLAN" + +if [ "$SKIPPED" -gt 0 ]; then + echo "experiment ${EXPERIMENT} phase ${PHASE} INCOMPLETE: ${SKIPPED} runs skipped or unusable; rerun to retry them" >&2 + exit 2 +fi +echo "experiment ${EXPERIMENT} phase ${PHASE} complete" diff --git a/publications/comnet/experiments/parallel/03f_binary.sh b/publications/comnet/experiments/parallel/03f_binary.sh new file mode 100755 index 00000000..f216cf07 --- /dev/null +++ b/publications/comnet/experiments/parallel/03f_binary.sh @@ -0,0 +1,45 @@ +#!/usr/bin/env bash +set -euo pipefail +source "$(dirname "${BASH_SOURCE[0]}")/common_parallel.sh" + +ACTION="${1:?usage: GROUP=G $0 install | use | status}" + +on_hosts() { + local host + for host in ssh_broker ssh_pub ssh_sub; do + printf '%-10s ' "$host" + "$host" "$@" < /dev/null + done +} + +case "$ACTION" in + install) + BINARY="${2:?local binary path}" + NAME="${3:?installed name}" + EXPECTED=$(shasum -a 256 "$BINARY" | cut -d' ' -f1) + for host in ssh_broker ssh_pub ssh_sub; do + "$host" "cat > /tmp/${NAME} && sudo install -m 755 /tmp/${NAME} /opt/mqtt-bin/${NAME} && rm -f /tmp/${NAME}" < "$BINARY" + done + on_hosts "sha256sum /opt/mqtt-bin/${NAME} | cut -d' ' -f1" | grep -c "$EXPECTED" | \ + awk -v want=3 '{ if ($1 != want) { print "checksum mismatch on " want - $1 " host(s)"; exit 1 } else { print "installed on 3 hosts" } }' + ;; + use) + NAME="${2:?installed name}" + EXPECTED="${3:?expected sha256}" + for host in ssh_broker ssh_pub ssh_sub; do + if ! "$host" "test -x /opt/mqtt-bin/${NAME} && sha256sum /opt/mqtt-bin/${NAME} | grep -q ^${EXPECTED}" < /dev/null; then + echo "${host}: /opt/mqtt-bin/${NAME} missing or wrong sha; nothing switched" >&2 + exit 1 + fi + done + on_hosts "sudo ln -sfn /opt/mqtt-bin/${NAME} /usr/local/bin/mqttv5 && sha256sum \$(readlink -f /usr/local/bin/mqttv5) | cut -d' ' -f1" | tee /dev/stderr | \ + grep -c "$EXPECTED" | awk '{ if ($1 != 3) { print "switch incomplete: " $1 " of 3 hosts on the expected binary"; exit 1 } else { print "all 3 hosts on the expected binary" } }' + ;; + status) + on_hosts "readlink /usr/local/bin/mqttv5; ls /opt/mqtt-bin" + ;; + *) + echo "unknown action ${ACTION}" >&2 + exit 1 + ;; +esac diff --git a/publications/comnet/experiments/parallel/04_transport_limits_v5.sh b/publications/comnet/experiments/parallel/04_transport_limits_v5.sh new file mode 100644 index 00000000..30b571ae --- /dev/null +++ b/publications/comnet/experiments/parallel/04_transport_limits_v5.sh @@ -0,0 +1,147 @@ +#!/usr/bin/env bash +set -euo pipefail +source "$(dirname "${BASH_SOURCE[0]}")/common_parallel.sh" + +EXPERIMENT="04_transport_limits" +: "${PART:?Set PART=A|B|C}" +RUNS_PER_DATAPOINT="${RUNS_PER_DATAPOINT:-10}" +COLLECT_QUIC_STATS=1 + +RESULTS_DIR="${ROOT_DIR}/results-v5" +OUTPUT_DIR="${RESULTS_DIR}/${EXPERIMENT}" +mkdir -p "$OUTPUT_DIR" + +CA="--ca-cert /opt/mqtt-certs/ca.pem" +BROKER_BASE="--tls-cert /opt/mqtt-certs/server.pem --tls-key /opt/mqtt-certs/server.key --quic-host 0.0.0.0:14567" + +case "$PART" in + A) + CELLS=() + for strategy in control-only per-topic per-publish; do + for topics in 1 2 4 8 16; do + CELLS+=("${strategy} ${topics} def def 25 2 tput 0") + done + done + ;; + B) + CELLS=( + "per-publish 8 25 def 25 2 tput 0" + "per-publish 8 100 def 25 2 tput 0" + "per-publish 8 250 def 25 2 tput 0" + "per-publish 8 1000 def 25 2 tput 0" + "control-only 8 100 def 25 2 tput 0" + "control-only 8 1000 def 25 2 tput 0" + "per-topic 8 100 def 25 2 tput 0" + "per-topic 8 1000 def 25 2 tput 0" + "per-publish 8 100 def 10 2 tput 0" + "per-publish 8 100 def 50 2 tput 0" + "control-only 8 100 def 10 2 tput 0" + "control-only 8 100 def 50 2 tput 0" + "per-publish 8 100 def 25 5 hol 2000" + "per-publish 8 1000 def 25 5 hol 2000" + ) + ;; + C) + CELLS=() + for window in 131072 262144 1048576; do + CELLS+=("control-only 8 def ${window} 25 2 tput 0") + CELLS+=("per-topic 1 def ${window} 25 2 tput 0") + CELLS+=("per-topic 8 def ${window} 25 2 tput 0") + done + CELLS+=("control-only 8 def 262144 25 0 tput 0") + CELLS+=("per-topic 8 def 262144 25 0 tput 0") + CELLS+=("control-only 8 def 1048576 25 0 tput 0") + ;; + *) + echo "unknown PART ${PART}" >&2 + exit 1 + ;; +esac + +collect_broker_quic_stats() { + local run_label="$1" + local index=0 + local remote + for remote in $(ssh_broker "ls /tmp/quic-stats/*.csv 2>/dev/null" || true); do + index=$((index + 1)) + scp -q -i "$SSH_KEY_PATH" "${SSH_USER}@${BROKER_SSH_IP}:${remote}" \ + "${OUTPUT_DIR}/${run_label}_broker_quic_${index}.csv" || true + done +} + +run_hol_colocated() { + local run_label="$1" + shift + ssh_pub "ulimit -n 65536; mqttv5 bench $*" > "${OUTPUT_DIR}/${run_label}.json" 2>/dev/null || true + warn_if_empty "${OUTPUT_DIR}/${run_label}.json" +} + +for cell in "${CELLS[@]}"; do + read -r strategy topics streams window delay loss mode rate <<< "$cell" + label="${strategy}_t${topics}_s${streams}_w${window}_d${delay}_l${loss}_${mode}" + if [ "$mode" = "hol" ]; then + label="${label}_r${rate}" + fi + + broker_flags="${BROKER_BASE} --quic-delivery-strategy ${strategy}" + client_flags="" + if [ "$streams" != "def" ]; then + broker_flags="${broker_flags} --quic-max-streams ${streams}" + client_flags="--quic-max-streams ${streams}" + fi + if [ "$window" != "def" ]; then + broker_flags="${broker_flags} --quic-stream-window ${window}" + fi + + if [ "$mode" = "hol" ]; then + bench_args="--url quic://${BROKER_IP}:14567 ${CA} --quic-stream-strategy ${strategy} ${client_flags} --mode hol-blocking --topics ${topics} --duration 60 --warmup 5 --payload-size 256 --rate ${rate}" + else + bench_args="--url quic://${BROKER_IP}:14567 ${CA} --quic-stream-strategy ${strategy} ${client_flags} --mode throughput --duration 60 --warmup 5 --payload-size 256 --publishers 1 --topics ${topics} --subscribers 1 --inflight 64" + fi + + pending=0 + for run in $(seq 1 "$RUNS_PER_DATAPOINT"); do + if [ ! -s "${OUTPUT_DIR}/${label}_run${run}.json" ]; then + pending=$((pending + 1)) + fi + done + if [ "$pending" = "0" ]; then + echo "[${EXPERIMENT}:${PART}] ${label} complete, skipping" + continue + fi + + clear_netem 2>/dev/null || true + stop_broker 2>/dev/null || true + if ! start_broker "$broker_flags"; then + echo "WARN: broker start failed for ${label}, skipping" >&2 + continue + fi + apply_netem "$delay" "$loss" + echo "[${EXPERIMENT}:${PART}] ${label} (${pending} runs pending)" + + for run in $(seq 1 "$RUNS_PER_DATAPOINT"); do + run_label="${label}_run${run}" + if [ -s "${OUTPUT_DIR}/${run_label}.json" ]; then + continue + fi + if [ "$BROKER_FRESH" = "1" ]; then + BROKER_FRESH=0 + elif ! restart_broker; then + echo "WARN: broker restart failed, skipping ${run_label}" >&2 + continue + fi + start_monitors + if [ "$mode" = "hol" ]; then + run_hol_colocated "$run_label" "$bench_args" + else + run_bench_split "$EXPERIMENT" "$run_label" "$bench_args" + fi + stop_monitors "$OUTPUT_DIR" "$run_label" + collect_broker_quic_stats "$run_label" + sleep 5 + done + clear_netem +done + +stop_broker +echo "experiment ${EXPERIMENT} part ${PART} complete (group ${GROUP})" diff --git a/publications/comnet/experiments/parallel/04b_stream_limit_sweep.sh b/publications/comnet/experiments/parallel/04b_stream_limit_sweep.sh new file mode 100644 index 00000000..a8b08add --- /dev/null +++ b/publications/comnet/experiments/parallel/04b_stream_limit_sweep.sh @@ -0,0 +1,68 @@ +#!/usr/bin/env bash +set -euo pipefail +source "$(dirname "${BASH_SOURCE[0]}")/common_parallel.sh" + +EXPERIMENT="04b_stream_limit_sweep" +STRATEGIES=("control-only" "per-topic" "per-publish") +STREAM_LIMITS=(100 250 1000) +TOPICS=8 +DELAY=25 +LOSS=2 +RUNS_PER_DATAPOINT="${RUNS_PER_DATAPOINT:-5}" + +RESULTS_DIR="${ROOT_DIR}/results-v5" +OUTPUT_DIR="${RESULTS_DIR}/${EXPERIMENT}" +mkdir -p "$OUTPUT_DIR" + +BROKER_TLS="--tls-cert /opt/mqtt-certs/server.pem --tls-key /opt/mqtt-certs/server.key" +CA="--ca-cert /opt/mqtt-certs/ca.pem" + +run_hol_colocated() { + local label="$1" + shift + echo " running (co-located): ${label}" + ssh_pub "ulimit -n 65536; mqttv5 bench $*" \ + > "${OUTPUT_DIR}/${label}.json" 2>/dev/null || true + warn_if_empty "${OUTPUT_DIR}/${label}.json" +} + +for limit in "${STREAM_LIMITS[@]}"; do + for strategy in "${STRATEGIES[@]}"; do + clear_netem 2>/dev/null || true + stop_broker 2>/dev/null || true + if ! start_broker "${BROKER_TLS} --quic-host 0.0.0.0:14567 --quic-delivery-strategy ${strategy} --quic-max-streams ${limit}"; then + echo "WARN: broker start failed for ${strategy} limit ${limit}, skipping" >&2 + continue + fi + + apply_netem "$DELAY" "$LOSS" + label="${strategy}_limit${limit}_throughput" + echo "[${EXPERIMENT}] ${label}" + run_monitored_split "$EXPERIMENT" "$label" \ + "--url quic://${BROKER_IP}:14567 ${CA} --quic-stream-strategy ${strategy} --quic-max-streams ${limit} --mode throughput --duration 60 --warmup 5 --payload-size 256 --publishers 1 --topics ${TOPICS} --subscribers 1 --inflight 64" + + apply_netem "$DELAY" 5 + hol_label="${strategy}_limit${limit}_hol_r2000_loss5pct" + echo "[${EXPERIMENT}] ${hol_label}" + for run in $(seq 1 "$RUNS_PER_DATAPOINT"); do + run_label="${hol_label}_run${run}" + if [ -s "${OUTPUT_DIR}/${run_label}.json" ]; then + echo " skip (exists): ${run_label}" + continue + fi + if [ "$BROKER_FRESH" = "1" ]; then + BROKER_FRESH=0 + elif ! restart_broker; then + echo "WARN: broker restart failed, skipping ${run_label}" >&2 + continue + fi + run_hol_colocated "$run_label" \ + "--url quic://${BROKER_IP}:14567 ${CA} --quic-stream-strategy ${strategy} --quic-max-streams ${limit} --mode hol-blocking --topics ${TOPICS} --duration 60 --warmup 5 --payload-size 256 --rate 2000" + sleep 5 + done + clear_netem + done +done + +stop_broker +echo "experiment ${EXPERIMENT} complete (group ${GROUP})" diff --git a/publications/comnet/experiments/parallel/E1_fairness_v5.sh b/publications/comnet/experiments/parallel/E1_fairness_v5.sh new file mode 100755 index 00000000..75c96d73 --- /dev/null +++ b/publications/comnet/experiments/parallel/E1_fairness_v5.sh @@ -0,0 +1,155 @@ +#!/usr/bin/env bash +set -euo pipefail +source "$(dirname "${BASH_SOURCE[0]}")/common_parallel.sh" + +: "${PUB_INTERNAL_IP:?Set PUB_INTERNAL_IP (bottlenecked subnet address of the sub/pub host) in group${GROUP}.env}" + +EXPERIMENT="E1_fairness" +DELAY=25 +RATE_MBIT="${RATE_MBIT:-50}" +QUEUE_PKTS="${QUEUE_PKTS:-1000}" +read -ra LOSSES <<< "${LOSSES_OVERRIDE:-0 1}" +NTOPICS="${NTOPICS:-8}" +OFFERED_RATE="${OFFERED_RATE:-40000}" +DURATION="${DURATION:-60}" +WARMUP="${WARMUP:-5}" +RUNS_PER_DATAPOINT="${RUNS_OVERRIDE:-10}" + +RESULTS_DIR="${ROOT_DIR}/results-v5" +mkdir -p "$RESULTS_DIR" + +BROKER_TLS="--tls-cert /opt/mqtt-certs/server.pem --tls-key /opt/mqtt-certs/server.key" +BROKER_QUIC="--quic-host 0.0.0.0:14567" +CA="--ca-cert /opt/mqtt-certs/ca.pem" + +declare -A ARM_URL ARM_FLAGS ARM_DELIVERY ARM_PUBCONN +ARM_URL[tcp-1conn]="mqtt://${BROKER_IP}:1883" +ARM_FLAGS[tcp-1conn]="" +ARM_DELIVERY[tcp-1conn]="" +ARM_PUBCONN[tcp-1conn]=1 + +ARM_URL[tcp-Nconn]="mqtt://${BROKER_IP}:1883" +ARM_FLAGS[tcp-Nconn]="" +ARM_DELIVERY[tcp-Nconn]="" +ARM_PUBCONN[tcp-Nconn]="$NTOPICS" + +ARM_URL[quic-control]="quic://${BROKER_IP}:14567" +ARM_FLAGS[quic-control]="--quic-stream-strategy control-only ${CA}" +ARM_DELIVERY[quic-control]="--quic-delivery-strategy control-only" +ARM_PUBCONN[quic-control]=1 + +ARM_URL[quic-pertopic]="quic://${BROKER_IP}:14567" +ARM_FLAGS[quic-pertopic]="--quic-stream-strategy per-topic ${CA}" +ARM_DELIVERY[quic-pertopic]="--quic-delivery-strategy per-topic" +ARM_PUBCONN[quic-pertopic]=1 + +: "${E1_ARMS:=tcp-1conn tcp-Nconn quic-control quic-pertopic}" +read -ra ARMS <<< "$E1_ARMS" + +apply_bottleneck() { + local delay_ms="$1" + local loss_pct="$2" + CUR_NETEM_DELAY="$delay_ms" + CUR_NETEM_LOSS="$loss_pct" + ssh_broker "sudo bash /opt/mqtt-lib/experiments/netem/apply_bottleneck.sh ${delay_ms} ${RATE_MBIT} ${QUEUE_PKTS} ${loss_pct}" +} + +restore_netem() { + if [ -n "$CUR_NETEM_DELAY" ]; then + if ! ssh_broker "sudo bash /opt/mqtt-lib/experiments/netem/apply_bottleneck.sh ${CUR_NETEM_DELAY} ${RATE_MBIT} ${QUEUE_PKTS} ${CUR_NETEM_LOSS}" 2>/dev/null; then + echo "WARN: failed to restore bottleneck (rate=${RATE_MBIT}mbit delay=${CUR_NETEM_DELAY}ms loss=${CUR_NETEM_LOSS}%)" >&2 + return 1 + fi + fi + return 0 +} + +start_competing_flow() { + local duration="$1" + ssh_pub "pkill -x iperf3 2>/dev/null; sleep 1; iperf3 -s -D" >/dev/null 2>&1 || true + sleep 1 + ssh_broker "pkill -x iperf3 2>/dev/null; nohup iperf3 -c ${PUB_INTERNAL_IP} -t ${duration} -J > /tmp/iperf_client.json 2>&1 & echo client-started" >/dev/null 2>&1 || true +} + +wait_competing_flow() { + ssh_broker "for _ in \$(seq 1 30); do pgrep -x iperf3 >/dev/null 2>&1 || break; sleep 1; done" 2>/dev/null || true +} + +collect_competing_flow() { + local output_dir="$1" + local run_label="$2" + scp -i "$SSH_KEY_PATH" $SSH_OPTS "${SSH_USER}@${BROKER_SSH_IP}:/tmp/iperf_client.json" \ + "${output_dir}/${run_label}_iperf.json" 2>/dev/null || true +} + +collect_traces() { + local output_dir="$1" + local run_label="$2" + for csv in messages.csv quinn_stats.csv; do + scp -i "$SSH_KEY_PATH" $SSH_OPTS "${SSH_USER}@${PUB_IP}:/tmp/e1-traces/${csv}" \ + "${output_dir}/${run_label}_${csv}" 2>/dev/null || true + done + ssh_pub "rm -rf /tmp/e1-traces" 2>/dev/null || true +} + +run_arm() { + local label="$1" + local bench_args="$2" + local output_dir="${RESULTS_DIR}/${EXPERIMENT}" + mkdir -p "$output_dir" + ssh_pub "ulimit -n 65536; mqttv5 bench ${bench_args}" \ + > "${output_dir}/${label}.json" 2>/dev/null || true + warn_if_empty "${output_dir}/${label}.json" + echo " saved: ${output_dir}/${label}.json" +} + +for arm in "${ARMS[@]}"; do + url="${ARM_URL[$arm]}" + flags="${ARM_FLAGS[$arm]}" + delivery="${ARM_DELIVERY[$arm]}" + pubconn="${ARM_PUBCONN[$arm]}" + + stop_broker 2>/dev/null || true + if ! start_broker "${BROKER_TLS} ${BROKER_QUIC} ${delivery}"; then + echo "WARN: broker start failed for ${arm}, skipping arm" >&2 + continue + fi + + for loss in "${LOSSES[@]}"; do + apply_bottleneck "$DELAY" "$loss" + label="${arm}_rate${RATE_MBIT}mbit_loss${loss}pct" + echo "[${EXPERIMENT}] ${label}" + + bench_args="--url ${url} ${flags} --mode hol-blocking --topics ${NTOPICS} \ + --pub-connections ${pubconn} --sub-connections ${pubconn} \ + --duration ${DURATION} --warmup ${WARMUP} --payload-size 256 --qos 0 --rate ${OFFERED_RATE} \ + --trace-dir /tmp/e1-traces" + + for run in $(seq 1 "$RUNS_PER_DATAPOINT"); do + run_label="${label}_run${run}" + if [ -f "${RESULTS_DIR}/${EXPERIMENT}/${run_label}_messages.csv" ]; then + echo " skip (already complete): ${run_label}" + continue + fi + if [ "$BROKER_FRESH" = "1" ]; then + BROKER_FRESH=0 + elif ! restart_broker; then + echo "WARN: broker restart failed, skipping ${run_label}" >&2 + continue + fi + start_monitors + start_competing_flow $((DURATION + WARMUP + 5)) + run_arm "$run_label" "$bench_args" + wait_competing_flow + stop_monitors "${RESULTS_DIR}/${EXPERIMENT}" "$run_label" + collect_competing_flow "${RESULTS_DIR}/${EXPERIMENT}" "$run_label" + collect_traces "${RESULTS_DIR}/${EXPERIMENT}" "$run_label" + sleep 5 + done + + clear_netem + done +done + +stop_broker +echo "experiment ${EXPERIMENT} v5 complete (group ${GROUP})" diff --git a/publications/comnet/experiments/parallel/common_parallel.sh b/publications/comnet/experiments/parallel/common_parallel.sh index 2f9ee14a..c232a07f 100755 --- a/publications/comnet/experiments/parallel/common_parallel.sh +++ b/publications/comnet/experiments/parallel/common_parallel.sh @@ -5,7 +5,22 @@ SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" ROOT_DIR="$(cd "${SCRIPT_DIR}/.." && pwd)" : "${GROUP:?Set GROUP=1|2|3 before sourcing common_parallel.sh}" + +_cfg_keys="BROKER_IP BROKER_SSH_IP PUB_IP PUB_INTERNAL_IP SUB_IP SUB_INTERNAL_IP SUB_PROXY ROUTER_IP ROUTER_HOP_US SSH_USER SSH_KEY_PATH RUNS_PER_DATAPOINT" +for _k in $_cfg_keys; do + if [ -n "${!_k+set}" ]; then + eval "_preset_${_k}=\${${_k}}" + fi +done source "${SCRIPT_DIR}/group${GROUP}.env" +for _k in $_cfg_keys; do + _p="_preset_${_k}" + if [ -n "${!_p+set}" ]; then + eval "${_k}=\${${_p}}" + unset "${_p}" + fi +done +unset _cfg_keys _k _p : "${BROKER_IP:?Set BROKER_IP in group${GROUP}.env}" : "${BROKER_SSH_IP:=${BROKER_IP}}" @@ -18,7 +33,7 @@ SSH_USER="${SSH_USER:-bench}" RESULTS_DIR="${ROOT_DIR}/results_v2" mkdir -p "$RESULTS_DIR" -SSH_OPTS="-o StrictHostKeyChecking=no -o ServerAliveInterval=30 -o ServerAliveCountMax=10" +SSH_OPTS="-o StrictHostKeyChecking=no -o ServerAliveInterval=30 -o ServerAliveCountMax=10 -o ConnectTimeout=20 -o ConnectionAttempts=6" SUB_SSH_OPTS="$SSH_OPTS" if [ -n "${SUB_PROXY:-}" ]; then SUB_SSH_OPTS="$SSH_OPTS -o ProxyJump=${SSH_USER}@${SUB_PROXY}" @@ -26,6 +41,41 @@ fi ssh_broker() { ssh -i "$SSH_KEY_PATH" $SSH_OPTS "${SSH_USER}@${BROKER_SSH_IP}" "$@"; } ssh_pub() { ssh -i "$SSH_KEY_PATH" $SSH_OPTS "${SSH_USER}@${PUB_IP}" "$@"; } ssh_sub() { ssh -i "$SSH_KEY_PATH" $SUB_SSH_OPTS "${SSH_USER}@${SUB_IP}" "$@"; } +REMOTE_EXPERIMENTS_DIR="/opt/mqtt-lib/experiments" +GCP_PROJECT="${GCP_PROJECT:-$(gcloud config get-value project 2>/dev/null)}" +GCP_ACCOUNT="${GCP_ACCOUNT:-$(gcloud config get-value account 2>/dev/null)}" +GCP_REGION="us-west1" + +gc() { gcloud "$@" --project="$GCP_PROJECT" --account="$GCP_ACCOUNT" --quiet; } + +ilb_health() { + gc compute backend-services get-health "exp3-bs-${GROUP}" --region="$GCP_REGION" \ + --format='value(status.healthStatus[].healthState)' 2>/dev/null | tr -s '[:space:];' ' ' | sed 's/ *$//' || echo "unknown" +} + +router_ingress_packets() { + ssh_router "sudo bash ${REMOTE_EXPERIMENTS_DIR}/netem/router_ingress_count.sh" +} + +router_path_state() { + local before after moved + before=$(router_ingress_packets) || { echo "unknown"; return 0; } + ssh_broker "ping -q -c 200 -i 0.005 ${SUB_INTERNAL_IP}; ping -q -c 200 -i 0.005 ${PUB_INTERNAL_IP}" >/dev/null 2>&1 || true + after=$(router_ingress_packets) || { echo "unknown"; return 0; } + moved=$((after - before)) + if [ "$moved" -ge 380 ]; then + echo "on" + elif [ "$moved" -lt 20 ]; then + echo "off" + else + echo "partial:${moved}" + fi +} + +ssh_router() { + : "${ROUTER_IP:?Set ROUTER_IP in group${GROUP}.env}" + ssh -i "$SSH_KEY_PATH" $SSH_OPTS -o ProxyJump="${SSH_USER}@${BROKER_SSH_IP}" "${SSH_USER}@${ROUTER_IP}" "$@" +} scp_from_sub() { local remote_path="$1" @@ -48,7 +98,12 @@ start_broker() { local attempt for attempt in 1 2 3; do echo "starting broker on ${BROKER_IP} (group ${GROUP}) [attempt ${attempt}]..." - BROKER_PID=$(ssh_broker "ulimit -n 65536; nohup mqttv5 broker --allow-anonymous --host 0.0.0.0:1883 --storage-backend memory --max-clients 50000 \ + local quic_stats_env="" + if [ "${COLLECT_QUIC_STATS:-0}" = "1" ]; then + ssh_broker "rm -rf /tmp/quic-stats; mkdir -p /tmp/quic-stats" 2>/dev/null || true + quic_stats_env="MQTT5_QUIC_STATS_DIR=/tmp/quic-stats " + fi + BROKER_PID=$(ssh_broker "ulimit -n 65536; ${quic_stats_env}nohup mqttv5 broker --allow-anonymous --host 0.0.0.0:1883 --storage-backend memory --max-clients 50000 \ ${extra_flags} > /tmp/broker.log 2>&1 & echo \$!") || BROKER_PID="" sleep 2 if [ -n "${BROKER_PID}" ] && ssh_broker "kill -0 ${BROKER_PID}" 2>/dev/null; then @@ -116,7 +171,14 @@ BROKER_MONITOR_PID="" PUB_MONITOR_PID="" SUB_MONITOR_PID="" +stop_stale_monitors() { + ssh_broker "pkill -f '[r]esource_monitor.sh'" 2>/dev/null || true + ssh_pub "pkill -f '[c]lient_monitor.sh'" 2>/dev/null || true + ssh_sub "pkill -f '[c]lient_monitor.sh'" 2>/dev/null || true +} + start_monitors() { + stop_stale_monitors BROKER_MONITOR_PID=$(ssh_broker "nohup bash /opt/mqtt-lib/experiments/monitor/resource_monitor.sh ${BROKER_PID} \ > /tmp/monitor.csv 2>&1 & echo \$!") || BROKER_MONITOR_PID="" PUB_MONITOR_PID=$(ssh_pub "nohup bash /opt/mqtt-lib/experiments/monitor/client_monitor.sh \ @@ -191,8 +253,11 @@ run_bench_split() { pub_args="${pub_args} --subscribers 0" fi - ssh_sub "rm -f /tmp/sub_bench.json; ulimit -n 65536; nohup mqttv5 bench ${sub_args} \ - > /tmp/sub_bench.json 2>/dev/null &" + if ! ssh_sub "rm -f /tmp/sub_bench.json; ulimit -n 65536; nohup mqttv5 bench ${sub_args} \ + > /tmp/sub_bench.json 2>/dev/null &"; then + echo "WARN: subscriber launch failed for ${label}" >&2 + return 0 + fi sleep 2 ssh_pub "ulimit -n 65536; mqttv5 bench ${pub_args}" \ @@ -201,7 +266,7 @@ run_bench_split() { local waited=0 while [ "$waited" -lt 60 ]; do local size - size=$(ssh_sub "stat -c%s /tmp/sub_bench.json 2>/dev/null || echo 0") + size=$(ssh_sub "stat -c%s /tmp/sub_bench.json 2>/dev/null || echo 0") || size=0 if [ "$size" -gt 0 ]; then break fi diff --git a/publications/comnet/experiments/parallel/group1.env b/publications/comnet/experiments/parallel/group1.env index abd2da1d..066b2740 100644 --- a/publications/comnet/experiments/parallel/group1.env +++ b/publications/comnet/experiments/parallel/group1.env @@ -1,5 +1,9 @@ -BROKER_IP="" -BROKER_SSH_IP="" -PUB_IP="" -SUB_IP="" +BROKER_IP="10.138.0.21" +BROKER_SSH_IP="35.197.5.208" +PUB_IP="136.86.210.25" +SUB_IP="136.67.121.123" SSH_USER="bench" +PUB_INTERNAL_IP="10.138.0.22" +SUB_INTERNAL_IP="10.138.0.23" +ROUTER_IP="10.138.0.31" +ROUTER_HOP_US="73" diff --git a/publications/comnet/experiments/parallel/group2.env b/publications/comnet/experiments/parallel/group2.env index abd2da1d..a213a81f 100644 --- a/publications/comnet/experiments/parallel/group2.env +++ b/publications/comnet/experiments/parallel/group2.env @@ -1,5 +1,9 @@ -BROKER_IP="" -BROKER_SSH_IP="" -PUB_IP="" -SUB_IP="" +BROKER_IP="10.138.0.24" +BROKER_SSH_IP="35.247.119.228" +PUB_IP="35.203.174.192" +SUB_IP="136.86.245.212" SSH_USER="bench" +PUB_INTERNAL_IP="10.138.0.25" +SUB_INTERNAL_IP="10.138.0.26" +ROUTER_IP="10.138.0.33" +ROUTER_HOP_US="57" diff --git a/publications/comnet/experiments/parallel/group3.env b/publications/comnet/experiments/parallel/group3.env index 930f838d..f4912b3f 100644 --- a/publications/comnet/experiments/parallel/group3.env +++ b/publications/comnet/experiments/parallel/group3.env @@ -1,6 +1,10 @@ -BROKER_IP="" -BROKER_SSH_IP="" -PUB_IP="" -SUB_IP="" -SUB_PROXY="" +BROKER_IP="10.138.0.27" +BROKER_SSH_IP="136.109.157.44" +PUB_IP="136.86.198.226" +SUB_IP="10.138.0.30" +SUB_PROXY="136.86.198.226" SSH_USER="bench" +PUB_INTERNAL_IP="10.138.0.28" +SUB_INTERNAL_IP="10.138.0.30" +ROUTER_IP="10.138.0.35" +ROUTER_HOP_US="58" diff --git a/publications/comnet/experiments/parallel/run_03e_pilot.sh b/publications/comnet/experiments/parallel/run_03e_pilot.sh new file mode 100755 index 00000000..34616f49 --- /dev/null +++ b/publications/comnet/experiments/parallel/run_03e_pilot.sh @@ -0,0 +1,28 @@ +#!/usr/bin/env bash +set -euo pipefail + +: "${GROUP:?Set GROUP=1|2|3}" +HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" + +run_phase() { + local attempt + for attempt in 1 2 3; do + if PHASE="$1" bash "${HERE}/03e_single_segment.sh"; then + return 0 + fi + echo "phase $1 attempt ${attempt} failed; resuming in 60 s" >&2 + sleep 60 + done + return 1 +} + +run_phase calib +bash "${HERE}/03e_infra.sh" route-off +if run_phase calib-direct; then + bash "${HERE}/03e_infra.sh" route-on +else + bash "${HERE}/03e_infra.sh" route-on + exit 1 +fi +run_phase accuracy +python3 "${HERE}/../analysis/single_segment_accept.py" "${HERE}/../results-v5/03e_single_segment_g${GROUP}" diff --git a/publications/comnet/experiments/parallel/run_03e_sweep.sh b/publications/comnet/experiments/parallel/run_03e_sweep.sh new file mode 100755 index 00000000..fedcf042 --- /dev/null +++ b/publications/comnet/experiments/parallel/run_03e_sweep.sh @@ -0,0 +1,23 @@ +#!/usr/bin/env bash +set -euo pipefail + +: "${GROUP:?Set GROUP=1|2|3}" +: "${ACCEPT_OVERRIDE:?Set ACCEPT_OVERRIDE=}" +HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +export ACCEPT_OVERRIDE + +run_phase() { + local attempt + for attempt in 1 2 3 4 5; do + if PHASE="$1" bash "${HERE}/03e_single_segment.sh"; then + return 0 + fi + echo "phase $1 attempt ${attempt} failed; resuming in 120 s" >&2 + sleep 120 + done + return 1 +} + +run_phase main +run_phase crosscheck +run_phase hol diff --git a/publications/comnet/experiments/parallel/run_reruns_resilient.sh b/publications/comnet/experiments/parallel/run_reruns_resilient.sh new file mode 100644 index 00000000..6b17ee89 --- /dev/null +++ b/publications/comnet/experiments/parallel/run_reruns_resilient.sh @@ -0,0 +1,34 @@ +#!/usr/bin/env bash +DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +source "${DIR}/group1.env" +KEY="$HOME/.ssh/id_ed25519" +OPTS="-i $KEY -o StrictHostKeyChecking=no -o BatchMode=yes -o ConnectTimeout=10" + +cleanup() { + ssh $OPTS "bench@${BROKER_SSH_IP}" "pkill -f '[m]qttv5 broker'; sudo tc qdisc del dev ens4 root" 2>/dev/null || true +} + +run_resilient() { + local label="$1" + shift + for attempt in $(seq 1 40); do + echo "===== ${label} attempt ${attempt} ($(date '+%H:%M:%S')) =====" + cleanup + sleep 3 + if "$@"; then + echo "===== ${label} COMPLETE on attempt ${attempt} =====" + cleanup + return 0 + fi + echo "----- ${label} attempt ${attempt} aborted (transient), retrying -----" + sleep 15 + done + echo "===== ${label} GAVE UP after 40 attempts =====" + cleanup + return 1 +} + +run_resilient "exp3-qos1" env QOS_OVERRIDE=1 GROUP=1 bash "${DIR}/03_throughput_under_loss_v5.sh" || exit 1 +run_resilient "exp3-qos1-tls" env QOS_OVERRIDE=1 GROUP=1 bash "${DIR}/03b_tls_throughput_v5.sh" || exit 1 +run_resilient "exp4" env GROUP=1 bash "${DIR}/04_stream_strategies_v5.sh" || exit 1 +echo "===== ALL RE-RUNS COMPLETE ($(date '+%H:%M:%S')) =====" diff --git a/publications/comnet/experiments/parallel/run_tls_resilient.sh b/publications/comnet/experiments/parallel/run_tls_resilient.sh new file mode 100644 index 00000000..3850df74 --- /dev/null +++ b/publications/comnet/experiments/parallel/run_tls_resilient.sh @@ -0,0 +1,24 @@ +#!/usr/bin/env bash +source "$(dirname "${BASH_SOURCE[0]}")/group1.env" +KEY="$HOME/.ssh/id_ed25519" +OPTS="-i $KEY -o StrictHostKeyChecking=no -o BatchMode=yes -o ConnectTimeout=10" + +cleanup() { + ssh $OPTS "bench@${BROKER_SSH_IP}" "pkill -f '[m]qttv5 broker'; sudo tc qdisc del dev ens4 root" 2>/dev/null || true +} + +for attempt in $(seq 1 40); do + echo "===== TLS sweep attempt ${attempt} ($(date '+%H:%M:%S')) =====" + cleanup + sleep 3 + if GROUP=1 bash "$(dirname "${BASH_SOURCE[0]}")/03b_tls_throughput_v5.sh"; then + echo "===== TLS sweep COMPLETE on attempt ${attempt} =====" + cleanup + exit 0 + fi + echo "----- attempt ${attempt} aborted (transient), retrying -----" + sleep 15 +done +echo "===== TLS sweep gave up after 40 attempts =====" +cleanup +exit 1 diff --git a/publications/comnet/experiments/run/02c_hol_topic_scaling.sh b/publications/comnet/experiments/run/02c_hol_topic_scaling.sh index ee2e0497..83ece940 100755 --- a/publications/comnet/experiments/run/02c_hol_topic_scaling.sh +++ b/publications/comnet/experiments/run/02c_hol_topic_scaling.sh @@ -54,7 +54,7 @@ for tname in "${TRANSPORTS[@]}"; do label="${tname}_${topics}topics" echo "[${EXPERIMENT}] ${label}" - bench_args="--url ${url} ${flags} --mode hol-blocking --topics ${topics} --duration 30 --warmup 5 --payload-size 512 --rate 5000 --trace-dir /tmp/hol-traces" + bench_args="--url ${url} ${flags} --mode hol-blocking --topics ${topics} --duration 60 --warmup 5 --payload-size 256 --rate 500 --trace-dir /tmp/hol-traces" output_dir="${RESULTS_DIR}/${EXPERIMENT}" mkdir -p "$output_dir" diff --git a/publications/comnet/experiments/run/common.sh b/publications/comnet/experiments/run/common.sh index 7a6ecc53..7fd7938f 100755 --- a/publications/comnet/experiments/run/common.sh +++ b/publications/comnet/experiments/run/common.sh @@ -3,7 +3,24 @@ set -euo pipefail SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" ROOT_DIR="$(cd "${SCRIPT_DIR}/.." && pwd)" + +# config.env is sourced for defaults, but a value already present in the +# environment wins, so `RUNS_PER_DATAPOINT=1 bash run/foo.sh` behaves as written. +_cfg_keys="GCP_PROJECT_ID BROKER_IP BROKER_SSH_IP CLIENT_IP REPO_URL REPO_BRANCH SSH_USER SSH_KEY_PATH RUNS_PER_DATAPOINT" +for _k in $_cfg_keys; do + if [ -n "${!_k+set}" ]; then + eval "_preset_${_k}=\${${_k}}" + fi +done source "${ROOT_DIR}/setup/config.env" +for _k in $_cfg_keys; do + _p="_preset_${_k}" + if [ -n "${!_p+set}" ]; then + eval "${_k}=\${${_p}}" + unset "${_p}" + fi +done +unset _cfg_keys _k _p : "${BROKER_IP:?Set BROKER_IP in config.env}" : "${BROKER_SSH_IP:=${BROKER_IP}}" @@ -63,11 +80,11 @@ restart_broker() { apply_netem() { local delay_ms="$1" local loss_pct="${2:-0}" - ssh_client "sudo bash /opt/mqtt-lib/experiments/netem/apply.sh ${delay_ms} ${loss_pct}" + ssh_broker "sudo bash /opt/mqtt-lib/experiments/netem/apply.sh ${delay_ms} ${loss_pct}" } clear_netem() { - ssh_client "sudo bash /opt/mqtt-lib/experiments/netem/clear.sh" + ssh_broker "sudo bash /opt/mqtt-lib/experiments/netem/clear.sh" } MONITOR_PID="" diff --git a/publications/comnet/experiments/setup/pin_kernel.sh b/publications/comnet/experiments/setup/pin_kernel.sh new file mode 100644 index 00000000..a7b597e1 --- /dev/null +++ b/publications/comnet/experiments/setup/pin_kernel.sh @@ -0,0 +1,32 @@ +#!/usr/bin/env bash +set -euo pipefail + +KERNEL="${1:?usage: $0 }" + +sudo systemctl disable --now unattended-upgrades.service apt-daily.timer apt-daily-upgrade.timer +while sudo fuser /var/lib/dpkg/lock-frontend /var/lib/dpkg/lock >/dev/null 2>&1; do + sleep 5 +done + +if [ ! -e "/boot/vmlinuz-${KERNEL}" ]; then + sudo apt-get update -qq + sudo DEBIAN_FRONTEND=noninteractive apt-get install -y -qq "linux-image-${KERNEL}" "linux-modules-${KERNEL}" +fi +sudo apt-mark hold linux-gcp linux-image-gcp linux-headers-gcp >/dev/null + +submenu=$(sudo grep -o "gnulinux-advanced-[^']*" /boot/grub/grub.cfg | head -1) +entry=$(sudo grep -o "gnulinux-${KERNEL}-advanced-[^']*" /boot/grub/grub.cfg | head -1) +if [ -z "$submenu" ] || [ -z "$entry" ]; then + sudo update-grub >/dev/null 2>&1 + submenu=$(sudo grep -o "gnulinux-advanced-[^']*" /boot/grub/grub.cfg | head -1) + entry=$(sudo grep -o "gnulinux-${KERNEL}-advanced-[^']*" /boot/grub/grub.cfg | head -1) +fi +if [ -z "$submenu" ] || [ -z "$entry" ]; then + echo "no grub entry for ${KERNEL}" >&2 + exit 1 +fi + +echo "GRUB_DEFAULT=\"${submenu}>${entry}\"" | sudo tee /etc/default/grub.d/99-pin-kernel.cfg >/dev/null +sudo update-grub >/dev/null 2>&1 +grep -h "^GRUB_DEFAULT" /etc/default/grub.d/99-pin-kernel.cfg +echo "pinned ${KERNEL}; running $(uname -r)" diff --git a/publications/comnet/submission/Figure_1.pdf b/publications/comnet/submission/Figure_1.pdf deleted file mode 100644 index 8b73ee5c..00000000 Binary files a/publications/comnet/submission/Figure_1.pdf and /dev/null differ diff --git a/publications/comnet/submission/Figure_2.pdf b/publications/comnet/submission/Figure_2.pdf deleted file mode 100644 index 97864fef..00000000 Binary files a/publications/comnet/submission/Figure_2.pdf and /dev/null differ diff --git a/publications/comnet/submission/Figure_3.pdf b/publications/comnet/submission/Figure_3.pdf index cc7d8f4c..51079789 100644 Binary files a/publications/comnet/submission/Figure_3.pdf and b/publications/comnet/submission/Figure_3.pdf differ diff --git a/publications/comnet/submission/Figure_4.pdf b/publications/comnet/submission/Figure_4.pdf index 784af9ed..6530fc03 100644 Binary files a/publications/comnet/submission/Figure_4.pdf and b/publications/comnet/submission/Figure_4.pdf differ diff --git a/publications/comnet/submission/Figure_5.pdf b/publications/comnet/submission/Figure_5.pdf index 854e2429..eaa381a1 100644 Binary files a/publications/comnet/submission/Figure_5.pdf and b/publications/comnet/submission/Figure_5.pdf differ diff --git a/publications/comnet/submission/Figure_6.pdf b/publications/comnet/submission/Figure_6.pdf index 46cb6af5..f8b02386 100644 Binary files a/publications/comnet/submission/Figure_6.pdf and b/publications/comnet/submission/Figure_6.pdf differ diff --git a/publications/comnet/submission/Figure_7.pdf b/publications/comnet/submission/Figure_7.pdf index dec899e8..9857b250 100644 Binary files a/publications/comnet/submission/Figure_7.pdf and b/publications/comnet/submission/Figure_7.pdf differ diff --git a/publications/comnet/submission/Figure_8.pdf b/publications/comnet/submission/Figure_8.pdf index 0d03f190..5094ebb2 100644 Binary files a/publications/comnet/submission/Figure_8.pdf and b/publications/comnet/submission/Figure_8.pdf differ diff --git a/publications/comnet/submission/biography.docx b/publications/comnet/submission/biography.docx index f810bdcb..2728c082 100644 Binary files a/publications/comnet/submission/biography.docx and b/publications/comnet/submission/biography.docx differ diff --git a/publications/comnet/submission/main.bbl b/publications/comnet/submission/main.bbl index d25ebf2d..4b2ba9af 100644 --- a/publications/comnet/submission/main.bbl +++ b/publications/comnet/submission/main.bbl @@ -1,4 +1,4 @@ -\begin{thebibliography}{18} +\begin{thebibliography}{24} \expandafter\ifx\csname natexlab\endcsname\relax\def\natexlab#1{#1}\fi \providecommand{\url}[1]{\texttt{#1}} \providecommand{\href}[2]{#2} @@ -27,13 +27,27 @@ \bibinfo{author}{Bracht, F.}, \bibinfo{year}{2026}. \newblock \bibinfo{title}{Experiment data for ``{Evaluating Stream Mapping Strategies for MQTT over QUIC}''}. -\newblock \DOIprefix\doi{10.5281/zenodo.19098820}. +\newblock \DOIprefix\doi{10.5281/zenodo.19098819}. +%Type = Article +\bibitem[{Chiu and Jain(1989)}]{chiu1989aimd} +\bibinfo{author}{Chiu, D.M.}, \bibinfo{author}{Jain, R.}, \bibinfo{year}{1989}. +\newblock \bibinfo{title}{Analysis of the increase and decrease algorithms for + congestion avoidance in computer networks}. +\newblock \bibinfo{journal}{Computer Networks and ISDN Systems} + \bibinfo{volume}{17}, \bibinfo{pages}{1--14}. +\newblock \DOIprefix\doi{10.1016/0169-7552(89)90019-6}. %Type = Misc -\bibitem[{{EMQ Technologies}(2024)}]{emqx} -\bibinfo{author}{{EMQ Technologies}}, \bibinfo{year}{2024}. +\bibitem[{{EMQ Technologies}(2024a)}]{emqx} +\bibinfo{author}{{EMQ Technologies}}, \bibinfo{year}{2024}a. \newblock \bibinfo{title}{{EMQX}: An open-source, highly scalable {MQTT} broker}. \newblock \URLprefix \url{https://www.emqx.io/}. +%Type = Misc +\bibitem[{{EMQ Technologies}(2024b)}]{nanomq-quic} +\bibinfo{author}{{EMQ Technologies}}, \bibinfo{year}{2024}b. +\newblock \bibinfo{title}{{NanoMQ}: {MQTT} over {QUIC} bridge}. +\newblock \URLprefix + \url{https://nanomq.io/docs/en/latest/bridges/quic-bridge.html}. %Type = Inproceedings \bibitem[{Fern{\'a}ndez et~al.(2020)Fern{\'a}ndez, Zverev, Garrido et~al.}]{fernandez2020iot-quic} @@ -76,7 +90,9 @@ \bibinfo{author}{Kumar, P.}, \bibinfo{author}{Dezfouli, B.}, \bibinfo{year}{2019}. \newblock \bibinfo{title}{Implementation and analysis of {QUIC} for {MQTT}}. -\newblock \bibinfo{journal}{Computer Networks} . +\newblock \bibinfo{journal}{Computer Networks} \bibinfo{volume}{150}, + \bibinfo{pages}{28--45}. +\newblock \DOIprefix\doi{10.1016/j.comnet.2018.12.012}. %Type = Article \bibitem[{Langley et~al.(2017)Langley, Riddoch et~al.}]{langley2017quic} \bibinfo{author}{Langley, A.}, \bibinfo{author}{Riddoch, A.}, et~al., @@ -92,6 +108,24 @@ \bibinfo{booktitle}{Workshop on the Evolution, Performance, and Interoperability of QUIC (EPIQ)}. %Type = Article +\bibitem[{Mathis et~al.(1997)Mathis, Semke, Mahdavi and + Ott}]{mathis1997macroscopic} +\bibinfo{author}{Mathis, M.}, \bibinfo{author}{Semke, J.}, + \bibinfo{author}{Mahdavi, J.}, \bibinfo{author}{Ott, T.}, + \bibinfo{year}{1997}. +\newblock \bibinfo{title}{The macroscopic behavior of the {TCP} congestion + avoidance algorithm}. +\newblock \bibinfo{journal}{ACM SIGCOMM Computer Communication Review} + \bibinfo{volume}{27}, \bibinfo{pages}{67--82}. +%Type = Misc +\bibitem[{{Microsoft}(2024)}]{azureiot} +\bibinfo{author}{{Microsoft}}, \bibinfo{year}{2024}. +\newblock \bibinfo{title}{Communicate with your {IoT} hub using the {MQTT} + protocol}. +\newblock \bibinfo{howpublished}{Microsoft Azure IoT Hub documentation}. +\newblock \URLprefix + \url{https://learn.microsoft.com/en-us/azure/iot-hub/iot-hub-mqtt-support}. +%Type = Article \bibitem[{Mishra and Kertesz(2020)}]{mishra2020mqtt-survey} \bibinfo{author}{Mishra, B.}, \bibinfo{author}{Kertesz, A.}, \bibinfo{year}{2020}. @@ -115,17 +149,39 @@ {Rust}}. \newblock \URLprefix \url{https://github.com/quinn-rs/quinn}. %Type = Misc -\bibitem[{Pauly et~al.(2022)Pauly, Kinnear and Schinazi}]{rfc9221} -\bibinfo{author}{Pauly, T.}, \bibinfo{author}{Kinnear, E.}, - \bibinfo{author}{Schinazi, D.}, \bibinfo{year}{2022}. -\newblock \bibinfo{title}{An unreliable datagram extension to {QUIC}}. -\newblock \bibinfo{howpublished}{RFC 9221}. +\bibitem[{Ott et~al.(2017)Ott, Even, Perkins and Singh}]{rtpfolks2017} +\bibinfo{author}{Ott, J.}, \bibinfo{author}{Even, R.}, + \bibinfo{author}{Perkins, C.}, \bibinfo{author}{Singh, V.}, + \bibinfo{year}{2017}. +\newblock \bibinfo{title}{{RTP} over {QUIC}}. +\newblock \bibinfo{howpublished}{Internet-Draft + draft-rtpfolks-quic-rtp-over-quic-01, IETF}. +\newblock \URLprefix + \url{https://datatracker.ietf.org/doc/html/draft-rtpfolks-quic-rtp-over-quic}. +%Type = Inproceedings +\bibitem[{Scharf and Kiesel(2006)}]{scharf2006hol} +\bibinfo{author}{Scharf, M.}, \bibinfo{author}{Kiesel, S.}, + \bibinfo{year}{2006}. +\newblock \bibinfo{title}{Head-of-line blocking in {TCP} and {SCTP}: Analysis + and measurements}, in: \bibinfo{booktitle}{Proceedings of the IEEE Global + Telecommunications Conference (GLOBECOM 2006)}, \bibinfo{publisher}{IEEE}, + \bibinfo{address}{San Francisco, CA, USA}. +\newblock \DOIprefix\doi{10.1109/GLOCOM.2006.333}. %Type = Misc \bibitem[{Thomson and Turner(2021)}]{rfc9001} \bibinfo{author}{Thomson, M.}, \bibinfo{author}{Turner, S.}, \bibinfo{year}{2021}. \newblock \bibinfo{title}{Using {TLS} to secure {QUIC}}. \newblock \bibinfo{howpublished}{RFC 9001}. +%Type = Misc +\bibitem[{Yang(2024)}]{yang2024mqttnext} +\bibinfo{author}{Yang, W.}, \bibinfo{year}{2024}. +\newblock \bibinfo{title}{{Spec MQTT-next}: A mapping of {MQTT} to the {QUIC} + transport protocol}. +\newblock \bibinfo{howpublished}{OASIS MQTT Technical Committee, document + 71729}. +\newblock \URLprefix + \url{https://groups.oasis-open.org/higherlogic/ws/public/download/71729/oasis_mqtt_over_quic.pdf}. %Type = Inproceedings \bibitem[{Yu and Benson(2021)}]{yu2021quic} \bibinfo{author}{Yu, A.}, \bibinfo{author}{Benson, T.A.}, \bibinfo{year}{2021}. diff --git a/publications/comnet/submission/main.pdf b/publications/comnet/submission/main.pdf index ca17f44a..1f1f7831 100644 Binary files a/publications/comnet/submission/main.pdf and b/publications/comnet/submission/main.pdf differ diff --git a/publications/comnet/submission/main.tex b/publications/comnet/submission/main.tex index 28c39b4d..f6a8e660 100644 --- a/publications/comnet/submission/main.tex +++ b/publications/comnet/submission/main.tex @@ -35,15 +35,15 @@ \cortext[1]{Corresponding author} \begin{abstract} -MQTT, the dominant IoT messaging protocol, operates over TCP, where head-of-line (HOL) blocking couples logically independent topic flows: a single packet loss delays delivery across all topics sharing a connection. We define three configurable stream mapping strategies---control-only, per-topic, and per-publish---for running MQTT over QUIC, each offering a different point in the isolation--overhead tradeoff space. We implement these strategies in \texttt{mqtt5}, an open-source MQTT library, and evaluate across five experiments on GCP infrastructure under controlled network impairment. Using inter-topic latency spread and spike isolation ratio to quantify HOL blocking, we find that per-topic QUIC provides meaningful topic-level isolation under loss: spread amplifies 200$\times$ at 5\% packet loss versus 1.2$\times$ for TCP, and 27\% of latency spikes affect individual topics independently versus less than 2\% for TCP. TCP outperforms QUIC in throughput at 0\% loss, but under loss ($\geq$1\%) QUIC outperforms TCP by up to 2.8$\times$ while providing encryption by default. We additionally evaluate QUIC datagram transport, finding no statistically significant performance difference from stream transport at representative IoT RTTs. These results provide concrete strategy selection guidelines for deploying MQTT over QUIC in IoT environments. +MQTT, the dominant IoT messaging protocol, is conventionally carried over TCP, where head-of-line (HOL) blocking couples logically independent topic flows: a single packet loss delays delivery across all topics sharing a connection. Three stream mapping strategies, namely control-only, per-topic, and per-publish, each offer a different point in the isolation--overhead tradeoff space for MQTT over QUIC. We implement them in \texttt{mqtt5}, an open-source MQTT library whose broker-autonomous per-topic delivery-stream mapping is absent from existing brokers, and evaluate them across four experiments on GCP infrastructure under controlled network impairment. Using a density-corrected spike-isolation metric and per-topic tail latency to quantify HOL blocking, we find that the transport, not stream mapping, sets the level of HOL blocking at heavy (5\%) loss: all three strategies keep the median per-topic tail below TCP at 5\% loss (control-only 132\,ms versus TCP's coupled 351\,ms). Multistream adds spike-timing isolation, but as a bounded tradeoff rather than a free benefit: per-topic's isolation holds only at small topic counts, reaching single-stream coupling from about four topics, while per-publish stays isolated, at a tail latency that depends on loss and rate. On a single connection the strategies differ ninefold in publish rate, but we show that gap is set by two QUIC transport constants, the concurrent stream limit and the per-stream receive window, rather than by the mappings themselves. At our 8-topic, 500\,msg/s reference point, control-only, adding the least packet overhead, has the lowest median and 90th-percentile tail among the QUIC strategies at 5\% loss; at 2\% it is indistinguishable from per-topic and at 1\% it is the highest of the three, and at 125\,msg/s under 5\% loss per-publish is lowest, so which strategy has the lowest tail depends on both loss and rate. Under loss, QUIC delivers 1.26$\times$ the throughput of an encryption-matched TCP+TLS~1.3 baseline at 1\% loss, rising to 1.78$\times$ at 10\%, once the emulated loss removes single packets; applying the same loss per kernel buffer on the sending host inflates that advantage to 2--3$\times$. These results provide concrete strategy selection guidelines for deploying MQTT over QUIC in IoT environments. \end{abstract} \begin{highlights} \item Three stream mapping strategies for MQTT over QUIC evaluated -\item Per-topic QUIC amplifies inter-topic spread 200$\times$ under 5\% loss -\item QUIC outperforms TCP in throughput under loss by up to 2.8$\times$ -\item Frame packing policy interacts asymmetrically with stream lifetime -\item QUIC datagrams show no significant advantage over stream transport +\item The transport, not stream mapping, sets the HOL blocking level at 5\% loss +\item Control-only mapping has the lowest QUIC latency tail at 5\% loss and 500\,msg/s +\item QUIC delivers 1.3--1.8$\times$ TCP+TLS~1.3 throughput under 1--10\% per-packet loss +\item Per-buffer loss emulation inflated QUIC throughput gains to 2--3$\times$ \end{highlights} \begin{keywords} @@ -59,20 +59,20 @@ \section{Introduction} MQTT is the dominant messaging protocol for Internet of Things (IoT) deployments, widely adopted for device-to-cloud communication across industrial, smart city, and consumer IoT platforms~\cite{mishra2020mqtt-survey}. MQTT's publish-subscribe model, lightweight framing, and multiple quality-of-service levels make it well-suited for constrained devices and unreliable networks~\cite{mqtt-v5-spec}. -However, MQTT operates over TCP, which introduces a fundamental limitation: \emph{head-of-line (HOL) blocking}. When an IoT device publishes to multiple topics over a single TCP connection---as is common for devices reporting temperature, humidity, GPS, and status simultaneously---a packet loss on any one topic's data blocks delivery of \emph{all} topics until retransmission completes. This creates artificial coupling between logically independent data flows, increasing tail latency and reducing the responsiveness of multi-topic workloads. +MQTT requires a reliable, ordered transport, and is most commonly carried over TCP. This standard transport, however, introduces a fundamental limitation: \emph{head-of-line (HOL) blocking}~\cite{scharf2006hol}. When an IoT device publishes to multiple topics over a single TCP connection, as is common for devices reporting temperature, humidity, GPS, and status simultaneously, a packet loss on any one topic's data blocks delivery of \emph{all} topics until retransmission completes. This creates artificial coupling between logically independent data flows, increasing tail latency and reducing the responsiveness of multi-topic workloads. -QUIC~\cite{rfc9000}, the transport protocol underlying HTTP/3, addresses HOL blocking through \emph{stream multiplexing}: multiple independent streams share a single connection, each with its own flow control and loss recovery. A lost packet on one stream does not block data on other streams. This property has been extensively studied for web traffic~\cite{langley2017quic, kakhki2017quic, marx2020hol, yu2021quic}, but its application to IoT messaging protocols remains largely unexplored. +QUIC~\cite{rfc9000}, the transport protocol underlying HTTP/3, addresses HOL blocking through \emph{stream multiplexing}: multiple independent streams share a single connection, each with its own flow control and in-order delivery, while loss detection and congestion control remain per-connection~\cite{rfc9002}. A lost packet blocks in-order delivery only on the streams whose data it carried, although the connection-wide congestion response can slow all of them. This property has been extensively studied for web traffic~\cite{langley2017quic, kakhki2017quic, marx2020hol, yu2021quic}, but its application to IoT messaging protocols remains largely unexplored. -Running MQTT over QUIC is not simply a matter of replacing the TCP socket. The question of \emph{how} to map MQTT's control and data packets onto QUIC streams admits multiple answers, each with different performance characteristics. A single-stream approach (equivalent to TCP) gains QUIC's security and connection management benefits but not HOL blocking elimination. A per-topic approach maps each MQTT topic to a dedicated QUIC stream, providing topic-level isolation. A per-message approach creates a new stream for each PUBLISH, maximizing isolation at the cost of stream management overhead. +Running MQTT over QUIC is not simply a matter of replacing the TCP socket. The question of \emph{how} to map MQTT's control and data packets onto QUIC streams admits multiple answers, each with different performance characteristics. A control-only approach keeps all data on one stream, as TCP does, gaining QUIC's security and connection management benefits but not HOL blocking elimination. A per-topic approach maps each MQTT topic to a dedicated QUIC stream, intended to provide topic-level isolation. A per-publish approach creates a new stream for each PUBLISH, maximizing isolation at the cost of stream management overhead. We use these three names throughout. -In this paper, we define three configurable stream mapping strategies for MQTT over QUIC, implement them in \texttt{mqtt5}, an open-source MQTT library, and evaluate them across five experiments on GCP infrastructure under controlled network impairment. +In this paper, we implement and evaluate three stream mapping strategies for MQTT over QUIC in \texttt{mqtt5}\footnote{\url{https://github.com/LabOverWire/mqtt-lib}}, an open-source MQTT library, across four experiments on GCP infrastructure under controlled network impairment. Our contributions are: \begin{enumerate} -\item Three stream mapping strategies (control-only, per-topic, per-publish) for MQTT over QUIC with a lightweight flow header protocol for stream demultiplexing. -\item The first empirical evaluation of MQTT stream mapping strategies under controlled network impairment, quantifying HOL blocking and throughput. -\item An evaluation of QUIC datagram transport for MQTT as an alternative to stream-based delivery. -\item An open-source implementation with integrated benchmarking tools. +\item The first implementation of \emph{broker-autonomous} per-topic and per-publish delivery-stream mapping for MQTT over QUIC. The operating modes are defined by EMQ's \emph{MQTT-next} specification~\cite{yang2024mqttnext}; we implement the broker side that specification describes but that no production broker realizes, in that the broker opens its own QUIC streams and maps each subscriber's deliveries per topic rather than only mirroring deliveries back onto client-initiated streams. +\item The first vendor-independent, controlled, statistically-treated evaluation of MQTT-over-QUIC operating modes, namely control-only, per-topic, and per-publish, under emulated network impairment, quantifying head-of-line blocking and throughput. +\item An argument for why QUIC stream multiplexing, rather than $N$ parallel TCP connections, is the concurrency mechanism available to an MQTT client: the v5.0 session-takeover rule evicts an existing connection when a second presents the same client identifier~\cite{mqtt-v5-spec}, so one identity cannot hold the parallel connections that would otherwise supply per-topic concurrency. +\item An open-source implementation and a reproducible benchmarking harness. \end{enumerate} The remainder of this paper is organized as follows. Section~\ref{sec:background} covers MQTT, QUIC, and related work. Section~\ref{sec:architecture} describes the stream mapping architecture. Section~\ref{sec:implementation} details the implementation. Section~\ref{sec:evaluation} presents experimental results. Section~\ref{sec:discussion} discusses strategy selection guidelines and future directions. Section~\ref{sec:conclusion} concludes. @@ -92,40 +92,39 @@ \subsection{QUIC Transport Protocol} \textbf{Integrated security.} QUIC integrates TLS~1.3 into the transport handshake, achieving a 1-RTT connection establishment that includes both transport and cryptographic setup~\cite{rfc9001}. This compares favorably to TCP's 1-RTT handshake plus TLS's additional 1--2 RTTs. -\textbf{Unreliable datagrams.} The DATAGRAM extension (RFC~9221)~\cite{rfc9221} allows sending unreliable data within a QUIC connection, benefiting from the connection's encryption and congestion control without reliability or ordering guarantees. - \textbf{Loss detection.} QUIC uses per-packet sequence numbers (unlike TCP's per-byte) and acknowledgment-based loss detection with probe timeout (PTO) for tail loss~\cite{rfc9002}. This enables more precise loss recovery than TCP's retransmission timeout mechanism. \subsection{Related Work} \begin{table} \centering -\caption{Feature comparison of MQTT-over-QUIC approaches.} +\caption{MQTT-over-QUIC stream mapping across implementations. The operating modes originate in EMQ's \emph{MQTT-next} specification~\cite{yang2024mqttnext}; implementations differ chiefly in \emph{where} topic-to-stream mapping occurs. The \texttt{mqtt5} column is verified against its source; the EMQX and NanoMQ columns against their published documentation and source. For EMQX and NanoMQ, multistream is a client-initiated capability rather than default behavior: EMQX runs single-stream unless the application opens data streams through its client API, and NanoMQ's multistream exists only in its outbound bridge and is disabled by default. The ``Published evaluation'' row records where each implementation's performance numbers come from: EMQX's are published by its own vendor, Kumar's are peer-reviewed, and \texttt{mqtt5}'s are the present manuscript, whose implementation and evaluation share an author; the open implementation and reproducible harness are intended to offset that. The EMQX and NanoMQ columns describe the sources and documentation cited here at the time of writing and may change across releases.} \label{tab:feature-comparison} -\begin{tabular}{lccc} +\resizebox{\columnwidth}{!}{% +\begin{tabular}{lcccc} \toprule -Feature & \texttt{mqtt5} & EMQX~\cite{emqx} & Kumar~\cite{kumar2019mqtt-quic} \\ + & EMQX~\cite{emqx} & NanoMQ~\cite{nanomq-quic} & Kumar~\cite{kumar2019mqtt-quic} & \texttt{mqtt5} \\ \midrule -Stream strategies & 3 & 1 & 1 \\ -Per-topic isolation & \checkmark & --- & --- \\ -Per-publish isolation & \checkmark & --- & --- \\ -Datagram transport & \checkmark & --- & --- \\ -MQTT v5.0 full & \checkmark & \checkmark & --- \\ -Open-source & \checkmark & \checkmark & \checkmark \\ -Empirical evaluation & \checkmark & partial & \checkmark \\ +MQTT over QUIC & \checkmark & \checkmark & \checkmark & \checkmark \\ +Multistream (client-initiated) & \checkmark & \checkmark & --- & \checkmark \\ +Client topic$\rightarrow$stream mapping & --- & \checkmark\,(bridge) & --- & \checkmark \\ +Broker-initiated delivery streams & --- & --- & --- & \checkmark \\ +Broker topic$\rightarrow$stream mapping & --- & --- & --- & \checkmark \\ +Published evaluation & vendor & --- & peer-reviewed & this work \\ \bottomrule -\end{tabular} +\end{tabular}% +} \end{table} -Prior work on MQTT over QUIC falls into three categories. +Prior work on MQTT over QUIC falls into two strands, which we take in turn. -\textbf{Implementations and evaluations.} Kumar and Dezfouli~\cite{kumar2019mqtt-quic} implemented MQTT over QUIC using a single bidirectional stream, demonstrating reduced connection overhead (up to 56\% fewer handshake packets) and lower delivery latency (up to 55\%) compared to MQTT over TCP across wired and wireless testbeds. Fern{\'a}ndez et al.~\cite{fernandez2020iot-quic} evaluated MQTT over QUIC under emulated wireless conditions (WiFi, LTE, satellite), finding that QUIC maintains stable throughput under packet loss while TCP degrades significantly. Neither study explored stream multiplexing for topic-level HOL blocking isolation. +\textbf{Implementations and evaluations.} Kumar and Dezfouli~\cite{kumar2019mqtt-quic} implemented MQTT over QUIC using a single bidirectional stream, demonstrating reduced connection overhead (up to 56\% fewer packets exchanged with the broker) and lower delivery latency (up to 55\%) compared to MQTT over TCP across wired and wireless testbeds. Fern{\'a}ndez et al.~\cite{fernandez2020iot-quic} evaluated MQTT over QUIC under emulated wireless conditions (WiFi, LTE, satellite), finding that QUIC maintains stable throughput under packet loss while TCP degrades. Neither study explored stream multiplexing for topic-level HOL blocking isolation. -\textbf{Industrial implementations.} EMQX~\cite{emqx} is the only production MQTT broker to support QUIC transport, but uses a single-stream approach without configurable stream mapping. Table~\ref{tab:feature-comparison} compares the feature sets. +\textbf{The multistream operating modes, and where mapping happens.} Mapping MQTT traffic onto multiple QUIC streams is established industrial practice, and its taxonomy predates this work. EMQ's \emph{MQTT-next} specification~\cite{yang2024mqttnext} defines three operating modes, namely Single Stream, Simple multistreams, and Advanced multistreams, and names ``mitigate HOLB'' as the multistream benefit; it also defines the flow header and, for the advanced mode, broker-initiated streams. The single/per-flow/per-frame stream-mapping taxonomy is older still: the 2017 RTP-over-QUIC draft~\cite{rtpfolks2017} mapped media onto QUIC streams at exactly these three granularities. We therefore do not claim to originate these strategies or the flow header. Existing implementations, however, realize only the \emph{client-driven} portion of the design. EMQX~\cite{emqx} accepts multiple client-initiated streams but opens none of its own, so any delivery-side isolation is driven entirely by the client rather than the broker; in a default session its own client uses only the control stream. NanoMQ~\cite{nanomq-quic} pairs configured topics to streams via a per-topic stream identifier in its bridge configuration, but only within its outbound bridge \emph{client}, and only as a default-off, work-in-progress feature. Table~\ref{tab:feature-comparison} summarizes where each implementation performs mapping. The HOL blocking problem in multiplexed protocols has been studied extensively in the context of HTTP/2~\cite{rfc7540} and HTTP/3~\cite{rfc9114}. Langley et al.~\cite{langley2017quic} demonstrated QUIC's benefits at Internet scale, reporting reduced search latency and video rebuffering across Google's infrastructure. Subsequent empirical work found nuanced results: Kakhki et al.~\cite{kakhki2017quic} showed that QUIC's practical effect on web page loading is modest despite reducing transport-layer HOL blocking, and Marx et al.~\cite{marx2020hol} demonstrated that implementation-level decisions (prioritization, frame scheduling) further affect the degree of isolation achieved. Yu and Benson~\cite{yu2021quic} corroborated these nuanced findings through production-scale measurements. Our work applies stream multiplexing to the messaging domain, where the ``resource'' granularity is the MQTT topic rather than the HTTP request, and independent topic flows may benefit more directly from stream isolation. -To our knowledge, no prior work has (1)~defined multiple stream mapping strategies for MQTT over QUIC, (2)~evaluated the HOL blocking implications of each strategy under controlled network impairment, or (3)~quantified the throughput tradeoffs across strategies. +To our knowledge, no MQTT-over-QUIC \emph{broker} autonomously maps topics to server-initiated delivery streams, which is the mechanism we implement and evaluate, and no peer-reviewed study independently evaluates and compares these operating modes under controlled network impairment. This paper addresses that gap. %% --- Architecture --- @@ -140,6 +139,7 @@ \subsection{Stream Mapping Strategies} \begin{figure} \centering +\resizebox{\columnwidth}{!}{% \begin{tikzpicture}[ stream/.style={draw, rounded corners=2pt, minimum width=1.6cm, minimum height=0.5cm, font=\footnotesize}, connbox/.style={draw, dashed, rounded corners=4pt, inner sep=6pt}, @@ -159,249 +159,289 @@ \subsection{Stream Mapping Strategies} \node[label] at (0.8, 2.0) {(b) Per-topic}; \node[connbox, minimum width=2.2cm, minimum height=2.8cm] at (0.8, 0.2) {}; \node[stream, fill=blue!15] at (0.8, 1.2) {Stream 0}; -\node[stream, fill=green!15] at (0.8, 0.5) {Stream 1}; -\node[stream, fill=orange!15] at (0.8, -0.2) {Stream 2}; +\node[stream, fill=green!15] at (0.8, 0.2) {Stream 1}; +\node[stream, fill=orange!15] at (0.8, -0.8) {Stream 2}; \node[font=\tiny] at (0.8, 1.55) {\textit{control}}; -\node[font=\tiny] at (0.8, 0.85) {\textit{topic/a}}; -\node[font=\tiny] at (0.8, 0.15) {\textit{topic/b}}; +\node[font=\tiny] at (0.8, 0.55) {\textit{topic/a}}; +\node[font=\tiny] at (0.8, -0.45) {\textit{topic/b}}; \end{scope} \begin{scope}[shift={(6.4,0)}] \node[label] at (0.8, 2.0) {(c) Per-publish}; \node[connbox, minimum width=2.2cm, minimum height=2.8cm] at (0.8, 0.2) {}; \node[stream, fill=blue!15] at (0.8, 1.2) {Stream 0}; -\node[stream, fill=yellow!20] at (0.8, 0.5) {Stream $n$}; -\node[stream, fill=yellow!20] at (0.8, -0.2) {Stream $n$+1}; +\node[stream, fill=yellow!20] at (0.8, 0.2) {Stream $n$}; +\node[stream, fill=yellow!20] at (0.8, -0.8) {Stream $n$+1}; \node[font=\tiny] at (0.8, 1.55) {\textit{control}}; -\node[font=\tiny] at (0.8, 0.85) {\textit{msg 1}}; -\node[font=\tiny] at (0.8, 0.15) {\textit{msg 2}}; +\node[font=\tiny] at (0.8, 0.55) {\textit{msg 1}}; +\node[font=\tiny] at (0.8, -0.45) {\textit{msg 2}}; \end{scope} -\end{tikzpicture} -\caption{The three stream mapping strategies. (a)~Control-only multiplexes everything on stream~0. (b)~Per-topic assigns a dedicated stream per topic, providing topic-level isolation. (c)~Per-publish creates a new stream for each PUBLISH, providing message-level isolation.} +\end{tikzpicture}% +} +\caption{The three stream mapping strategies. (a)~Control-only multiplexes everything on stream~0. (b)~Per-topic assigns a dedicated stream per topic, intended to provide topic-level isolation. (c)~Per-publish creates a new stream for each PUBLISH, providing message-level isolation.} \label{fig:strategies} \end{figure} -\textbf{Control-only} (Fig.~\ref{fig:strategies}a) uses a single bidirectional QUIC stream for all MQTT traffic. This is functionally equivalent to running MQTT over a single TCP connection and provides no HOL blocking mitigation. It serves as the QUIC baseline, isolating the effect of QUIC's transport mechanisms (encryption, congestion control) from stream multiplexing. +\textbf{Control-only} (Fig.~\ref{fig:strategies}a) uses a single bidirectional QUIC stream for all MQTT traffic. This is structurally equivalent to running MQTT over a single TCP connection and provides no stream-level HOL blocking mitigation, though it still keeps a far lower tail than TCP at 5\% loss (Section~\ref{sec:exp2}). It serves as the QUIC baseline, isolating the effect of the QUIC transport as a whole from stream multiplexing. -\textbf{Per-topic} (Fig.~\ref{fig:strategies}b) assigns each MQTT topic to a dedicated QUIC stream. Control packets (CONNECT, SUBSCRIBE, PINGREQ, etc.) use stream~0, while PUBLISH packets are routed to topic-specific streams. A hash map maintains the topic-to-stream mapping, creating streams lazily on first publish. This provides topic-level HOL blocking isolation: a retransmission delay on one topic's stream does not block delivery on other topics' streams. +\textbf{Per-topic} (Fig.~\ref{fig:strategies}b) assigns each MQTT topic to a dedicated QUIC stream. Control packets (CONNECT, SUBSCRIBE, PINGREQ, etc.) use stream~0, while PUBLISH packets are routed to topic-specific streams. A hash map maintains the topic-to-stream mapping, creating streams lazily on first publish. This provides topic-level HOL blocking isolation: a retransmission delay on one topic's stream does not block delivery on other topics' streams, though the isolation narrows as the topic count grows, for the reason Section~\ref{sec:evaluation} measures. -\textbf{Per-publish} (Fig.~\ref{fig:strategies}c) creates a new unidirectional QUIC stream for each PUBLISH packet. This provides the finest-grained isolation---every message is independent---but at the highest overhead cost, as each message requires opening a new QUIC stream. +\textbf{Per-publish} (Fig.~\ref{fig:strategies}c) creates a new unidirectional QUIC stream for each PUBLISH packet. This provides the finest-grained isolation, because every message is independent, but at the highest overhead cost, as each message requires opening a new QUIC stream. \subsection{Flow Header Protocol} -When using per-topic or per-publish strategies, the receiver must determine how to demultiplex incoming data on non-control streams. We introduce a lightweight \emph{flow header} prepended to the MQTT payload on data streams. +Under per-topic and per-publish, the receiver must know how to demultiplex the data arriving on each non-control stream. For this purpose the MQTT-next specification~\cite{yang2024mqttnext} defines a \emph{flow header}, placed at the start of a data stream ahead of the MQTT packets it carries, and our implementation supports it. \begin{figure} \centering \begin{tikzpicture}[font=\footnotesize] -\draw (0,0) rectangle (7, 0.8); +\draw (0,0) rectangle (7.8, 0.8); + +\draw (1.0,0) -- (1.0,0.8); +\draw (2.3,0) -- (2.3,0.8); +\draw (3.5,0) -- (3.5,0.8); +\draw (4.5,0) -- (4.5,0.8); -\draw (1.2,0) -- (1.2,0.8); -\draw (2.4,0) -- (2.4,0.8); -\draw (7.0,0) -- (7.0,0.8); +\node at (0.5, 0.4) {Type}; +\node at (1.65, 0.4) {Flow ID}; +\node at (2.9, 0.4) {Expire}; +\node at (4.0, 0.4) {Flags}; +\node at (6.15, 0.4) {MQTT Packet Data}; -\node at (0.6, 0.4) {Type}; -\node at (1.8, 0.4) {Flags}; -\node at (4.7, 0.4) {MQTT Packet Data}; +\node[font=\tiny, text=gray] at (0.5, -0.25) {varint}; +\node[font=\tiny, text=gray] at (1.65, -0.25) {varint}; +\node[font=\tiny, text=gray] at (2.9, -0.25) {varint}; +\node[font=\tiny, text=gray] at (4.0, -0.25) {1 byte}; +\node[font=\tiny, text=gray] at (6.15, -0.25) {variable}; -\node[font=\tiny, text=gray] at (0.6, -0.25) {1 byte}; -\node[font=\tiny, text=gray] at (1.8, -0.25) {1 byte}; -\node[font=\tiny, text=gray] at (4.7, -0.25) {variable}; +\draw[decorate, decoration={brace, amplitude=4pt}] (0, 0.95) -- (4.5, 0.95) + node[midway, above=4pt, font=\tiny] {flow header, $\geq$4 bytes}; -\draw[decorate, decoration={brace, amplitude=4pt, mirror}] (0, -0.6) -- (7, -0.6) +\draw[decorate, decoration={brace, amplitude=4pt, mirror}] (0, -0.6) -- (7.8, -0.6) node[midway, below=5pt, font=\tiny] {QUIC stream payload}; \end{tikzpicture} -\caption{Flow header format. The 2-byte header identifies the packet type and flags, enabling the receiver to parse data streams without full MQTT framing on stream~0.} +\caption{Data flow header format, as defined by the MQTT-next specification~\cite{yang2024mqttnext}: variable-length integer \texttt{Type}, \texttt{Flow ID} and \texttt{Expire} fields and a one-byte \texttt{Flags} field, four bytes at minimum. On the client's publish path the header is optional and we leave it disabled throughout; on the broker's delivery path it is always written, so every server-opened stream under per-topic and per-publish carries one.} \label{fig:flow-header} \end{figure} -The flow header (Fig.~\ref{fig:flow-header}) consists of two bytes: a type field identifying the MQTT packet type (PUBLISH for data streams) and a flags byte encoding QoS level, retain, and duplicate delivery (DUP) indicators. The receiver reads this header to determine how to parse the remainder of the stream. This is more efficient than transmitting the full MQTT fixed header on every stream, since the stream itself provides the framing boundary. - -\subsection{Datagram Transport} - -As an alternative to stream-based transport, \texttt{mqtt5} supports QUIC datagrams (RFC~9221) for QoS~0 messages. Datagrams bypass QUIC's stream layer entirely: messages are sent as unreliable frames within the QUIC connection, benefiting from QUIC's encryption and connection management but without reliability or ordering guarantees. - -Datagram transport is designed for latency-sensitive telemetry where occasional message loss is acceptable. Since QUIC datagrams are subject to the connection's congestion window but not to per-stream flow control, lost messages are skipped rather than retransmitted. However, as both transports share the same per-connection congestion controller, the practical latency difference is small (see Section~\ref{sec:evaluation}). +The header (Fig.~\ref{fig:flow-header}) is optional. It carries a \texttt{Type} field distinguishing client from server data, a \texttt{Flow ID} that names the flow for recovery on a later stream, and an \texttt{Expire} interval, each a variable-length integer, followed by a one-byte \texttt{Flags} field whose bits are the specification's session-persistence controls (persistent subscriptions, topic aliases and QoS state, error tolerance, and clean-start semantics) rather than per-message MQTT flags. The receiver detects the header from the first byte and parses plain MQTT framing when it is absent. The two directions differ in our implementation. On the client's publish path the header is optional and we leave it disabled throughout, so client-side data streams carry plain MQTT. On the broker's delivery path it is unconditional: every stream the broker opens for delivery is prefixed with one. Control-only opens none, per-topic one per topic, and per-publish one per message, so the header is itself part of the per-message overhead that Section~\ref{sec:evaluation} measures on that path. \subsection{Broker Integration} -The client and broker configure their stream strategies independently. On the client side, the application selects a strategy (control-only, per-topic, or per-publish) before connecting; this governs how outbound PUBLISH packets are dispatched onto QUIC streams. On the broker side, a server-wide delivery strategy governs how the broker sends messages to subscribers. In a typical deployment, both sides are configured to the same strategy, but the architecture does not enforce this---a per-topic client can connect to a control-only broker, or vice versa. +The client and broker configure their stream strategies independently. On the client side, the application selects a strategy (control-only, per-topic, or per-publish) before connecting; this governs how outbound PUBLISH packets are dispatched onto QUIC streams. On the broker side, a server-wide delivery strategy governs how the broker sends messages to subscribers. In a typical deployment, both sides are configured to the same strategy, but the architecture does not enforce this, so a per-topic client can connect to a control-only broker, or vice versa. -For outbound delivery, the broker's stream manager maintains a per-connection topic-to-stream mapping. When configured for per-topic delivery, streams are created lazily on first use and reused for subsequent messages on the same topic, with LRU eviction to bound stream count. For per-publish delivery, the broker opens a new unidirectional stream for each message. On the receive side, the broker's QUIC acceptor classifies incoming streams by ID: stream~0 carries the full MQTT packet parser, while non-zero streams use the flow header parser to extract PUBLISH packets and route them into the topic distribution engine. +For outbound delivery, the broker's stream manager maintains a per-connection topic-to-stream mapping. When configured for per-topic delivery, streams are created lazily on first use and reused for subsequent messages on the same topic, with LRU eviction to bound stream count. For per-publish delivery, the broker opens a new unidirectional stream for each message. On the receive side, the broker's QUIC acceptor classifies streams by kind and arrival order rather than by numeric identifier: the first bidirectional stream accepted becomes the control stream and carries the full MQTT packet parser, while later data streams are probed for a flow header and otherwise parsed as plain MQTT, their PUBLISH packets routed into the topic distribution engine. %% --- Implementation --- \section{Implementation} \label{sec:implementation} -\texttt{mqtt5}\footnote{\url{https://github.com/LabOverWire/mqtt-lib}} is an open-source MQTT library written in Rust. The library provides both a client and broker, supporting TCP, TLS~1.3, WebSocket, and QUIC transports. The QUIC implementation uses Quinn~\cite{quinn}, a Rust QUIC library built on the \texttt{tokio} async runtime. +\texttt{mqtt5} is an open-source MQTT library written in Rust. The library provides both a client and broker, supporting TCP, TLS~1.3, WebSocket, and QUIC transports. The QUIC implementation uses Quinn~\cite{quinn}, a Rust QUIC library built on the \texttt{tokio} async runtime. \subsection{Transport Abstraction} -The library defines a \texttt{Transport} trait abstracting read/write operations over any underlying connection. For TCP and TLS, this wraps a single byte stream. For QUIC, the transport layer manages multiple streams through a \texttt{StreamDispatcher} that implements the configured strategy. +The library defines a \texttt{Transport} trait abstracting read/write operations over any underlying connection. For TCP and TLS, this wraps a single byte stream. For QUIC, the transport layer manages multiple streams through a \texttt{QuicStreamManager} that implements the configured strategy. -The stream dispatcher maintains a \texttt{HashMap} for per-topic mapping, creating streams lazily on first use and reusing them for subsequent messages on the same topic. For per-publish, the dispatcher opens a new unidirectional stream for each PUBLISH and closes it after sending. Stream~0 (the initial bidirectional stream) is always reserved for control packets. Quinn's default behavior packs STREAM frames from multiple streams into single QUIC packets for efficiency. To isolate the effect of this frame packing on inter-stream coupling, we extended Quinn with a configurable frame packing policy: the default \emph{Greedy} policy preserves the original batching behavior, while a new \emph{StreamIsolated} policy restricts each packet to a single stream's data, preventing cross-stream coupling at the packet level. +The stream manager maintains a per-topic map from topic name to its send stream, creating streams lazily on first use and reusing them for subsequent messages on the same topic. For per-publish, it opens a new unidirectional stream for each PUBLISH and closes it after sending. The initial bidirectional stream carries control packets, and under control-only it carries the PUBLISH traffic as well. \subsection{Connection Lifecycle} QUIC connections follow the same MQTT session lifecycle as TCP connections. The CONNECT packet is always sent on stream~0. After CONNACK, subsequent PUBLISH packets are dispatched according to the client's configured stream strategy. -On the broker side, the QUIC acceptor spawns a dedicated task per connection. Incoming streams are classified by their stream ID: stream~0 receives the full MQTT packet parser, while non-zero streams use the flow header parser to extract PUBLISH packets and route them into the broker's topic distribution engine. - -\subsection{Datagram Path} - -When datagram mode is enabled, QoS~0 PUBLISH packets are sent as QUIC DATAGRAM frames. The MQTT packet is serialized with the flow header prepended and submitted via Quinn's \texttt{send\_datagram} API. The maximum datagram size is bounded by the connection's \texttt{max\_datagram\_size} parameter, which is negotiated during the QUIC handshake. Messages exceeding this limit fall back to stream transport. - -On the receive path, datagrams are delivered via a separate channel from stream data. The broker's QUIC handler polls both the stream accept loop and the datagram receive loop concurrently, feeding both into the same message processing pipeline. +On the broker side, the QUIC acceptor spawns a dedicated task per connection, classifying streams as described in Section~\ref{sec:architecture}. \subsection{Benchmarking Infrastructure} +\label{sec:bench-infra} The library includes an integrated benchmarking tool (\texttt{mqttv5 bench}) that measures latency and throughput. Latency mode embeds microsecond timestamps in payloads and computes p50/p95/p99 percentiles. Throughput mode counts messages per second over a configurable duration with warmup. Both modes support configurable publisher/subscriber counts, QoS levels, payload sizes, and transport parameters. -Network impairment is applied externally via \texttt{tc-netem}, and broker/client resource usage (RSS, CPU, network I/O) is sampled at 1\,Hz during each run using procfs monitors. +The evaluation uses two builds of \texttt{mqtt5}. The \emph{pacing build} is release 0.41 of the library, whose benchmarking tool can pace publishers at a fixed aggregate rate instead of letting them send as fast as they can. The \emph{evaluation build} lacks this option; its binary checksum is recorded with the data~\cite{bracht2026zenodo}. The two builds' absolute rates differ, by amounts Section~\ref{sec:exp3} reports, so every conclusion rests on comparisons within one build; where the two builds' results appear together, the text reports the difference, and a side-by-side comparison at 10\% loss measures it directly. The pacing build produces three sets of Experiment~3 runs: the offered-load check (paced publishers and the unpaced runs they are compared against), the broker delivery-strategy runs at fixed publish rates with their unpaced counterparts, and a side-by-side comparison of both builds at 10\% loss. The evaluation build produces every other result. + +Network impairment is applied externally via \texttt{tc-netem}, and broker/client resource usage (RSS, CPU, network I/O) is sampled at 1\,Hz during each run using procfs monitors. To observe the loss-bearing path directly, the 5\% loss condition of Experiment~2's three QUIC arms also has an instrumented arm of five runs each, with the broker sampling its own QUIC connection statistics at 10\,Hz on each connection it holds, giving congestion window, losses and congestion events; the subscriber-facing connection is identified per run as the one carrying more packets. The instrumented arm uses the Experiment~2 configuration and matches the uninstrumented arm's tail ordering and level (median per-topic p99 133.7, 161.5 and 209.1\,ms for control-only, per-topic and per-publish, against 131.9, 164.3 and 194.7\,ms); tail figures come from the uninstrumented arm and connection statistics from the instrumented one. For Experiment~3's primary arms the impairment is instead applied on a forwarding VM placed between the broker and its clients by a policy-based route, with receive coalescing disabled so that each loss removes exactly one packet (Section~\ref{sec:loss-method}). %% --- Evaluation --- \section{Evaluation} \label{sec:evaluation} -We evaluate \texttt{mqtt5} across five experiments measuring connection establishment, head-of-line blocking, throughput, stream strategy behavior, and datagram transport. All experiments use QoS~0 unless otherwise noted. +We conduct four experiments, measuring \texttt{mqtt5}'s connection establishment, head-of-line blocking, throughput, and stream strategy behavior. All four use QoS~0, so that the comparison measures the transports' own loss recovery and nothing layered above it. QoS~1 and~2 add MQTT-level acknowledgements that travel the same impaired path, and the Receive Maximum in-flight window~\cite{mqtt-v5-spec} caps unacknowledged messages, so every transport-level delay to an acknowledgement would also throttle the publisher at the MQTT layer and mix application-level flow control into the transport comparison. On a live connection MQTT~v5.0 leaves retransmission to the transport, resending unacknowledged messages only after a reconnect. + +Experiments~2 and~4 set the broker's delivery strategy to match the client's. Experiment~3's main comparison holds the broker at per-topic delivery across the QUIC arms, isolating the client's publish-side mapping, and a second set of arms varies only the broker: control-only against per-topic delivery with a control-only client, and per-publish against per-topic delivery with a per-publish client. Because client and broker strategies co-vary in Experiment~2, the broker-side packet and loss counts reported there compare matched client-and-broker strategies and cannot isolate the broker's mapping. \subsection{Experimental Setup} \begin{table} \centering -\caption{Experimental parameters across all five experiments.} +\caption{Common parameters for the four experiments.} \label{tab:params} -\begin{tabular}{ll} +\small +\begin{tabular}{@{}>{\raggedright\arraybackslash}p{0.33\columnwidth}>{\raggedright\arraybackslash}p{0.60\columnwidth}@{}} \toprule Parameter & Value \\ \midrule Hardware & GCP n2-standard-4 (4 vCPU, 16 GB RAM) \\ -Topology & Co-located pub/sub (Exp 2), separate VMs (others) \\ +Topology & Co-located pub/sub (Exp 2), separate VMs (others); forwarding VM (Exp 3) \\ Region & us-west1-b (same zone, $<$1ms baseline RTT) \\ -OS & Debian 12, kernel 6.1 \\ -Network emulation & \texttt{tc-netem} on broker (symmetric) \\ -QUIC implementation & Quinn 0.11 (Rust) \\ +OS & Ubuntu 24.04 LTS (kernel 6.17.0-1008-gcp pinned for Exp 3's per-packet and paced runs) \\ +One-way delay & 0--200\,ms swept (Exp 1), 25\,ms (Exp 2, 4; also swept 10--50 in Exp 4), 10\,ms (Exp 3) \\ +Network emulation & \texttt{tc-netem}, Bernoulli loss: per buffer on broker egress (Exp 2, 4; Exp 1 delay only); per packet on a forwarding VM (Exp 3; Section~\ref{sec:loss-method}) \\ +QUIC library & Quinn 0.11 (Rust) \\ TLS & rustls 0.23 (TLS 1.3 only) \\ -Congestion control & QUIC: NewReno / TCP: CUBIC \\ -Payload size & 256 bytes (all experiments) \\ -Publish rate & 500 msg/s fixed (Exp 2), max throughput (others) \\ -Measurement & 60s duration, 5s warmup (30s/0s warmup for Exp 1) \\ -Runs per config & 15 (all experiments) \\ -Frame packing & Greedy (default); StreamIsolated (Exp 2 ablation) \\ +Congestion control & CUBIC (both QUIC and TCP) \\ +Implementation & \texttt{mqtt5} evaluation build; pacing build for Exp 3 paced checks (Section~\ref{sec:bench-infra}) \\ +Frame packing & Greedy (the default) in every arm \\ +Concurrent stream limit & 100 (the QUIC library default) except where Exp 4 sweeps it \\ +Per-stream receive window & 256\,KB except where Exp 4 sweeps it; connection window 1\,MB \\ +Flow headers & Off on the client publish path; always present on broker-opened delivery streams \\ +Payload size & 256 bytes (Exp 2--4; Exp 1 establishes connections only) \\ +Publish rate & 500 msg/s base, swept 125--2000 (Exp 2); max throughput (others), paced 1.25K--20K in Exp 3 paced checks \\ +Measurement & 60\,s (Exp 2), 70\,s subscriber (Exp 3), 60\,s publisher (Exp 4), 30\,s (Exp 1); 5\,s warmup (0\,s for Exp 1) \\ +Runs per config & 15 (9--15 Exp 2 sweep; 5 Exp 2 instrumented arm; 10 Exp 4; 5--15 Exp 3 robustness and paced runs; exclusions noted where reported) \\ +HOL spike window & 10\,ms (Exp 2 spike isolation) \\ +Broker subscriber queue & 10{,}000 messages per subscriber; QoS~0 messages are dropped when full \\ \bottomrule \end{tabular} \end{table} -Table~\ref{tab:params} summarizes the common parameters. Network impairment is applied symmetrically on the broker interface using \texttt{tc-netem}. Each experiment varies one or two factors while holding others constant. Experiment~2 uses co-located publisher and subscriber on a single client VM to eliminate network variability between pub/sub; all other experiments use separate VMs. +Table~\ref{tab:params} summarizes the common parameters. Two details behind it matter for reproduction. Experiment 3's delivered throughput, and the single-connection delivered rates quoted in Experiment 4, are the mean of the per-second delivered rate over the central 80\% of the window, the first and last 10\% of samples being dropped as ramp-up and drain; Experiment 4's publish rates and Experiment 2's rates are instead a message count divided by the full window. The per-packet Experiment~3 runs are randomized across loss levels, loss-application arms and transports. Network impairment is applied with \texttt{tc-netem} on the broker's egress interface in Experiments~1, 2 and~4 (Experiment~1 applies delay only), and on a forwarding VM in Experiment~3. Experiment~3's robustness arms and the per-packet variant of Experiment~2 keep the delay on the broker and apply the loss on the forwarding VM (Section~\ref{sec:loss-method}). Each experiment varies one or two factors while holding others constant. Experiment~2 uses co-located publisher and subscriber on a single client VM to eliminate network variability between pub/sub; all other experiments use separate VMs. + +\textbf{Statistical reporting.} Unless a caption states otherwise, each plotted data point aggregates 15 independent runs. The line plots show the mean with its 95\% confidence interval, computed as the Student~$t$ interval $\bar{x}\pm t_{0.975,\,n-1}\,s/\sqrt{n}$ and rendered as capped error bars or a shaded band. The connection-latency bars of Fig.~\ref{fig:conn-latency} are different in kind: the bar is the p50 and the whisker reaches the p95 of the within-run latency distribution, not a confidence interval of the mean. Because the delay is fixed and the Bernoulli loss rate is held constant, run-to-run variation in the throughput and strategy cells is low, so these intervals are frequently narrower than the plotted marker and can be hard to see. They are drawn at every point and never omitted; their half-width stays below 4\% of the mean at the large majority of points on the throughput and strategy plots, and is visibly wider only on the density-corrected excess of Fig.~\ref{fig:hol-excess}, where the underlying values are small and most cells aggregate fewer runs (9--10; 15 for the 8-topic, 500\,msg/s cells) than the 15-run default. An interval that is not discernible is present but smaller than the plotting symbol, which reflects measurement stability rather than missing data. + +\subsection{Emulating Packet Loss} +\label{sec:loss-method} + +\texttt{tc-netem} makes one loss decision per buffer it is handed and drops the whole buffer. With segmentation offload the kernel hands the qdisc buffers that carry several packets, and QUIC's datagram batching does the same for UDP, so loss applied on a sending host clusters into fewer, larger events whose size depends on the transport and the load. The qdisc statistics hide this, since they count the packets inside each buffer while counting a dropped buffer once; the clustering shows up only as a drop fraction below the configured rate. Applying loss this way on the broker's egress, as Experiments~2 and~4 do, dropped 7.7\% of QUIC's packets and 9.0\% of TCP+TLS's at 10\% configured loss in Experiment~3's fan-out configuration (three runs per transport). + +For Experiment~3 we therefore apply both the delay and the loss on a separate forwarding VM, through which a policy-based route sends the broker's traffic to its clients. Receive coalescing is disabled on that VM, so each buffer its netem sees is one wire packet; a kernel probe on its enqueue path found no multi-packet buffer and a largest frame of 1474 bytes. The measured drop fraction matched the configured rate within its 99.9\% binomial interval in all but 2 of 972 lossy runs. With a 99.9\% interval about one run in a thousand falls outside by chance, so two is consistent with correct emulation; we excluded both (4.93\% and 4.94\% at 5\%). The forwarding hop adds 28--58\,$\mu$s one way, which we subtract from the emulated delay, and at 0\% loss routing through it changes no transport's throughput by more than 2\% relative to the direct path. Loss remains Bernoulli, each packet being dropped independently with the configured probability, and applies only to broker-to-client traffic. Experiments~2 and~4 keep the broker-egress method; Section~\ref{sec:exp2} also reports Experiment~2's reference cell under per-packet loss. \subsection{Experiment 1: Connection Establishment Latency} -We measure cold-start connection latency across TCP, TLS~1.3, and QUIC at five one-way delays (0, 25, 50, 100, 200\,ms) using 100 sequential connections per run. +We measure cold-start connection latency across TCP, TLS~1.3, and QUIC at five one-way delays (0, 25, 50, 100, 200\,ms) over 30\,s of back-to-back cold connections per run. The 0\,ms condition is the only one in which the client opens connections faster than the network paces them, so its results are shaped by client-side socket resources (ephemeral ports, \texttt{TIME\_WAIT} state, accept backlog), none of which we tuned; we read that condition as a property of the test host as much as of the transport. \begin{figure} \centering -\includegraphics[width=\columnwidth]{Figure_4} -\caption{Connection latency (p50 bars, p95 whiskers) across network delays. QUIC's integrated TLS handshake achieves comparable latency to raw TCP despite cryptographic overhead, and outperforms the separate TCP+TLS~1.3 handshake at all RTTs.} +\includegraphics[width=\columnwidth]{Figure_3} +\caption{Connection latency (p50 bars, p95 whiskers) across network delays. QUIC's integrated TLS handshake achieves comparable latency to raw TCP despite cryptographic overhead, and outperforms the separate TCP+TLS~1.3 handshake at every emulated delay of 25\,ms or more. At 0\,ms, where the handshake is bound by local processing rather than the network, all three complete in under two milliseconds, QUIC and TLS are indistinguishable, and raw TCP is fastest.} \label{fig:conn-latency} \end{figure} -Fig.~\ref{fig:conn-latency} shows that QUIC 1-RTT connection establishment matches TCP's cold-connect latency at all tested delays. TLS~1.3 over TCP consistently adds one additional round trip. At 200\,ms delay, QUIC connects in 402\,ms (p50) versus TCP's 400\,ms and TLS's 842\,ms. The integrated handshake eliminates the sequential TCP+TLS penalty without sacrificing security. We did not test QUIC 0-RTT resumption, which could further reduce connection latency for returning clients. +Fig.~\ref{fig:conn-latency} shows that QUIC 1-RTT connection establishment matches TCP's cold-connect latency at every emulated delay of 25\,ms or more. At 0\,ms the comparison is dominated by local processing rather than the network: QUIC (1.55\,ms) and TLS (1.59\,ms) are indistinguishable and both sit about 2.4--2.5$\times$ above raw TCP's mean of 0.64\,ms, which runs no cryptographic handshake; raw TCP's runs split between 0.26--0.30\,ms and 1.0--1.2\,ms, so that mean describes no single run. That cell is also the only one in which connections failed, and the transports fare oppositely there: TCP lost 26.3\% of its attempts and TLS~1.3 4.4\%, while QUIC completed every one. At 25\,ms and above no arm failed a connection. TLS~1.3 over TCP consistently adds one additional round trip. At 200\,ms delay, QUIC connects in 402\,ms (p50) versus TCP's 400\,ms and TLS's 602\,ms. The p95 whiskers are tight beyond the 0\,ms case, with p95 within 2\% of p50 at every delay of 25\,ms or more (within 0.6\% for TCP and TLS~1.3); only at 0\,ms, where the handshake is bound by local processing rather than the network, is the distribution appreciably wider. The integrated handshake eliminates the sequential TCP+TLS penalty without sacrificing security. We did not test QUIC 0-RTT resumption, which could further reduce connection latency for returning clients. \subsection{Experiment 2: Head-of-Line Blocking} +\label{sec:exp2} -This experiment is the core motivation for MQTT-over-QUIC multistream transport. We publish to 8 topics simultaneously at a fixed rate of 500\,msg/s and measure whether latency fluctuations on one topic propagate to others. The fixed rate is chosen to remain well below connection capacity at all loss rates (the minimum sustained capacity under 5\% loss at 25\,ms RTT exceeds 500\,msg/s), ensuring that any observed latency variation reflects transport-layer behavior rather than queue buildup from rate saturation. +This experiment is the core motivation for MQTT-over-QUIC multistream transport. Head-of-line blocking couples logically independent topics: when packets share one ordered connection, a single loss delays every topic queued behind it~\cite{scharf2006hol}. We publish across many topics at once and measure whether this cross-topic propagation occurs. In each QUIC arm the broker's delivery strategy matches the client's, so the arms exercise the broker-autonomous delivery mapping as well as the client's publish mapping. Around a reference point of 8 topics and an aggregate 500\,msg/s, we vary two factors independently: holding the rate at 500\,msg/s we sweep the topic count from 2 to 32 topics, and holding the count at 8 topics we sweep the offered rate from 125 to 2000\,msg/s; the two sweeps meet at this reference point, which we also use for the tail-latency comparison. At 500\,msg/s the connection is unsaturated, so latency variation reflects head-of-line blocking rather than queue buildup, while the highest swept rates deliberately drive a lossy connection into saturation. We emulate Bernoulli packet loss at 0\%, 1\%, 2\% and 5\%. The tail-latency comparison spans all four, with 0\% as an unimpaired baseline; the density-corrected timing analysis uses the 1\% setting alone, since it needs spikes to be sparse (as explained below) and at 0\% loss no retransmissions occur, so head-of-line blocking is absent by construction. -We quantify HOL blocking using two complementary metrics. \emph{Inter-topic spread} measures the maximum difference between per-topic mean latencies within 100\,ms time windows: high spread means topics experience different latencies (isolation), while low spread means topics are coupled. \emph{Spike isolation ratio} measures the fraction of latency spikes that co-occur across all topics: a ratio of 1.0 means every spike affects all topics simultaneously (full HOL blocking), while lower values indicate independent spikes. +We measure HOL blocking directly, through the \emph{timing} of latency spikes: does a spike on one topic co-occur on the others (coupling) or stay confined to it (isolation)? A latency spike is a message whose latency exceeds twice the median of the preceding 50 messages on its topic, and the \emph{spike isolation ratio} is the fraction of spikes that co-occur with a spike on at least one other topic within a 10\,ms window: 1.0 means every spike propagates across topics (full HOL blocking), lower values mean spikes stay confined. Because a raw co-occurrence count is inflated by spike \emph{density} (when spikes are frequent, some coincide by chance), we report the density-corrected \emph{excess}: the observed ratio minus a null obtained by independently circularly shifting each topic's spike train (averaged over 100 shifts), which preserves each topic's spike density and autocorrelation while removing genuine cross-topic alignment. The excess is meaningful only while the density null stays well below 1: because the observed ratio cannot exceed 1, the excess is bounded by one minus the null, and once chance coincidences become common that bound squeezes coupled and isolated strategies together. We ran the sweep at 1\% and 5\% loss only. \begin{figure} \centering -\includegraphics[width=\columnwidth]{Figure_1} -\caption{Inter-topic latency spread across packet loss rates (25\,ms RTT, 8 topics, 15 runs per configuration). Per-topic QUIC spread amplifies 200$\times$ from baseline to 5\% loss, consistent with independent per-stream loss recovery. TCP spread increases only 1.2$\times$, indicating that all topics remain tightly coupled under loss.} -\label{fig:spread} +\includegraphics[width=\columnwidth]{Figure_4} +\caption{Density-corrected head-of-line isolation (means over 9--15 runs per cell, 15 for the 8-topic, 500\,msg/s cells, 95\% confidence intervals). (a)~Excess spike co-occurrence (observed ratio minus a density null; lower is more isolated) versus topic count at 500\,msg/s and 1\% loss. TCP and control-only stay coupled at every topic count; per-publish is the most isolated; per-topic isolates only at small topic counts and reaches single-stream coupling from 4 topics on. (b)~The density null grows with offered rate, so the raw ratio cannot be compared across rates or loss levels.} +\label{fig:hol-excess} \end{figure} -Fig.~\ref{fig:spread} shows inter-topic spread at 25\,ms RTT across four loss rates with 15 runs per configuration. At 0\% loss, all transports show low spread: TCP at 6.7\,ms, QUIC per-topic at 0.09\,ms, per-publish at 0.11\,ms, and control-only at 0.09\,ms. Under 5\% loss, per-topic QUIC spread amplifies to 17.8\,ms (200$\times$ baseline), while TCP increases to only 8.4\,ms (1.2$\times$). Per-publish shows 107$\times$ amplification and control-only shows 71$\times$. +Fig.~\ref{fig:hol-excess}(a) confirms both the blocking and its isolation. The single-stream transports couple the topics: on TCP and QUIC control-only a loss's spike co-occurs across topics (excess 0.83--0.93 across the topic sweep). Splitting topics onto separate streams decouples them, and at 2 topics per-publish and per-topic isolate (excess 0.33 and 0.59). The isolation is real but bounded: it erodes as topics are added, and from 4 topics on per-topic's excess reaches single-stream coupling (0.82 against TCP's 0.83 at 4 topics, 0.84 against 0.84 at 8), and at 16--32 topics it is at control-only's level, above TCP's, leaving only per-publish still isolating (0.55 at 8 topics). This comparison is confined to 1\% loss, where the density null stays at or below 0.14 across the whole topic sweep (Fig.~\ref{fig:hol-excess}(b) shows it growing broadly with offered rate instead), so excess is well determined. At 5\% loss the null is two to six times larger at 4--16 topics (up to ten times at 32), reaching 0.44--0.50 at 8 topics and 0.53--0.65 at 16, and 0.59--0.73 for the QUIC arms at 1000\,msg/s, which leaves too little room below the 1.0 ceiling for excess to be separable from density. At 5\% loss and 2000\,msg/s the failure runs the other way, because the transports are overloaded. TCP's single connection to the subscriber carries only about 880\,msg/s whether 1000 or 2000\,msg/s is offered, within the loss-limited range of Mathis et al.~\cite{mathis1997macroscopic} for this path (about 810--1140\,msg/s for a model constant between 0.87 and 1.22, a 25\,ms round trip, 1408-byte segments and about 270 bytes per message), so the rate is a property of one connection rather than of TCP in aggregate. The excess piles up in the broker's queue for that subscriber, which holds 10{,}000 messages and drops QoS~0 messages once full: roughly half of the offered messages are discarded there, and every delivered message waits about as long as the full queue takes to drain, roughly 12\,s. A spike is defined relative to recent latency, so a retransmission stall of tens of milliseconds no longer registers against that baseline, and the detector finds no spikes at all, in every TCP run and in nine of ten per-publish runs. The observed ratio and the null are then both zero because there is nothing to count, not because the topics are isolated. TCP at 1000\,msg/s is already capped at the same rate but its queue does not fill within a run; it finds no spikes in five of its ten runs and a single, unpaired spike in a sixth. Control-only and per-topic are queueing too at 2000\,msg/s and lose all spikes in one and two runs respectively, so their excess is too erratic to estimate (confidence intervals of $\pm$0.27--0.28). -The high spread for multi-stream QUIC strategies under loss is consistent with QUIC's independent per-stream loss recovery~\cite{rfc9000}: when a packet carrying data for one topic is lost, only that topic's stream should experience retransmission delay, while other streams continue unaffected. TCP's single byte stream provides no such isolation, which is consistent with the low spread amplification observed. In absolute terms, TCP's coupled degradation is also more severe: at 5\% loss, TCP p95 latency reaches 180\,ms and p99 reaches 355\,ms across all topics uniformly, while QUIC per-topic achieves 99\,ms p95 and 183\,ms p99 with the impact confined to individually affected streams. +How that isolation shapes tail latency (Fig.~\ref{fig:hol-tail}) is governed by two effects. First, the transport, not stream mapping, sets the level: even single-stream control-only, which shares TCP's single ordered-stream structure, holds a median 132\,ms at 5\% loss against TCP's 351\,ms. We did not isolate which transport difference is responsible. Second, the strategies carry the same message load but differ in how many packets they put on the \emph{delivery} path, and under this loss method the tail ordering tracks that packet count. Control-only adds no per-stream framing and has both the lowest median (132\,ms) and the lowest 90th-percentile tail (162\,ms); per-topic follows at a 164\,ms median, and per-publish, which opens a stream for every message, is worst at 195\,ms (under per-packet loss it ties per-topic; see below). Both gaps are significant over the 15 runs per arm (Mann--Whitney $U$, two-sided: control-only versus per-topic $p=0.006$, versus per-publish $p=0.0003$). The broker-side statistics from the instrumented runs (Section~\ref{sec:bench-infra}; medians over five runs per arm) show a chain consistent with that ordering: on the subscriber-facing connection the broker sends about 10.8K packets per run under control-only, 11.8K under per-topic and 20.9K under per-publish. The emulated loss rate is identical for all three arms, so the extra packets become proportionally more losses (540, 598 and 1076) and disproportionately more congestion events (347, 407 and 853). The congestion window responds only at the extreme: per-publish's median is measurably depressed (6.9K versus control-only's 7.7K bytes, $p=0.008$) while per-topic's is indistinguishable from control-only's ($p=0.69$). Under this loss method, per-publish's longer tail thus coincides with a smaller window as well as more recovery events, whereas per-topic's coincides with the extra recovery events alone. Experiment~4 shows per-publish's one-stream-per-message pattern separately cutting a single connection's publish rate roughly ninefold, through the peer's stream limit rather than through packet count, a single-connection limit that does not reach the broker's aggregate throughput when the broker delivers per-topic, although a ceiling consistent with it appears in broker-side per-publish delivery per subscriber (Experiment~3). Individual runs can spike well above their arm's median. One control-only run at 5\% loss reached 342\,ms, 2.6 times the arm's median of 132\,ms; its second-worst run reached 168\,ms, and among the fifteen runs per arm the worst multistream runs reached 174\,ms (per-topic) and 254\,ms (per-publish). The five instrumented and fifteen per-packet runs of this cell produced no control-only run above 159\,ms (the instrumented per-publish runs reached 292\,ms, 1.4 times their median). Smaller but still large outliers, 1.6--1.8 times the cell median, appear in all three QUIC strategies elsewhere in the 5\%-loss sweeps at rates up to 500\,msg/s, so we treat the 342\,ms run as an outlier rather than as a property of control-only; fifteen runs per arm cannot estimate how often such outliers occur. Below 5\% loss the ordering is different. At 0\% the three sit within 0.4\,ms of one another. At 1\% control-only is in fact the highest of the three (58.3\,ms against per-topic's 55.8 and per-publish's 54.0; $p=0.007$ and $p=0.003$), so finer stream granularity does help at light loss. At 2\% control-only and per-topic are indistinguishable (68.1 versus 68.4\,ms, $p=0.15$) while per-publish has separated upward (82.6\,ms; $p<0.001$ against control-only, $p=0.002$ against per-topic). At 500\,msg/s, control-only's tail advantage is therefore specific to the heaviest loss we emulate; the rate sweep also shows it at 1\% loss and 1000\,msg/s (61 against 70 and 82\,ms). \begin{figure} \centering -\includegraphics[width=\columnwidth]{Figure_2} -\caption{Spike isolation ratio across packet loss rates. A ratio of 1.0 means all latency spikes co-occur across topics (full HOL blocking). TCP remains near 1.0 at all loss rates, while per-topic (0.73) and per-publish (0.53) QUIC show progressively more independent spike behavior at 1\% loss.} -\label{fig:spike-iso} +\includegraphics[width=\columnwidth]{Figure_5} +\caption{Mean per-topic p99 latency versus packet loss at 8 topics (line is the median over runs; shaded band spans the 10th--90th percentile of runs). At 5\% loss all QUIC strategies stay below TCP, and at 2\% control-only and per-topic do. At 1\% per-topic and per-publish are below TCP and control-only is indistinguishable from it; in the 1\% rate and topic sweeps TCP's tail is far higher at 1000--2000\,msg/s and at 32 topics, and nominally (never significantly) lower in a few cells. At 5\% loss control-only has the lowest median and 90th-percentile tail and per-publish the highest (under per-packet loss per-publish ties per-topic; Section~\ref{sec:exp2}); at 2\% control-only and per-topic are indistinguishable and only per-publish has separated upward; at 1\% control-only is the highest of the three. A single control-only run at 5\% loss reached 342\,ms (outside the band); Section~\ref{sec:exp2} discusses such outliers.} +\label{fig:hol-tail} \end{figure} -Fig.~\ref{fig:spike-iso} provides a complementary view through spike isolation ratio. TCP maintains a ratio near 1.0 at all loss rates (0.98 at 1\% loss), confirming that virtually all latency spikes propagate to every topic. Per-topic QUIC drops to 0.73 and per-publish to 0.53 at 1\% loss, indicating that 27--47\% of spikes affect individual topics independently. Control-only QUIC remains at 0.98, consistent with its single-stream architecture that provides no inter-topic isolation. +Under this loss method, coarser mapping (control-only) minimizes packet overhead and has the lowest median and 90th-percentile tail at 5\% loss. Finer mapping decorrelates spike timing but adds framing packets, and per-publish's per-message streams roughly double the packet count and, under this method, raise its median. At 2000\,msg/s under 5\% loss it also delivers less than the other QUIC arms, 1.64K against 1.94K\,msg/s. That shortfall is mostly the concurrent stream limit rather than the framing: the same cell in Experiment~4's setup gives 1.61K at the 100-stream default and 1.90K once the limit is raised to 1000 (three runs each). Both the isolation and the tail therefore turn on one variable, stream granularity. Because the overhead term scales with rate, the ordering crosses over: at 125\,msg/s under 5\% loss per-publish's extra packets are few in absolute number and it has the lowest tail of the three (96\,ms, $p=0.0002$ against control-only), whereas by 1000\,msg/s it is nominally the worst (361 against 302 and 313\,ms, not significant; the rate sweep uses broker-egress loss only). The spike isolation itself erodes with topic count because, at a fixed rate, a single packet increasingly coalesces several topics' data, so one loss recouples them. QUIC loss detection is per-connection rather than per-stream~\cite{rfc9002}, so we could not separate per-stream recovery timing from packet count and coalescing. -QUIC's congestion control operates per-connection~\cite{rfc9002, rfc9308}, sharing a single congestion window across all streams. We hypothesize that this shared congestion response limits the degree of isolation: a loss event reduces the congestion window for all concurrent streams, even though stream-level loss recovery is independent. This is consistent with per-topic isolation being partial (0.73) rather than complete ($\approx$0). +Experiment~2 applies its loss with the broker-egress method of Section~\ref{sec:loss-method}, so we also ran its 8-topic, 500\,msg/s reference cell at 0\% and 5\% loss with per-packet loss on the forwarding VM (15 runs per arm; at 0\% two TCP runs were excluded for retransmissions on a clean path, leaving 13). At 0\% every arm stays within 0.3\,ms of its level in the broker-egress runs (27.3--27.7\,ms). At 5\% the ends of the ordering hold: control-only remains the lowest (108\,ms; $p=0.0004$ against either multistream arm), and TCP stays far above every QUIC arm (458\,ms, with runs spread from 200 to 1297\,ms). Only one arm moved significantly at 5\%: per-publish fell from 195 to 143\,ms ($p<0.001$) and no longer trails per-topic (141\,ms, $p=0.65$), although it still sends about 1.5 times per-topic's packets on the loss-bearing link. Its penalty against control-only shrinks from 63 to 35\,ms but remains. TCP (351 against 458\,ms, $p=0.28$), control-only (132 against 108, $p=0.06$) and per-topic (164 against 141, $p=0.09$) stay within run-to-run variation between the two loss methods. Control-only also has the lowest worst run and 90th percentile (146 and 138\,ms, against 189 and 169\,ms for per-topic and 168 and 155\,ms for per-publish). This variant changes more than the loss granularity (the loss moved to the forwarding VM and the broker's delay queue ran at 100{,}000 packets rather than 1{,}000, which also changes TCP's packetization; Section~\ref{sec:exp3}), so we cannot attribute the per-publish change to clustering alone. The 1\% and 2\% loss levels and the topic and rate sweeps use broker-egress loss only. -\textbf{Ablation: frame packing policy.} To determine whether the partial correlation observed at 0\% loss reflects QUIC's congestion control architecture or an implementation-level artifact, we conducted a control experiment varying Quinn's frame packing behavior. By default, Quinn packs STREAM frames from multiple streams into single QUIC packets---a hardcoded optimization we term \emph{Greedy} packing. When a single packet is lost, all streams whose data it carried experience retransmission delay simultaneously, potentially creating artificial inter-stream coupling. We extended Quinn with a \emph{StreamIsolated} policy that restricts each packet to a single stream's data, eliminating this cross-stream coupling at the packet level. We tested both policies using the same parameters as the main experiment (500\,msg/s, 8 topics, 25\,ms RTT, 15 runs per configuration). - -At 0\% loss, the baseline correlation persists under StreamIsolated packing: per-topic windowed correlation is 0.64 versus 0.55 with Greedy. Without packet loss, no retransmissions occur, so packet-level frame co-location cannot be the cause. QUIC specifies a single congestion controller per connection~\cite{rfc9002}, and all streams share a single UDP socket. We hypothesize that this shared per-connection state---pacer, congestion window, and socket---creates correlated jitter across streams regardless of how frames are packed into packets. Under loss, however, StreamIsolated per-topic packing shows 26\% higher inter-topic spread (22.5\,ms versus 17.8\,ms at 5\% loss) and 6\% lower spike isolation ratio (0.85 versus 0.91 at 5\% loss), demonstrating genuine per-stream isolation improvement when retransmission events cannot cross stream boundaries within a single packet (Fig.~\ref{fig:frame-packing}). +\subsection{Experiment 3: Throughput Under Loss} +\label{sec:exp3} -The effect is asymmetric across strategies. For per-publish, StreamIsolated packing is counterproductive: p50 latency at 5\% loss rises from 29\,ms to 111\,ms. We hypothesize that isolated packing creates more QUIC packets for ephemeral streams (each publish opens and closes a new stream), increasing the volume subject to loss and retransmission, which may overwhelm connection capacity under high loss. This interaction between frame packing policy and stream lifetime---beneficial for persistent streams, harmful for ephemeral ones---has implications for adaptive transport configuration (Section~\ref{sec:discussion}). +We measure sustained delivered throughput at QoS~0 using 16 publishers and 8 subscribers at 10\,ms one-way delay across loss rates from 0\% to 10\%, driving each transport to saturation. Every subscriber receives every message, so the broker performs an eight-way fan-out; we report throughput as \emph{unique} messages delivered per second (deliveries divided by eight). The publishers are unpaced, so under loss the delivered rate is whatever the impaired transport can carry, and the offered load differs by transport: under loss the TCP+TLS publishers offer 315--332K\,msg/s, TCP's 264--273K and QUIC control-only's 228--243K. The publishers' transmitted bytes confirm these rates (271--304\,B per message under loss). TCP's publisher hosts run only about 45\% busy, so for TCP the broker, at its CPU ceiling, sets the offered rate by how fast it reads its publishers; it takes TCP+TLS publishers faster still, whose hosts run 66--71\% busy, a difference we did not isolate. QUIC control-only's publisher hosts run 80--83\% busy, so a contribution from the load generator to QUIC's offered rate cannot be excluded. Delay and loss are applied per packet on the forwarding VM (Section~\ref{sec:loss-method}). For TCP, TCP+TLS and the QUIC arms of Fig.~\ref{fig:throughput-loss} the broker process runs near its host's four-core ceiling in every condition (340--390\% of 400\%), and under loss the subscriber hosts retain headroom (median 91--99\% CPU idle), so on impaired links the delivery path rather than the receiver sets the rate. On the clean link the subscribers run 21--42\% idle and raw TCP saturates its subscriber host in 11 of 15 runs, which we exclude, so TCP's 0\% level rests on 4 runs (the excluded runs delivered slightly more; all 15 average 59.6K, so the retained level slightly understates raw TCP); we read the 0\% levels as a joint receiver-and-broker bound. Every cell runs in parallel on three identical groups of broker, publisher, subscriber and forwarding VMs, and ranges quoted ``in every group'' are over the per-group means. Because the brokers, and for QUIC possibly the generator, bound the offered load, the delivered rates are lower bounds on broker capacity. All Experiment~3 results use the evaluation build except the paced checks at the end of this section, which use the pacing build (Section~\ref{sec:bench-infra}). \begin{figure} \centering -\includegraphics[width=\columnwidth]{Figure_8} -\caption{Effect of frame packing policy on HOL blocking metrics. \emph{Greedy} packing (default) batches multiple streams per packet; \emph{StreamIsolated} restricts each packet to one stream. For per-topic (left), StreamIsolated improves spread and reduces correlation under loss. For per-publish (right), StreamIsolated increases overhead, degrading performance at high loss rates.} -\label{fig:frame-packing} +\includegraphics[width=\columnwidth]{Figure_6} +\caption{Unique delivered throughput (log scale) vs.\ packet loss at QoS~0 under per-packet loss (solid), with the eight-way subscriber fan-out normalized out; faded dashed lines show the same experiment under per-buffer loss on the broker's egress (Section~\ref{sec:loss-method}). The QUIC arms (client-side publish strategies; broker delivery per-topic) stay within 1.5\% of one another under loss. 95\% confidence intervals over 15 runs (14 for per-topic at 5\%, 4 for TCP at 0\%) are drawn at every point and are frequently narrower than the markers.} +\label{fig:throughput-loss} \end{figure} -\begin{figure} -\centering -\includegraphics[width=\columnwidth]{Figure_3} -\caption{Latency percentiles at 1\% loss, 25\,ms RTT. TCP exhibits higher tail latency than all QUIC strategies, with the gap widening at higher percentiles. Multi-stream QUIC strategies (per-topic, per-publish) achieve the lowest p95 and p99 values.} -\label{fig:lat-percentiles} -\end{figure} +Fig.~\ref{fig:throughput-loss} shows that all the transports lose throughput as loss rises, differing in magnitude rather than shape. Within QUIC, the client-side publish strategies stay within 1.5\% of one another at every impaired loss rate and within 10\% on the clean link. Changing only the broker's delivery from per-topic to control-only, with a control-only client, changes delivered throughput by 6\% on the clean link and by at most 1.5\% under loss. -Fig.~\ref{fig:lat-percentiles} shows latency percentiles at 1\% loss. TCP exhibits the highest tail latency across all percentiles, with the gap between TCP and QUIC widening from p50 to p99. Multi-stream QUIC strategies (per-topic, per-publish) achieve the lowest p95 and p99 values, consistent with independent per-stream loss recovery reducing tail latency when retransmission delays are confined to individual streams rather than stalling the entire connection. +Raw TCP's unique delivered throughput falls from 58.0K to 0.82K\,msg/s between 0\% and 10\% loss ($\approx$71-fold), the encryption-matched TCP+TLS~1.3 baseline from 49.1K to 0.77K ($\approx$64-fold), and QUIC control-only from 54.3K to 1.37K ($\approx$40-fold). That clean-to-lossy ratio mixes a clean-link level bounded jointly by broker and receiver CPU with the response to loss, so we compare the transports at equal loss instead: QUIC delivers 1.26$\times$ TCP+TLS's rate at 1\% loss, 1.37$\times$ at 2\%, 1.52$\times$ at 5\% and 1.78$\times$ at 10\% (1.19--1.66$\times$ against raw TCP). Over the loss-bound range from 1\% to 10\% its throughput falls 1.41$\times$ less steeply than TCP+TLS's (95\% CI 1.40--1.43). Both transports run CUBIC and we did not instrument their loss recovery, so we do not attribute the difference to a specific mechanism. -\subsection{Experiment 3: Throughput Under Loss} +The comparison is at equal loss but not at equal offered load, since the unpaced publishers offer different loads and every broker runs at its CPU ceiling. To test whether that shapes the lossy rates, we also ran the 1\% and 10\% loss conditions with the 16 publishers paced to a fixed aggregate rate of 2.5K or 20K\,msg/s and compared them with unpaced runs. Pacing needs the pacing build (Section~\ref{sec:bench-infra}), so both the paced and the unpaced runs of this check use it, and neither is compared with Fig.~\ref{fig:throughput-loss}, which uses the evaluation build. Each transport, rate and loss level has 5--6 runs; two runs whose monitoring failed (one lost its broker CPU samples, one its router loss counters) are excluded. Paced at 20K\,msg/s the brokers use 27--73\% CPU out of 400\%, yet every transport delivers within 3\% of its unpaced rate in the same build (TCP 1.00 and 0.99 of unpaced at 1\% and 10\% loss, TCP+TLS 0.99 and 0.99, QUIC control-only 1.02 and 1.03). At 2.5K\,msg/s the 1\% cells deliver everything offered, and at 10\% loss even that load exceeds what the loss allows and each transport delivers within 2.5\% of its unpaced rate. With the pacing build, at 1\% and 10\% loss, offered load above the loss-bound rate therefore does not change delivery. -We measure sustained throughput using 4 publishers and 4 subscribers at 10\,ms one-way delay across loss rates from 0\% to 10\%. +With a matched offered load of 20K\,msg/s, QUIC delivers 1.28$\times$ the rate of TCP+TLS at 1\% loss (1.25$\times$ unpaced with the pacing build, 1.26$\times$ with the evaluation build). At 10\% loss the ratio depends on the build: 2.05$\times$ paced and 1.97$\times$ unpaced in the offered-load check, against 1.78$\times$ with the evaluation build. A separate side-by-side comparison of both builds at 10\% loss (6 runs per build and transport) gives 1.81$\times$ with the evaluation build against 2.02$\times$ with the pacing build: the pacing build's QUIC delivers 9\% more (1.51K against 1.39K\,msg/s, 8--9\% in every group), its TCP+TLS 1--3\% less and its TCP the same. The pacing build's unpaced publishers also offer less load (148K against 245K\,msg/s for QUIC), and only the pacing build can be paced, so we cannot say whether the gain lies in QUIC's send path or in lighter overload; we report the evaluation build's ratios and treat their magnitude at 10\% loss as build-dependent; the builds were not compared at 2\% or 5\%. -\begin{figure} +Under loss the delivery path itself narrows, and shedding is where the excess goes: the broker drops what it cannot deliver once a subscriber's queue is full (Table~\ref{tab:params}), at 10\% loss delivering 0.23\% of the offered messages over TCP+TLS, 0.30\% over TCP and 0.56\% over QUIC. + +Per-buffer loss on the broker's egress, the method of Experiments~2 and~4, gives a different picture. With it, QUIC delivers 2.06--3.06$\times$ TCP+TLS's rate at equal loss and appears to degrade 2.6$\times$ more gently from 0\% to 10\% (Table~\ref{tab:loss-method}, from the 0\% and 10\% rows). With per-packet loss on the forwarding VM, QUIC delivers 42--49\% of its per-buffer rate and TCP and TCP+TLS 67--86\%. Clustered loss therefore flattered QUIC far more than TCP. The loss-bound comparison survives at a smaller magnitude (1.41 against 1.48), and it does not depend on where per-packet loss is applied: keeping the delay on the broker and applying the loss on the forwarding VM gives 1.36 (1.34--1.38; 15 runs per cell), and applying the loss on the broker behind a token-bucket filter that splits every buffer into packets gives 1.39 (1.36--1.41; 6 runs per cell). A model in which clustering acts only by lowering the effective loss rate, with throughput scaling as $1/\sqrt{p}$~\cite{mathis1997macroscopic} and the measured per-buffer drop fractions taken as the effective rates, predicted that correcting it would raise the ratio by a factor of 1.04. It instead fell by a factor of 0.91 (0.89--0.94) with the delay kept on the broker (to 1.36), by 0.95 (0.93--0.98) with the forwarding-VM arm and by 0.93 (0.91--0.96) with the token-bucket arm. The clustering therefore acted through more than the loss rate, and we do not identify the mechanism. A single-connection ablation (one publisher and one subscriber at 10\,ms one-way delay, with loss per buffer on the broker's egress; data in the deposit~\cite{bracht2026zenodo}) pointed the same way (disabling QUIC's datagram batching roughly halved its delivered rate under loss), and on that unimpaired single connection TCP+TLS delivered 111.9K\,msg/s against QUIC's 58.1K. + +Where the delay is applied also matters on the clean link. With the delay on the broker, the broker-egress netem queue of Experiments~2 and~4 (1000 packets) tail-dropped about 0.15\% of packets even at 0\% loss, and TCP packed several messages into each packet (about 1.1\,KB per packet). Enlarging that queue so that it never drops, with nothing else changed, gives one segment per message (about 400\,B) and TCP+TLS falls to 25.0K\,msg/s; with the delay on the forwarding VM instead, TCP packs messages normally (about 1.2\,KB per packet) and TCP+TLS delivers 49.1K. The broker's TCP stack disables Nagle's algorithm and writes each message separately; we did not isolate why a large delay queue on the sending host removes the packing. We report the forwarding-VM arm, which avoids both effects; under loss the two delay placements agree within 3\% for every configuration except broker per-publish delivery, which with the delay on the broker delivers a further 34\% less at 1--2\% loss and 18\% less at 5\%, and the same at 10\%. + +\begin{table}[!t] \centering -\includegraphics[width=\columnwidth]{Figure_5} -\caption{Throughput vs.\ packet loss for QoS~0 (top) and QoS~1 (bottom). TCP outperforms all QUIC strategies at baseline (0\% loss), but under loss ($\geq$1\%) all QUIC strategies outperform TCP.} -\label{fig:throughput-loss} -\end{figure} +\caption{Unique delivered throughput (K\,msg/s, 15 runs) under per-buffer loss on the broker's egress and under per-packet loss on the forwarding VM, and QUIC control-only's ratio to TCP+TLS at equal loss. $G'$ is the ratio of the two transports' 1\%-to-10\% degradation, with 95\% intervals.} +\label{tab:loss-method} +\small +\setlength{\tabcolsep}{3.5pt} +\begin{tabular}{@{}lrrrrrr@{}} +\toprule +& \multicolumn{2}{c}{TCP+TLS} & \multicolumn{2}{c}{QUIC control-only} & \multicolumn{2}{c}{QUIC/TLS} \\ +\cmidrule(lr){2-3}\cmidrule(lr){4-5}\cmidrule(lr){6-7} +Loss & buffer & packet & buffer & packet & buffer & packet \\ +\midrule +0\% & 45.2 & 49.1 & 53.7 & 54.3 & 1.19 & 1.11 \\ +1\% & 5.89 & 4.02 & 12.16 & 5.05 & 2.06 & 1.26 \\ +2\% & 3.76 & 2.62 & 8.26 & 3.59 & 2.20 & 1.37 \\ +5\% & 1.92 & 1.42 & 4.75 & 2.16 & 2.48 & 1.52 \\ +10\% & 0.91 & 0.77 & 2.80 & 1.37 & 3.06 & 1.78 \\ +\midrule +\multicolumn{5}{@{}l}{$G'$ (1\%$\rightarrow$10\%), 95\% interval} & 1.48 & 1.41 \\ +\multicolumn{5}{@{}l}{} & {\scriptsize[1.45,1.51]} & {\scriptsize[1.40,1.43]} \\ +\bottomrule +\end{tabular} +\end{table} -Fig.~\ref{fig:throughput-loss} reveals that at 0\% loss, TCP outperforms all QUIC strategies in raw throughput. With QoS~0, TCP achieves 311K\,msg/s versus QUIC control-only at 132K\,msg/s, per-topic at 82K\,msg/s, and per-publish at 77K\,msg/s. The throughput gap at baseline is consistent with QUIC's additional per-packet processing overhead, though we did not isolate individual contributing factors (see Section~\ref{sec:discussion} for further analysis). +Encryption carries a predictable throughput cost on the clean link: TCP+TLS runs 15--18\% below raw TCP at 0\% loss (49.1K against 58.0K unique msg/s over the 4 unsaturated TCP runs, or 59.6K over all 15), a CPU-bound overhead that shrinks to about 6\% under loss (4.02K versus 4.26K at 1\%, 0.77K versus 0.82K at 10\%). QUIC, encrypted throughout, sits above TCP+TLS at 0\% loss (54.3K) and 6--9\% below raw TCP, so encryption does not explain the loss-response gap. -Under loss ($\geq$1\%), the ordering reverses: all QUIC strategies outperform TCP, achieving 1.9$\times$ TCP's throughput at 1\% loss and 2.8$\times$ at 10\% loss. Per-publish QUIC matches control-only throughput under loss, suggesting that stream creation overhead becomes negligible relative to loss recovery delays. +Broker-side per-publish delivery is the exception to the strategies' near-equality. With a per-publish client and the broker also delivering one stream per message, delivered throughput stays nearly flat at 1.3--2.0K unique msg/s (10--16K deliveries/s) across loss levels. That is far below per-topic delivery on the clean and lightly lossy link (1.74K against 49.6K at 0\%, 1.93K against 5.03K at 1\%), and the gap closes as loss becomes the binding limit (86\% of per-topic at 5\%, 94\% at 10\%). Under loss both arms receive about 135K\,msg/s from unpaced per-publish publishers, each running at its own stream-credit ceiling (Experiment~4), but per-publish delivery stays near 1.3--2.0K\,msg/s at every loss level, so at 1--2\% loss it delivers only 38--57\% of per-topic's rate; on the clean link its slow delivery coincides with the publishers running at that ceiling, while with per-topic delivery they offer about 50K. Pacing the publishers separates delivery cost from overload. For this we used the pacing build. Its unpaced per-topic and control-only deliveries on the clean link run 10--12\% below the evaluation build's, so we compare delivery strategies only among pacing-build runs. At 0\% loss, per-publish delivery keeps up with the eight-way fan-out up to 40K deliveries/s, then stalls near 72K deliveries/s (about 9K\,msg/s per subscriber), consistent with a per-subscriber stream-credit ceiling at this round trip; we did not vary the stream limit or the round trip for delivery. Below that ceiling it costs 1.17$\times$ the broker CPU per delivered message of per-topic delivery at 10K deliveries/s, 1.33$\times$ at 20K and 1.41$\times$ at 40K. It sends about one datagram per message (1.06--1.08 broker packets per delivered message over 10--40K deliveries/s, against 0.89 falling to 0.54 for per-topic delivery over the same range). Under 1\% loss its packets per delivered message fall to 0.35--0.93 as messages share datagrams, yet it still costs 1.15--1.33$\times$ per-topic's CPU per delivered message, so the per-datagram cost does not fully explain the premium. When paced at 1\% loss, every strategy saturates near 5K\,msg/s per subscriber (per-publish 4.6K, per-topic and control-only 5.1K), so there loss rather than credit sets most of the ceiling. Unpaced, per-topic delivery reaches 93\% of its paced ceiling, while per-publish delivery reaches only 17\% of its own (about 0.8K\,msg/s per subscriber, the same as on the clean link in the pacing build, where unpaced per-publish delivery runs about 55\% below the evaluation build's 1.74K), so its unpaced collapse comes from overload, not loss. We did not test whether batching per-publish frames removes the premium. -QoS~1 throughput drops to approximately 1.3K\,msg/s across all transports at 0\% loss, as the per-message acknowledgment round-trip dominates. Under loss, QUIC maintains higher QoS~1 throughput than TCP (880 versus 409\,msg/s at 10\% loss). -\subsection{Experiment 4: Stream Strategy Comparison} +\subsection{Experiment 4: What Limits a Single Connection} -We compare the three stream mapping strategies (control-only, per-publish, per-topic) while varying topic count (1, 4, 8, 16) at 25\,ms one-way delay and 2\% packet loss to understand how strategy selection interacts with workload structure under representative IoT conditions. +Experiment~3 measures the broker at saturation, where many client connections share it and, under loss, the delivery path sets the rate. A single MQTT client cannot spread its own load that way: the protocol ties each connection to a client identifier, and a second connection presenting an identifier already in use causes the broker to disconnect the first (Section~\ref{sec:parallel}). Short of provisioning further identities, everything one client publishes therefore travels over a single QUIC connection, and that connection's limits bound what the client can send. Two of those limits are transport parameters an implementation chooses, the number of streams the peer lets it open at once and the flow-control window of each stream; above them we also find a rate ceiling that we did not instrument. The three strategies differ sharply in what one connection can push, and this experiment asks why. We vary the number of topics (1, 2, 4, 8, 16) on a \emph{single} connection, and then vary the two QUIC transport limits the strategies interact with: the number of streams a peer may keep open at once, and the per-stream receive window. The broker's delivery strategy matches the client's in each arm. Unless stated, cells use Experiment~2's delay and broker-egress loss method, with 25\,ms one-way delay and 2\% loss, one of Experiment~2's loss levels; publisher and subscriber run on separate VMs and the publishers are unpaced. Loss is not this experiment's variable. Because \texttt{tc-netem} shapes the broker's egress, the publisher's data travels an unimpaired path, and the delay and loss apply to the acknowledgements and flow-control credit returning to it (and to the subscriber's deliveries); since the delay applies in one direction only, the publisher's round trip equals the 25\,ms one-way delay. Cells run at 0\% loss check that the loss on that returning path does not shape the limits: control-only at the default 256\,KB window, where the window binds, sends 32.97K\,msg/s at 0\% against 32.90--33.23K in the equivalent 2\% cells, and removing the loss does not raise the per-connection ceiling (below). No per-publish cell was run at 0\% loss, so the effect of loss on its credit-bound rate is untested; QUIC's credit frames carry cumulative limits~\cite{rfc9000}, so a lost update delays credit but does not forfeit it. + +Varying topic count alone shows the gap between the strategies on a single connection. Control-only sustains about 33K\,msg/s at every topic count and per-publish about 3.8K, roughly ninefold below it. Per-topic is the only strategy that moves with topic count, and it moves once: it matches control-only on one topic (33.1 against 32.7K), where it opens the same single data stream, reaches 56.0K at two topics, and then stays flat to sixteen (53.9K). Adding streams beyond the second buys nothing, which already rules out an explanation in terms of write concurrency scaling with stream count. \begin{figure} \centering \includegraphics[width=\columnwidth]{Figure_7} -\caption{Strategy comparison across topic counts. (a)~Latency is stable across all strategies and topic counts. (b)~Throughput scales linearly with topic count for all strategies.} -\label{fig:strategy-comparison} +\caption{The two transport limits behind those rates, 8 topics unless noted. (a)~Per-publish opens one stream per message, so its rate follows the concurrent stream limit divided by the round trip (dashed) across a fortyfold range, while control-only and per-topic, which open one and eight streams, do not respond to it. (b)~Control-only and per-topic on a single topic both open one data stream and coincide while the per-stream receive window binds, following window divided by round trip (dotted, using the 313\,B per-message stream cost measured from these runs); at 1\,MB the window no longer binds and they separate, only control-only reaching the rate ceiling near 54K\,msg/s. Per-topic on eight topics is already at that ceiling and is insensitive to the window. 95\% confidence intervals over 10 runs per point.} +\label{fig:transport-limits} \end{figure} -Fig.~\ref{fig:strategy-comparison} shows that latency is stable at approximately 35\,ms across all strategies and topic counts, indicating that stream management overhead does not materially affect per-message latency in this workload. Throughput scales linearly with topic count for all strategies. At 16 topics, all three strategies converge to approximately 50K\,msg/s with no measurable throughput difference between them, indicating that stream management overhead is negligible at this workload scale. +Fig.~\ref{fig:transport-limits} identifies the limits. Per-publish is bounded by stream credit. Sweeping the limit over 25, 100, 250 and 1000 streams moves its rate to 0.93, 3.75, 9.19 and 30.87K\,msg/s, and the streams it keeps in flight, the rate multiplied by the round trip, stay at 93\%, 94\%, 93\% and 78\% of the limit. Holding the limit at 100 and sweeping the delay over 10, 25 and 50\,ms leaves that count at 93.2, 94.4 and 95.4 streams: the rate falls as $1/\mathrm{RTT}$, which is what a credit ceiling does and what a per-message processing cost would not. Control-only and per-topic, needing one and eight streams, barely respond to the stream limit itself: raising it from 100 to 1000 moves control-only by 0.8\% (33.13 to 32.85K) and per-topic by 7.3\% (50.96 to 54.67K, $p=0.09$). -\subsection{Experiment 5: Datagram vs.\ Stream Transport} +Control-only is bounded by the per-stream receive window. Halving it to 128\,KB halves the rate to 16.79K and raising it to 1\,MB lifts it to 54.30K. Per-topic on one topic, which also opens a single data stream, tracks it to within 0.2\% wherever the window binds (16.76 against 16.79K at 128\,KB, 32.92 against 32.90K at 256\,KB). At 1\,MB the window no longer binds either arm and they separate, per-topic on one topic reaching 49.12K against control-only's 54.30K ($p=1\times10^{-6}$); the two are single-stream but not identical, since control-only carries its data on the bidirectional control stream while per-topic opens a unidirectional one, and that difference surfaces only once both are pressed against the ceiling below. Across the six cells of the window and round-trip sweeps in which the window binds, spanning a factor of two in window and a factor of two in round trip, and including one configuration measured on two VM groups, the implied per-message cost on the stream is constant at 312.8$\pm$3.0\,B (a 1.0\% spread), consistent with a 256\,B payload plus MQTT and QUIC stream framing. That constancy is the evidence; the absolute value is derived from the same measurements and is not an independent prediction. Per-topic on eight topics has eight such windows and is therefore insensitive to the setting (55.7, 55.7 and 53.8K). -QUIC datagrams (RFC~9221) bypass the stream abstraction entirely, sending messages as unreliable datagrams within the QUIC connection. We compare datagram transport against stream-based transport at 50\,ms delay across four loss rates. +Above those two limits sits a third, a rate ceiling on one connection near 54K\,msg/s: the cells it bounds deliver 51.0--56.0K at 25\,ms. Removing the loss does not raise it (control-only at a 1\,MB window sends 52.1K at 0\% against 54.3K at 2\%, and per-topic's 53.6K at 0\% lies within the 51.0--55.7K of its equivalent 2\% cells). It is also roughly independent of the round trip, reaching 57.8K at 10\,ms, so it is neither a window nor a credit effect. We did not instrument the client further, and attribute it only to per-connection processing. \begin{figure} \centering -\includegraphics[width=\columnwidth]{Figure_6} -\caption{Datagram vs.\ stream transport at 50\,ms RTT. (a)~Latency distributions overlap at all loss rates, with high run-to-run variance under loss. (b)~Throughput is comparable since both transports share the same QUIC connection and congestion controller.} -\label{fig:datagram} +\includegraphics[width=\columnwidth]{Figure_8} +\caption{Every publish-rate cell of Experiment~4 against the rate its binding limit predicts: stream credit divided by the round trip for per-publish, and the stream count times the per-stream window divided by the round trip for the others, each capped at the 54K\,msg/s per-connection ceiling. The 39 cells span a sixtyfold range of measured rate and sit on the diagonal with a median absolute error of 2.3\%. The two constants in the model, the 313\,B per-message stream cost and the ceiling, are measured from these runs rather than fitted per cell.} +\label{fig:model-collapse} \end{figure} -Fig.~\ref{fig:datagram} shows that at 50\,ms RTT, datagram and stream transport are statistically indistinguishable at all loss rates. At 0\% loss, both achieve 12.9K\,msg/s throughput and 50.3\,ms p50 latency. Under loss, both transports degrade similarly: at 5\% loss, stream p50 is 1,188$\pm$1,070\,ms and datagram p50 is 854$\pm$308\,ms---the stream distribution's high variance (90\% coefficient of variation) means these values overlap within error bars. At 10\% loss, both exhibit multi-second latencies (stream 12.9$\pm$2.7\,s, datagram 12.0$\pm$1.3\,s). Throughput is comparable at all conditions ($\approx$1,000\,msg/s at 5\% loss), as both share the same QUIC connection and congestion controller. The theoretical advantage of datagrams---skipping retransmission of lost messages---is not observed in our measurements, possibly because both transports share the same per-connection congestion controller~\cite{rfc9002}. +The strategies therefore do not differ in what they can carry; they differ in which limit they meet first. Per-publish meets the stream limit, control-only the per-stream window, and per-topic, which escapes the window with its second stream, meets the per-connection ceiling. Lift each in turn and they move toward that ceiling: control-only reaches 54.3K at a 1\,MB window and per-topic already sits at 53.8K, while per-publish, at 30.9K with a limit of 1000, is still climbing and would need a larger limit still to reach it. The ninefold gap between per-publish and control-only at our defaults, and per-topic's $\approx$1.6$\times$ advantage over control-only, are both consequences of two constants we chose, not properties of the mappings. Fig.~\ref{fig:model-collapse} states the position compactly: across all 39 publish-rate cells, spanning stream limits from 25 to 1000, windows from 128\,KB to 1\,MB, delays from 10 to 50\,ms and both loss settings, the measured rate is within a median 2.3\% of whichever limit binds, and the strategy identity adds little once that limit is known. The largest departures are per-publish at a 1000-stream limit, 22\% below the credit line as it approaches the ceiling; per-topic on one topic at a 1\,MB window, 9\% below the ceiling, the single-stream separation discussed above; per-publish at a 250-stream limit, 7\% below its credit line, as per-publish sits 5--7\% below it at every limit up to 250; and control-only at 10\,ms, 7\% above a ceiling that is itself slightly round-trip dependent. + +Two caveats bound this. The limits are ours: \texttt{mqtt5} advertises a 100-stream limit, which is the default of the QUIC library it builds on~\cite{quinn}, and a 256\,KB per-stream window, and an implementation choosing differently would order the strategies differently. And the finding is about one connection. At broker capacity the shared delivery path sets the rate, which is why Experiment~3's client-side strategies differ by at most 1.5\% under loss and 10\% on the clean link; we report the publish rate here because a single subscriber connection receives only about 3.4K\,msg/s under control-only and per-topic and 2.1K under per-publish, its host staying 97\% CPU idle, so the delivered rate in this one-to-one setup measures the impaired delivery path rather than the send-side strategy. %% --- Discussion --- @@ -412,49 +452,61 @@ \subsection{Strategy Selection Guidelines} Our results suggest the following guidelines for practitioners selecting a stream mapping strategy: -\textbf{Use control-only} when throughput is the primary concern and topics are not independent. This strategy adds QUIC's security benefits (integrated TLS~1.3) and connection management (migration, 0-RTT potential) with minimal overhead compared to TCP. It provides no inter-topic isolation (spike isolation ratio 0.98, comparable to TCP's 0.98), making it appropriate for single-topic workloads or scenarios where HOL blocking is acceptable. +\textbf{Use control-only as the default.} Adding no per-stream packet overhead, it has the lowest median and 90th-percentile tail latency at 5\% loss, 8 topics and 500\,msg/s (132 and 162\,ms; 108 and 138\,ms under per-packet loss). At 1\% loss and 500\,msg/s both multistream arms beat it, and at 125\,msg/s under 5\% loss per-publish does (96 versus 116\,ms); at 1\% loss and 1000\,msg/s control-only is again lowest (61 against 70 and 82\,ms), so which strategy has the lowest tail depends on both loss and rate. As the client publish strategy it matches the other QUIC arms' delivered throughput within 1.5\% under loss and 10\% on the clean link; as the broker's delivery strategy with a control-only client it is within 1.5\% under loss and 6\% lower on the clean link (Experiment~3). It inherits TLS~1.3 in a one-round-trip handshake that matches raw TCP and saves a round trip over TCP+TLS at delays of 25\,ms and above (Experiment~1). Its one throughput concession is on a single connection, where per-topic sends about 1.6--1.7$\times$ faster once a second topic is in play; Experiment~4 shows that gap is the per-stream receive window, and raising it removes the gap. Like the other strategies, it has occasional outlier runs (Experiment~2). + +\textbf{Consider per-topic when one client must publish at high rate across several topics and the per-stream window cannot be raised.} With the default 256\,KB per-stream window, per-topic reaches the per-connection ceiling once a second topic is in play (about 56K against control-only's 33K\,msg/s; Experiment~4), because each topic gets its own window; raising control-only's window closes that gap. Under broker-egress loss at 500\,msg/s its tail sits between the other two, below control-only at 1\% loss (55.8 against 58.3\,ms) and above it at 5\% (164 against 132\,ms); under per-packet 5\% loss it ties per-publish (141 against 143\,ms), both above control-only (108\,ms). In the 1\%-loss topic sweep it isolates spike timing only at two topics and reaches single-stream coupling from four topics on (Fig.~\ref{fig:hol-excess}). + +\textbf{Use per-publish for low-rate or light-loss channels where per-message isolation matters.} It isolates spikes best at every rate we tested under 1\% loss. It has the lowest tail at 1\% loss up to 500\,msg/s (54.0\,ms at the reference point), and at 125\,msg/s under 5\% loss (96\,ms). On a single connection its publish rate tracks the concurrent stream limit divided by the round trip, roughly ninefold below control-only at our 100-stream limit and 256\,KB window. That rate rises nearly in proportion as the limit is raised, but only toward the per-connection ceiling of about 54K\,msg/s (30.9K at a 1000-stream limit; Experiment~4). As the broker's delivery strategy it is capped per subscriber near 9K\,msg/s on the clean link at a 10\,ms round trip (under 1\% loss all strategies saturate near 5K), costs 1.2--1.4$\times$ the broker CPU per delivered message of per-topic delivery at 10--40K deliveries/s, and under unpaced overload delivers a small fraction of what per-topic delivery achieves on a clean link (Experiment~3). Size the stream limit to the per-subscriber message rate. -\textbf{Use per-topic} when topic independence matters and the number of topics is bounded. At 1\% loss---representative of typical wireless IoT conditions---per-topic mapping already provides measurable isolation: inter-topic spread amplifies 42$\times$ from baseline, and 27\% of latency spikes affect individual topics independently (spike isolation ratio 0.73 versus TCP's 0.98). At higher loss (5\%), spread amplification reaches 200$\times$. Per-topic throughput is 62\% of control-only at baseline. The overhead scales with the number of active topics, not the message rate, making it efficient for workloads with moderate topic counts (tens to hundreds). StreamIsolated frame packing further improves isolation (+26\% spread, $-$6\% spike ratio at 5\% loss) with modest latency cost, making it recommended for loss-prone deployments. This is the recommended default for multi-topic IoT deployments on lossy networks. +\subsection{Loss Resilience and the Cost of Encryption} -\textbf{Use per-publish} only when message-level isolation is required and throughput is secondary. Per-publish shows the strongest spike isolation (ratio 0.53 at 1\% loss, meaning 47\% of spikes are topic-independent), but the per-message stream creation overhead makes this strategy unsuitable for high-throughput workloads. StreamIsolated frame packing should be avoided for this strategy, as our measurements show significant latency degradation under loss (p50 rises from 29\,ms to 111\,ms at 5\% loss). It may be appropriate for low-rate command-and-control channels where every message must be independently deliverable. +Because our QUIC arms are encrypted, we compare them against an encryption-matched TCP+TLS~1.3 baseline, so the difference reflects the transport rather than the cipher. Under per-packet loss QUIC delivers 1.26$\times$ TCP+TLS's rate at 1\% loss, growing to 1.78$\times$ at 10\%, and over that range its throughput falls 1.41$\times$ less steeply. Both transports run CUBIC and we did not instrument their loss recovery, so the mechanism remains open. With the pacing build, pacing all three transports to a matched offered load far below the broker's CPU ceiling left their rates at 1\% and 10\% loss within 3\% of unpaced, so in the pacing build the gap is not an artifact of unequal load. The evaluation build, whose ratios we report, has no rate pacing and could not be checked this way, and at 10\% loss the gap depends on the build (2.02$\times$ against 1.81$\times$ when both are run side by side). The advantage is real but smaller than the 2.06--3.06$\times$ that per-buffer loss emulation produced with the evaluation build. Under per-buffer loss QUIC also appears to degrade 2.6$\times$ more gently from 0\% to 10\% loss, but that ratio rests on the method and on a clean-link level bounded jointly by broker and receiver CPU; with per-packet loss the same 0-to-10\% ratio is 1.61 with the delay on the forwarding VM, but 0.82 with the delay on the broker and the loss on the forwarding VM and 0.86 behind the broker's token-bucket filter, which is why we compare at equal loss. On the clean link the ordering depends on configuration: at fan-out saturation QUIC delivers 54.3K\,msg/s against TCP+TLS's 49.1K, whereas on a single unimpaired connection at 10\,ms TCP+TLS reaches about 112K while QUIC stays near 58K, at the per-connection ceiling of Experiment~4. The benefit is therefore conditional on the loss a deployment sees and on how its traffic is spread across connections. Encryption is a predictable cost: TCP+TLS runs 15--18\% below raw TCP on the clean link and about 6\% below under loss. -\textbf{Use datagrams} for latency-sensitive telemetry with QoS~0 where occasional message loss is acceptable. At 50\,ms RTT, datagrams show no statistically significant latency or throughput advantage over streams. Datagrams remain a valid design choice for applications that explicitly tolerate message loss, but our results do not demonstrate a measurable performance benefit at representative IoT RTTs. They are unsuitable for QoS~1 or QoS~2 workloads that require acknowledgment semantics. +\subsection{The Cost of Parallel Connections} +\label{sec:parallel} -\subsection{Throughput--Isolation Tradeoff} +A TCP client can obtain per-topic concurrency simply by opening several connections, and doing so carries a bandwidth consequence: each connection is a separate congestion controller, so $N$ of them converge to roughly $N$ times a single flow's share of a congested link~\cite{rfc9308, chiu1989aimd}. For a generic TCP application this is a real alternative to stream multiplexing. -At baseline (0\% loss), unencrypted TCP achieves 2.4--4.0$\times$ the throughput of QUIC strategies (Experiment~3). This comparison is inherently asymmetric: QUIC mandates encryption, while our TCP baseline operates without TLS. A fairer throughput comparison would use TCP with TLS~1.3, which would narrow the gap by adding comparable per-packet encryption overhead to TCP. However, the absolute throughput difference at 0\% loss is less important than the practical picture: in real IoT deployments, packet loss is the norm rather than the exception, and QUIC already outperforms even unencrypted TCP at just 1\% loss (1.9$\times$), reaching 2.8$\times$ at 10\% loss. The implication is that QUIC provides a net advantage in realistic conditions: it delivers higher throughput under loss, provides stream-level isolation for independent topic flows, and includes encryption by default---eliminating the need for a separate TLS layer. +MQTT forbids it. The protocol ties each connection to a client identifier, and a second connection presenting an identifier already in use causes the broker to disconnect the first (normative statement \mbox{[MQTT-3.1.4-3]}~\cite{mqtt-v5-spec}). An MQTT device therefore cannot open the $N$ connections that would claim the larger share; it would have to provision $N$ distinct identities, which is a fleet-management and, on managed platforms, a billing cost rather than a socket call. Managed IoT platforms enforce this rule in practice; Azure IoT Hub, for one, permits a single active connection per device and drops the existing one when a second presents the same identity~\cite{azureiot}. Stream multiplexing is therefore the concurrency mechanism actually available to an MQTT client: it supplies per-topic streams inside the single identity the protocol expects. -\subsection{HOL Blocking Isolation: Partial but Meaningful} +\subsection{HOL Blocking: Transport and Stream Mapping} -Our HOL blocking results show that QUIC stream multiplexing provides \emph{partial} rather than \emph{complete} isolation between topics. The spike isolation ratio for per-topic QUIC drops to 0.73 at 1\% loss rather than approaching 0 (full independence). QUIC's congestion control operates per-connection~\cite{rfc9002}: a loss event reduces the congestion window for all concurrent streams, even though stream-level loss recovery is independent. We hypothesize that this shared congestion response is responsible for the partial isolation. The inter-topic spread metric is consistent with this interpretation---topics experience different retransmission delays (high spread) but partially correlated congestion responses (spike co-occurrence). +At 5\% loss, and at 2\% for control-only and per-topic, our HOL blocking results locate the benefit in the transport rather than in stream mapping. TCP's single ordered byte stream couples every topic (351\,ms median p99 at 5\%; 458\,ms in the per-packet variant, whose broker-side delay queue also changes TCP's packetization; Section~\ref{sec:exp3}). QUIC control-only couples topics just as strongly at 1\% loss (Fig.~\ref{fig:hol-excess}), yet it and every other QUIC strategy keep the median tail below 200\,ms at the 8-topic, 500\,msg/s reference point; we did not isolate which transport difference is responsible. At 1\% loss control-only is indistinguishable from TCP (58.3 against 61.5\,ms, $p=0.12$), while finer stream mapping lowers the tail significantly (per-topic 55.8, per-publish 54.0\,ms; $p=0.007$ and $p=0.003$ against control-only). Stream granularity is a tradeoff knob within QUIC. Finer streams shrink the set of messages one loss can block, which decorrelates spike timing: per-publish isolates most, while in the 1\%-loss sweeps per-topic isolates only at two topics and at 2000\,msg/s. But finer streams also add framing packets. Under per-buffer 5\% loss, per-publish sent about twice control-only's packets on the delivery path, had 2.5 times its congestion events and a smaller congestion window, and had the highest tail (195 against 132\,ms). Per-topic added only about 9\% more packets, yet its median sat midway between the two (164\,ms). Under per-packet loss per-publish no longer trails per-topic (143 against 141\,ms) although it still sends about 1.5 times per-topic's packets, and both sit about 30\% above control-only. Packet count therefore does not fully explain the ordering under either method; QUIC loss detection is per-connection rather than per-stream~\cite{rfc9002}, and we could not separate per-stream recovery effects from packet count. -We tested whether Quinn's default frame packing---batching STREAM frames from multiple streams into single QUIC packets---contributes to this baseline correlation by running the same experiment with StreamIsolated packing, which restricts each packet to a single stream's data. The correlation persists (0.64 versus 0.55 at 0\% loss), confirming that it is not caused by frame co-location. Without packet loss, no retransmissions occur, so frame co-location has no effect; all streams still share a single congestion controller, pacer, and UDP socket~\cite{rfc9002}, which we hypothesize creates correlated jitter independent of frame packing. Under loss, however, StreamIsolated measurably improves per-topic isolation at 5\% loss, with 26\% higher inter-topic spread and a 6\% lower spike isolation ratio---both consistent with reduced cross-stream coupling at the packet level. However, for per-publish, isolated packing is counterproductive: p50 latency rises from 29\,ms to 111\,ms at 5\% loss with StreamIsolated versus Greedy, possibly because the increased packet count from ephemeral streams amplifies retransmission overhead. +\subsection{Threats to Validity} +\label{sec:threats} + +Our results come from one implementation, evaluated by its author, on one QUIC library (Quinn 0.11) whose default 100-stream limit, together with our 256\,KB per-stream window, decides the single-connection ordering (Experiment~4). Loss is random and independent from packet to packet (Bernoulli), without bursts, and applies only to broker-to-client traffic, so the publish path is never lossy and bursty loss is untested. Experiment~2 applies loss per kernel buffer on the broker's egress, a method Experiment~3 shows inflates QUIC's throughput advantage; Experiment~4's publish rates are measured on an unimpaired publish path, with the loss on the returning acknowledgements and credit; its 0\%-loss cells show little effect on the window-bound and ceiling cells, and no per-publish cell was run at 0\%. The two 5\%-loss per-publish delivery cells that Experiment~2 cites carry per-buffer loss on their delivery path. Only Experiment~2's reference cell at 0\% and 5\% was also run with per-packet loss; there QUIC's tail advantage over TCP did not shrink but the per-publish tail penalty over per-topic disappeared, so its other cells may carry similar distortions. All VMs share one GCP zone with emulated delay and CUBIC, and all runs use 256-byte QoS~0 payloads, so the acknowledgement traffic of QoS~1 and~2 is untested. Brokers are CPU-bound at saturation, QUIC control-only's publisher hosts run 80--83\% busy under loss, and TCP's clean-link level in Experiment~3 rests on the 4 runs whose subscriber did not saturate. At 10\% loss QUIC's advantage over TCP+TLS depends on the implementation build (1.81$\times$ with the evaluation build against 2.02$\times$ with the pacing build, side by side), and we did not identify the change responsible. The pacing build produces the offered-load check (1\% and 10\% loss only), the broker per-publish CPU premium, per-subscriber ceiling and paced ceilings, their unpaced comparisons and the side-by-side 10\% comparison; its unpaced clean-link per-topic and control-only deliveries run 10--12\% below the evaluation build's, and its unpaced per-publish delivery 55--59\% below (0\% and 1\% loss). Experiment~2 co-locates publisher and subscriber, and the broker's delivery mapping is varied independently of the client's only in Experiment~3's fan-out configuration. Fifteen runs per arm cannot estimate the rate of rare tail outliers. \subsection{Future Work} -Several directions emerge from this work. First, evaluating 0-RTT connection resumption would quantify the benefit for mobile clients that reconnect frequently. Second, testing connection migration could demonstrate QUIC's advantage for devices that change network interfaces. Third, exploring adaptive strategy selection---where the broker dynamically switches strategies based on observed network conditions---could automate the throughput--isolation tradeoff. Additionally, exploring adaptive frame packing that automatically selects Greedy packing for per-publish and StreamIsolated for per-topic based on the negotiated stream strategy could optimize the transport layer without manual configuration. Finally, extending the evaluation to constrained devices (e.g., ESP32, Raspberry Pi) would assess whether QUIC's computational overhead is viable for resource-limited IoT endpoints. +This work opens several directions. Per-publish delivery could pack the frames of several streams into shared datagrams, which would show whether its broker CPU premium is inherent to one stream per message or a consequence of sending one datagram per message. Evaluating the strategies at QoS~1 and~2, where acknowledgements share the impaired path, and under bursty loss would extend the comparison to conditions common in IoT deployments, and long-running measurements could estimate how often tail outliers occur and whether they depend on the stream mapping. Because the tail ordering of the strategies reverses with both loss and rate, a client or broker could select stream granularity adaptively from the loss and rate it observes. Finally, constrained devices, 0-RTT resumption and connection migration would test QUIC's other advertised benefits for IoT. %% --- Conclusion --- \section{Conclusion} \label{sec:conclusion} -We presented three configurable stream mapping strategies for running MQTT over QUIC. Through five experiments on controlled infrastructure, we demonstrated that per-topic QUIC stream mapping provides measurable HOL blocking isolation: inter-topic latency spread amplifies 200$\times$ under 5\% packet loss, consistent with independent stream loss recovery allowing topics to experience different retransmission delays rather than stalling together as TCP does (1.2$\times$ spread amplification). Spike isolation ratio corroborates this---at 1\% loss, 27\% of per-topic QUIC latency spikes affect individual topics independently, versus less than 2\% for TCP. The isolation is partial rather than complete, which we attribute to QUIC's per-connection congestion control that shares a single congestion window across all streams. +We implemented and evaluated three stream mapping strategies for running MQTT over QUIC, including broker-autonomous per-topic delivery-stream mapping absent from existing brokers. At 5\% loss the transport, not stream mapping, sets the level of HOL blocking: control-only and per-topic keep the median per-topic tail below TCP at 2\% and 5\% loss, and per-publish does so at 5\% (at 2\% its 82.6\,ms is not distinguishable from TCP's 89.0\,ms, $p=0.06$), and at 5\% control-only holds 132\,ms against TCP's 351\,ms, and 108 against 458\,ms in the per-packet variant. Within QUIC, control-only adds the least packet overhead and has the lowest typical and 90th-percentile tail at 5\% loss and 500\,msg/s; at 1\% loss and 500\,msg/s both multistream arms come out ahead, and at 125\,msg/s under 5\% loss per-publish does. -Our evaluation reveals a nuanced throughput tradeoff: at baseline (0\% loss), TCP achieves 2.4--4.0$\times$ the throughput of QUIC strategies depending on workload, but under loss ($\geq$1\%) QUIC outperforms TCP by up to 2.8$\times$. Per-topic mapping achieves 62\% of single-stream QUIC throughput at baseline. QUIC datagrams show no statistically significant performance difference from streams at representative IoT RTTs. +Under loss QUIC delivers more than an encryption-matched TCP+TLS~1.3 baseline, 1.26$\times$ at 1\% loss rising to 1.78$\times$ at 10\%, once the emulated loss removes single packets; the evaluation build that produced these ratios could not be paced, and at 10\% loss the ratio depends on the implementation build, rising to 1.97--2.05$\times$ with the pacing build (Sections~\ref{sec:exp3} and~\ref{sec:threats}). Loss applied per kernel buffer on the sending host inflated that advantage to 2--3$\times$; we show how to detect this from the qdisc's own counters and how to avoid it with a forwarding hop. Broker-side per-publish delivery has a per-subscriber ceiling on the clean link consistent with stream credit, costs more broker CPU per delivered message than per-topic delivery, and under unpaced load at light loss delivers a fraction of per-topic delivery's rate (38\% at 1\% loss), a gap that closes as loss becomes the binding limit. On a single connection the strategies differ ninefold, but that difference is set by two QUIC transport constants, the concurrent stream limit and the per-stream receive window, rather than by the mappings. -These results provide practitioners with concrete guidance for deploying MQTT over QUIC in IoT systems: use per-topic mapping for multi-topic workloads on lossy networks where independent stream recovery provides meaningful isolation and QUIC's throughput advantage under loss is beneficial, and use TCP when baseline throughput on reliable networks is the dominant concern. The implementation is available as open-source software. +For practitioners: use control-only as the default for the lowest typical tail latency at heavy loss and moderate-to-high rate (5\% loss, 500\,msg/s); consider per-topic when a single client must publish beyond control-only's window-bound rate (about 33K\,msg/s here) across several topics and the per-stream window cannot be raised; use per-publish for low-rate or light-loss channels that need per-message isolation, sizing the stream limit to the per-subscriber message rate. Whether QUIC's throughput gain matters depends on the loss a deployment sees: on a clean link both transports are bound jointly by broker and receiver CPU, and the gap between them depends on configuration. The implementation is available as open-source software. -\section*{Declaration of Generative AI and AI-Assisted Technologies in the Manuscript Preparation Process} +\section*{Funding} -During the preparation of this work the author used Claude (Anthropic) in order to assist with manuscript drafting, data analysis scripting, and editing. After using this tool, the author reviewed and edited the content as needed and takes full responsibility for the content of the published article. +This research did not receive any specific grant from funding agencies in the public, commercial, or not-for-profit sectors. \section*{Data Availability} -The experimental data used in this study is archived at Zenodo~\cite{bracht2026zenodo}. Analysis scripts and experiment infrastructure are available in the project repository at \url{https://github.com/LabOverWire/mqtt-lib}. +All experimental data reported in this study, together with the scripts that produced and analyzed them, are archived at Zenodo~\cite{bracht2026zenodo}. The implementation, analysis scripts, figure generators and experiment infrastructure are also available in the project repository at \url{https://github.com/LabOverWire/mqtt-lib}. \printcredits +\section*{Declaration of generative AI and AI-assisted technologies in the manuscript preparation process} + +During the preparation of this work the author used Claude (Anthropic) in order to assist with experiment design and scripting, data analysis, and manuscript drafting and editing. After using this tool, the author reviewed and edited the content as needed and takes full responsibility for the content of the published article. + \bibliographystyle{cas-model2-names} \bibliography{references} diff --git a/publications/comnet/submission/references.bib b/publications/comnet/submission/references.bib index 42e035e1..17c0e9c8 100644 --- a/publications/comnet/submission/references.bib +++ b/publications/comnet/submission/references.bib @@ -38,20 +38,25 @@ @misc{rfc9308 publisher = {IETF} } -@misc{rfc9221, - title = {An Unreliable Datagram Extension to {QUIC}}, - author = {Pauly, Tommy and Kinnear, Eric and Schinazi, David}, - year = {2022}, - howpublished = {RFC 9221}, - publisher = {IETF} -} - @article{kumar2019mqtt-quic, title = {Implementation and Analysis of {QUIC} for {MQTT}}, author = {Kumar, Puneet and Dezfouli, Behnam}, journal = {Computer Networks}, + volume = {150}, + pages = {28--45}, year = {2019}, - publisher = {Elsevier} + publisher = {Elsevier}, + doi = {10.1016/j.comnet.2018.12.012} +} + +@inproceedings{scharf2006hol, + title = {Head-of-line Blocking in {TCP} and {SCTP}: Analysis and Measurements}, + author = {Scharf, Michael and Kiesel, Sebastian}, + booktitle = {Proceedings of the IEEE Global Telecommunications Conference (GLOBECOM 2006)}, + year = {2006}, + address = {San Francisco, CA, USA}, + publisher = {IEEE}, + doi = {10.1109/GLOCOM.2006.333} } @inproceedings{fernandez2020iot-quic, @@ -127,7 +132,7 @@ @misc{bracht2026zenodo title = {Experiment Data for ``{Evaluating Stream Mapping Strategies for MQTT over QUIC}''}, author = {Bracht, Fabricio}, year = {2026}, - doi = {10.5281/zenodo.19098820}, + doi = {10.5281/zenodo.19098819}, publisher = {Zenodo} } @@ -138,3 +143,55 @@ @misc{rfc7540 howpublished = {RFC 7540}, publisher = {IETF} } + +@misc{yang2024mqttnext, + title = {{Spec MQTT-next}: A Mapping of {MQTT} to the {QUIC} Transport Protocol}, + author = {Yang, William}, + year = {2024}, + howpublished = {OASIS MQTT Technical Committee, document 71729}, + url = {https://groups.oasis-open.org/higherlogic/ws/public/download/71729/oasis_mqtt_over_quic.pdf} +} + +@misc{nanomq-quic, + title = {{NanoMQ}: {MQTT} over {QUIC} Bridge}, + author = {{EMQ Technologies}}, + year = {2024}, + url = {https://nanomq.io/docs/en/latest/bridges/quic-bridge.html} +} + +@misc{rtpfolks2017, + title = {{RTP} over {QUIC}}, + author = {Ott, J{\"o}rg and Even, Roni and Perkins, Colin and Singh, Varun}, + year = {2017}, + howpublished = {Internet-Draft draft-rtpfolks-quic-rtp-over-quic-01, IETF}, + url = {https://datatracker.ietf.org/doc/html/draft-rtpfolks-quic-rtp-over-quic} +} + +@article{chiu1989aimd, + title = {Analysis of the Increase and Decrease Algorithms for Congestion Avoidance in Computer Networks}, + author = {Chiu, Dah-Ming and Jain, Raj}, + journal = {Computer Networks and ISDN Systems}, + volume = {17}, + number = {1}, + pages = {1--14}, + year = {1989}, + doi = {10.1016/0169-7552(89)90019-6} +} + +@misc{azureiot, + title = {Communicate with your {IoT} Hub using the {MQTT} protocol}, + author = {{Microsoft}}, + year = {2024}, + howpublished = {Microsoft Azure IoT Hub documentation}, + url = {https://learn.microsoft.com/en-us/azure/iot-hub/iot-hub-mqtt-support} +} + +@article{mathis1997macroscopic, + title = {The Macroscopic Behavior of the {TCP} Congestion Avoidance Algorithm}, + author = {Mathis, Matthew and Semke, Jeffrey and Mahdavi, Jamshid and Ott, Teunis}, + journal = {ACM SIGCOMM Computer Communication Review}, + volume = {27}, + number = {3}, + pages = {67--82}, + year = {1997} +}