"""Sharper tests for FedDPO / DecDPO convergence claims: fitted rate exponents and parameter-free ratio checks, at larger scale than the first pass. Log-linear DPO: reward r(x) = theta^T phi(x); on a preference pair (w, l) the loss is -log sigmoid(beta (r_w - r_l)). Clients hold heterogeneous preference data generated by client-specific ground-truth rewards. """ import json import numpy as np RESULTS = {} D, BETA = 32, 1.0 def make_clients(N, kappa, seed, n_per=200): rng = np.random.default_rng(seed) base = rng.normal(size=D); base /= np.linalg.norm(base) cl = [] for i in range(N): t = base + kappa * rng.normal(size=D) / np.sqrt(D) t /= np.linalg.norm(t) W = rng.normal(size=(n_per, D)); L = rng.normal(size=(n_per, D)) flip = (W - L) @ t < 0 W2 = np.where(flip[:, None], L, W); L2 = np.where(flip[:, None], W, L) cl.append((W2, L2, t)) return cl, base def grad(th, W, L): z = BETA * ((W - L) @ th) s = 1.0 / (1.0 + np.exp(z)) return -BETA * ((W - L) * s[:, None]).mean(axis=0) def loss(th, cl): tot = 0.0 for W, L, _ in cl: z = BETA * ((W - L) @ th) tot += float(np.mean(np.log1p(np.exp(-z)))) return tot / len(cl) def fed_dpo(cl, R=200, E=5, S=None, lr=0.5, q_max=0, seed=0): N = len(cl); S = S or N rng = np.random.default_rng(seed) th = np.zeros(D); buf = {} hist = [] for r in range(R): sel = rng.choice(N, size=S, replace=False) deltas = [] for i in sel: local = th.copy() for _ in range(E): local -= lr * grad(local, cl[i][0], cl[i][1]) d = local - th delay = int(rng.integers(0, q_max + 1)) if q_max else 0 buf.setdefault(r + delay, []).append(d) for d in buf.pop(r, []): deltas.append(d) if deltas: th = th + np.mean(deltas, axis=0) hist.append(loss(th, cl)) return np.array(hist) def metropolis(adj): n = adj.shape[0]; deg = adj.sum(1) Wm = np.zeros((n, n)) for i in range(n): for j in range(n): if i != j and adj[i, j]: Wm[i, j] = 1.0 / (1 + max(deg[i], deg[j])) Wm[i, i] = 1 - Wm[i].sum() ev = np.sort(np.abs(np.linalg.eigvals(Wm)))[::-1] return Wm, float(ev[1]) def dec_dpo(cl, Wm, R=200, E=5, lr=0.5): N = len(cl) TH = np.zeros((N, D)); hist = [] for r in range(R): for i in range(N): for _ in range(E): TH[i] -= lr * grad(TH[i], cl[i][0], cl[i][1]) TH = Wm @ TH hist.append(float(np.mean(np.linalg.norm(TH - TH.mean(0), axis=1)))) return np.array(hist) def claim2_participation(): rows = [] N = 20 cl, _ = make_clients(N, 0.8, seed=1) for S in (2, 5, 10, 20): finals = [fed_dpo(cl, R=150, S=S, seed=100 + s)[-20:].mean() for s in range(5)] var = [np.var(fed_dpo(cl, R=150, S=S, seed=200 + s)[-20:]) for s in range(5)] rows.append({"S": S, "N": N, "final_loss": float(np.mean(finals)), "tail_variance": float(np.mean(var)), "one_over_S": 1.0 / S}) print(" S=%-3d final loss=%.6f tail var=%.3e (1/S=%.3f)" % (S, rows[-1]["final_loss"], rows[-1]["tail_variance"], 1.0 / S), flush=True) ls = np.log([r["one_over_S"] for r in rows]); lv = np.log([max(r["tail_variance"], 1e-16) for r in rows]) RESULTS["claim2_participation"] = { "rows": rows, "loglog_slope_var_vs_1_over_S": round(float(np.polyfit(ls, lv, 1)[0]), 4), "var_ratio_S2_over_SN": round(rows[0]["tail_variance"] / max(rows[-1]["tail_variance"], 1e-16), 2)} print(" variance slope vs 1/S = %.3f ; S=2 vs S=N ratio = %.1fx" % (RESULTS["claim2_participation"]["loglog_slope_var_vs_1_over_S"], RESULTS["claim2_participation"]["var_ratio_S2_over_SN"]), flush=True) def claim3_staleness(): rows = [] cl, _ = make_clients(10, 0.8, seed=3) for q in (0, 1, 2, 5, 10): f = [fed_dpo(cl, R=150, q_max=q, seed=300 + s)[-20:].mean() for s in range(5)] rows.append({"q_max": q, "final_loss": float(np.mean(f)), "sd": float(np.std(f))}) print(" q_max=%-3d final loss=%.6f +- %.6f" % (q, rows[-1]["final_loss"], rows[-1]["sd"]), flush=True) base = rows[0]["final_loss"] RESULTS["claim3_staleness"] = { "rows": rows, "monotone_in_q": all(rows[i+1]["final_loss"] >= rows[i]["final_loss"] - 1e-9 for i in range(len(rows) - 1)), "penalty_at_qmax10": round(rows[-1]["final_loss"] - base, 6)} def claim5_topology(): N = 8 cl, _ = make_clients(N, 0.8, seed=5) tops = {} ring = np.zeros((N, N), int) for i in range(N): ring[i, (i + 1) % N] = ring[(i + 1) % N, i] = 1 tops["ring"] = ring star = np.zeros((N, N), int); star[0, 1:] = star[1:, 0] = 1 tops["star"] = star full = np.ones((N, N), int) - np.eye(N, dtype=int) tops["complete"] = full path = np.zeros((N, N), int) for i in range(N - 1): path[i, i + 1] = path[i + 1, i] = 1 tops["path"] = path rows = [] for name, adj in tops.items(): Wm, rho = metropolis(adj) h = dec_dpo(cl, Wm, R=120) rows.append({"topology": name, "rho": round(rho, 4), "one_over_1_minus_rho2": round(1.0 / (1 - rho ** 2), 3), "final_consensus_error": float(h[-1]), "mean_tail_consensus": float(h[-20:].mean())}) print(" %-9s rho=%.4f 1/(1-rho^2)=%8.2f consensus err=%.4e" % (name, rho, rows[-1]["one_over_1_minus_rho2"], rows[-1]["mean_tail_consensus"]), flush=True) x = np.log([r["one_over_1_minus_rho2"] for r in rows]) y = np.log([max(r["mean_tail_consensus"], 1e-16) for r in rows]) sl, ic = np.polyfit(x, y, 1) r2 = 1 - np.var(y - (sl * x + ic)) / np.var(y) RESULTS["claim5_topology"] = {"rows": rows, "loglog_slope_consensus_vs_1_over_1_minus_rho2": round(float(sl), 4), "r2": round(float(r2), 4)} print(" consensus error vs 1/(1-rho^2): slope %.3f, R2 %.3f" % (sl, r2), flush=True) if __name__ == "__main__": claim2_participation(); claim3_staleness(); claim5_topology() json.dump(RESULTS, open("dpo_results.json", "w"), indent=1)