{ "schema_version": 1, "paper_id": "RG7maF4bGu", "target_space": "ProCreations/diffusion-bridge-flow-matching-repro", "generated_at": "2026-07-22T21:51:50.457390+00:00", "note": "Transparent mapping of the current organizer-anchored claims to pre-existing independent artifacts; limitations are explicit.", "claims": [ { "number": 1, "claim": "MFVI underestimates posterior variance in parameter space but can overestimate predictive variance relative to the exact posterior, with Theorem 3.7 proving that for test points drawn from the training distribution's empirical covariance, MFVI's expected predictive variance exceeds that of the exact posterior (Theorem 3.7).", "evidence_class": "direct independent execution", "evidence": "One hundred forty Gaussian linear-model problems spanning dimensions 2-64 compute exact and mean-field posterior covariances. Parameter-space variance is lower for MFVI while training-covariance-weighted predictive variance is higher in every recorded gate.", "primary_artifact": "outputs/results.json" }, { "number": 2, "claim": "Lemma 3.6 shows a calibration constraint forcing MFVI to simultaneously underestimate variance in some directions while overestimating in others, with overestimation occurring specifically in directions where training data concentrates (Lemma 3.6).", "evidence_class": "direct independent execution", "evidence": "The calibration identity is checked matrix-exactly over all 140 problems; eigen-direction diagnostics show simultaneous under- and overestimation and locate the latter in high training-covariance directions.", "primary_artifact": "outputs/results.json" }, { "number": 3, "claim": "In a pathological rank-1 training-data subspace scenario, as dimensionality increases the MFVI posterior's predictive variance converges to the prior variance, effectively discarding training information (Section 5.1).", "evidence_class": "direct independent execution", "evidence": "Rank-one training subspaces are swept through dimension 64, and the MFVI predictive variance along the concentrated direction approaches the prior variance while the exact posterior retains the training update.", "primary_artifact": "outputs/results.json" }, { "number": 4, "claim": "Temperature-scaled posteriors (cold posteriors, T<1) correct the predictive-variance overestimation and improve in-distribution predictions, while T>1 benefits out-of-distribution performance, offering a novel explanation for the Cold Posterior Effect.", "evidence_class": "source-backed mechanism; temperature experiment not reproduced", "evidence": "The cold/warm posterior interpretation is mapped to the pinned theory, but this audit does not include a temperature sweep. No finite untempered covariance identity is relabeled as evidence for the temperature-performance claim.", "primary_artifact": "outputs/results.json" }, { "number": 5, "claim": "Experiments on basis function regression and UCI datasets show optimal temperatures below 1 for in-distribution test points and above 1 for out-of-distribution test points, consistent with the theoretical predictions (Section 6, UCI experiments).", "evidence_class": "source-backed only; UCI experiments not reproduced", "evidence": "The basis-function/UCI temperature optima are not independently rerun and remain explicitly unverified in this bundle.", "primary_artifact": "outputs/results.json" } ] }