From 6892ffe1e11024e5a5b80225ec385db44b0bdd14 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 20:22:54 +0000 Subject: [PATCH 01/13] [ENH] verify tau^2, heterogeneity and the tau^2 interval against metafor PyMARE reproduces metafor's Knapp-Hartung adjustment to machine precision and has since #139, but that is one of four things rma.uni reports. The other three -- Cochran's Q and its p-value, I^2 and H, and the Q-profile interval for tau^2 -- were never compared against anything, and neither was the tau^2 point estimate outside the two closed-form estimators. Measuring them turned up one defect and located three divergences. q_profile computed both bounds by minimizing (Q(tau^2) - crit)**2 with scipy.optimize.minimize. Squaring turns a transversal crossing into a tangential minimum, so the gradient vanishes as the critical value is approached and the search stops early where Q is flattest. The upper bound was wrong by up to 3.6e-2 relative against metafor's confint.rma.uni -- 59.6127 where the root is 59.6160 on the test_stats design. Solving for the root with brentq instead brings both bounds to 1.3e-13 of metafor across the grid. The two tests that pinned the old upper bound asserted it to two decimals, which is why this never showed up as a failure. run_metafor.R now also records QE, QEp, I2, H2 and confint's tau^2 bounds for the existing 180-case grid. confint is asked for to convergence rather than at its default uniroot tolerance of eps^0.25 (~1.2e-4 relative), which would otherwise pin metafor's display precision rather than the bound it solves for. test_metafor_random_effects.py compares all of it. What agrees: Q, p(Q), logp(Q) 1.5e-13, all 60 design/model/method cells I^2, H for FE and DL 4.9e-14 tau^2 Q-profile interval 1.3e-13 Hedges tau^2, no moderators 2.2e-16 ML, REML tau^2 2.7e-5 (both profile numerically) Three divergences are pinned down by asserting their cause rather than their size, so the tests stay meaningful if a tolerance moves: - I^2 and H: PyMARE always reports the Q-based Higgins-Thompson pair. metafor reports that pair only for FE and DL, where it coincides with tau^2 / (tau^2 + v_t), and switches to the tau^2-based pair otherwise. - Hedges tau^2: metafor subtracts tr(PV) / (K - P), PyMARE subtracts sum(v) / K. Equal when the intercept is the only predictor, out by up to 0.14 relative with moderators. This refines what validation/metafor/README.md recorded: the divergence is specific to meta-regression, not general. - ML on extreme_k10 with one moderator: metafor stops at tau^2 = 0 and PyMARE at 0.0114. The README had this cell's direction backwards. The reference file is regenerated in full under the same pinned stack it already named (R 4.4.1, metafor 4.6-0) but a different BLAS, so every previously pinned number moved: at most 9.5e-14 relative on the inference path and 3.3e-9 on an ML tau^2, with dof unchanged. That is below every tolerance in the suite, but it does mean `git diff --exit-code` can no longer police this file; a numeric comparator follows. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011o2j5ZNAdzPFL6LBqDqUsU --- pymare/stats.py | 94 +- pymare/tests/data/metafor_reference.json | 2407 ++++++++++++++----- pymare/tests/test_metafor_random_effects.py | 376 +++ pymare/tests/test_results.py | 7 +- pymare/tests/test_stats.py | 40 +- validation/metafor/run_metafor.R | 46 +- 6 files changed, 2290 insertions(+), 680 deletions(-) create mode 100644 pymare/tests/test_metafor_random_effects.py diff --git a/pymare/stats.py b/pymare/stats.py index 14609b0..06f46e8 100644 --- a/pymare/stats.py +++ b/pymare/stats.py @@ -19,7 +19,7 @@ import numpy as np import scipy.stats as ss -from scipy.optimize import Bounds, minimize +from scipy.optimize import brentq from scipy.special import gammaln # At or below this many clusters, robust variance estimation is known to be @@ -2291,6 +2291,72 @@ def ensure_2d(arr): return arr +#: Doublings allowed when widening the bracket for a Q-profile bound. Q(tau^2) +#: falls to zero as tau^2 grows, so a root exists whenever Q(0) exceeds the +#: critical value and the search terminates long before this; it is a stop for +#: the degenerate case of a zero critical value, which no root can reach. +_Q_PROFILE_MAX_DOUBLINGS = 200 + +#: Relative tolerance asked of the Q-profile root finder. Four times machine +#: epsilon is the smallest :func:`scipy.optimize.brentq` accepts, which is what +#: makes the bounds agree with metafor's ``confint.rma.uni`` to the last few +#: bits rather than to the third decimal. +_Q_PROFILE_RTOL = 4 * np.finfo(np.float64).eps + + +def _invert_q(excess, crit, scale): + """Solve ``Q(tau^2) = crit`` for tau^2 >= 0. + + Parameters + ---------- + excess : callable + ``(tau2, crit) -> Q(tau2) - crit``. + crit : :obj:`float` + The chi-squared quantile being inverted. + scale : :obj:`float` + A point estimate of tau^2, used to size the first bracket tried. + + Returns + ------- + :obj:`float` + The root, ``0.0`` when ``Q(0) <= crit`` so that no positive root + exists, or :obj:`numpy.nan` when the bracket could not be widened far + enough to contain one. + + Notes + ----- + Solved as a root rather than as the minimum of ``(Q(tau^2) - crit)**2``, + which is what this function replaced. Squaring is what makes the difference: + it turns a transversal crossing into a tangential minimum, so the gradient + the minimizer follows vanishes as ``crit`` is approached and it stops while + still far from the root in the flat upper tail, where ``Q`` changes slowly. + The upper bound was wrong by up to 4% relative against + ``metafor::confint.rma.uni`` on the designs in + ``pymare/tests/data/metafor_small_sample.csv`` for that reason; it now + agrees to a few multiples of machine epsilon. + + ``Q`` is monotonically decreasing in tau^2, but Brent's method needs only a + sign change over the bracket, so nothing here relies on that. + """ + if excess(0.0, crit) <= 0: + # Q at tau^2 = 0 is already at or below the critical value, so the + # profile never crosses it. metafor reports the boundary here too. + return 0.0 + + # Q(tau^2) -> 0 as tau^2 -> infinity, since the weights approach a common + # 1 / tau^2 that scales the residual sum of squares away. So a root exists + # for any positive crit, and doubling finds it. + upper = max(abs(scale), 1.0) + for _ in range(_Q_PROFILE_MAX_DOUBLINGS): + if excess(upper, crit) <= 0: + break + upper *= 2.0 + else: + return np.nan + + return brentq(excess, 0.0, upper, args=(crit,), rtol=_Q_PROFILE_RTOL, maxiter=200) + + def q_profile(y, v, X, alpha=0.05, groups=None): """Get the CI for tau^2 via the Q-Profile method. @@ -2343,22 +2409,28 @@ def q_profile(y, v, X, alpha=0.05, groups=None): l_crit = ss.chi2.ppf(1 - alpha / 2, df) u_crit = ss.chi2.ppf(alpha / 2, df) args = (ensure_2d(y), ensure_2d(v), X) - bds = Bounds([0], [np.inf], keep_feasible=True) - # Use a point estimate of tau^2 as a starting point; when using a fixed - # value, minimize() sometimes fails to stay in bounds. It has to be the - # estimator that matches the Q being inverted, or the search can start on - # the wrong side of the upper root. + def excess(tau2, crit): + """Q(tau^2) - crit, the function whose root is a bound.""" + return float(np.ravel(q_gen(*args, float(tau2), groups))[0]) - crit + + # A scale for the bracket search, not a starting point: the root finder + # below needs an interval that contains the root, and the point estimate + # says what order of magnitude tau^2 lives at. It has to be the estimator + # that matches the Q being inverted, so that the first bracket tried is + # usually already wide enough. if groups is None: from .estimators import DerSimonianLaird - ub_start = 2 * DerSimonianLaird().fit(y, v, X).params_["tau2"] + scale = DerSimonianLaird().fit(y, v, X).params_["tau2"] else: - ub_start = 2 * correlated_effects_tau2(*args, groups) + scale = correlated_effects_tau2(*args, groups) - lb = minimize(lambda x: (q_gen(*args, x, groups) - l_crit) ** 2, [0], bounds=bds).x[0] - ub = minimize(lambda x: (q_gen(*args, x, groups) - u_crit) ** 2, ub_start, bounds=bds).x[0] - return {"ci_l": lb, "ci_u": ub} + scale = float(np.ravel(scale)[0]) + return { + "ci_l": _invert_q(excess, l_crit, scale), + "ci_u": _invert_q(excess, u_crit, scale), + } #: Iterations allowed in the continued fraction of :func:`log_chi2_sf`. A safety diff --git a/pymare/tests/data/metafor_reference.json b/pymare/tests/data/metafor_reference.json index 81cf6e3..1171db4 100644 --- a/pymare/tests/data/metafor_reference.json +++ b/pymare/tests/data/metafor_reference.json @@ -2,17 +2,24 @@ "source": { "data": "metafor_small_sample.csv", "call": "rma.uni(y, v, mods = , data = , method = , test = )", + "tau2_ci_call": "confint(, control = list(tol = 1e-12, maxiter = 1000))$random, row tau^2", "metafor_version": "4.6.0", "r_version": "4.4.1" }, "cases": [ {"design": "equal_k5", "model": "intercept", "method": "FE", "test": "z", "tau2": 0, - "beta": [0.5912488674785934], + "beta": [0.59124886747859351], "se": [0.44265664589063253], - "pval": [0.18165297415447393], - "ci_lb": [-0.27634221598434638], + "pval": [0.18165297415447376], + "ci_lb": [-0.27634221598434627], "ci_ub": [1.4588399509415333], + "QE": 8.3983359150613381, + "QEp": 0.078029419298244626, + "I2": 52.371516923650162, + "H2": 2.0995839787653345, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": null}, {"design": "unequal_k5", "model": "intercept", "method": "FE", "test": "z", "tau2": 0, @@ -21,14 +28,26 @@ "pval": [0.0026933945863149944], "ci_lb": [-0.51105313280333575], "ci_ub": [-0.10721965639349787], + "QE": 20.315605412654854, + "QEp": 0.00043261444283937314, + "I2": 80.310702444002246, + "H2": 5.0789013531637135, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": null}, {"design": "extreme_k10", "model": "intercept", "method": "FE", "test": "z", "tau2": 0, - "beta": [0.2484353958726899], + "beta": [0.24843539587268987], "se": [0.027996739563533962], - "pval": [7.0738472818813947e-19], - "ci_lb": [0.1935627946436157], - "ci_ub": [0.30330799710176409], + "pval": [7.0738472818815074e-19], + "ci_lb": [0.19356279464361567], + "ci_ub": [0.30330799710176404], + "QE": 528.36542813689618, + "QEp": 4.8272097789668306e-108, + "I2": 98.296633443309204, + "H2": 58.707269792988463, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": null}, {"design": "moderate_k20", "model": "intercept", "method": "FE", "test": "z", "tau2": 0, @@ -37,30 +56,54 @@ "pval": [1.4342046963366567e-08], "ci_lb": [0.51728641988318369], "ci_ub": [1.0639477495833147], + "QE": 27.443467066041269, + "QEp": 0.094740008781753912, + "I2": 30.766765167544271, + "H2": 1.4443930034758563, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": null}, {"design": "equal_k5", "model": "one", "method": "FE", "test": "z", "tau2": 0, - "beta": [1.2765478897207587, 1.5577764687433346], - "se": [0.53219419519755828, 0.67157839168830602], - "pval": [0.016455611193178728, 0.020363877879022316], - "ci_lb": [0.23346643435226522, 0.24150700823892146], - "ci_ub": [2.3196293450892522, 2.8740459292477478], + "beta": [1.2765478897207592, 1.5577764687433355], + "se": [0.5321941951975584, 0.67157839168830613], + "pval": [0.016455611193178711, 0.02036387787902226], + "ci_lb": [0.23346643435226544, 0.24150700823892213], + "ci_ub": [2.3196293450892531, 2.8740459292477487], + "QE": 3.0179075904568835, + "QEp": 0.38887241030467207, + "I2": 0.5933776936547106, + "H2": 1.0059691968189612, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": null}, {"design": "unequal_k5", "model": "one", "method": "FE", "test": "z", "tau2": 0, - "beta": [-0.28720869740317378, -0.35624980573686255], - "se": [0.10568710690329607, 0.38326506514697062], - "pval": [0.0065769662190249571, 0.35262336525639432], - "ci_lb": [-0.49435162056386861, -1.1074355299573224], - "ci_ub": [-0.080065774242478988, 0.39493591848359727], + "beta": [-0.28720869740317373, -0.35624980573686266], + "se": [0.10568710690329605, 0.38326506514697067], + "pval": [0.0065769662190249571, 0.35262336525639426], + "ci_lb": [-0.4943516205638685, -1.1074355299573226], + "ci_ub": [-0.08006577424247896, 0.39493591848359727], + "QE": 19.451611261114039, + "QEp": 0.00022048004795655233, + "I2": 84.577113125855348, + "H2": 6.4838704203713462, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": null}, {"design": "extreme_k10", "model": "one", "method": "FE", "test": "z", "tau2": 0, - "beta": [0.36308889801604494, 0.65095292619066303], + "beta": [0.36308889801604494, 0.65095292619066292], "se": [0.028451161183932691, 0.028755150789987197], - "pval": [2.678321482283576e-37, 1.8406210952213409e-113], - "ci_lb": [0.30732564677719287, 0.59459386627226962], - "ci_ub": [0.418852149254897, 0.70731198610905643], + "pval": [2.678321482283576e-37, 1.8406210952214895e-113], + "ci_lb": [0.30732564677719287, 0.59459386627226951], + "ci_ub": [0.418852149254897, 0.70731198610905632], + "QE": 15.896110614196777, + "QEp": 0.043891457052016768, + "I2": 49.673223883739084, + "H2": 1.9870138267745971, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": null}, {"design": "moderate_k20", "model": "one", "method": "FE", "test": "z", "tau2": 0, @@ -69,62 +112,110 @@ "pval": [3.9548102645066542e-07, 0.00010770940386508056], "ci_lb": [0.43818989196649638, 0.32356831858529306], "ci_ub": [0.99029101918584939, 0.98674097650419279], + "QE": 12.446943265071944, + "QEp": 0.82332537450409671, + "I2": 0, + "H2": 0.69149684805955247, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": null}, {"design": "equal_k5", "model": "two", "method": "FE", "test": "z", "tau2": 0, - "beta": [1.2531305457285846, 1.5129861636211883, 0.23824827277034544], - "se": [0.5382906100438315, 0.68912464663627082, 0.82190123836347695], - "pval": [0.01991308959269936, 0.028126399001332784, 0.77191219256141608], - "ci_lb": [0.19810033682658035, 0.1623266753552064, -1.3726485532709394], - "ci_ub": [2.3081607546305891, 2.86364565188717, 1.8491450988116303], + "beta": [1.2531305457285848, 1.5129861636211892, 0.2382482727703448], + "se": [0.53829061004383161, 0.68912464663627104, 0.82190123836347695], + "pval": [0.01991308959269936, 0.028126399001332753, 0.77191219256141663], + "ci_lb": [0.19810033682658035, 0.16232667535520684, -1.37264855327094], + "ci_ub": [2.3081607546305891, 2.8636456518871718, 1.8491450988116296], + "QE": 2.9338803207967543, + "QEp": 0.23063009774267032, + "I2": 31.830893515899799, + "H2": 1.4669401603983772, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": null}, {"design": "unequal_k5", "model": "two", "method": "FE", "test": "z", "tau2": 0, - "beta": [-0.31818517955900727, -1.4858777781006078, -1.4718397892444939], - "se": [0.10600770201892341, 0.48697806397185173, 0.39143711218949201], - "pval": [0.0026862755330678378, 0.0022791262368038611, 0.00016985047321758549], - "ci_lb": [-0.52595645759995113, -2.4403372447464795, -2.2390424313482629], - "ci_ub": [-0.11041390151806343, -0.53141831145473606, -0.70463714714072501], + "beta": [-0.31818517955900727, -1.4858777781006076, -1.4718397892444932], + "se": [0.10600770201892343, 0.48697806397185173, 0.39143711218949195], + "pval": [0.0026862755330678417, 0.0022791262368038654, 0.00016985047321758646], + "ci_lb": [-0.52595645759995113, -2.4403372447464795, -2.239042431348262], + "ci_ub": [-0.11041390151806341, -0.53141831145473584, -0.70463714714072445], + "QE": 5.3133148940953756, + "QEp": 0.070182418569350757, + "I2": 62.358715042043215, + "H2": 2.6566574470476878, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": null}, {"design": "extreme_k10", "model": "two", "method": "FE", "test": "z", "tau2": 0, - "beta": [0.33951715132107646, 0.65156082084807287, -0.12127406276771753], - "se": [0.034367023554020541, 0.028759448117667584, 0.099180863165782243], - "pval": [5.1263369109800041e-23, 1.2302772325697296e-113, 0.22142161260838295], + "beta": [0.3395171513210764, 0.65156082084807287, -0.12127406276771752], + "se": [0.034367023554020541, 0.028759448117667584, 0.099180863165782257], + "pval": [5.1263369109800952e-23, 1.2302772325697296e-113, 0.22142161260838308], "ci_lb": [0.27215902289935645, 0.59519333832219612, -0.31566498252824593], - "ci_ub": [0.40687527974279647, 0.70792830337394963, 0.073116856992810883], + "ci_ub": [0.40687527974279636, 0.70792830337394963, 0.073116856992810925], + "QE": 14.400976720554908, + "QEp": 0.044492241224326719, + "I2": 51.39218585077839, + "H2": 2.0572823886507012, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": null}, {"design": "moderate_k20", "model": "two", "method": "FE", "test": "z", "tau2": 0, - "beta": [0.73709810315645485, 0.64747703612718221, 0.19044587474139929], - "se": [0.14672220323157109, 0.16974243343585627, 0.34253541418439043], - "pval": [5.0667809224461877e-07, 0.00013647676387872824, 0.57821846967745549], - "ci_lb": [0.4495278690902092, 0.31478797994471652, -0.4809112004895163], - "ci_ub": [1.0246683372227006, 0.98016609230964791, 0.86180294997231488], + "beta": [0.73709810315645485, 0.64747703612718244, 0.19044587474139935], + "se": [0.14672220323157109, 0.16974243343585629, 0.34253541418439043], + "pval": [5.0667809224461877e-07, 0.000136476763878728, 0.57821846967745549], + "ci_lb": [0.4495278690902092, 0.31478797994471669, -0.48091120048951624], + "ci_ub": [1.0246683372227006, 0.98016609230964824, 0.86180294997231499], + "QE": 12.137819651655924, + "QEp": 0.79172076097379995, + "I2": 0, + "H2": 0.71398939127387795, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": null}, {"design": "equal_k5", "model": "intercept", "method": "DL", "test": "z", "tau2": 1.08260596526162, - "beta": [0.60497665381666166], + "beta": [0.60497665381666177], "se": [0.64379518699524474], - "pval": [0.34736961888930801], - "ci_lb": [-0.65683872611424721], - "ci_ub": [1.8667920337475705], + "pval": [0.34736961888930795], + "ci_lb": [-0.6568387261142471], + "ci_ub": [1.8667920337475707], + "QE": 8.3983359150613381, + "QEp": 0.078029419298244626, + "I2": 52.371516923650162, + "H2": 2.0995839787653345, + "tau2_ci_lb": 0, + "tau2_ci_ub": 15.562063970721971, "dof": null}, {"design": "unequal_k5", "model": "intercept", "method": "DL", "test": "z", "tau2": 0.28382585271252081, - "beta": [-0.25834451420148663], + "beta": [-0.25834451420148669], "se": [0.31823753095856316], - "pval": [0.41690768865610456], - "ci_lb": [-0.88207861340922078], - "ci_ub": [0.36538958500624752], + "pval": [0.41690768865610461], + "ci_lb": [-0.88207861340922089], + "ci_ub": [0.36538958500624746], + "QE": 20.315605412654854, + "QEp": 0.00043261444283937314, + "I2": 80.310702444002246, + "H2": 5.0789013531637144, + "tau2_ci_lb": 0.37437910706976263, + "tau2_ci_ub": 62.648818709153389, "dof": null}, {"design": "extreme_k10", "model": "intercept", "method": "DL", "test": "z", - "tau2": 1.0914460604195861, + "tau2": 1.0914460604195859, "beta": [0.69453702188798982], "se": [0.39079170141387481], "pval": [0.075526076516440638], "ci_lb": [-0.071400638340335276], "ci_ub": [1.460474682116315], + "QE": 528.36542813689618, + "QEp": 4.8272097789668306e-108, + "I2": 98.296633443309219, + "H2": 58.707269792988463, + "tau2_ci_lb": 0.27621241313967376, + "tau2_ci_ub": 4.0683787385044337, "dof": null}, {"design": "moderate_k20", "model": "intercept", "method": "DL", "test": "z", "tau2": 0.18430288219628541, @@ -133,30 +224,54 @@ "pval": [2.7230633698122343e-05], "ci_lb": [0.42053203336548378], "ci_ub": [1.157928968499971], + "QE": 27.443467066041269, + "QEp": 0.094740008781753912, + "I2": 30.766765167544271, + "H2": 1.4443930034758563, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.94040725109002843, "dof": null}, {"design": "equal_k5", "model": "one", "method": "DL", "test": "z", - "tau2": 0.0059427274506224873, - "beta": [1.2764449434579284, 1.5576340773383561], - "se": [0.53378359948569698, 0.67366912517087141], - "pval": [0.016788123831894555, 0.020768600060376924], - "ci_lb": [0.23024831292780945, 0.23726685450684282], - "ci_ub": [2.3226415739880473, 2.8780013001698697], + "tau2": 0.0059427274506217518, + "beta": [1.2764449434579284, 1.5576340773383557], + "se": [0.53378359948569687, 0.67366912517087096], + "pval": [0.016788123831894534, 0.020768600060376882], + "ci_lb": [0.23024831292780967, 0.23726685450684326], + "ci_ub": [2.3226415739880473, 2.8780013001698679], + "QE": 3.0179075904568835, + "QEp": 0.38887241030467207, + "I2": 0.5933776936547106, + "H2": 1.0059691968189612, + "tau2_ci_lb": 0, + "tau2_ci_ub": 12.956611642049028, "dof": null}, {"design": "unequal_k5", "model": "one", "method": "DL", "test": "z", - "tau2": 0.44482420458951977, - "beta": [-0.090448214875049893, -1.9230388371068361], - "se": [0.38939265054554362, 0.94903884952366147], - "pval": [0.81632036583877821, 0.042733898429604587], - "ci_lb": [-0.85364378578890632, -3.7831208021025402], - "ci_ub": [0.67274735603880664, -0.06295687211113199], + "tau2": 0.44482420458951955, + "beta": [-0.090448214875049893, -1.923038837106835], + "se": [0.38939265054554356, 0.94903884952366124], + "pval": [0.81632036583877809, 0.042733898429604678], + "ci_lb": [-0.85364378578890632, -3.7831208021025384], + "ci_ub": [0.67274735603880642, -0.062956872111131323], + "QE": 19.451611261114039, + "QEp": 0.00022048004795655233, + "I2": 84.577113125855362, + "H2": 6.4838704203713462, + "tau2_ci_lb": 0.19746423631486623, + "tau2_ci_ub": 45.327128288195816, "dof": null}, {"design": "extreme_k10", "model": "one", "method": "DL", "test": "z", - "tau2": 0.029699600901791464, - "beta": [0.27000256176805176, 0.67393162666171846], - "se": [0.096628598036758412, 0.067317117048797101], - "pval": [0.0052023406214139895, 1.359587846629236e-23], - "ci_lb": [0.080613989739407532, 0.54199250170300894], - "ci_ub": [0.45939113379669599, 0.80587075162042798], + "tau2": 0.029699600901791457, + "beta": [0.27000256176805176, 0.67393162666171824], + "se": [0.096628598036758384, 0.067317117048797087], + "pval": [0.0052023406214139764, 1.3595878466292604e-23], + "ci_lb": [0.080613989739407588, 0.54199250170300872], + "ci_ub": [0.45939113379669594, 0.80587075162042776], + "QE": 15.896110614196777, + "QEp": 0.043891457052016768, + "I2": 49.673223883739091, + "H2": 1.9870138267745971, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.98404808836485924, "dof": null}, {"design": "moderate_k20", "model": "one", "method": "DL", "test": "z", "tau2": 0, @@ -165,46 +280,82 @@ "pval": [3.9548102645066542e-07, 0.00010770940386508056], "ci_lb": [0.43818989196649638, 0.32356831858529306], "ci_ub": [0.99029101918584939, 0.98674097650419279], + "QE": 12.446943265071944, + "QEp": 0.82332537450409671, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.23460881621137722, "dof": null}, {"design": "equal_k5", "model": "two", "method": "DL", "test": "z", - "tau2": 0.48520036034108888, - "beta": [1.2464498791089595, 1.4947307524436979, 0.29015096278611763], - "se": [0.65425571520006054, 0.84732066452122279, 1.0153006551870316], - "pval": [0.056761647441435165, 0.077720634099653046, 0.77504787842989153], - "ci_lb": [-0.035867759362653739, -0.1659872333744441, -1.6998017548603841], - "ci_ub": [2.5287675175805728, 3.1554487382618399, 2.2801036804326191], + "tau2": 0.48520036034108743, + "beta": [1.2464498791089589, 1.4947307524436986, 0.29015096278611674], + "se": [0.6542557152000601, 0.84732066452122257, 1.0153006551870312], + "pval": [0.056761647441435109, 0.077720634099652852, 0.7750478784298922], + "ci_lb": [-0.035867759362653517, -0.16598723337444299, -1.6998017548603841], + "ci_ub": [2.5287675175805715, 3.1554487382618399, 2.2801036804326174], + "QE": 2.9338803207967543, + "QEp": 0.23063009774267032, + "I2": 31.830893515899795, + "H2": 1.4669401603983772, + "tau2_ci_lb": 0, + "tau2_ci_ub": 54.363997637568161, "dof": null}, {"design": "unequal_k5", "model": "two", "method": "DL", "test": "z", "tau2": 0.4717863239469125, - "beta": [-0.076102991931023045, -3.0910321898139177, -2.525201560728938], - "se": [0.39901562395069623, 1.0841059140587561, 1.1138538939878788], - "pval": [0.84873960377832236, 0.0043550848712978641, 0.023385027934051256], - "ci_lb": [-0.85815924414316536, -5.2158407367959541, -4.7083150769848761], - "ci_ub": [0.70595326028111938, -0.96622364283188089, -0.34208804447300034], + "beta": [-0.076102991931022809, -3.0910321898139173, -2.5252015607289384], + "se": [0.39901562395069623, 1.0841059140587559, 1.113853893987879], + "pval": [0.8487396037783228, 0.0043550848712978641, 0.023385027934051256], + "ci_lb": [-0.85815924414316513, -5.2158407367959541, -4.7083150769848761], + "ci_ub": [0.7059532602811196, -0.96622364283188089, -0.34208804447300034], + "QE": 5.3133148940953756, + "QEp": 0.070182418569350757, + "I2": 62.358715042043215, + "H2": 2.6566574470476878, + "tau2_ci_lb": 0, + "tau2_ci_ub": 17.40992233276512, "dof": null}, {"design": "extreme_k10", "model": "two", "method": "DL", "test": "z", - "tau2": 0.041031566660919984, - "beta": [0.26915780054154376, 0.67216962808157321, -0.13691153321000049], - "se": [0.10897817440249509, 0.075996321669380168, 0.18563292537830384], - "pval": [0.013517645607301123, 9.1720894442967697e-19, 0.46079459922276667], - "ci_lb": [0.055564503611728572, 0.52321957465206725, -0.50074538129628743], - "ci_ub": [0.48275109747135891, 0.82111968151107917, 0.22692231487628639], + "tau2": 0.041031566660919866, + "beta": [0.26915780054154353, 0.67216962808157354, -0.13691153321000071], + "se": [0.10897817440249498, 0.075996321669380112, 0.18563292537830364], + "pval": [0.013517645607301104, 9.1720894442958953e-19, 0.4607945992227655], + "ci_lb": [0.055564503611728572, 0.5232195746520677, -0.50074538129628721], + "ci_ub": [0.48275109747135847, 0.82111968151107939, 0.22692231487628578], + "QE": 14.400976720554908, + "QEp": 0.044492241224326719, + "I2": 51.392185850778375, + "H2": 2.0572823886507012, + "tau2_ci_lb": 0, + "tau2_ci_ub": 1.4208174995455576, "dof": null}, {"design": "moderate_k20", "model": "two", "method": "DL", "test": "z", "tau2": 0, - "beta": [0.73709810315645485, 0.64747703612718221, 0.19044587474139929], - "se": [0.14672220323157109, 0.16974243343585627, 0.34253541418439043], - "pval": [5.0667809224461877e-07, 0.00013647676387872824, 0.57821846967745549], - "ci_lb": [0.4495278690902092, 0.31478797994471652, -0.4809112004895163], - "ci_ub": [1.0246683372227006, 0.98016609230964791, 0.86180294997231488], + "beta": [0.73709810315645485, 0.64747703612718244, 0.19044587474139935], + "se": [0.14672220323157109, 0.16974243343585629, 0.34253541418439043], + "pval": [5.0667809224461877e-07, 0.000136476763878728, 0.57821846967745549], + "ci_lb": [0.4495278690902092, 0.31478797994471669, -0.48091120048951624], + "ci_ub": [1.0246683372227006, 0.98016609230964824, 0.86180294997231499], + "QE": 12.137819651655924, + "QEp": 0.79172076097379995, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.30347543125718318, "dof": null}, {"design": "equal_k5", "model": "intercept", "method": "HE", "test": "z", - "tau2": 1.0017690475532999, - "beta": [0.6045998164625348], - "se": [0.6310554502646879], - "pval": [0.33802384939427677], - "ci_lb": [-0.63224613830396059], - "ci_ub": [1.8414457712290302], + "tau2": 1.0017690475533003, + "beta": [0.60459981646253491], + "se": [0.63105545026468812], + "pval": [0.33802384939427688], + "ci_lb": [-0.63224613830396093], + "ci_ub": [1.8414457712290306], + "QE": 8.3983359150613381, + "QEp": 0.078029419298244626, + "I2": 50.433197317508395, + "H2": 2.0174793327010949, + "tau2_ci_lb": 0, + "tau2_ci_ub": 15.562063970721971, "dof": null}, {"design": "unequal_k5", "model": "intercept", "method": "HE", "test": "z", "tau2": 6.3777689219320006, @@ -213,6 +364,12 @@ "pval": [0.81353257507132448], "ci_lb": [-2.1135438871396253], "ci_ub": [2.6918471741750269], + "QE": 20.315605412654854, + "QEp": 0.00043261444283937314, + "I2": 98.920736940938127, + "H2": 92.655816541078309, + "tau2_ci_lb": 0.37437910706976263, + "tau2_ci_ub": 62.648818709153389, "dof": null}, {"design": "extreme_k10", "model": "intercept", "method": "HE", "test": "z", "tau2": 0.69581387232204428, @@ -221,6 +378,12 @@ "pval": [0.033189106515707202], "ci_lb": [0.054390893784053929], "ci_ub": [1.3098563395634844], + "QE": 528.36542813689618, + "QEp": 4.8272097789668306e-108, + "I2": 97.353747147956526, + "H2": 37.789283787744829, + "tau2_ci_lb": 0.27621241313967376, + "tau2_ci_ub": 4.0683787385044337, "dof": null}, {"design": "moderate_k20", "model": "intercept", "method": "HE", "test": "z", "tau2": 0, @@ -229,30 +392,54 @@ "pval": [1.4342046963366567e-08], "ci_lb": [0.51728641988318369], "ci_ub": [1.0639477495833147], + "QE": 27.443467066041269, + "QEp": 0.094740008781753912, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.94040725109002843, "dof": null}, {"design": "equal_k5", "model": "one", "method": "HE", "test": "z", "tau2": 0, - "beta": [1.2765478897207587, 1.5577764687433346], - "se": [0.53219419519755828, 0.67157839168830602], - "pval": [0.016455611193178728, 0.020363877879022316], - "ci_lb": [0.23346643435226522, 0.24150700823892146], - "ci_ub": [2.3196293450892522, 2.8740459292477478], + "beta": [1.2765478897207592, 1.5577764687433355], + "se": [0.5321941951975584, 0.67157839168830613], + "pval": [0.016455611193178711, 0.02036387787902226], + "ci_lb": [0.23346643435226544, 0.24150700823892213], + "ci_ub": [2.3196293450892531, 2.8740459292477487], + "QE": 3.0179075904568835, + "QEp": 0.38887241030467207, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 12.956611642049028, "dof": null}, {"design": "unequal_k5", "model": "one", "method": "HE", "test": "z", - "tau2": 1.9505050157064565, - "beta": [-0.023749990802700544, -2.898365534025575], - "se": [0.73429181259895149, 1.4473400305507784], - "pval": [0.97419765934907188, 0.045226000486082836], + "tau2": 1.9505050157064572, + "beta": [-0.023749990802700443, -2.898365534025575], + "se": [0.7342918125989516, 1.4473400305507784], + "pval": [0.97419765934907199, 0.045226000486082836], "ci_lb": [-1.4629354976392801, -5.7350998672882021], - "ci_ub": [1.415435516033879, -0.061631200762948257], + "ci_ub": [1.4154355160338794, -0.061631200762948257], + "QE": 19.451611261114039, + "QEp": 0.00022048004795655233, + "I2": 96.007372979243186, + "H2": 25.046166215907853, + "tau2_ci_lb": 0.19746423631486623, + "tau2_ci_ub": 45.327128288195816, "dof": null}, {"design": "extreme_k10", "model": "one", "method": "HE", "test": "z", "tau2": 0, - "beta": [0.36308889801604494, 0.65095292619066303], + "beta": [0.36308889801604494, 0.65095292619066292], "se": [0.028451161183932691, 0.028755150789987197], - "pval": [2.678321482283576e-37, 1.8406210952213409e-113], - "ci_lb": [0.30732564677719287, 0.59459386627226962], - "ci_ub": [0.418852149254897, 0.70731198610905643], + "pval": [2.678321482283576e-37, 1.8406210952214895e-113], + "ci_lb": [0.30732564677719287, 0.59459386627226951], + "ci_ub": [0.418852149254897, 0.70731198610905632], + "QE": 15.896110614196777, + "QEp": 0.043891457052016768, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.98404808836485924, "dof": null}, {"design": "moderate_k20", "model": "one", "method": "HE", "test": "z", "tau2": 0, @@ -261,54 +448,96 @@ "pval": [3.9548102645066542e-07, 0.00010770940386508056], "ci_lb": [0.43818989196649638, 0.32356831858529306], "ci_ub": [0.99029101918584939, 0.98674097650419279], + "QE": 12.446943265071944, + "QEp": 0.82332537450409671, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.23460881621137722, "dof": null}, {"design": "equal_k5", "model": "two", "method": "HE", "test": "z", - "tau2": 0.35326005474691957, - "beta": [1.2477742750156047, 1.4983564437097099, 0.27984719123805257], - "se": [0.62487062066186538, 0.80740986986537866, 0.96666056275058276], - "pval": [0.045841240815695711, 0.06348821525993617, 0.77219960440940372], - "ci_lb": [0.023050363521158523, -0.084137821988603978, -1.6147726970283105], - "ci_ub": [2.4724981865100508, 3.0808507094080237, 2.1744670795044154], + "tau2": 0.35326005474691868, + "beta": [1.2477742750156047, 1.4983564437097099, 0.27984719123805285], + "se": [0.62487062066186516, 0.80740986986537822, 0.96666056275058243], + "pval": [0.045841240815695655, 0.063488215259936004, 0.77219960440940338], + "ci_lb": [0.023050363521158967, -0.08413782198860309, -1.6147726970283094], + "ci_ub": [2.4724981865100504, 3.0808507094080229, 2.174467079504415], + "QE": 2.9338803207967543, + "QEp": 0.23063009774267032, + "I2": 25.371204136934111, + "H2": 1.3399653423791933, + "tau2_ci_lb": 0, + "tau2_ci_ub": 54.363997637568161, "dof": null}, {"design": "unequal_k5", "model": "two", "method": "HE", "test": "z", "tau2": 0, - "beta": [-0.31818517955900727, -1.4858777781006078, -1.4718397892444939], - "se": [0.10600770201892341, 0.48697806397185173, 0.39143711218949201], - "pval": [0.0026862755330678378, 0.0022791262368038611, 0.00016985047321758549], - "ci_lb": [-0.52595645759995113, -2.4403372447464795, -2.2390424313482629], - "ci_ub": [-0.11041390151806343, -0.53141831145473606, -0.70463714714072501], + "beta": [-0.31818517955900727, -1.4858777781006076, -1.4718397892444932], + "se": [0.10600770201892343, 0.48697806397185173, 0.39143711218949195], + "pval": [0.0026862755330678417, 0.0022791262368038654, 0.00016985047321758646], + "ci_lb": [-0.52595645759995113, -2.4403372447464795, -2.239042431348262], + "ci_ub": [-0.11041390151806341, -0.53141831145473584, -0.70463714714072445], + "QE": 5.3133148940953756, + "QEp": 0.070182418569350757, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 17.40992233276512, "dof": null}, {"design": "extreme_k10", "model": "two", "method": "HE", "test": "z", "tau2": 0, - "beta": [0.33951715132107646, 0.65156082084807287, -0.12127406276771753], - "se": [0.034367023554020541, 0.028759448117667584, 0.099180863165782243], - "pval": [5.1263369109800041e-23, 1.2302772325697296e-113, 0.22142161260838295], + "beta": [0.3395171513210764, 0.65156082084807287, -0.12127406276771752], + "se": [0.034367023554020541, 0.028759448117667584, 0.099180863165782257], + "pval": [5.1263369109800952e-23, 1.2302772325697296e-113, 0.22142161260838308], "ci_lb": [0.27215902289935645, 0.59519333832219612, -0.31566498252824593], - "ci_ub": [0.40687527974279647, 0.70792830337394963, 0.073116856992810883], + "ci_ub": [0.40687527974279636, 0.70792830337394963, 0.073116856992810925], + "QE": 14.400976720554908, + "QEp": 0.044492241224326719, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 1.4208174995455576, "dof": null}, {"design": "moderate_k20", "model": "two", "method": "HE", "test": "z", "tau2": 0, - "beta": [0.73709810315645485, 0.64747703612718221, 0.19044587474139929], - "se": [0.14672220323157109, 0.16974243343585627, 0.34253541418439043], - "pval": [5.0667809224461877e-07, 0.00013647676387872824, 0.57821846967745549], - "ci_lb": [0.4495278690902092, 0.31478797994471652, -0.4809112004895163], - "ci_ub": [1.0246683372227006, 0.98016609230964791, 0.86180294997231488], + "beta": [0.73709810315645485, 0.64747703612718244, 0.19044587474139935], + "se": [0.14672220323157109, 0.16974243343585629, 0.34253541418439043], + "pval": [5.0667809224461877e-07, 0.000136476763878728, 0.57821846967745549], + "ci_lb": [0.4495278690902092, 0.31478797994471669, -0.48091120048951624], + "ci_ub": [1.0246683372227006, 0.98016609230964824, 0.86180294997231499], + "QE": 12.137819651655924, + "QEp": 0.79172076097379995, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.30347543125718318, "dof": null}, {"design": "equal_k5", "model": "intercept", "method": "ML", "test": "z", - "tau2": 0.68027628202071799, - "beta": [0.60259585469760302], - "se": [0.5775530123469218], + "tau2": 0.68027628202071777, + "beta": [0.60259585469760291], + "se": [0.57755301234692169], "pval": [0.2967814767680888], - "ci_lb": [-0.52938724866498077], - "ci_ub": [1.7345789580601867], + "ci_lb": [-0.52938724866498066], + "ci_ub": [1.7345789580601865], + "QE": 8.3983359150613381, + "QEp": 0.078029419298244626, + "I2": 40.8614619775202, + "H2": 1.6909447433750882, + "tau2_ci_lb": 0, + "tau2_ci_ub": 15.562063970721971, "dof": null}, {"design": "unequal_k5", "model": "intercept", "method": "ML", "test": "z", - "tau2": 0.07110732122668309, - "beta": [-0.33948195894057742], - "se": [0.18698733173636548], - "pval": [0.069441803254777362], - "ci_lb": [-0.70597039470909717], - "ci_ub": [0.027006476827942327], + "tau2": 0.071107321226682979, + "beta": [-0.33948195894057737], + "se": [0.18698733173636534], + "pval": [0.069441803254777196], + "ci_lb": [-0.70597039470909684], + "ci_ub": [0.027006476827942105], + "QE": 20.315605412654854, + "QEp": 0.00043261444283937314, + "I2": 50.54140691154651, + "H2": 2.021893340579997, + "tau2_ci_lb": 0.37437910706976263, + "tau2_ci_ub": 62.648818709153389, "dof": null}, {"design": "extreme_k10", "model": "intercept", "method": "ML", "test": "z", "tau2": 0.71407535451805593, @@ -317,6 +546,12 @@ "pval": [0.035041769029996887], "ci_lb": [0.04791733008865473], "ci_ub": [1.3178158564995393], + "QE": 528.36542813689618, + "QEp": 4.8272097789668306e-108, + "I2": 97.419675162397368, + "H2": 38.754810457473226, + "tau2_ci_lb": 0.27621241313967376, + "tau2_ci_ub": 4.0683787385044337, "dof": null}, {"design": "moderate_k20", "model": "intercept", "method": "ML", "test": "z", "tau2": 0.19754570451620143, @@ -325,30 +560,54 @@ "pval": [3.5556649576652921e-05], "ci_lb": [0.41470167676600977], "ci_ub": [1.1622278629375096], + "QE": 27.443467066041269, + "QEp": 0.094740008781753912, + "I2": 32.264202007657786, + "H2": 1.4763242327388744, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.94040725109002843, "dof": null}, {"design": "equal_k5", "model": "one", "method": "ML", "test": "z", "tau2": 0, - "beta": [1.2765478897207587, 1.5577764687433346], - "se": [0.53219419519755828, 0.67157839168830602], - "pval": [0.016455611193178728, 0.020363877879022316], - "ci_lb": [0.23346643435226522, 0.24150700823892146], - "ci_ub": [2.3196293450892522, 2.8740459292477478], + "beta": [1.2765478897207592, 1.5577764687433355], + "se": [0.5321941951975584, 0.67157839168830613], + "pval": [0.016455611193178711, 0.02036387787902226], + "ci_lb": [0.23346643435226544, 0.24150700823892213], + "ci_ub": [2.3196293450892531, 2.8740459292477487], + "QE": 3.0179075904568835, + "QEp": 0.38887241030467207, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 12.956611642049028, "dof": null}, {"design": "unequal_k5", "model": "one", "method": "ML", "test": "z", - "tau2": 0.5468260898151458, - "beta": [-0.069360915104283674, -2.0828519418136895], - "se": [0.42430348919260485, 1.0043485303739079], - "pval": [0.870148347911886, 0.038094747743996349], - "ci_lb": [-0.90098047243646917, -4.051338889272281], - "ci_ub": [0.76225864222790185, -0.11436499435509773], + "tau2": 0.54682608981514602, + "beta": [-0.069360915104283785, -2.0828519418136899], + "se": [0.42430348919260508, 1.0043485303739081], + "pval": [0.87014834791188589, 0.038094747743996349], + "ci_lb": [-0.90098047243646973, -4.0513388892722819], + "ci_ub": [0.76225864222790218, -0.11436499435509773], + "QE": 19.451611261114039, + "QEp": 0.00022048004795655233, + "I2": 87.082385574134719, + "H2": 7.7413674617634705, + "tau2_ci_lb": 0.19746423631486623, + "tau2_ci_ub": 45.327128288195816, "dof": null}, {"design": "extreme_k10", "model": "one", "method": "ML", "test": "z", "tau2": 0, - "beta": [0.36308889801604494, 0.65095292619066303], + "beta": [0.36308889801604494, 0.65095292619066292], "se": [0.028451161183932691, 0.028755150789987197], - "pval": [2.678321482283576e-37, 1.8406210952213409e-113], - "ci_lb": [0.30732564677719287, 0.59459386627226962], - "ci_ub": [0.418852149254897, 0.70731198610905643], + "pval": [2.678321482283576e-37, 1.8406210952214895e-113], + "ci_lb": [0.30732564677719287, 0.59459386627226951], + "ci_ub": [0.418852149254897, 0.70731198610905632], + "QE": 15.896110614196777, + "QEp": 0.043891457052016768, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.98404808836485924, "dof": null}, {"design": "moderate_k20", "model": "one", "method": "ML", "test": "z", "tau2": 0, @@ -357,54 +616,96 @@ "pval": [3.9548102645066542e-07, 0.00010770940386508056], "ci_lb": [0.43818989196649638, 0.32356831858529306], "ci_ub": [0.99029101918584939, 0.98674097650419279], + "QE": 12.446943265071944, + "QEp": 0.82332537450409671, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.23460881621137722, "dof": null}, {"design": "equal_k5", "model": "two", "method": "ML", "test": "z", - "tau2": 1.2606432107381935e-07, - "beta": [1.2531305431068513, 1.5129861564728475, 0.23824829310485862], - "se": [0.53829064344970834, 0.68912469252440955, 0.82190129476131846], - "pval": [0.019913097523122543, 0.028126410219828524, 0.77191218885089763], - "ci_lb": [0.19810026873053155, 0.16232657826776631, -1.3726486434741643], - "ci_ub": [2.3081608174831709, 2.8636457346779287, 1.8491452296838815], + "tau2": 1.2606432066241397e-07, + "beta": [1.2531305431068516, 1.5129861564728473, 0.2382482931048589], + "se": [0.53829064344970834, 0.68912469252440933, 0.82190129476131846], + "pval": [0.019913097523122522, 0.028126410219828486, 0.77191218885089741], + "ci_lb": [0.19810026873053177, 0.16232657826776653, -1.372648643474164], + "ci_ub": [2.3081608174831714, 2.8636457346779283, 1.8491452296838817], + "QE": 2.9338803207967543, + "QEp": 0.23063009774267032, + "I2": 1.2131995723952348e-05, + "H2": 1.0000001213199721, + "tau2_ci_lb": 0, + "tau2_ci_ub": 54.363997637568161, "dof": null}, {"design": "unequal_k5", "model": "two", "method": "ML", "test": "z", "tau2": 0, - "beta": [-0.31818517955900727, -1.4858777781006078, -1.4718397892444939], - "se": [0.10600770201892341, 0.48697806397185173, 0.39143711218949201], - "pval": [0.0026862755330678378, 0.0022791262368038611, 0.00016985047321758549], - "ci_lb": [-0.52595645759995113, -2.4403372447464795, -2.2390424313482629], - "ci_ub": [-0.11041390151806343, -0.53141831145473606, -0.70463714714072501], + "beta": [-0.31818517955900727, -1.4858777781006076, -1.4718397892444932], + "se": [0.10600770201892343, 0.48697806397185173, 0.39143711218949195], + "pval": [0.0026862755330678417, 0.0022791262368038654, 0.00016985047321758646], + "ci_lb": [-0.52595645759995113, -2.4403372447464795, -2.239042431348262], + "ci_ub": [-0.11041390151806341, -0.53141831145473584, -0.70463714714072445], + "QE": 5.3133148940953756, + "QEp": 0.070182418569350757, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 17.40992233276512, "dof": null}, {"design": "extreme_k10", "model": "two", "method": "ML", "test": "z", "tau2": 0, - "beta": [0.33951715132107646, 0.65156082084807287, -0.12127406276771753], - "se": [0.034367023554020541, 0.028759448117667584, 0.099180863165782243], - "pval": [5.1263369109800041e-23, 1.2302772325697296e-113, 0.22142161260838295], + "beta": [0.3395171513210764, 0.65156082084807287, -0.12127406276771752], + "se": [0.034367023554020541, 0.028759448117667584, 0.099180863165782257], + "pval": [5.1263369109800952e-23, 1.2302772325697296e-113, 0.22142161260838308], "ci_lb": [0.27215902289935645, 0.59519333832219612, -0.31566498252824593], - "ci_ub": [0.40687527974279647, 0.70792830337394963, 0.073116856992810883], + "ci_ub": [0.40687527974279636, 0.70792830337394963, 0.073116856992810925], + "QE": 14.400976720554908, + "QEp": 0.044492241224326719, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 1.4208174995455576, "dof": null}, {"design": "moderate_k20", "model": "two", "method": "ML", "test": "z", "tau2": 0, - "beta": [0.73709810315645485, 0.64747703612718221, 0.19044587474139929], - "se": [0.14672220323157109, 0.16974243343585627, 0.34253541418439043], - "pval": [5.0667809224461877e-07, 0.00013647676387872824, 0.57821846967745549], - "ci_lb": [0.4495278690902092, 0.31478797994471652, -0.4809112004895163], - "ci_ub": [1.0246683372227006, 0.98016609230964791, 0.86180294997231488], + "beta": [0.73709810315645485, 0.64747703612718244, 0.19044587474139935], + "se": [0.14672220323157109, 0.16974243343585629, 0.34253541418439043], + "pval": [5.0667809224461877e-07, 0.000136476763878728, 0.57821846967745549], + "ci_lb": [0.4495278690902092, 0.31478797994471669, -0.48091120048951624], + "ci_ub": [1.0246683372227006, 0.98016609230964824, 0.86180294997231499], + "QE": 12.137819651655924, + "QEp": 0.79172076097379995, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.30347543125718318, "dof": null}, {"design": "equal_k5", "model": "intercept", "method": "REML", "test": "z", - "tau2": 1.0830683217408148, - "beta": [0.60497869770208212], - "se": [0.64386731554321897], + "tau2": 1.0830683217408152, + "beta": [0.60497869770208224], + "se": [0.64386731554321908], "pval": [0.34742200481618563], - "ci_lb": [-0.65697805158511335], - "ci_ub": [1.8669354469892776], + "ci_lb": [-0.65697805158511347], + "ci_ub": [1.8669354469892778], + "QE": 8.3983359150613381, + "QEp": 0.078029419298244626, + "I2": 52.382167455829745, + "H2": 2.1000535861694267, + "tau2_ci_lb": 0, + "tau2_ci_ub": 15.562063970721971, "dof": null}, {"design": "unequal_k5", "model": "intercept", "method": "REML", "test": "z", - "tau2": 3.0539045155524742, - "beta": [0.18480405951699175], - "se": [0.88814021281473976], + "tau2": 3.0539045155524738, + "beta": [0.18480405951699172], + "se": [0.88814021281473954], "pval": [0.83516664018309494], - "ci_lb": [-1.5559187708216369], - "ci_ub": [1.9255268898556204], + "ci_lb": [-1.5559187708216364], + "ci_ub": [1.9255268898556199], + "QE": 20.315605412654854, + "QEp": 0.00043261444283937314, + "I2": 97.772237701301052, + "H2": 44.888092440742795, + "tau2_ci_lb": 0.37437910706976263, + "tau2_ci_ub": 62.648818709153389, "dof": null}, {"design": "extreme_k10", "model": "intercept", "method": "REML", "test": "z", "tau2": 0.8224002783454577, @@ -413,38 +714,68 @@ "pval": [0.046360786532183673], "ci_lb": [0.011078554483622383], "ci_ub": [1.3627008496642645], + "QE": 528.36542813689618, + "QEp": 4.8272097789668306e-108, + "I2": 97.751909699504225, + "H2": 44.482198948123568, + "tau2_ci_lb": 0.27621241313967376, + "tau2_ci_ub": 4.0683787385044337, "dof": null}, {"design": "moderate_k20", "model": "intercept", "method": "REML", "test": "z", - "tau2": 0.23909863987739999, - "beta": [0.78602512965241333], - "se": [0.19840151801276226], + "tau2": 0.23909863987740002, + "beta": [0.78602512965241322], + "se": [0.19840151801276223], "pval": [7.4389988359311137e-05], - "ci_lb": [0.39716529986932453], - "ci_ub": [1.1748849594355022], + "ci_lb": [0.39716529986932447], + "ci_ub": [1.174884959435502], + "QE": 27.443467066041269, + "QEp": 0.094740008781753912, + "I2": 36.569035528086367, + "H2": 1.5765170974860179, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.94040725109002843, "dof": null}, {"design": "equal_k5", "model": "one", "method": "REML", "test": "z", - "tau2": 0.012139238147945345, - "beta": [1.2763383912384598, 1.5574875090169169], - "se": [0.53543567441262074, 0.67584219308871096], - "pval": [0.017137796330448211, 0.021193831168416759], - "ci_lb": [0.22690375335180879, 0.23286115133047858], - "ci_ub": [2.3257730291251111, 2.8821138667033552], + "tau2": 0.012139238147945644, + "beta": [1.2763383912384594, 1.5574875090169171], + "se": [0.53543567441262063, 0.67584219308871119], + "pval": [0.017137796330448225, 0.021193831168416773], + "ci_lb": [0.22690375335180857, 0.23286115133047836], + "ci_ub": [2.3257730291251102, 2.8821138667033557], + "QE": 3.0179075904568835, + "QEp": 0.38887241030467207, + "I2": 1.2046421535754204, + "H2": 1.0121933072548612, + "tau2_ci_lb": 0, + "tau2_ci_ub": 12.956611642049028, "dof": null}, {"design": "unequal_k5", "model": "one", "method": "REML", "test": "z", - "tau2": 1.6012284825460623, - "beta": [-0.01956806718199039, -2.7975303090464272], - "se": [0.67305605377360478, 1.3615844091907945], - "pval": [0.97680600397343731, 0.039916308053007131], - "ci_lb": [-1.3387336921549096, -5.4661867129716324], - "ci_ub": [1.2995975577909287, -0.12887390512122243], + "tau2": 1.6012284825460621, + "beta": [-0.019568067181989932, -2.7975303090464259], + "se": [0.67305605377360489, 1.361584409190794], + "pval": [0.97680600397343786, 0.039916308053007131], + "ci_lb": [-1.3387336921549093, -5.4661867129716297], + "ci_ub": [1.2995975577909293, -0.12887390512122199], + "QE": 19.451611261114039, + "QEp": 0.00022048004795655233, + "I2": 95.178451341004532, + "H2": 20.74022416292167, + "tau2_ci_lb": 0.19746423631486623, + "tau2_ci_ub": 45.327128288195816, "dof": null}, {"design": "extreme_k10", "model": "one", "method": "REML", "test": "z", - "tau2": 0.036951880858045835, - "beta": [0.26541839994868621, 0.67408851856411911], - "se": [0.10458480147412126, 0.072987984380740473], - "pval": [0.011154229474546256, 2.5681976501640085e-20], - "ci_lb": [0.060435955729137014, 0.53103469787369584], - "ci_ub": [0.47040084416823541, 0.81714233925454238], + "tau2": 0.036951880858045794, + "beta": [0.26541839994868621, 0.67408851856411889], + "se": [0.10458480147412121, 0.072987984380740445], + "pval": [0.011154229474546215, 2.5681976501640085e-20], + "ci_lb": [0.060435955729137125, 0.53103469787369562], + "ci_ub": [0.4704008441682353, 0.81714233925454216], + "QE": 15.896110614196777, + "QEp": 0.043891457052016768, + "I2": 55.117312082709873, + "H2": 2.2280305534347704, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.98404808836485924, "dof": null}, {"design": "moderate_k20", "model": "one", "method": "REML", "test": "z", "tau2": 0, @@ -453,46 +784,82 @@ "pval": [3.9548102645066542e-07, 0.00010770940386508056], "ci_lb": [0.43818989196649638, 0.32356831858529306], "ci_ub": [0.99029101918584939, 0.98674097650419279], + "QE": 12.446943265071944, + "QEp": 0.82332537450409671, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.23460881621137722, "dof": null}, {"design": "equal_k5", "model": "two", "method": "REML", "test": "z", - "tau2": 0.5280809767005048, - "beta": [1.2460703551935211, 1.4936911119117537, 0.29310505071275272], - "se": [0.66352418592804685, 0.85988959667497467, 1.0306035555149087], - "pval": [0.06038695174250746, 0.082374263743252268, 0.77610281844366702], - "ci_lb": [-0.054413152096709272, -0.19166152825186944, -1.7268408004353941], - "ci_ub": [2.5465538624837514, 3.1790437520753771, 2.3130509018608998], + "tau2": 0.52808097670050402, + "beta": [1.2460703551935206, 1.4936911119117531, 0.29310505071275289], + "se": [0.66352418592804663, 0.85988959667497433, 1.0306035555149085], + "pval": [0.060386951742507487, 0.082374263743252268, 0.77610281844366691], + "ci_lb": [-0.054413152096709272, -0.19166152825186944, -1.7268408004353937], + "ci_ub": [2.5465538624837505, 3.1790437520753754, 2.3130509018608993], + "QE": 2.9338803207967543, + "QEp": 0.23063009774267032, + "I2": 33.696103726458212, + "H2": 1.5082069926545849, + "tau2_ci_lb": 0, + "tau2_ci_ub": 54.363997637568161, "dof": null}, {"design": "unequal_k5", "model": "two", "method": "REML", "test": "z", - "tau2": 0.47326742184966425, - "beta": [-0.075886999301457397, -3.092299300315581, -2.5259267977879882], - "se": [0.39953440935260487, 1.0847585343866863, 1.1149099958718196], - "pval": [0.84935725843681675, 0.004362587003623416, 0.023476615539460104], - "ci_lb": [-0.85896005221704574, -5.2183869596359393, -4.7111102357004544], - "ci_ub": [0.707186053614131, -0.96621164099522217, -0.34074335987552162], + "tau2": 0.47326742184966131, + "beta": [-0.075886999301457744, -3.092299300315577, -2.5259267977879856], + "se": [0.39953440935260387, 1.0847585343866848, 1.1149099958718174], + "pval": [0.84935725843681564, 0.0043625870036234117, 0.023476615539459979], + "ci_lb": [-0.85896005221704408, -5.2183869596359322, -4.7111102357004473], + "ci_ub": [0.70718605361412867, -0.96621164099522128, -0.3407433598755234], + "QE": 5.3133148940953756, + "QEp": 0.070182418569350757, + "I2": 62.43225961847147, + "H2": 2.6618582588259274, + "tau2_ci_lb": 0, + "tau2_ci_ub": 17.40992233276512, "dof": null}, {"design": "extreme_k10", "model": "two", "method": "REML", "test": "z", "tau2": 0.054688742532978674, - "beta": [0.26582139792705878, 0.67314112868512155, -0.14647088836558972], - "se": [0.12151930104983935, 0.084990666437201268, 0.20414935748367657], - "pval": [0.028707287284557952, 2.3717255434188226e-15, 0.47308459759637245], - "ci_lb": [0.027647944442893285, 0.50656248344614996, -0.54659627650058829], - "ci_ub": [0.50399485141122424, 0.83971977392409314, 0.2536544997694089], + "beta": [0.26582139792705889, 0.67314112868512144, -0.14647088836558994], + "se": [0.12151930104983936, 0.084990666437201254, 0.20414935748367652], + "pval": [0.028707287284557889, 2.3717255434188226e-15, 0.47308459759637161], + "ci_lb": [0.027647944442893368, 0.50656248344614985, -0.5465962765005884], + "ci_ub": [0.50399485141122446, 0.83971977392409303, 0.25365449976940857], + "QE": 14.400976720554908, + "QEp": 0.044492241224326719, + "I2": 58.492345102622807, + "H2": 2.4091941654434166, + "tau2_ci_lb": 0, + "tau2_ci_ub": 1.4208174995455576, "dof": null}, {"design": "moderate_k20", "model": "two", "method": "REML", "test": "z", "tau2": 0, - "beta": [0.73709810315645485, 0.64747703612718221, 0.19044587474139929], - "se": [0.14672220323157109, 0.16974243343585627, 0.34253541418439043], - "pval": [5.0667809224461877e-07, 0.00013647676387872824, 0.57821846967745549], - "ci_lb": [0.4495278690902092, 0.31478797994471652, -0.4809112004895163], - "ci_ub": [1.0246683372227006, 0.98016609230964791, 0.86180294997231488], + "beta": [0.73709810315645485, 0.64747703612718244, 0.19044587474139935], + "se": [0.14672220323157109, 0.16974243343585629, 0.34253541418439043], + "pval": [5.0667809224461877e-07, 0.000136476763878728, 0.57821846967745549], + "ci_lb": [0.4495278690902092, 0.31478797994471669, -0.48091120048951624], + "ci_ub": [1.0246683372227006, 0.98016609230964824, 0.86180294997231499], + "QE": 12.137819651655924, + "QEp": 0.79172076097379995, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.30347543125718318, "dof": null}, {"design": "equal_k5", "model": "intercept", "method": "FE", "test": "knha", "tau2": 0, - "beta": [0.5912488674785934], - "se": [0.64140687997215995], - "pval": [0.40879940095473755], - "ci_lb": [-1.1895821248602994], + "beta": [0.59124886747859351], + "se": [0.64140687997215984], + "pval": [0.40879940095473738], + "ci_lb": [-1.1895821248602987], "ci_ub": [2.3720798598174859], + "QE": 8.3983359150613381, + "QEp": 0.078029419298244626, + "I2": 52.371516923650162, + "H2": 2.0995839787653345, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 4}, {"design": "unequal_k5", "model": "intercept", "method": "FE", "test": "knha", "tau2": 0, @@ -501,14 +868,26 @@ "pval": [0.25383725912891136], "ci_lb": [-0.95374813477543374], "ci_ub": [0.33547534557860015], + "QE": 20.315605412654854, + "QEp": 0.00043261444283937314, + "I2": 80.310702444002246, + "H2": 5.0789013531637135, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 4}, {"design": "extreme_k10", "model": "intercept", "method": "FE", "test": "knha", "tau2": 0, - "beta": [0.2484353958726899], + "beta": [0.24843539587268987], "se": [0.21451289263744763], - "pval": [0.27661622739882197], - "ci_lb": [-0.23682648071967458], + "pval": [0.2766162273988223], + "ci_lb": [-0.2368264807196746], "ci_ub": [0.73369727246505434], + "QE": 528.36542813689618, + "QEp": 4.8272097789668306e-108, + "I2": 98.296633443309204, + "H2": 58.707269792988463, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 9}, {"design": "moderate_k20", "model": "intercept", "method": "FE", "test": "knha", "tau2": 0, @@ -517,30 +896,54 @@ "pval": [0.00014999107399305813], "ci_lb": [0.43981903170095121], "ci_ub": [1.1414151377655473], + "QE": 27.443467066041269, + "QEp": 0.094740008781753912, + "I2": 30.766765167544271, + "H2": 1.4443930034758563, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 19}, {"design": "equal_k5", "model": "one", "method": "FE", "test": "knha", "tau2": 0, - "beta": [1.2765478897207587, 1.5577764687433346], + "beta": [1.2765478897207592, 1.5577764687433355], "se": [0.53378021784724261, 0.67357980122991301], - "pval": [0.096608698810429064, 0.10377595124373379], - "ci_lb": [-0.42217899240073198, -0.58585508099453709], - "ci_ub": [2.9752747718422494, 3.7014080184812066], + "pval": [0.096608698810429022, 0.10377595124373369], + "ci_lb": [-0.42217899240073153, -0.5858550809945362], + "ci_ub": [2.9752747718422499, 3.7014080184812075], + "QE": 3.0179075904568835, + "QEp": 0.38887241030467207, + "I2": 0.5933776936547106, + "H2": 1.0059691968189612, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 3}, {"design": "unequal_k5", "model": "one", "method": "FE", "test": "knha", "tau2": 0, - "beta": [-0.28720869740317378, -0.35624980573686255], - "se": [0.26911578561900218, 0.97592489878373545], + "beta": [-0.28720869740317373, -0.35624980573686266], + "se": [0.26911578561900207, 0.97592489878373545], "pval": [0.36412597358958804, 0.73929985863352288], - "ci_lb": [-1.1436552350398901, -3.4620783941055393], - "ci_ub": [0.5692378402335424, 2.7495787826318141], + "ci_lb": [-1.1436552350398896, -3.4620783941055393], + "ci_ub": [0.56923784023354207, 2.7495787826318141], + "QE": 19.451611261114039, + "QEp": 0.00022048004795655233, + "I2": 84.577113125855348, + "H2": 6.4838704203713462, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 3}, {"design": "extreme_k10", "model": "one", "method": "FE", "test": "knha", "tau2": 0, - "beta": [0.36308889801604494, 0.65095292619066303], - "se": [0.040105177300723688, 0.040533685542252965], + "beta": [0.36308889801604494, 0.65095292619066292], + "se": [0.040105177300723674, 0.040533685542252958], "pval": [1.7741894997235766e-05, 2.2676624900582826e-07], - "ci_lb": [0.27060619331747987, 0.55748207971516239], - "ci_ub": [0.45557160271461, 0.74442377266616366], + "ci_lb": [0.27060619331747987, 0.55748207971516228], + "ci_ub": [0.45557160271461, 0.74442377266616355], + "QE": 15.896110614196777, + "QEp": 0.043891457052016768, + "I2": 49.673223883739084, + "H2": 1.9870138267745971, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 8}, {"design": "moderate_k20", "model": "one", "method": "FE", "test": "knha", "tau2": 0, @@ -549,62 +952,110 @@ "pval": [9.2206602208637616e-06, 0.00019612224367436334], "ci_lb": [0.4681778679719355, 0.35958926614846282], "ci_ub": [0.96030304318041027, 0.95072002894102314], + "QE": 12.446943265071944, + "QEp": 0.82332537450409671, + "I2": 0, + "H2": 0.69149684805955247, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 18}, {"design": "equal_k5", "model": "two", "method": "FE", "test": "knha", "tau2": 0, - "beta": [1.2531305457285846, 1.5129861636211883, 0.23824827277034544], - "se": [0.65196308069804165, 0.83464920105024409, 0.99546463080482717], - "pval": [0.19453170844887055, 0.21155911129607638, 0.83313811041724883], - "ci_lb": [-1.5520401831327142, -2.0782194996608534, -4.0448903383310864], - "ci_ub": [4.0583012745898834, 5.1041918269032305, 4.5213868838717772], + "beta": [1.2531305457285848, 1.5129861636211892, 0.2382482727703448], + "se": [0.65196308069804176, 0.83464920105024443, 0.99546463080482717], + "pval": [0.19453170844887055, 0.21155911129607632, 0.83313811041724928], + "ci_lb": [-1.5520401831327144, -2.0782194996608538, -4.0448903383310872], + "ci_ub": [4.0583012745898843, 5.1041918269032323, 4.5213868838717763], + "QE": 2.9338803207967543, + "QEp": 0.23063009774267032, + "I2": 31.830893515899799, + "H2": 1.4669401603983772, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 2}, {"design": "unequal_k5", "model": "two", "method": "FE", "test": "knha", "tau2": 0, - "beta": [-0.31818517955900727, -1.4858777781006078, -1.4718397892444939], - "se": [0.17278466684668742, 0.79373801094189367, 0.63801336820807031], - "pval": [0.20689031180802245, 0.20209441300762737, 0.14744825091247518], - "ci_lb": [-1.0616175980257585, -4.9010567975856567, -4.2169897495815967], - "ci_ub": [0.42524723890774396, 1.9293012413844406, 1.2733101710926094], + "beta": [-0.31818517955900727, -1.4858777781006076, -1.4718397892444932], + "se": [0.17278466684668745, 0.79373801094189367, 0.63801336820807031], + "pval": [0.20689031180802256, 0.20209441300762737, 0.14744825091247529], + "ci_lb": [-1.0616175980257587, -4.9010567975856558, -4.2169897495815967], + "ci_ub": [0.42524723890774407, 1.9293012413844408, 1.27331017109261], + "QE": 5.3133148940953756, + "QEp": 0.070182418569350757, + "I2": 62.358715042043215, + "H2": 2.6566574470476878, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 2}, {"design": "extreme_k10", "model": "two", "method": "FE", "test": "knha", "tau2": 0, - "beta": [0.33951715132107646, 0.65156082084807287, -0.12127406276771753], + "beta": [0.3395171513210764, 0.65156082084807287, -0.12127406276771752], "se": [0.049293411370510194, 0.041250337103667649, 0.14225739044373062], - "pval": [0.00023390627660365608, 9.8783751861973846e-07, 0.42214449510262975], - "ci_lb": [0.22295675535062853, 0.5540192733463627, -0.45765933817926657], - "ci_ub": [0.45607754729152439, 0.74910236834978305, 0.21511121264383154], + "pval": [0.00023390627660365652, 9.8783751861973846e-07, 0.42214449510262986], + "ci_lb": [0.22295675535062848, 0.5540192733463627, -0.45765933817926657], + "ci_ub": [0.45607754729152433, 0.74910236834978305, 0.21511121264383154], + "QE": 14.400976720554908, + "QEp": 0.044492241224326719, + "I2": 51.39218585077839, + "H2": 2.0572823886507012, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 7}, {"design": "moderate_k20", "model": "two", "method": "FE", "test": "knha", "tau2": 0, - "beta": [0.73709810315645485, 0.64747703612718221, 0.19044587474139929], - "se": [0.12397717021831628, 0.14342877969283499, 0.28943520758828034], - "pval": [1.5970552811389704e-05, 0.00030625937068610938, 0.51935533515130516], - "ci_lb": [0.47552913813415842, 0.34486876242161613, -0.42020903500177431], - "ci_ub": [0.99866706817875128, 0.9500853098327483, 0.80110078448457289], + "beta": [0.73709810315645485, 0.64747703612718244, 0.19044587474139935], + "se": [0.12397717021831625, 0.14342877969283496, 0.28943520758828029], + "pval": [1.5970552811389646e-05, 0.00030625937068610765, 0.51935533515130494], + "ci_lb": [0.47552913813415854, 0.3448687624216164, -0.42020903500177414], + "ci_ub": [0.99866706817875117, 0.95008530983274841, 0.80110078448457278], + "QE": 12.137819651655924, + "QEp": 0.79172076097379995, + "I2": 0, + "H2": 0.71398939127387795, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 17}, {"design": "equal_k5", "model": "intercept", "method": "DL", "test": "knha", "tau2": 1.08260596526162, - "beta": [0.60497665381666166], + "beta": [0.60497665381666177], "se": [0.63782138309867997], - "pval": [0.3965820609370212], - "ci_lb": [-1.1658994032781556], + "pval": [0.39658206093702131], + "ci_lb": [-1.1658994032781553], "ci_ub": [2.3758527109114791], + "QE": 8.3983359150613381, + "QEp": 0.078029419298244626, + "I2": 52.371516923650162, + "H2": 2.0995839787653345, + "tau2_ci_lb": 0, + "tau2_ci_ub": 15.562063970721971, "dof": 4}, {"design": "unequal_k5", "model": "intercept", "method": "DL", "test": "knha", "tau2": 0.28382585271252081, - "beta": [-0.25834451420148663], + "beta": [-0.25834451420148669], "se": [0.54387142220951235], "pval": [0.65955362976715803], "ci_lb": [-1.7683736622520501], "ci_ub": [1.2516846338490768], + "QE": 20.315605412654854, + "QEp": 0.00043261444283937314, + "I2": 80.310702444002246, + "H2": 5.0789013531637144, + "tau2_ci_lb": 0.37437910706976263, + "tau2_ci_ub": 62.648818709153389, "dof": 4}, {"design": "extreme_k10", "model": "intercept", "method": "DL", "test": "knha", - "tau2": 1.0914460604195861, + "tau2": 1.0914460604195859, "beta": [0.69453702188798982], "se": [0.33516575613229493], "pval": [0.068118658473445767], "ci_lb": [-0.06366059407135749], "ci_ub": [1.4527346378473371], + "QE": 528.36542813689618, + "QEp": 4.8272097789668306e-108, + "I2": 98.296633443309219, + "H2": 58.707269792988463, + "tau2_ci_lb": 0.27621241313967376, + "tau2_ci_ub": 4.0683787385044337, "dof": 9}, {"design": "moderate_k20", "model": "intercept", "method": "DL", "test": "knha", "tau2": 0.18430288219628541, @@ -613,30 +1064,54 @@ "pval": [0.00039616420479101001], "ci_lb": [0.40408943147421506], "ci_ub": [1.1743715703912398], + "QE": 27.443467066041269, + "QEp": 0.094740008781753912, + "I2": 30.766765167544271, + "H2": 1.4443930034758563, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.94040725109002843, "dof": 19}, {"design": "equal_k5", "model": "one", "method": "DL", "test": "knha", - "tau2": 0.0059427274506224873, - "beta": [1.2764449434579284, 1.5576340773383561], - "se": [0.53377360418207731, 0.67365651045687158], + "tau2": 0.0059427274506217518, + "beta": [1.2764449434579284, 1.5576340773383557], + "se": [0.53377360418207731, 0.67365651045687136], "pval": [0.096622904670168874, 0.10382102491970405], - "ci_lb": [-0.42226089102929265, -0.58624159539543119], - "ci_ub": [2.9751507779451494, 3.7015097500721437], + "ci_lb": [-0.42226089102929265, -0.58624159539543075], + "ci_ub": [2.9751507779451494, 3.7015097500721419], + "QE": 3.0179075904568835, + "QEp": 0.38887241030467207, + "I2": 0.5933776936547106, + "H2": 1.0059691968189612, + "tau2_ci_lb": 0, + "tau2_ci_ub": 12.956611642049028, "dof": 3}, {"design": "unequal_k5", "model": "one", "method": "DL", "test": "knha", - "tau2": 0.44482420458951977, - "beta": [-0.090448214875049893, -1.9230388371068361], - "se": [0.58109556806197715, 1.4162626557132092], - "pval": [0.88619158014892241, 0.2676225332293149], - "ci_lb": [-1.9397536584706272, -6.4302186930926331], - "ci_ub": [1.7588572287205275, 2.5841410188789604], + "tau2": 0.44482420458951955, + "beta": [-0.090448214875049893, -1.923038837106835], + "se": [0.58109556806197704, 1.416262655713209], + "pval": [0.88619158014892241, 0.26762253322931512], + "ci_lb": [-1.9397536584706268, -6.4302186930926304], + "ci_ub": [1.7588572287205271, 2.5841410188789609], + "QE": 19.451611261114039, + "QEp": 0.00022048004795655233, + "I2": 84.577113125855362, + "H2": 6.4838704203713462, + "tau2_ci_lb": 0.19746423631486623, + "tau2_ci_ub": 45.327128288195816, "dof": 3}, {"design": "extreme_k10", "model": "one", "method": "DL", "test": "knha", - "tau2": 0.029699600901791464, - "beta": [0.27000256176805176, 0.67393162666171846], - "se": [0.095911365017903413, 0.066817450696780301], - "pval": [0.022663382618669067, 7.9612483104985655e-06], - "ci_lb": [0.048830557423690274, 0.5198503090511426], - "ci_ub": [0.49117456611241328, 0.82801294427229433], + "tau2": 0.029699600901791457, + "beta": [0.27000256176805176, 0.67393162666171824], + "se": [0.095911365017903386, 0.066817450696780287], + "pval": [0.022663382618669025, 7.9612483104985655e-06], + "ci_lb": [0.048830557423690329, 0.51985030905114238], + "ci_ub": [0.49117456611241317, 0.8280129442722941], + "QE": 15.896110614196777, + "QEp": 0.043891457052016768, + "I2": 49.673223883739091, + "H2": 1.9870138267745971, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.98404808836485924, "dof": 8}, {"design": "moderate_k20", "model": "one", "method": "DL", "test": "knha", "tau2": 0, @@ -645,46 +1120,82 @@ "pval": [9.2206602208637616e-06, 0.00019612224367436334], "ci_lb": [0.4681778679719355, 0.35958926614846282], "ci_ub": [0.96030304318041027, 0.95072002894102314], + "QE": 12.446943265071944, + "QEp": 0.82332537450409671, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.23460881621137722, "dof": 18}, {"design": "equal_k5", "model": "two", "method": "DL", "test": "knha", - "tau2": 0.48520036034108888, - "beta": [1.2464498791089595, 1.4947307524436979, 0.29015096278611763], - "se": [0.64524394538678886, 0.83564960287163714, 1.0013158238997548], - "pval": [0.19311661916925363, 0.2155606307552608, 0.79927217746155022], - "ci_lb": [-1.5298107438638215, -2.1007792924660071, -4.0181633002574966], - "ci_ub": [4.0227105020817406, 5.0902407973534025, 4.5984652258297309], + "tau2": 0.48520036034108743, + "beta": [1.2464498791089589, 1.4947307524436986, 0.29015096278611674], + "se": [0.64524394538678864, 0.83564960287163725, 1.0013158238997548], + "pval": [0.19311661916925368, 0.21556063075526063, 0.79927217746155077], + "ci_lb": [-1.5298107438638213, -2.1007792924660067, -4.0181633002574966], + "ci_ub": [4.0227105020817389, 5.0902407973534043, 4.5984652258297309], + "QE": 2.9338803207967543, + "QEp": 0.23063009774267032, + "I2": 31.830893515899795, + "H2": 1.4669401603983772, + "tau2_ci_lb": 0, + "tau2_ci_ub": 54.363997637568161, "dof": 2}, {"design": "unequal_k5", "model": "two", "method": "DL", "test": "knha", "tau2": 0.4717863239469125, - "beta": [-0.076102991931023045, -3.0910321898139177, -2.525201560728938], - "se": [0.32834240697558154, 0.89209024377075097, 0.91656929357807471], - "pval": [0.83826501943505127, 0.074148199027364595, 0.1103613546939354], - "ci_lb": [-1.4888463455970182, -6.9293867123570037, -6.4688809337471787], - "ci_ub": [1.3366403617349722, 0.74732233272916826, 1.4184778122893027], + "beta": [-0.076102991931022809, -3.0910321898139173, -2.5252015607289384], + "se": [0.32834240697558159, 0.89209024377075097, 0.91656929357807493], + "pval": [0.83826501943505183, 0.074148199027364609, 0.1103613546939354], + "ci_lb": [-1.4888463455970182, -6.9293867123570028, -6.4688809337471795], + "ci_ub": [1.3366403617349727, 0.7473223327291687, 1.4184778122893031], + "QE": 5.3133148940953756, + "QEp": 0.070182418569350757, + "I2": 62.358715042043215, + "H2": 2.6566574470476878, + "tau2_ci_lb": 0, + "tau2_ci_ub": 17.40992233276512, "dof": 2}, {"design": "extreme_k10", "model": "two", "method": "DL", "test": "knha", - "tau2": 0.041031566660919984, - "beta": [0.26915780054154376, 0.67216962808157321, -0.13691153321000049], - "se": [0.10495311877063331, 0.073189434655413874, 0.17877666396770886], - "pval": [0.037303050571847352, 3.7389626388064028e-05, 0.46881200134135692], - "ci_lb": [0.020983110616206363, 0.49910411593501619, -0.55965116844689877], - "ci_ub": [0.51733249046688112, 0.84523514022813018, 0.28582810202689779], + "tau2": 0.041031566660919866, + "beta": [0.26915780054154353, 0.67216962808157354, -0.13691153321000071], + "se": [0.10495311877063326, 0.07318943465541386, 0.17877666396770878], + "pval": [0.037303050571847352, 3.7389626388063899e-05, 0.46881200134135581], + "ci_lb": [0.020983110616206224, 0.49910411593501658, -0.55965116844689877], + "ci_ub": [0.51733249046688079, 0.84523514022813051, 0.28582810202689735], + "QE": 14.400976720554908, + "QEp": 0.044492241224326719, + "I2": 51.392185850778375, + "H2": 2.0572823886507012, + "tau2_ci_lb": 0, + "tau2_ci_ub": 1.4208174995455576, "dof": 7}, {"design": "moderate_k20", "model": "two", "method": "DL", "test": "knha", "tau2": 0, - "beta": [0.73709810315645485, 0.64747703612718221, 0.19044587474139929], - "se": [0.12397717021831628, 0.14342877969283499, 0.28943520758828034], - "pval": [1.5970552811389704e-05, 0.00030625937068610938, 0.51935533515130516], - "ci_lb": [0.47552913813415842, 0.34486876242161613, -0.42020903500177431], - "ci_ub": [0.99866706817875128, 0.9500853098327483, 0.80110078448457289], + "beta": [0.73709810315645485, 0.64747703612718244, 0.19044587474139935], + "se": [0.12397717021831625, 0.14342877969283496, 0.28943520758828029], + "pval": [1.5970552811389646e-05, 0.00030625937068610765, 0.51935533515130494], + "ci_lb": [0.47552913813415854, 0.3448687624216164, -0.42020903500177414], + "ci_ub": [0.99866706817875117, 0.95008530983274841, 0.80110078448457278], + "QE": 12.137819651655924, + "QEp": 0.79172076097379995, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.30347543125718318, "dof": 17}, {"design": "equal_k5", "model": "intercept", "method": "HE", "test": "knha", - "tau2": 1.0017690475532999, - "beta": [0.6045998164625348], - "se": [0.63799517738619749], + "tau2": 1.0017690475533003, + "beta": [0.60459981646253491], + "se": [0.6379951773861976], "pval": [0.39696568122989262], - "ci_lb": [-1.1667587709311718], + "ci_lb": [-1.166758770931172], "ci_ub": [2.3759584038562416], + "QE": 8.3983359150613381, + "QEp": 0.078029419298244626, + "I2": 50.433197317508395, + "H2": 2.0174793327010949, + "tau2_ci_lb": 0, + "tau2_ci_ub": 15.562063970721971, "dof": 4}, {"design": "unequal_k5", "model": "intercept", "method": "HE", "test": "knha", "tau2": 6.3777689219320006, @@ -693,6 +1204,12 @@ "pval": [0.81212676600160261], "ci_lb": [-2.8734139410862261], "ci_ub": [3.4517172281216277], + "QE": 20.315605412654854, + "QEp": 0.00043261444283937314, + "I2": 98.920736940938127, + "H2": 92.655816541078309, + "tau2_ci_lb": 0.37437910706976263, + "tau2_ci_ub": 62.648818709153389, "dof": 4}, {"design": "extreme_k10", "model": "intercept", "method": "HE", "test": "knha", "tau2": 0.69581387232204428, @@ -701,6 +1218,12 @@ "pval": [0.064712291421993279], "ci_lb": [-0.051326129247421637], "ci_ub": [1.4155733625949598], + "QE": 528.36542813689618, + "QEp": 4.8272097789668306e-108, + "I2": 97.353747147956526, + "H2": 37.789283787744829, + "tau2_ci_lb": 0.27621241313967376, + "tau2_ci_ub": 4.0683787385044337, "dof": 9}, {"design": "moderate_k20", "model": "intercept", "method": "HE", "test": "knha", "tau2": 0, @@ -709,30 +1232,54 @@ "pval": [0.00014999107399305813], "ci_lb": [0.43981903170095121], "ci_ub": [1.1414151377655473], + "QE": 27.443467066041269, + "QEp": 0.094740008781753912, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.94040725109002843, "dof": 19}, {"design": "equal_k5", "model": "one", "method": "HE", "test": "knha", "tau2": 0, - "beta": [1.2765478897207587, 1.5577764687433346], + "beta": [1.2765478897207592, 1.5577764687433355], "se": [0.53378021784724261, 0.67357980122991301], - "pval": [0.096608698810429064, 0.10377595124373379], - "ci_lb": [-0.42217899240073198, -0.58585508099453709], - "ci_ub": [2.9752747718422494, 3.7014080184812066], + "pval": [0.096608698810429022, 0.10377595124373369], + "ci_lb": [-0.42217899240073153, -0.5858550809945362], + "ci_ub": [2.9752747718422499, 3.7014080184812075], + "QE": 3.0179075904568835, + "QEp": 0.38887241030467207, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 12.956611642049028, "dof": 3}, {"design": "unequal_k5", "model": "one", "method": "HE", "test": "knha", - "tau2": 1.9505050157064565, - "beta": [-0.023749990802700544, -2.898365534025575], - "se": [0.71511250076526311, 1.4095362782835927], - "pval": [0.97559200316962991, 0.13196919006268668], - "ci_lb": [-2.2995571267253054, -7.384139055012545], - "ci_ub": [2.2520571451199047, 1.5874079869613951], + "tau2": 1.9505050157064572, + "beta": [-0.023749990802700443, -2.898365534025575], + "se": [0.715112500765263, 1.4095362782835925], + "pval": [0.97559200316963002, 0.13196919006268668], + "ci_lb": [-2.299557126725305, -7.3841390550125441], + "ci_ub": [2.2520571451199043, 1.5874079869613942], + "QE": 19.451611261114039, + "QEp": 0.00022048004795655233, + "I2": 96.007372979243186, + "H2": 25.046166215907853, + "tau2_ci_lb": 0.19746423631486623, + "tau2_ci_ub": 45.327128288195816, "dof": 3}, {"design": "extreme_k10", "model": "one", "method": "HE", "test": "knha", "tau2": 0, - "beta": [0.36308889801604494, 0.65095292619066303], - "se": [0.040105177300723688, 0.040533685542252965], + "beta": [0.36308889801604494, 0.65095292619066292], + "se": [0.040105177300723674, 0.040533685542252958], "pval": [1.7741894997235766e-05, 2.2676624900582826e-07], - "ci_lb": [0.27060619331747987, 0.55748207971516239], - "ci_ub": [0.45557160271461, 0.74442377266616366], + "ci_lb": [0.27060619331747987, 0.55748207971516228], + "ci_ub": [0.45557160271461, 0.74442377266616355], + "QE": 15.896110614196777, + "QEp": 0.043891457052016768, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.98404808836485924, "dof": 8}, {"design": "moderate_k20", "model": "one", "method": "HE", "test": "knha", "tau2": 0, @@ -741,54 +1288,96 @@ "pval": [9.2206602208637616e-06, 0.00019612224367436334], "ci_lb": [0.4681778679719355, 0.35958926614846282], "ci_ub": [0.96030304318041027, 0.95072002894102314], + "QE": 12.446943265071944, + "QEp": 0.82332537450409671, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.23460881621137722, "dof": 18}, {"design": "equal_k5", "model": "two", "method": "HE", "test": "knha", - "tau2": 0.35326005474691957, - "beta": [1.2477742750156047, 1.4983564437097099, 0.27984719123805257], - "se": [0.64659879577191193, 0.8354853505455756, 1.0002735528399129], - "pval": [0.19340833321942136, 0.21477116753295861, 0.80593330384572393], - "ci_lb": [-1.5343157986651281, -2.0964468804808982, -4.0239825413847932], - "ci_ub": [4.0298643486963375, 5.0931597679003184, 4.5836769238608976], + "tau2": 0.35326005474691868, + "beta": [1.2477742750156047, 1.4983564437097099, 0.27984719123805285], + "se": [0.64659879577191182, 0.83548535054557538, 1.0002735528399127], + "pval": [0.1934083332194213, 0.2147711675329585, 0.8059333038457237], + "ci_lb": [-1.5343157986651277, -2.0964468804808973, -4.0239825413847914], + "ci_ub": [4.0298643486963375, 5.0931597679003175, 4.5836769238608976], + "QE": 2.9338803207967543, + "QEp": 0.23063009774267032, + "I2": 25.371204136934111, + "H2": 1.3399653423791933, + "tau2_ci_lb": 0, + "tau2_ci_ub": 54.363997637568161, "dof": 2}, {"design": "unequal_k5", "model": "two", "method": "HE", "test": "knha", "tau2": 0, - "beta": [-0.31818517955900727, -1.4858777781006078, -1.4718397892444939], - "se": [0.17278466684668742, 0.79373801094189367, 0.63801336820807031], - "pval": [0.20689031180802245, 0.20209441300762737, 0.14744825091247518], - "ci_lb": [-1.0616175980257585, -4.9010567975856567, -4.2169897495815967], - "ci_ub": [0.42524723890774396, 1.9293012413844406, 1.2733101710926094], + "beta": [-0.31818517955900727, -1.4858777781006076, -1.4718397892444932], + "se": [0.17278466684668745, 0.79373801094189367, 0.63801336820807031], + "pval": [0.20689031180802256, 0.20209441300762737, 0.14744825091247529], + "ci_lb": [-1.0616175980257587, -4.9010567975856558, -4.2169897495815967], + "ci_ub": [0.42524723890774407, 1.9293012413844408, 1.27331017109261], + "QE": 5.3133148940953756, + "QEp": 0.070182418569350757, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 17.40992233276512, "dof": 2}, {"design": "extreme_k10", "model": "two", "method": "HE", "test": "knha", "tau2": 0, - "beta": [0.33951715132107646, 0.65156082084807287, -0.12127406276771753], + "beta": [0.3395171513210764, 0.65156082084807287, -0.12127406276771752], "se": [0.049293411370510194, 0.041250337103667649, 0.14225739044373062], - "pval": [0.00023390627660365608, 9.8783751861973846e-07, 0.42214449510262975], - "ci_lb": [0.22295675535062853, 0.5540192733463627, -0.45765933817926657], - "ci_ub": [0.45607754729152439, 0.74910236834978305, 0.21511121264383154], + "pval": [0.00023390627660365652, 9.8783751861973846e-07, 0.42214449510262986], + "ci_lb": [0.22295675535062848, 0.5540192733463627, -0.45765933817926657], + "ci_ub": [0.45607754729152433, 0.74910236834978305, 0.21511121264383154], + "QE": 14.400976720554908, + "QEp": 0.044492241224326719, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 1.4208174995455576, "dof": 7}, {"design": "moderate_k20", "model": "two", "method": "HE", "test": "knha", "tau2": 0, - "beta": [0.73709810315645485, 0.64747703612718221, 0.19044587474139929], - "se": [0.12397717021831628, 0.14342877969283499, 0.28943520758828034], - "pval": [1.5970552811389704e-05, 0.00030625937068610938, 0.51935533515130516], - "ci_lb": [0.47552913813415842, 0.34486876242161613, -0.42020903500177431], - "ci_ub": [0.99866706817875128, 0.9500853098327483, 0.80110078448457289], + "beta": [0.73709810315645485, 0.64747703612718244, 0.19044587474139935], + "se": [0.12397717021831625, 0.14342877969283496, 0.28943520758828029], + "pval": [1.5970552811389646e-05, 0.00030625937068610765, 0.51935533515130494], + "ci_lb": [0.47552913813415854, 0.3448687624216164, -0.42020903500177414], + "ci_ub": [0.99866706817875117, 0.95008530983274841, 0.80110078448457278], + "QE": 12.137819651655924, + "QEp": 0.79172076097379995, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.30347543125718318, "dof": 17}, {"design": "equal_k5", "model": "intercept", "method": "ML", "test": "knha", - "tau2": 0.68027628202071799, - "beta": [0.60259585469760302], - "se": [0.63880735789098919], + "tau2": 0.68027628202071777, + "beta": [0.60259585469760291], + "se": [0.63880735789098908], "pval": [0.39893240287347198], "ci_lb": [-1.1710177072831693], - "ci_ub": [2.3762094166783756], + "ci_ub": [2.3762094166783752], + "QE": 8.3983359150613381, + "QEp": 0.078029419298244626, + "I2": 40.8614619775202, + "H2": 1.6909447433750882, + "tau2_ci_lb": 0, + "tau2_ci_ub": 15.562063970721971, "dof": 4}, {"design": "unequal_k5", "model": "intercept", "method": "ML", "test": "knha", - "tau2": 0.07110732122668309, - "beta": [-0.33948195894057742], - "se": [0.35332439644604918], - "pval": [0.39105206712603408], - "ci_lb": [-1.3204677500001756], - "ci_ub": [0.64150383211902073], + "tau2": 0.071107321226682979, + "beta": [-0.33948195894057737], + "se": [0.35332439644604896], + "pval": [0.39105206712603396], + "ci_lb": [-1.3204677500001749], + "ci_ub": [0.64150383211902029], + "QE": 20.315605412654854, + "QEp": 0.00043261444283937314, + "I2": 50.54140691154651, + "H2": 2.021893340579997, + "tau2_ci_lb": 0.37437910706976263, + "tau2_ci_ub": 62.648818709153389, "dof": 4}, {"design": "extreme_k10", "model": "intercept", "method": "ML", "test": "knha", "tau2": 0.71407535451805593, @@ -797,6 +1386,12 @@ "pval": [0.064841751772212602], "ci_lb": [-0.051812017767040808], "ci_ub": [1.417545204355235], + "QE": 528.36542813689618, + "QEp": 4.8272097789668306e-108, + "I2": 97.419675162397368, + "H2": 38.754810457473226, + "tau2_ci_lb": 0.27621241313967376, + "tau2_ci_ub": 4.0683787385044337, "dof": 9}, {"design": "moderate_k20", "model": "intercept", "method": "ML", "test": "knha", "tau2": 0.19754570451620143, @@ -805,30 +1400,54 @@ "pval": [0.00041283004861265735], "ci_lb": [0.40206838620227658], "ci_ub": [1.1748611535012428], + "QE": 27.443467066041269, + "QEp": 0.094740008781753912, + "I2": 32.264202007657786, + "H2": 1.4763242327388744, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.94040725109002843, "dof": 19}, {"design": "equal_k5", "model": "one", "method": "ML", "test": "knha", "tau2": 0, - "beta": [1.2765478897207587, 1.5577764687433346], + "beta": [1.2765478897207592, 1.5577764687433355], "se": [0.53378021784724261, 0.67357980122991301], - "pval": [0.096608698810429064, 0.10377595124373379], - "ci_lb": [-0.42217899240073198, -0.58585508099453709], - "ci_ub": [2.9752747718422494, 3.7014080184812066], + "pval": [0.096608698810429022, 0.10377595124373369], + "ci_lb": [-0.42217899240073153, -0.5858550809945362], + "ci_ub": [2.9752747718422499, 3.7014080184812075], + "QE": 3.0179075904568835, + "QEp": 0.38887241030467207, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 12.956611642049028, "dof": 3}, {"design": "unequal_k5", "model": "one", "method": "ML", "test": "knha", - "tau2": 0.5468260898151458, - "beta": [-0.069360915104283674, -2.0828519418136895], - "se": [0.60171902845827929, 1.4243003824457388], - "pval": [0.91551270675726582, 0.23981353377540848], - "ci_lb": [-1.9842994140402377, -6.6156114315423054], - "ci_ub": [1.8455775838316701, 2.4499075479149259], + "tau2": 0.54682608981514602, + "beta": [-0.069360915104283785, -2.0828519418136899], + "se": [0.60171902845827963, 1.424300382445739], + "pval": [0.9155127067572657, 0.23981353377540837], + "ci_lb": [-1.9842994140402388, -6.6156114315423054], + "ci_ub": [1.8455775838316713, 2.4499075479149255], + "QE": 19.451611261114039, + "QEp": 0.00022048004795655233, + "I2": 87.082385574134719, + "H2": 7.7413674617634705, + "tau2_ci_lb": 0.19746423631486623, + "tau2_ci_ub": 45.327128288195816, "dof": 3}, {"design": "extreme_k10", "model": "one", "method": "ML", "test": "knha", "tau2": 0, - "beta": [0.36308889801604494, 0.65095292619066303], - "se": [0.040105177300723688, 0.040533685542252965], + "beta": [0.36308889801604494, 0.65095292619066292], + "se": [0.040105177300723674, 0.040533685542252958], "pval": [1.7741894997235766e-05, 2.2676624900582826e-07], - "ci_lb": [0.27060619331747987, 0.55748207971516239], - "ci_ub": [0.45557160271461, 0.74442377266616366], + "ci_lb": [0.27060619331747987, 0.55748207971516228], + "ci_ub": [0.45557160271461, 0.74442377266616355], + "QE": 15.896110614196777, + "QEp": 0.043891457052016768, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.98404808836485924, "dof": 8}, {"design": "moderate_k20", "model": "one", "method": "ML", "test": "knha", "tau2": 0, @@ -837,54 +1456,96 @@ "pval": [9.2206602208637616e-06, 0.00019612224367436334], "ci_lb": [0.4681778679719355, 0.35958926614846282], "ci_ub": [0.96030304318041027, 0.95072002894102314], + "QE": 12.446943265071944, + "QEp": 0.82332537450409671, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.23460881621137722, "dof": 18}, {"design": "equal_k5", "model": "two", "method": "ML", "test": "knha", - "tau2": 1.2606432107381935e-07, - "beta": [1.2531305431068513, 1.5129861564728475, 0.23824829310485862], - "se": [0.65196307811546828, 0.83464920152483679, 0.99546463339138747], - "pval": [0.19453170792011315, 0.21155911287513976, 0.83313809699359309], - "ci_lb": [-1.5520401746425312, -2.0782195088512019, -4.0448903291256446], - "ci_ub": [4.0583012608562337, 5.1041918217968965, 4.521386915335361], + "tau2": 1.2606432066241397e-07, + "beta": [1.2531305431068516, 1.5129861564728473, 0.2382482931048589], + "se": [0.6519630781154685, 0.83464920152483668, 0.99546463339138769], + "pval": [0.19453170792011318, 0.21155911287513976, 0.83313809699359287], + "ci_lb": [-1.5520401746425319, -2.0782195088512014, -4.0448903291256446], + "ci_ub": [4.0583012608562345, 5.1041918217968956, 4.5213869153353627], + "QE": 2.9338803207967543, + "QEp": 0.23063009774267032, + "I2": 1.2131995723952348e-05, + "H2": 1.0000001213199721, + "tau2_ci_lb": 0, + "tau2_ci_ub": 54.363997637568161, "dof": 2}, {"design": "unequal_k5", "model": "two", "method": "ML", "test": "knha", "tau2": 0, - "beta": [-0.31818517955900727, -1.4858777781006078, -1.4718397892444939], - "se": [0.17278466684668742, 0.79373801094189367, 0.63801336820807031], - "pval": [0.20689031180802245, 0.20209441300762737, 0.14744825091247518], - "ci_lb": [-1.0616175980257585, -4.9010567975856567, -4.2169897495815967], - "ci_ub": [0.42524723890774396, 1.9293012413844406, 1.2733101710926094], + "beta": [-0.31818517955900727, -1.4858777781006076, -1.4718397892444932], + "se": [0.17278466684668745, 0.79373801094189367, 0.63801336820807031], + "pval": [0.20689031180802256, 0.20209441300762737, 0.14744825091247529], + "ci_lb": [-1.0616175980257587, -4.9010567975856558, -4.2169897495815967], + "ci_ub": [0.42524723890774407, 1.9293012413844408, 1.27331017109261], + "QE": 5.3133148940953756, + "QEp": 0.070182418569350757, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 17.40992233276512, "dof": 2}, {"design": "extreme_k10", "model": "two", "method": "ML", "test": "knha", "tau2": 0, - "beta": [0.33951715132107646, 0.65156082084807287, -0.12127406276771753], + "beta": [0.3395171513210764, 0.65156082084807287, -0.12127406276771752], "se": [0.049293411370510194, 0.041250337103667649, 0.14225739044373062], - "pval": [0.00023390627660365608, 9.8783751861973846e-07, 0.42214449510262975], - "ci_lb": [0.22295675535062853, 0.5540192733463627, -0.45765933817926657], - "ci_ub": [0.45607754729152439, 0.74910236834978305, 0.21511121264383154], + "pval": [0.00023390627660365652, 9.8783751861973846e-07, 0.42214449510262986], + "ci_lb": [0.22295675535062848, 0.5540192733463627, -0.45765933817926657], + "ci_ub": [0.45607754729152433, 0.74910236834978305, 0.21511121264383154], + "QE": 14.400976720554908, + "QEp": 0.044492241224326719, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 1.4208174995455576, "dof": 7}, {"design": "moderate_k20", "model": "two", "method": "ML", "test": "knha", "tau2": 0, - "beta": [0.73709810315645485, 0.64747703612718221, 0.19044587474139929], - "se": [0.12397717021831628, 0.14342877969283499, 0.28943520758828034], - "pval": [1.5970552811389704e-05, 0.00030625937068610938, 0.51935533515130516], - "ci_lb": [0.47552913813415842, 0.34486876242161613, -0.42020903500177431], - "ci_ub": [0.99866706817875128, 0.9500853098327483, 0.80110078448457289], + "beta": [0.73709810315645485, 0.64747703612718244, 0.19044587474139935], + "se": [0.12397717021831625, 0.14342877969283496, 0.28943520758828029], + "pval": [1.5970552811389646e-05, 0.00030625937068610765, 0.51935533515130494], + "ci_lb": [0.47552913813415854, 0.3448687624216164, -0.42020903500177414], + "ci_ub": [0.99866706817875117, 0.95008530983274841, 0.80110078448457278], + "QE": 12.137819651655924, + "QEp": 0.79172076097379995, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.30347543125718318, "dof": 17}, {"design": "equal_k5", "model": "intercept", "method": "REML", "test": "knha", - "tau2": 1.0830683217408148, - "beta": [0.60497869770208212], + "tau2": 1.0830683217408152, + "beta": [0.60497869770208224], "se": [0.63782041947034707], "pval": [0.39657996644396559], - "ci_lb": [-1.1658946839315669], - "ci_ub": [2.3758520793357309], + "ci_lb": [-1.1658946839315667], + "ci_ub": [2.3758520793357314], + "QE": 8.3983359150613381, + "QEp": 0.078029419298244626, + "I2": 52.382167455829745, + "H2": 2.1000535861694267, + "tau2_ci_lb": 0, + "tau2_ci_ub": 15.562063970721971, "dof": 4}, {"design": "unequal_k5", "model": "intercept", "method": "REML", "test": "knha", - "tau2": 3.0539045155524742, - "beta": [0.18480405951699175], + "tau2": 3.0539045155524738, + "beta": [0.18480405951699172], "se": [1.0399695871305989], "pval": [0.86759352145897251], - "ci_lb": [-2.7026144102263308], - "ci_ub": [3.0722225292603147], + "ci_lb": [-2.7026144102263312], + "ci_ub": [3.0722225292603142], + "QE": 20.315605412654854, + "QEp": 0.00043261444283937314, + "I2": 97.772237701301052, + "H2": 44.888092440742795, + "tau2_ci_lb": 0.37437910706976263, + "tau2_ci_ub": 62.648818709153389, "dof": 4}, {"design": "extreme_k10", "model": "intercept", "method": "REML", "test": "knha", "tau2": 0.8224002783454577, @@ -893,38 +1554,68 @@ "pval": [0.065676909884426357], "ci_lb": [-0.054899656296211186], "ci_ub": [1.4286790604440982], + "QE": 528.36542813689618, + "QEp": 4.8272097789668306e-108, + "I2": 97.751909699504225, + "H2": 44.482198948123568, + "tau2_ci_lb": 0.27621241313967376, + "tau2_ci_ub": 4.0683787385044337, "dof": 9}, {"design": "moderate_k20", "model": "intercept", "method": "REML", "test": "knha", - "tau2": 0.23909863987739999, - "beta": [0.78602512965241333], + "tau2": 0.23909863987740002, + "beta": [0.78602512965241322], "se": [0.1862742805111853], - "pval": [0.00046397052810461762], - "ci_lb": [0.39614857982490148], - "ci_ub": [1.1759016794799253], + "pval": [0.00046397052810461973], + "ci_lb": [0.39614857982490137], + "ci_ub": [1.1759016794799251], + "QE": 27.443467066041269, + "QEp": 0.094740008781753912, + "I2": 36.569035528086367, + "H2": 1.5765170974860179, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.94040725109002843, "dof": 19}, {"design": "equal_k5", "model": "one", "method": "REML", "test": "knha", - "tau2": 0.012139238147945345, - "beta": [1.2763383912384598, 1.5574875090169169], - "se": [0.53376681304335971, 0.67373570862818322], + "tau2": 0.012139238147945644, + "beta": [1.2763383912384594, 1.5574875090169171], + "se": [0.53376681304335949, 0.67373570862818311], "pval": [0.096637632308782154, 0.10386751267684148], - "ci_lb": [-0.42234583081444055, -0.58664020764454627], - "ci_ub": [2.97502261329136, 3.7016152256783803], + "ci_lb": [-0.42234583081444033, -0.58664020764454561], + "ci_ub": [2.9750226132913591, 3.7016152256783799], + "QE": 3.0179075904568835, + "QEp": 0.38887241030467207, + "I2": 1.2046421535754204, + "H2": 1.0121933072548612, + "tau2_ci_lb": 0, + "tau2_ci_ub": 12.956611642049028, "dof": 3}, {"design": "unequal_k5", "model": "one", "method": "REML", "test": "knha", - "tau2": 1.6012284825460623, - "beta": [-0.01956806718199039, -2.7975303090464272], - "se": [0.69894371863702431, 1.4139548479837643], - "pval": [0.97942311732082699, 0.14226284037670175], - "ci_lb": [-2.2439189221596445, -7.2973656908503468], - "ci_ub": [2.2047827877956641, 1.702305072757492], + "tau2": 1.6012284825460621, + "beta": [-0.019568067181989932, -2.7975303090464259], + "se": [0.69894371863702442, 1.4139548479837638], + "pval": [0.97942311732082754, 0.14226284037670175], + "ci_lb": [-2.2439189221596445, -7.2973656908503433], + "ci_ub": [2.2047827877956649, 1.7023050727574915], + "QE": 19.451611261114039, + "QEp": 0.00022048004795655233, + "I2": 95.178451341004532, + "H2": 20.74022416292167, + "tau2_ci_lb": 0.19746423631486623, + "tau2_ci_ub": 45.327128288195816, "dof": 3}, {"design": "extreme_k10", "model": "one", "method": "REML", "test": "knha", - "tau2": 0.036951880858045835, - "beta": [0.26541839994868621, 0.67408851856411911], - "se": [0.099940539477159474, 0.069746831581132304], - "pval": [0.028995401822003115, 1.0942399425632038e-05], - "ci_lb": [0.034955102639821239, 0.51325203652063944], - "ci_ub": [0.49588169725755116, 0.83492500060759878], + "tau2": 0.036951880858045794, + "beta": [0.26541839994868621, 0.67408851856411889], + "se": [0.099940539477159418, 0.069746831581132276], + "pval": [0.028995401822003056, 1.0942399425632038e-05], + "ci_lb": [0.034955102639821378, 0.51325203652063933], + "ci_ub": [0.49588169725755105, 0.83492500060759844], + "QE": 15.896110614196777, + "QEp": 0.043891457052016768, + "I2": 55.117312082709873, + "H2": 2.2280305534347704, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.98404808836485924, "dof": 8}, {"design": "moderate_k20", "model": "one", "method": "REML", "test": "knha", "tau2": 0, @@ -933,46 +1624,82 @@ "pval": [9.2206602208637616e-06, 0.00019612224367436334], "ci_lb": [0.4681778679719355, 0.35958926614846282], "ci_ub": [0.96030304318041027, 0.95072002894102314], + "QE": 12.446943265071944, + "QEp": 0.82332537450409671, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.23460881621137722, "dof": 18}, {"design": "equal_k5", "model": "two", "method": "REML", "test": "knha", - "tau2": 0.5280809767005048, - "beta": [1.2460703551935211, 1.4936911119117537, 0.29310505071275272], - "se": [0.64485350395294627, 0.83569345502150838, 1.0016037516863359], - "pval": [0.19303195146818933, 0.21578647929090319, 0.79736795088589474], - "ci_lb": [-1.5285103338781296, -2.1020076135702999, -4.0164480656077641], - "ci_ub": [4.0206510442651719, 5.0893898373938073, 4.6026581670332698], + "tau2": 0.52808097670050402, + "beta": [1.2460703551935206, 1.4936911119117531, 0.29310505071275289], + "se": [0.64485350395294616, 0.83569345502150805, 1.0016037516863359], + "pval": [0.19303195146818944, 0.21578647929090319, 0.79736795088589463], + "ci_lb": [-1.5285103338781296, -2.102007613570299, -4.0164480656077641], + "ci_ub": [4.0206510442651711, 5.0893898373938056, 4.6026581670332698], + "QE": 2.9338803207967543, + "QEp": 0.23063009774267032, + "I2": 33.696103726458212, + "H2": 1.5082069926545849, + "tau2_ci_lb": 0, + "tau2_ci_ub": 54.363997637568161, "dof": 2}, {"design": "unequal_k5", "model": "two", "method": "REML", "test": "knha", - "tau2": 0.47326742184966425, - "beta": [-0.075886999301457397, -3.092299300315581, -2.5259267977879882], - "se": [0.32838947754286407, 0.89159601783641562, 0.91637842068452247], - "pval": [0.8387346092175143, 0.074020843641829287, 0.1102694703699738], - "ci_lb": [-1.4888328812722618, -6.9285273402931864, -6.4687849110297506], - "ci_ub": [1.3370588826693468, 0.743928739662024, 1.4169313154537746], + "tau2": 0.47326742184966131, + "beta": [-0.075886999301457744, -3.092299300315577, -2.5259267977879856], + "se": [0.32838947754286396, 0.89159601783641629, 0.91637842068452258], + "pval": [0.83873460921751342, 0.074020843641829537, 0.11026947036997402], + "ci_lb": [-1.4888328812722615, -6.9285273402931846, -6.4687849110297488], + "ci_ub": [1.3370588826693461, 0.7439287396620311, 1.4169313154537777], + "QE": 5.3133148940953756, + "QEp": 0.070182418569350757, + "I2": 62.43225961847147, + "H2": 2.6618582588259274, + "tau2_ci_lb": 0, + "tau2_ci_ub": 17.40992233276512, "dof": 2}, {"design": "extreme_k10", "model": "two", "method": "REML", "test": "knha", "tau2": 0.054688742532978674, - "beta": [0.26582139792705878, 0.67314112868512155, -0.14647088836558972], - "se": [0.11075752164324341, 0.077463871961633901, 0.1860698398082371], - "pval": [0.047464046369096465, 5.3559514750015534e-05, 0.45699519904780517], - "ci_lb": [0.0039214762031327122, 0.48996817842236373, -0.58645614406613167], - "ci_ub": [0.5277213196509849, 0.85631407894787936, 0.29351436733495229], + "beta": [0.26582139792705889, 0.67314112868512144, -0.14647088836558994], + "se": [0.11075752164324341, 0.077463871961633901, 0.18606983980823702], + "pval": [0.047464046369096402, 5.3559514750015534e-05, 0.45699519904780439], + "ci_lb": [0.0039214762031328232, 0.48996817842236362, -0.58645614406613167], + "ci_ub": [0.5277213196509849, 0.85631407894787925, 0.29351436733495184], + "QE": 14.400976720554908, + "QEp": 0.044492241224326719, + "I2": 58.492345102622807, + "H2": 2.4091941654434166, + "tau2_ci_lb": 0, + "tau2_ci_ub": 1.4208174995455576, "dof": 7}, {"design": "moderate_k20", "model": "two", "method": "REML", "test": "knha", "tau2": 0, - "beta": [0.73709810315645485, 0.64747703612718221, 0.19044587474139929], - "se": [0.12397717021831628, 0.14342877969283499, 0.28943520758828034], - "pval": [1.5970552811389704e-05, 0.00030625937068610938, 0.51935533515130516], - "ci_lb": [0.47552913813415842, 0.34486876242161613, -0.42020903500177431], - "ci_ub": [0.99866706817875128, 0.9500853098327483, 0.80110078448457289], + "beta": [0.73709810315645485, 0.64747703612718244, 0.19044587474139935], + "se": [0.12397717021831625, 0.14342877969283496, 0.28943520758828029], + "pval": [1.5970552811389646e-05, 0.00030625937068610765, 0.51935533515130494], + "ci_lb": [0.47552913813415854, 0.3448687624216164, -0.42020903500177414], + "ci_ub": [0.99866706817875117, 0.95008530983274841, 0.80110078448457278], + "QE": 12.137819651655924, + "QEp": 0.79172076097379995, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.30347543125718318, "dof": 17}, {"design": "equal_k5", "model": "intercept", "method": "FE", "test": "adhoc", "tau2": 0, - "beta": [0.5912488674785934], - "se": [0.64140687997215995], - "pval": [0.40879940095473755], - "ci_lb": [-1.1895821248602994], + "beta": [0.59124886747859351], + "se": [0.64140687997215984], + "pval": [0.40879940095473738], + "ci_lb": [-1.1895821248602987], "ci_ub": [2.3720798598174859], + "QE": 8.3983359150613381, + "QEp": 0.078029419298244626, + "I2": 52.371516923650162, + "H2": 2.0995839787653345, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 4}, {"design": "unequal_k5", "model": "intercept", "method": "FE", "test": "adhoc", "tau2": 0, @@ -981,14 +1708,26 @@ "pval": [0.25383725912891136], "ci_lb": [-0.95374813477543374], "ci_ub": [0.33547534557860015], + "QE": 20.315605412654854, + "QEp": 0.00043261444283937314, + "I2": 80.310702444002246, + "H2": 5.0789013531637135, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 4}, {"design": "extreme_k10", "model": "intercept", "method": "FE", "test": "adhoc", "tau2": 0, - "beta": [0.2484353958726899], + "beta": [0.24843539587268987], "se": [0.21451289263744763], - "pval": [0.27661622739882197], - "ci_lb": [-0.23682648071967458], + "pval": [0.2766162273988223], + "ci_lb": [-0.2368264807196746], "ci_ub": [0.73369727246505434], + "QE": 528.36542813689618, + "QEp": 4.8272097789668306e-108, + "I2": 98.296633443309204, + "H2": 58.707269792988463, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 9}, {"design": "moderate_k20", "model": "intercept", "method": "FE", "test": "adhoc", "tau2": 0, @@ -997,30 +1736,54 @@ "pval": [0.00014999107399305813], "ci_lb": [0.43981903170095121], "ci_ub": [1.1414151377655473], + "QE": 27.443467066041269, + "QEp": 0.094740008781753912, + "I2": 30.766765167544271, + "H2": 1.4443930034758563, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 19}, {"design": "equal_k5", "model": "one", "method": "FE", "test": "adhoc", "tau2": 0, - "beta": [1.2765478897207587, 1.5577764687433346], + "beta": [1.2765478897207592, 1.5577764687433355], "se": [0.53378021784724261, 0.67357980122991301], - "pval": [0.096608698810429064, 0.10377595124373379], - "ci_lb": [-0.42217899240073198, -0.58585508099453709], - "ci_ub": [2.9752747718422494, 3.7014080184812066], + "pval": [0.096608698810429022, 0.10377595124373369], + "ci_lb": [-0.42217899240073153, -0.5858550809945362], + "ci_ub": [2.9752747718422499, 3.7014080184812075], + "QE": 3.0179075904568835, + "QEp": 0.38887241030467207, + "I2": 0.5933776936547106, + "H2": 1.0059691968189612, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 3}, {"design": "unequal_k5", "model": "one", "method": "FE", "test": "adhoc", "tau2": 0, - "beta": [-0.28720869740317378, -0.35624980573686255], - "se": [0.26911578561900218, 0.97592489878373545], + "beta": [-0.28720869740317373, -0.35624980573686266], + "se": [0.26911578561900207, 0.97592489878373545], "pval": [0.36412597358958804, 0.73929985863352288], - "ci_lb": [-1.1436552350398901, -3.4620783941055393], - "ci_ub": [0.5692378402335424, 2.7495787826318141], + "ci_lb": [-1.1436552350398896, -3.4620783941055393], + "ci_ub": [0.56923784023354207, 2.7495787826318141], + "QE": 19.451611261114039, + "QEp": 0.00022048004795655233, + "I2": 84.577113125855348, + "H2": 6.4838704203713462, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 3}, {"design": "extreme_k10", "model": "one", "method": "FE", "test": "adhoc", "tau2": 0, - "beta": [0.36308889801604494, 0.65095292619066303], - "se": [0.040105177300723688, 0.040533685542252965], + "beta": [0.36308889801604494, 0.65095292619066292], + "se": [0.040105177300723674, 0.040533685542252958], "pval": [1.7741894997235766e-05, 2.2676624900582826e-07], - "ci_lb": [0.27060619331747987, 0.55748207971516239], - "ci_ub": [0.45557160271461, 0.74442377266616366], + "ci_lb": [0.27060619331747987, 0.55748207971516228], + "ci_ub": [0.45557160271461, 0.74442377266616355], + "QE": 15.896110614196777, + "QEp": 0.043891457052016768, + "I2": 49.673223883739084, + "H2": 1.9870138267745971, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 8}, {"design": "moderate_k20", "model": "one", "method": "FE", "test": "adhoc", "tau2": 0, @@ -1029,62 +1792,110 @@ "pval": [7.9616439404665205e-05, 0.0011156542841317522], "ci_lb": [0.4183366951585395, 0.29972106190311493], "ci_ub": [1.0101442159938063, 1.0105882331863709], + "QE": 12.446943265071944, + "QEp": 0.82332537450409671, + "I2": 0, + "H2": 0.69149684805955247, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 18}, {"design": "equal_k5", "model": "two", "method": "FE", "test": "adhoc", "tau2": 0, - "beta": [1.2531305457285846, 1.5129861636211883, 0.23824827277034544], - "se": [0.65196308069804165, 0.83464920105024409, 0.99546463080482717], - "pval": [0.19453170844887055, 0.21155911129607638, 0.83313811041724883], - "ci_lb": [-1.5520401831327142, -2.0782194996608534, -4.0448903383310864], - "ci_ub": [4.0583012745898834, 5.1041918269032305, 4.5213868838717772], + "beta": [1.2531305457285848, 1.5129861636211892, 0.2382482727703448], + "se": [0.65196308069804176, 0.83464920105024443, 0.99546463080482717], + "pval": [0.19453170844887055, 0.21155911129607632, 0.83313811041724928], + "ci_lb": [-1.5520401831327144, -2.0782194996608538, -4.0448903383310872], + "ci_ub": [4.0583012745898843, 5.1041918269032323, 4.5213868838717763], + "QE": 2.9338803207967543, + "QEp": 0.23063009774267032, + "I2": 31.830893515899799, + "H2": 1.4669401603983772, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 2}, {"design": "unequal_k5", "model": "two", "method": "FE", "test": "adhoc", "tau2": 0, - "beta": [-0.31818517955900727, -1.4858777781006078, -1.4718397892444939], - "se": [0.17278466684668742, 0.79373801094189367, 0.63801336820807031], - "pval": [0.20689031180802245, 0.20209441300762737, 0.14744825091247518], - "ci_lb": [-1.0616175980257585, -4.9010567975856567, -4.2169897495815967], - "ci_ub": [0.42524723890774396, 1.9293012413844406, 1.2733101710926094], + "beta": [-0.31818517955900727, -1.4858777781006076, -1.4718397892444932], + "se": [0.17278466684668745, 0.79373801094189367, 0.63801336820807031], + "pval": [0.20689031180802256, 0.20209441300762737, 0.14744825091247529], + "ci_lb": [-1.0616175980257587, -4.9010567975856558, -4.2169897495815967], + "ci_ub": [0.42524723890774407, 1.9293012413844408, 1.27331017109261], + "QE": 5.3133148940953756, + "QEp": 0.070182418569350757, + "I2": 62.358715042043215, + "H2": 2.6566574470476878, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 2}, {"design": "extreme_k10", "model": "two", "method": "FE", "test": "adhoc", "tau2": 0, - "beta": [0.33951715132107646, 0.65156082084807287, -0.12127406276771753], + "beta": [0.3395171513210764, 0.65156082084807287, -0.12127406276771752], "se": [0.049293411370510194, 0.041250337103667649, 0.14225739044373062], - "pval": [0.00023390627660365608, 9.8783751861973846e-07, 0.42214449510262975], - "ci_lb": [0.22295675535062853, 0.5540192733463627, -0.45765933817926657], - "ci_ub": [0.45607754729152439, 0.74910236834978305, 0.21511121264383154], + "pval": [0.00023390627660365652, 9.8783751861973846e-07, 0.42214449510262986], + "ci_lb": [0.22295675535062848, 0.5540192733463627, -0.45765933817926657], + "ci_ub": [0.45607754729152433, 0.74910236834978305, 0.21511121264383154], + "QE": 14.400976720554908, + "QEp": 0.044492241224326719, + "I2": 51.39218585077839, + "H2": 2.0572823886507012, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 7}, {"design": "moderate_k20", "model": "two", "method": "FE", "test": "adhoc", "tau2": 0, - "beta": [0.73709810315645485, 0.64747703612718221, 0.19044587474139929], - "se": [0.14672220323157109, 0.16974243343585627, 0.34253541418439043], - "pval": [0.00010426559139560056, 0.0013866039152123568, 0.58546261659721521], - "ci_lb": [0.42754131316446037, 0.28935180584487785, -0.53224067806441477], - "ci_ub": [1.0466548931484494, 1.0056022664094866, 0.91313242754721335], + "beta": [0.73709810315645485, 0.64747703612718244, 0.19044587474139935], + "se": [0.14672220323157109, 0.16974243343585629, 0.34253541418439043], + "pval": [0.00010426559139560056, 0.0013866039152123544, 0.58546261659721499], + "ci_lb": [0.42754131316446037, 0.28935180584487802, -0.53224067806441466], + "ci_ub": [1.0466548931484494, 1.0056022664094868, 0.91313242754721347], + "QE": 12.137819651655924, + "QEp": 0.79172076097379995, + "I2": 0, + "H2": 0.71398939127387795, + "tau2_ci_lb": null, + "tau2_ci_ub": null, "dof": 17}, {"design": "equal_k5", "model": "intercept", "method": "DL", "test": "adhoc", "tau2": 1.08260596526162, - "beta": [0.60497665381666166], + "beta": [0.60497665381666177], "se": [0.64379518699524474], - "pval": [0.400574022875056], - "ci_lb": [-1.1824853418661843], + "pval": [0.40057402287505595], + "ci_lb": [-1.182485341866184], "ci_ub": [2.3924386494995078], + "QE": 8.3983359150613381, + "QEp": 0.078029419298244626, + "I2": 52.371516923650162, + "H2": 2.0995839787653345, + "tau2_ci_lb": 0, + "tau2_ci_ub": 15.562063970721971, "dof": 4}, {"design": "unequal_k5", "model": "intercept", "method": "DL", "test": "adhoc", "tau2": 0.28382585271252081, - "beta": [-0.25834451420148663], + "beta": [-0.25834451420148669], "se": [0.54387142220951235], "pval": [0.65955362976715803], "ci_lb": [-1.7683736622520501], "ci_ub": [1.2516846338490768], + "QE": 20.315605412654854, + "QEp": 0.00043261444283937314, + "I2": 80.310702444002246, + "H2": 5.0789013531637144, + "tau2_ci_lb": 0.37437910706976263, + "tau2_ci_ub": 62.648818709153389, "dof": 4}, {"design": "extreme_k10", "model": "intercept", "method": "DL", "test": "adhoc", - "tau2": 1.0914460604195861, + "tau2": 1.0914460604195859, "beta": [0.69453702188798982], "se": [0.39079170141387481], "pval": [0.10924843342085938], "ci_lb": [-0.18949522462750445], "ci_ub": [1.5785692684034842], + "QE": 528.36542813689618, + "QEp": 4.8272097789668306e-108, + "I2": 98.296633443309219, + "H2": 58.707269792988463, + "tau2_ci_lb": 0.27621241313967376, + "tau2_ci_ub": 4.0683787385044337, "dof": 9}, {"design": "moderate_k20", "model": "intercept", "method": "DL", "test": "adhoc", "tau2": 0.18430288219628541, @@ -1093,30 +1904,54 @@ "pval": [0.000490358039209688], "ci_lb": [0.39550144900689976], "ci_ub": [1.182959552858555], + "QE": 27.443467066041269, + "QEp": 0.094740008781753912, + "I2": 30.766765167544271, + "H2": 1.4443930034758563, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.94040725109002843, "dof": 19}, {"design": "equal_k5", "model": "one", "method": "DL", "test": "adhoc", - "tau2": 0.0059427274506224873, - "beta": [1.2764449434579284, 1.5576340773383561], - "se": [0.53378359948569698, 0.67366912517087141], - "pval": [0.096626802156227209, 0.10382513709338259], - "ci_lb": [-0.42229270054636725, -0.58628174104539199], - "ci_ub": [2.975182587462224, 3.7015498957221045], + "tau2": 0.0059427274506217518, + "beta": [1.2764449434579284, 1.5576340773383557], + "se": [0.53378359948569687, 0.67366912517087096], + "pval": [0.096626802156227168, 0.10382513709338252], + "ci_lb": [-0.42229270054636681, -0.5862817410453911], + "ci_ub": [2.9751825874622235, 3.7015498957221027], + "QE": 3.0179075904568835, + "QEp": 0.38887241030467207, + "I2": 0.5933776936547106, + "H2": 1.0059691968189612, + "tau2_ci_lb": 0, + "tau2_ci_ub": 12.956611642049028, "dof": 3}, {"design": "unequal_k5", "model": "one", "method": "DL", "test": "adhoc", - "tau2": 0.44482420458951977, - "beta": [-0.090448214875049893, -1.9230388371068361], - "se": [0.58109556806197715, 1.4162626557132092], - "pval": [0.88619158014892241, 0.2676225332293149], - "ci_lb": [-1.9397536584706272, -6.4302186930926331], - "ci_ub": [1.7588572287205275, 2.5841410188789604], + "tau2": 0.44482420458951955, + "beta": [-0.090448214875049893, -1.923038837106835], + "se": [0.58109556806197704, 1.416262655713209], + "pval": [0.88619158014892241, 0.26762253322931512], + "ci_lb": [-1.9397536584706268, -6.4302186930926304], + "ci_ub": [1.7588572287205271, 2.5841410188789609], + "QE": 19.451611261114039, + "QEp": 0.00022048004795655233, + "I2": 84.577113125855362, + "H2": 6.4838704203713462, + "tau2_ci_lb": 0.19746423631486623, + "tau2_ci_ub": 45.327128288195816, "dof": 3}, {"design": "extreme_k10", "model": "one", "method": "DL", "test": "adhoc", - "tau2": 0.029699600901791464, - "beta": [0.27000256176805176, 0.67393162666171846], - "se": [0.096628598036758412, 0.067317117048797101], - "pval": [0.023405119460235518, 8.416927756113611e-06], - "ci_lb": [0.047176615116305692, 0.51869807637716947], - "ci_ub": [0.49282850841979786, 0.82916517694626746], + "tau2": 0.029699600901791457, + "beta": [0.27000256176805176, 0.67393162666171824], + "se": [0.096628598036758384, 0.067317117048797087], + "pval": [0.02340511946023547, 8.4169277561136245e-06], + "ci_lb": [0.047176615116305748, 0.51869807637716925], + "ci_ub": [0.49282850841979775, 0.82916517694626724], + "QE": 15.896110614196777, + "QEp": 0.043891457052016768, + "I2": 49.673223883739091, + "H2": 1.9870138267745971, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.98404808836485924, "dof": 8}, {"design": "moderate_k20", "model": "one", "method": "DL", "test": "adhoc", "tau2": 0, @@ -1125,46 +1960,82 @@ "pval": [7.9616439404665205e-05, 0.0011156542841317522], "ci_lb": [0.4183366951585395, 0.29972106190311493], "ci_ub": [1.0101442159938063, 1.0105882331863709], + "QE": 12.446943265071944, + "QEp": 0.82332537450409671, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.23460881621137722, "dof": 18}, {"design": "equal_k5", "model": "two", "method": "DL", "test": "adhoc", - "tau2": 0.48520036034108888, - "beta": [1.2464498791089595, 1.4947307524436979, 0.29015096278611763], - "se": [0.65425571520006054, 0.84732066452122279, 1.0153006551870316], - "pval": [0.19704746752727681, 0.21977014722156479, 0.80192781908755628], - "ci_lb": [-1.5685852598507686, -2.1509958177316708, -4.0783351727707835], - "ci_ub": [4.0614850180686872, 5.1404573226190671, 4.6586370983430179], + "tau2": 0.48520036034108743, + "beta": [1.2464498791089589, 1.4947307524436986, 0.29015096278611674], + "se": [0.6542557152000601, 0.84732066452122257, 1.0153006551870312], + "pval": [0.19704746752727675, 0.21977014722156468, 0.80192781908755684], + "ci_lb": [-1.5685852598507675, -2.150995817731669, -4.0783351727707817], + "ci_ub": [4.0614850180686854, 5.1404573226190662, 4.6586370983430161], + "QE": 2.9338803207967543, + "QEp": 0.23063009774267032, + "I2": 31.830893515899795, + "H2": 1.4669401603983772, + "tau2_ci_lb": 0, + "tau2_ci_ub": 54.363997637568161, "dof": 2}, {"design": "unequal_k5", "model": "two", "method": "DL", "test": "adhoc", "tau2": 0.4717863239469125, - "beta": [-0.076102991931023045, -3.0910321898139177, -2.525201560728938], - "se": [0.39901562395069623, 1.0841059140587561, 1.1138538939878788], - "pval": [0.86634575195466668, 0.10414464740871864, 0.15154540628273758], - "ci_lb": [-1.7929286555351716, -7.7555634602763623, -7.3177280582379547], - "ci_ub": [1.6407226716731256, 1.5734990806485269, 2.2673249367800787], + "beta": [-0.076102991931022809, -3.0910321898139173, -2.5252015607289384], + "se": [0.39901562395069623, 1.0841059140587559, 1.113853893987879], + "pval": [0.86634575195466701, 0.10414464740871864, 0.15154540628273758], + "ci_lb": [-1.7929286555351713, -7.7555634602763615, -7.3177280582379556], + "ci_ub": [1.6407226716731258, 1.5734990806485265, 2.2673249367800792], + "QE": 5.3133148940953756, + "QEp": 0.070182418569350757, + "I2": 62.358715042043215, + "H2": 2.6566574470476878, + "tau2_ci_lb": 0, + "tau2_ci_ub": 17.40992233276512, "dof": 2}, {"design": "extreme_k10", "model": "two", "method": "DL", "test": "adhoc", - "tau2": 0.041031566660919984, - "beta": [0.26915780054154376, 0.67216962808157321, -0.13691153321000049], - "se": [0.10897817440249509, 0.075996321669380168, 0.18563292537830384], - "pval": [0.042843519715198466, 4.7761373755999592e-05, 0.48477775267268064], - "ci_lb": [0.011465366455095882, 0.49246688283031059, -0.5758636504536514], - "ci_ub": [0.52685023462799163, 0.85187237333283583, 0.30204058403365042], + "tau2": 0.041031566660919866, + "beta": [0.26915780054154353, 0.67216962808157354, -0.13691153321000071], + "se": [0.10897817440249498, 0.075996321669380112, 0.18563292537830364], + "pval": [0.042843519715198466, 4.776137375599926e-05, 0.48477775267267953], + "ci_lb": [0.011465366455095882, 0.49246688283031109, -0.57586365045365118], + "ci_ub": [0.52685023462799119, 0.85187237333283594, 0.30204058403364975], + "QE": 14.400976720554908, + "QEp": 0.044492241224326719, + "I2": 51.392185850778375, + "H2": 2.0572823886507012, + "tau2_ci_lb": 0, + "tau2_ci_ub": 1.4208174995455576, "dof": 7}, {"design": "moderate_k20", "model": "two", "method": "DL", "test": "adhoc", "tau2": 0, - "beta": [0.73709810315645485, 0.64747703612718221, 0.19044587474139929], - "se": [0.14672220323157109, 0.16974243343585627, 0.34253541418439043], - "pval": [0.00010426559139560056, 0.0013866039152123568, 0.58546261659721521], - "ci_lb": [0.42754131316446037, 0.28935180584487785, -0.53224067806441477], - "ci_ub": [1.0466548931484494, 1.0056022664094866, 0.91313242754721335], + "beta": [0.73709810315645485, 0.64747703612718244, 0.19044587474139935], + "se": [0.14672220323157109, 0.16974243343585629, 0.34253541418439043], + "pval": [0.00010426559139560056, 0.0013866039152123544, 0.58546261659721499], + "ci_lb": [0.42754131316446037, 0.28935180584487802, -0.53224067806441466], + "ci_ub": [1.0466548931484494, 1.0056022664094868, 0.91313242754721347], + "QE": 12.137819651655924, + "QEp": 0.79172076097379995, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.30347543125718318, "dof": 17}, {"design": "equal_k5", "model": "intercept", "method": "HE", "test": "adhoc", - "tau2": 1.0017690475532999, - "beta": [0.6045998164625348], - "se": [0.63799517738619749], + "tau2": 1.0017690475533003, + "beta": [0.60459981646253491], + "se": [0.6379951773861976], "pval": [0.39696568122989262], - "ci_lb": [-1.1667587709311718], + "ci_lb": [-1.166758770931172], "ci_ub": [2.3759584038562416], + "QE": 8.3983359150613381, + "QEp": 0.078029419298244626, + "I2": 50.433197317508395, + "H2": 2.0174793327010949, + "tau2_ci_lb": 0, + "tau2_ci_ub": 15.562063970721971, "dof": 4}, {"design": "unequal_k5", "model": "intercept", "method": "HE", "test": "adhoc", "tau2": 6.3777689219320006, @@ -1173,6 +2044,12 @@ "pval": [0.82511748492888681], "ci_lb": [-3.114457962573927], "ci_ub": [3.6927612496093287], + "QE": 20.315605412654854, + "QEp": 0.00043261444283937314, + "I2": 98.920736940938127, + "H2": 92.655816541078309, + "tau2_ci_lb": 0.37437910706976263, + "tau2_ci_ub": 62.648818709153389, "dof": 4}, {"design": "extreme_k10", "model": "intercept", "method": "HE", "test": "adhoc", "tau2": 0.69581387232204428, @@ -1181,6 +2058,12 @@ "pval": [0.064712291421993279], "ci_lb": [-0.051326129247421637], "ci_ub": [1.4155733625949598], + "QE": 528.36542813689618, + "QEp": 4.8272097789668306e-108, + "I2": 97.353747147956526, + "H2": 37.789283787744829, + "tau2_ci_lb": 0.27621241313967376, + "tau2_ci_ub": 4.0683787385044337, "dof": 9}, {"design": "moderate_k20", "model": "intercept", "method": "HE", "test": "adhoc", "tau2": 0, @@ -1189,30 +2072,54 @@ "pval": [0.00014999107399305813], "ci_lb": [0.43981903170095121], "ci_ub": [1.1414151377655473], + "QE": 27.443467066041269, + "QEp": 0.094740008781753912, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.94040725109002843, "dof": 19}, {"design": "equal_k5", "model": "one", "method": "HE", "test": "adhoc", "tau2": 0, - "beta": [1.2765478897207587, 1.5577764687433346], + "beta": [1.2765478897207592, 1.5577764687433355], "se": [0.53378021784724261, 0.67357980122991301], - "pval": [0.096608698810429064, 0.10377595124373379], - "ci_lb": [-0.42217899240073198, -0.58585508099453709], - "ci_ub": [2.9752747718422494, 3.7014080184812066], + "pval": [0.096608698810429022, 0.10377595124373369], + "ci_lb": [-0.42217899240073153, -0.5858550809945362], + "ci_ub": [2.9752747718422499, 3.7014080184812075], + "QE": 3.0179075904568835, + "QEp": 0.38887241030467207, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 12.956611642049028, "dof": 3}, {"design": "unequal_k5", "model": "one", "method": "HE", "test": "adhoc", - "tau2": 1.9505050157064565, - "beta": [-0.023749990802700544, -2.898365534025575], - "se": [0.73429181259895149, 1.4473400305507784], - "pval": [0.97622922680061164, 0.13898265825886177], + "tau2": 1.9505050157064572, + "beta": [-0.023749990802700443, -2.898365534025575], + "se": [0.7342918125989516, 1.4473400305507784], + "pval": [0.97622922680061175, 0.13898265825886177], "ci_lb": [-2.3605942568083114, -7.5044474667411105], "ci_ub": [2.3130942752029107, 1.7077163986899606], + "QE": 19.451611261114039, + "QEp": 0.00022048004795655233, + "I2": 96.007372979243186, + "H2": 25.046166215907853, + "tau2_ci_lb": 0.19746423631486623, + "tau2_ci_ub": 45.327128288195816, "dof": 3}, {"design": "extreme_k10", "model": "one", "method": "HE", "test": "adhoc", "tau2": 0, - "beta": [0.36308889801604494, 0.65095292619066303], - "se": [0.040105177300723688, 0.040533685542252965], + "beta": [0.36308889801604494, 0.65095292619066292], + "se": [0.040105177300723674, 0.040533685542252958], "pval": [1.7741894997235766e-05, 2.2676624900582826e-07], - "ci_lb": [0.27060619331747987, 0.55748207971516239], - "ci_ub": [0.45557160271461, 0.74442377266616366], + "ci_lb": [0.27060619331747987, 0.55748207971516228], + "ci_ub": [0.45557160271461, 0.74442377266616355], + "QE": 15.896110614196777, + "QEp": 0.043891457052016768, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.98404808836485924, "dof": 8}, {"design": "moderate_k20", "model": "one", "method": "HE", "test": "adhoc", "tau2": 0, @@ -1221,54 +2128,96 @@ "pval": [7.9616439404665205e-05, 0.0011156542841317522], "ci_lb": [0.4183366951585395, 0.29972106190311493], "ci_ub": [1.0101442159938063, 1.0105882331863709], + "QE": 12.446943265071944, + "QEp": 0.82332537450409671, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.23460881621137722, "dof": 18}, {"design": "equal_k5", "model": "two", "method": "HE", "test": "adhoc", - "tau2": 0.35326005474691957, - "beta": [1.2477742750156047, 1.4983564437097099, 0.27984719123805257], - "se": [0.64659879577191193, 0.8354853505455756, 1.0002735528399129], - "pval": [0.19340833321942136, 0.21477116753295861, 0.80593330384572393], - "ci_lb": [-1.5343157986651281, -2.0964468804808982, -4.0239825413847932], - "ci_ub": [4.0298643486963375, 5.0931597679003184, 4.5836769238608976], + "tau2": 0.35326005474691868, + "beta": [1.2477742750156047, 1.4983564437097099, 0.27984719123805285], + "se": [0.64659879577191182, 0.83548535054557538, 1.0002735528399127], + "pval": [0.1934083332194213, 0.2147711675329585, 0.8059333038457237], + "ci_lb": [-1.5343157986651277, -2.0964468804808973, -4.0239825413847914], + "ci_ub": [4.0298643486963375, 5.0931597679003175, 4.5836769238608976], + "QE": 2.9338803207967543, + "QEp": 0.23063009774267032, + "I2": 25.371204136934111, + "H2": 1.3399653423791933, + "tau2_ci_lb": 0, + "tau2_ci_ub": 54.363997637568161, "dof": 2}, {"design": "unequal_k5", "model": "two", "method": "HE", "test": "adhoc", "tau2": 0, - "beta": [-0.31818517955900727, -1.4858777781006078, -1.4718397892444939], - "se": [0.17278466684668742, 0.79373801094189367, 0.63801336820807031], - "pval": [0.20689031180802245, 0.20209441300762737, 0.14744825091247518], - "ci_lb": [-1.0616175980257585, -4.9010567975856567, -4.2169897495815967], - "ci_ub": [0.42524723890774396, 1.9293012413844406, 1.2733101710926094], + "beta": [-0.31818517955900727, -1.4858777781006076, -1.4718397892444932], + "se": [0.17278466684668745, 0.79373801094189367, 0.63801336820807031], + "pval": [0.20689031180802256, 0.20209441300762737, 0.14744825091247529], + "ci_lb": [-1.0616175980257587, -4.9010567975856558, -4.2169897495815967], + "ci_ub": [0.42524723890774407, 1.9293012413844408, 1.27331017109261], + "QE": 5.3133148940953756, + "QEp": 0.070182418569350757, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 17.40992233276512, "dof": 2}, {"design": "extreme_k10", "model": "two", "method": "HE", "test": "adhoc", "tau2": 0, - "beta": [0.33951715132107646, 0.65156082084807287, -0.12127406276771753], + "beta": [0.3395171513210764, 0.65156082084807287, -0.12127406276771752], "se": [0.049293411370510194, 0.041250337103667649, 0.14225739044373062], - "pval": [0.00023390627660365608, 9.8783751861973846e-07, 0.42214449510262975], - "ci_lb": [0.22295675535062853, 0.5540192733463627, -0.45765933817926657], - "ci_ub": [0.45607754729152439, 0.74910236834978305, 0.21511121264383154], + "pval": [0.00023390627660365652, 9.8783751861973846e-07, 0.42214449510262986], + "ci_lb": [0.22295675535062848, 0.5540192733463627, -0.45765933817926657], + "ci_ub": [0.45607754729152433, 0.74910236834978305, 0.21511121264383154], + "QE": 14.400976720554908, + "QEp": 0.044492241224326719, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 1.4208174995455576, "dof": 7}, {"design": "moderate_k20", "model": "two", "method": "HE", "test": "adhoc", "tau2": 0, - "beta": [0.73709810315645485, 0.64747703612718221, 0.19044587474139929], - "se": [0.14672220323157109, 0.16974243343585627, 0.34253541418439043], - "pval": [0.00010426559139560056, 0.0013866039152123568, 0.58546261659721521], - "ci_lb": [0.42754131316446037, 0.28935180584487785, -0.53224067806441477], - "ci_ub": [1.0466548931484494, 1.0056022664094866, 0.91313242754721335], + "beta": [0.73709810315645485, 0.64747703612718244, 0.19044587474139935], + "se": [0.14672220323157109, 0.16974243343585629, 0.34253541418439043], + "pval": [0.00010426559139560056, 0.0013866039152123544, 0.58546261659721499], + "ci_lb": [0.42754131316446037, 0.28935180584487802, -0.53224067806441466], + "ci_ub": [1.0466548931484494, 1.0056022664094868, 0.91313242754721347], + "QE": 12.137819651655924, + "QEp": 0.79172076097379995, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.30347543125718318, "dof": 17}, {"design": "equal_k5", "model": "intercept", "method": "ML", "test": "adhoc", - "tau2": 0.68027628202071799, - "beta": [0.60259585469760302], - "se": [0.63880735789098919], + "tau2": 0.68027628202071777, + "beta": [0.60259585469760291], + "se": [0.63880735789098908], "pval": [0.39893240287347198], "ci_lb": [-1.1710177072831693], - "ci_ub": [2.3762094166783756], + "ci_ub": [2.3762094166783752], + "QE": 8.3983359150613381, + "QEp": 0.078029419298244626, + "I2": 40.8614619775202, + "H2": 1.6909447433750882, + "tau2_ci_lb": 0, + "tau2_ci_ub": 15.562063970721971, "dof": 4}, {"design": "unequal_k5", "model": "intercept", "method": "ML", "test": "adhoc", - "tau2": 0.07110732122668309, - "beta": [-0.33948195894057742], - "se": [0.35332439644604918], - "pval": [0.39105206712603408], - "ci_lb": [-1.3204677500001756], - "ci_ub": [0.64150383211902073], + "tau2": 0.071107321226682979, + "beta": [-0.33948195894057737], + "se": [0.35332439644604896], + "pval": [0.39105206712603396], + "ci_lb": [-1.3204677500001749], + "ci_ub": [0.64150383211902029], + "QE": 20.315605412654854, + "QEp": 0.00043261444283937314, + "I2": 50.54140691154651, + "H2": 2.021893340579997, + "tau2_ci_lb": 0.37437910706976263, + "tau2_ci_ub": 62.648818709153389, "dof": 4}, {"design": "extreme_k10", "model": "intercept", "method": "ML", "test": "adhoc", "tau2": 0.71407535451805593, @@ -1277,6 +2226,12 @@ "pval": [0.064841751772212602], "ci_lb": [-0.051812017767040808], "ci_ub": [1.417545204355235], + "QE": 528.36542813689618, + "QEp": 4.8272097789668306e-108, + "I2": 97.419675162397368, + "H2": 38.754810457473226, + "tau2_ci_lb": 0.27621241313967376, + "tau2_ci_ub": 4.0683787385044337, "dof": 9}, {"design": "moderate_k20", "model": "intercept", "method": "ML", "test": "adhoc", "tau2": 0.19754570451620143, @@ -1285,30 +2240,54 @@ "pval": [0.00056342759017041126], "ci_lb": [0.3893272598519989], "ci_ub": [1.1876022798515202], + "QE": 27.443467066041269, + "QEp": 0.094740008781753912, + "I2": 32.264202007657786, + "H2": 1.4763242327388744, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.94040725109002843, "dof": 19}, {"design": "equal_k5", "model": "one", "method": "ML", "test": "adhoc", "tau2": 0, - "beta": [1.2765478897207587, 1.5577764687433346], + "beta": [1.2765478897207592, 1.5577764687433355], "se": [0.53378021784724261, 0.67357980122991301], - "pval": [0.096608698810429064, 0.10377595124373379], - "ci_lb": [-0.42217899240073198, -0.58585508099453709], - "ci_ub": [2.9752747718422494, 3.7014080184812066], + "pval": [0.096608698810429022, 0.10377595124373369], + "ci_lb": [-0.42217899240073153, -0.5858550809945362], + "ci_ub": [2.9752747718422499, 3.7014080184812075], + "QE": 3.0179075904568835, + "QEp": 0.38887241030467207, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 12.956611642049028, "dof": 3}, {"design": "unequal_k5", "model": "one", "method": "ML", "test": "adhoc", - "tau2": 0.5468260898151458, - "beta": [-0.069360915104283674, -2.0828519418136895], - "se": [0.60171902845827929, 1.4243003824457388], - "pval": [0.91551270675726582, 0.23981353377540848], - "ci_lb": [-1.9842994140402377, -6.6156114315423054], - "ci_ub": [1.8455775838316701, 2.4499075479149259], + "tau2": 0.54682608981514602, + "beta": [-0.069360915104283785, -2.0828519418136899], + "se": [0.60171902845827963, 1.424300382445739], + "pval": [0.9155127067572657, 0.23981353377540837], + "ci_lb": [-1.9842994140402388, -6.6156114315423054], + "ci_ub": [1.8455775838316713, 2.4499075479149255], + "QE": 19.451611261114039, + "QEp": 0.00022048004795655233, + "I2": 87.082385574134719, + "H2": 7.7413674617634705, + "tau2_ci_lb": 0.19746423631486623, + "tau2_ci_ub": 45.327128288195816, "dof": 3}, {"design": "extreme_k10", "model": "one", "method": "ML", "test": "adhoc", "tau2": 0, - "beta": [0.36308889801604494, 0.65095292619066303], - "se": [0.040105177300723688, 0.040533685542252965], + "beta": [0.36308889801604494, 0.65095292619066292], + "se": [0.040105177300723674, 0.040533685542252958], "pval": [1.7741894997235766e-05, 2.2676624900582826e-07], - "ci_lb": [0.27060619331747987, 0.55748207971516239], - "ci_ub": [0.45557160271461, 0.74442377266616366], + "ci_lb": [0.27060619331747987, 0.55748207971516228], + "ci_ub": [0.45557160271461, 0.74442377266616355], + "QE": 15.896110614196777, + "QEp": 0.043891457052016768, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.98404808836485924, "dof": 8}, {"design": "moderate_k20", "model": "one", "method": "ML", "test": "adhoc", "tau2": 0, @@ -1317,54 +2296,96 @@ "pval": [7.9616439404665205e-05, 0.0011156542841317522], "ci_lb": [0.4183366951585395, 0.29972106190311493], "ci_ub": [1.0101442159938063, 1.0105882331863709], + "QE": 12.446943265071944, + "QEp": 0.82332537450409671, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.23460881621137722, "dof": 18}, {"design": "equal_k5", "model": "two", "method": "ML", "test": "adhoc", - "tau2": 1.2606432107381935e-07, - "beta": [1.2531305431068513, 1.5129861564728475, 0.23824829310485862], - "se": [0.65196307811546828, 0.83464920152483679, 0.99546463339138747], - "pval": [0.19453170792011315, 0.21155911287513976, 0.83313809699359309], - "ci_lb": [-1.5520401746425312, -2.0782195088512019, -4.0448903291256446], - "ci_ub": [4.0583012608562337, 5.1041918217968965, 4.521386915335361], + "tau2": 1.2606432066241397e-07, + "beta": [1.2531305431068516, 1.5129861564728473, 0.2382482931048589], + "se": [0.6519630781154685, 0.83464920152483668, 0.99546463339138769], + "pval": [0.19453170792011318, 0.21155911287513976, 0.83313809699359287], + "ci_lb": [-1.5520401746425319, -2.0782195088512014, -4.0448903291256446], + "ci_ub": [4.0583012608562345, 5.1041918217968956, 4.5213869153353627], + "QE": 2.9338803207967543, + "QEp": 0.23063009774267032, + "I2": 1.2131995723952348e-05, + "H2": 1.0000001213199721, + "tau2_ci_lb": 0, + "tau2_ci_ub": 54.363997637568161, "dof": 2}, {"design": "unequal_k5", "model": "two", "method": "ML", "test": "adhoc", "tau2": 0, - "beta": [-0.31818517955900727, -1.4858777781006078, -1.4718397892444939], - "se": [0.17278466684668742, 0.79373801094189367, 0.63801336820807031], - "pval": [0.20689031180802245, 0.20209441300762737, 0.14744825091247518], - "ci_lb": [-1.0616175980257585, -4.9010567975856567, -4.2169897495815967], - "ci_ub": [0.42524723890774396, 1.9293012413844406, 1.2733101710926094], + "beta": [-0.31818517955900727, -1.4858777781006076, -1.4718397892444932], + "se": [0.17278466684668745, 0.79373801094189367, 0.63801336820807031], + "pval": [0.20689031180802256, 0.20209441300762737, 0.14744825091247529], + "ci_lb": [-1.0616175980257587, -4.9010567975856558, -4.2169897495815967], + "ci_ub": [0.42524723890774407, 1.9293012413844408, 1.27331017109261], + "QE": 5.3133148940953756, + "QEp": 0.070182418569350757, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 17.40992233276512, "dof": 2}, {"design": "extreme_k10", "model": "two", "method": "ML", "test": "adhoc", "tau2": 0, - "beta": [0.33951715132107646, 0.65156082084807287, -0.12127406276771753], + "beta": [0.3395171513210764, 0.65156082084807287, -0.12127406276771752], "se": [0.049293411370510194, 0.041250337103667649, 0.14225739044373062], - "pval": [0.00023390627660365608, 9.8783751861973846e-07, 0.42214449510262975], - "ci_lb": [0.22295675535062853, 0.5540192733463627, -0.45765933817926657], - "ci_ub": [0.45607754729152439, 0.74910236834978305, 0.21511121264383154], + "pval": [0.00023390627660365652, 9.8783751861973846e-07, 0.42214449510262986], + "ci_lb": [0.22295675535062848, 0.5540192733463627, -0.45765933817926657], + "ci_ub": [0.45607754729152433, 0.74910236834978305, 0.21511121264383154], + "QE": 14.400976720554908, + "QEp": 0.044492241224326719, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 1.4208174995455576, "dof": 7}, {"design": "moderate_k20", "model": "two", "method": "ML", "test": "adhoc", "tau2": 0, - "beta": [0.73709810315645485, 0.64747703612718221, 0.19044587474139929], - "se": [0.14672220323157109, 0.16974243343585627, 0.34253541418439043], - "pval": [0.00010426559139560056, 0.0013866039152123568, 0.58546261659721521], - "ci_lb": [0.42754131316446037, 0.28935180584487785, -0.53224067806441477], - "ci_ub": [1.0466548931484494, 1.0056022664094866, 0.91313242754721335], + "beta": [0.73709810315645485, 0.64747703612718244, 0.19044587474139935], + "se": [0.14672220323157109, 0.16974243343585629, 0.34253541418439043], + "pval": [0.00010426559139560056, 0.0013866039152123544, 0.58546261659721499], + "ci_lb": [0.42754131316446037, 0.28935180584487802, -0.53224067806441466], + "ci_ub": [1.0466548931484494, 1.0056022664094868, 0.91313242754721347], + "QE": 12.137819651655924, + "QEp": 0.79172076097379995, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.30347543125718318, "dof": 17}, {"design": "equal_k5", "model": "intercept", "method": "REML", "test": "adhoc", - "tau2": 1.0830683217408148, - "beta": [0.60497869770208212], - "se": [0.64386731554321897], + "tau2": 1.0830683217408152, + "beta": [0.60497869770208224], + "se": [0.64386731554321908], "pval": [0.40062052901477246], - "ci_lb": [-1.182683558934732], - "ci_ub": [2.392640954338896], + "ci_lb": [-1.1826835589347322], + "ci_ub": [2.3926409543388965], + "QE": 8.3983359150613381, + "QEp": 0.078029419298244626, + "I2": 52.382167455829745, + "H2": 2.1000535861694267, + "tau2_ci_lb": 0, + "tau2_ci_ub": 15.562063970721971, "dof": 4}, {"design": "unequal_k5", "model": "intercept", "method": "REML", "test": "adhoc", - "tau2": 3.0539045155524742, - "beta": [0.18480405951699175], + "tau2": 3.0539045155524738, + "beta": [0.18480405951699172], "se": [1.0399695871305989], "pval": [0.86759352145897251], - "ci_lb": [-2.7026144102263308], - "ci_ub": [3.0722225292603147], + "ci_lb": [-2.7026144102263312], + "ci_ub": [3.0722225292603142], + "QE": 20.315605412654854, + "QEp": 0.00043261444283937314, + "I2": 97.772237701301052, + "H2": 44.888092440742795, + "tau2_ci_lb": 0.37437910706976263, + "tau2_ci_ub": 62.648818709153389, "dof": 4}, {"design": "extreme_k10", "model": "intercept", "method": "REML", "test": "adhoc", "tau2": 0.8224002783454577, @@ -1373,38 +2394,68 @@ "pval": [0.077534609550980627], "ci_lb": [-0.09312005334974216], "ci_ub": [1.4668994574976293], + "QE": 528.36542813689618, + "QEp": 4.8272097789668306e-108, + "I2": 97.751909699504225, + "H2": 44.482198948123568, + "tau2_ci_lb": 0.27621241313967376, + "tau2_ci_ub": 4.0683787385044337, "dof": 9}, {"design": "moderate_k20", "model": "intercept", "method": "REML", "test": "adhoc", - "tau2": 0.23909863987739999, - "beta": [0.78602512965241333], - "se": [0.19840151801276226], + "tau2": 0.23909863987740002, + "beta": [0.78602512965241322], + "se": [0.19840151801276223], "pval": [0.0008360734328113793], - "ci_lb": [0.37076598002057837], - "ci_ub": [1.2012842792842484], + "ci_lb": [0.37076598002057831], + "ci_ub": [1.2012842792842482], + "QE": 27.443467066041269, + "QEp": 0.094740008781753912, + "I2": 36.569035528086367, + "H2": 1.5765170974860179, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.94040725109002843, "dof": 19}, {"design": "equal_k5", "model": "one", "method": "REML", "test": "adhoc", - "tau2": 0.012139238147945345, - "beta": [1.2763383912384598, 1.5574875090169169], - "se": [0.53543567441262074, 0.67584219308871096], - "pval": [0.09728908575857538, 0.10455493298301109], - "ci_lb": [-0.42765689251307615, -0.59334398133309052], - "ci_ub": [2.9803336749899958, 3.7083189993669246], + "tau2": 0.012139238147945644, + "beta": [1.2763383912384594, 1.5574875090169171], + "se": [0.53543567441262063, 0.67584219308871119], + "pval": [0.09728908575857538, 0.10455493298301112], + "ci_lb": [-0.42765689251307615, -0.59334398133309074], + "ci_ub": [2.9803336749899949, 3.708318999366925], + "QE": 3.0179075904568835, + "QEp": 0.38887241030467207, + "I2": 1.2046421535754204, + "H2": 1.0121933072548612, + "tau2_ci_lb": 0, + "tau2_ci_ub": 12.956611642049028, "dof": 3}, {"design": "unequal_k5", "model": "one", "method": "REML", "test": "adhoc", - "tau2": 1.6012284825460623, - "beta": [-0.01956806718199039, -2.7975303090464272], - "se": [0.69894371863702431, 1.4139548479837643], - "pval": [0.97942311732082699, 0.14226284037670175], - "ci_lb": [-2.2439189221596445, -7.2973656908503468], - "ci_ub": [2.2047827877956641, 1.702305072757492], + "tau2": 1.6012284825460621, + "beta": [-0.019568067181989932, -2.7975303090464259], + "se": [0.69894371863702442, 1.4139548479837638], + "pval": [0.97942311732082754, 0.14226284037670175], + "ci_lb": [-2.2439189221596445, -7.2973656908503433], + "ci_ub": [2.2047827877956649, 1.7023050727574915], + "QE": 19.451611261114039, + "QEp": 0.00022048004795655233, + "I2": 95.178451341004532, + "H2": 20.74022416292167, + "tau2_ci_lb": 0.19746423631486623, + "tau2_ci_ub": 45.327128288195816, "dof": 3}, {"design": "extreme_k10", "model": "one", "method": "REML", "test": "adhoc", - "tau2": 0.036951880858045835, - "beta": [0.26541839994868621, 0.67408851856411911], - "se": [0.10458480147412126, 0.072987984380740473], - "pval": [0.034828262305551382, 1.5318803102979648e-05], - "ci_lb": [0.024245415269855797, 0.50577792476191452], - "ci_ub": [0.50659138462751663, 0.8423991123663237], + "tau2": 0.036951880858045794, + "beta": [0.26541839994868621, 0.67408851856411889], + "se": [0.10458480147412121, 0.072987984380740445], + "pval": [0.034828262305551312, 1.5318803102979648e-05], + "ci_lb": [0.024245415269855936, 0.5057779247619143], + "ci_ub": [0.50659138462751652, 0.84239911236632348], + "QE": 15.896110614196777, + "QEp": 0.043891457052016768, + "I2": 55.117312082709873, + "H2": 2.2280305534347704, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.98404808836485924, "dof": 8}, {"design": "moderate_k20", "model": "one", "method": "REML", "test": "adhoc", "tau2": 0, @@ -1413,38 +2464,68 @@ "pval": [7.9616439404665205e-05, 0.0011156542841317522], "ci_lb": [0.4183366951585395, 0.29972106190311493], "ci_ub": [1.0101442159938063, 1.0105882331863709], + "QE": 12.446943265071944, + "QEp": 0.82332537450409671, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.23460881621137722, "dof": 18}, {"design": "equal_k5", "model": "two", "method": "REML", "test": "adhoc", - "tau2": 0.5280809767005048, - "beta": [1.2460703551935211, 1.4936911119117537, 0.29310505071275272], - "se": [0.66352418592804685, 0.85988959667497467, 1.0306035555149087], - "pval": [0.20117458555987894, 0.22450725072502373, 0.80284503591563905], - "ci_lb": [-1.6088437946445804, -2.2061152085049915, -4.1412241507129721], - "ci_ub": [4.1009845050316223, 5.193497432328499, 4.7274342521384778], + "tau2": 0.52808097670050402, + "beta": [1.2460703551935206, 1.4936911119117531, 0.29310505071275289], + "se": [0.66352418592804663, 0.85988959667497433, 1.0306035555149085], + "pval": [0.20117458555987897, 0.22450725072502373, 0.80284503591563894], + "ci_lb": [-1.6088437946445799, -2.2061152085049907, -4.1412241507129712], + "ci_ub": [4.1009845050316214, 5.1934974323284964, 4.7274342521384769], + "QE": 2.9338803207967543, + "QEp": 0.23063009774267032, + "I2": 33.696103726458212, + "H2": 1.5082069926545849, + "tau2_ci_lb": 0, + "tau2_ci_ub": 54.363997637568161, "dof": 2}, {"design": "unequal_k5", "model": "two", "method": "REML", "test": "adhoc", - "tau2": 0.47326742184966425, - "beta": [-0.075886999301457397, -3.092299300315581, -2.5259267977879882], - "se": [0.39953440935260487, 1.0847585343866863, 1.1149099958718196], - "pval": [0.86688833098322637, 0.10417860618601761, 0.15170245988993752], - "ci_lb": [-1.7949448163312824, -7.7596385694134842, -7.3229973349508359], - "ci_ub": [1.6431708177283675, 1.5750399687823222, 2.2711437393748599], + "tau2": 0.47326742184966131, + "beta": [-0.075886999301457744, -3.092299300315577, -2.5259267977879856], + "se": [0.39953440935260387, 1.0847585343866848, 1.1149099958718174], + "pval": [0.86688833098322537, 0.10417860618601758, 0.15170245988993727], + "ci_lb": [-1.7949448163312784, -7.7596385694134735, -7.3229973349508235], + "ci_ub": [1.643170817728363, 1.57503996878232, 2.2711437393748528], + "QE": 5.3133148940953756, + "QEp": 0.070182418569350757, + "I2": 62.43225961847147, + "H2": 2.6618582588259274, + "tau2_ci_lb": 0, + "tau2_ci_ub": 17.40992233276512, "dof": 2}, {"design": "extreme_k10", "model": "two", "method": "REML", "test": "adhoc", "tau2": 0.054688742532978674, - "beta": [0.26582139792705878, 0.67314112868512155, -0.14647088836558972], - "se": [0.12151930104983935, 0.084990666437201268, 0.20414935748367657], - "pval": [0.064919555709613525, 9.7171036845294565e-05, 0.49632302750135482], - "ci_lb": [-0.021526088371995877, 0.47217013766868254, -0.62920741001857627], - "ci_ub": [0.55316888422611343, 0.87411211970156055, 0.33626563328739678], + "beta": [0.26582139792705889, 0.67314112868512144, -0.14647088836558994], + "se": [0.12151930104983936, 0.084990666437201254, 0.20414935748367652], + "pval": [0.064919555709613441, 9.7171036845294565e-05, 0.49632302750135426], + "ci_lb": [-0.021526088371995766, 0.47217013766868243, -0.62920741001857639], + "ci_ub": [0.55316888422611354, 0.87411211970156044, 0.33626563328739645], + "QE": 14.400976720554908, + "QEp": 0.044492241224326719, + "I2": 58.492345102622807, + "H2": 2.4091941654434166, + "tau2_ci_lb": 0, + "tau2_ci_ub": 1.4208174995455576, "dof": 7}, {"design": "moderate_k20", "model": "two", "method": "REML", "test": "adhoc", "tau2": 0, - "beta": [0.73709810315645485, 0.64747703612718221, 0.19044587474139929], - "se": [0.14672220323157109, 0.16974243343585627, 0.34253541418439043], - "pval": [0.00010426559139560056, 0.0013866039152123568, 0.58546261659721521], - "ci_lb": [0.42754131316446037, 0.28935180584487785, -0.53224067806441477], - "ci_ub": [1.0466548931484494, 1.0056022664094866, 0.91313242754721335], + "beta": [0.73709810315645485, 0.64747703612718244, 0.19044587474139935], + "se": [0.14672220323157109, 0.16974243343585629, 0.34253541418439043], + "pval": [0.00010426559139560056, 0.0013866039152123544, 0.58546261659721499], + "ci_lb": [0.42754131316446037, 0.28935180584487802, -0.53224067806441466], + "ci_ub": [1.0466548931484494, 1.0056022664094868, 0.91313242754721347], + "QE": 12.137819651655924, + "QEp": 0.79172076097379995, + "I2": 0, + "H2": 1, + "tau2_ci_lb": 0, + "tau2_ci_ub": 0.30347543125718318, "dof": 17} ] } diff --git a/pymare/tests/test_metafor_random_effects.py b/pymare/tests/test_metafor_random_effects.py new file mode 100644 index 0000000..4514a18 --- /dev/null +++ b/pymare/tests/test_metafor_random_effects.py @@ -0,0 +1,376 @@ +"""Alignment between PyMARE and metafor on tau^2 and the statistics derived from Q. + +:mod:`pymare.tests.test_metafor_alignment` covers the fixed-effect inference path +-- what ``rma.uni`` reports under each ``test`` setting, given a tau^2. This +module covers the rest of what ``rma.uni`` reports and PyMARE also computes: + +- Cochran's ``Q`` and its p-value, from + :meth:`~pymare.results.MetaRegressionResults.get_heterogeneity_stats`. +- ``I^2`` and ``H``, from the same method. +- The Q-profile confidence interval for tau^2, from + :meth:`~pymare.results.MetaRegressionResults.get_re_stats`. +- The tau^2 point estimate itself, for each estimator PyMARE and metafor share. + +The reference values are pinned in ``data/metafor_reference.json`` alongside the +ones the other module reads, and regenerated by the same script. Both come from +the same 180-case grid; the quantities here do not depend on metafor's ``test`` +setting, which :func:`test_heterogeneity_does_not_depend_on_the_correction` +checks rather than assumes before the rest of the module drops down to the 60 +distinct cases. + +What agrees, and how exactly, is recorded in ``validation/metafor/README.md``. +Three divergences are pinned down here rather than merely tolerated, each by a +test that asserts *why* the two differ instead of how much: + +- ``I^2`` and ``H``: PyMARE always reports the Q-based definition of + :footcite:t:`higgins2002quantifying`. metafor reports that pair only for + ``FE`` and ``DL`` -- where it coincides with tau^2 / (tau^2 + v_t) -- and + the tau^2-based pair otherwise. See + :func:`test_i2_and_h_are_the_q_based_definition`. +- :class:`~pymare.estimators.Hedges` tau^2: PyMARE subtracts the mean sampling + variance, metafor subtracts ``tr(PV) / (K - P)``. The two are equal when the + only predictor is the intercept and differ otherwise. See + :func:`test_hedges_tau2_divergence_is_the_trace_term`. +- ``ML`` and ``REML`` tau^2: both profile the likelihood numerically, to + different tolerances, and on one cell of the grid they stop on opposite + sides of the tau^2 = 0 boundary. See :data:`RTOL_PROFILED` and + :data:`ML_BOUNDARY_CASE`. + +References +---------- +.. footbibliography:: + +""" + +import numpy as np +import pytest + +from pymare.estimators import ( + DerSimonianLaird, + Hedges, + VarianceBasedLikelihoodEstimator, + WeightedLeastSquares, +) +from pymare.tests.test_metafor_alignment import MODELS, build_dataset, case_id +from pymare.tests.utils import load_metafor_reference + +pytestmark = pytest.mark.metafor + +REFERENCE = load_metafor_reference() + +#: Tolerance for the quantities both implementations reach in closed form. Q is +#: a weighted residual sum of squares on either side and I^2 and H are algebraic +#: functions of it, so they agree to a few multiples of machine epsilon; the +#: p-value is the loosest at 1.5e-13, being a chi-squared tail. +RTOL = 1e-11 + +#: Tolerance for the Q-profile interval. Both implementations invert the same Q +#: to a root -- ``scipy.optimize.brentq`` here, ``uniroot`` there -- and agree to +#: 1.3e-13. Note that the reference is generated with ``confint``'s tolerance +#: tightened: its default is ``uniroot``'s ``.Machine$double.eps^0.25``, about +#: 1.2e-4 relative, so pinning the default would check PyMARE against metafor's +#: display precision rather than against the bound it solves for. +RTOL_PROFILE = 1e-11 + +#: Tolerance for tau^2 where *both* implementations profile a likelihood +#: numerically. Worst observed 2.7e-5 relative, on ``extreme_k10`` with one +#: moderator under ``REML``: PyMARE profiles to ``xtol=1e-6`` on tau^2 and +#: metafor runs its own optimizer to its own tolerance, so this bound measures +#: the two search tolerances and not the objective. Tightening it is a change to +#: :func:`~pymare.stats.bounded_scalar_min`, not to this test. +RTOL_PROFILED = 1e-4 + +#: Absolute companion to :data:`RTOL_PROFILED`, for the cells that land on the +#: tau^2 = 0 boundary. On ``equal_k5`` with two moderators metafor's ``ML`` +#: search stops at 1.3e-7 where PyMARE's returns exactly zero: the same answer +#: to seven decimals, but no relative tolerance can say so against a zero. +ATOL_PROFILED = 1e-6 + +#: The one cell in the grid where the two ``ML`` searches land on genuinely +#: different sides of the tau^2 = 0 boundary: metafor reports 0 and PyMARE +#: 0.0114, which is a real difference rather than a tolerance. A profile +#: likelihood is flattest at the boundary, and ``extreme_k10`` is the design +#: whose weights are most unequal. +#: :func:`test_ml_boundary_case_is_the_only_disagreement` keeps this from +#: silently growing into a list. +ML_BOUNDARY_CASE = ("extreme_k10", "one") + +#: The estimators compared, in metafor's names. Keyed the same way as +#: ``methods`` in ``validation/metafor/run_metafor.R``. +ESTIMATORS = { + "FE": lambda: WeightedLeastSquares(tau2=0.0), + "DL": DerSimonianLaird, + "HE": Hedges, + "ML": lambda: VarianceBasedLikelihoodEstimator(method="ML"), + "REML": lambda: VarianceBasedLikelihoodEstimator(method="REML"), +} + +#: metafor ``method`` values whose ``I^2`` and ``H^2`` are the Q-based pair, and +#: which PyMARE can therefore be compared against on those two. For ``DL`` the +#: coincidence is an identity rather than a special case: DL's tau^2 is +#: ``(Q - df) / sum(diag(P))`` and metafor's typical variance is +#: ``df / sum(diag(P))``, so ``tau^2 / v_t`` reduces to ``(Q - df) / df``. +Q_BASED_METHODS = ("FE", "DL") + +#: One case per (design, model, tau^2 estimator). The quantities in this module +#: do not depend on metafor's ``test``, so the grid collapses by a factor of +#: three; ``test="z"`` is the slice kept because it is the uncorrected one. +CASES = [case for case in REFERENCE["cases"] if case["test"] == "z"] + +RANDOM_EFFECT_CASES = [case for case in CASES if case["method"] != "FE"] + +HEDGES_CASES = [case for case in CASES if case["method"] == "HE"] + +PROFILED_CASES = [case for case in CASES if case["method"] in ("ML", "REML")] + + +def fit(frame, case): + """Fit the estimator metafor's ``method`` names to this case's design.""" + return ESTIMATORS[case["method"]]().fit_dataset(build_dataset(frame, case)).summary() + + +def design_arrays(frame, case): + """Return ``(y, v, X)`` for one case, with the intercept already in ``X``.""" + rows = frame[frame["case"] == case["design"]] + y = rows["y"].to_numpy() + columns = MODELS[case["model"]] + X = np.column_stack([np.ones_like(y)] + [rows[name].to_numpy() for name in columns]) + return y, rows["v"].to_numpy(), X + + +def test_heterogeneity_does_not_depend_on_the_correction(): + """The quantities below must be the same under all three ``test`` settings. + + Every other test in this module reads only the ``test="z"`` slice of the + pinned grid. That is sound only because ``rma.uni``'s heterogeneity block and + its ``confint`` do not move when the small-sample correction is switched on + -- the correction rescales the coefficient covariance and nothing else. This + asserts that from the reference itself, so a future metafor that did change + one of them would be caught here rather than hidden by the slicing. + """ + grouped = {} + for case in REFERENCE["cases"]: + key = (case["design"], case["model"], case["method"]) + recorded = tuple( + case[field] for field in ("QE", "QEp", "I2", "H2", "tau2", "tau2_ci_lb", "tau2_ci_ub") + ) + grouped.setdefault(key, set()).add(recorded) + + assert len(grouped) == len(CASES) + varying = sorted(key for key, values in grouped.items() if len(values) > 1) + assert not varying, varying + + +@pytest.mark.parametrize("case", CASES, ids=[case_id(case) for case in CASES]) +def test_q_matches_metafor(case, metafor_dataset): + """Cochran's Q and its p-value must be metafor's ``QE`` and ``QEp``. + + Both are properties of the design rather than of the estimator -- the Q + PyMARE reports is always taken at tau^2 = 0, and so is metafor's -- so this + runs over every estimator in the grid and expects the same numbers from each. + """ + stats = fit(metafor_dataset, case).get_heterogeneity_stats() + assert np.allclose(np.ravel(stats["Q"]), case["QE"], rtol=RTOL) + assert np.allclose(np.ravel(stats["p(Q)"]), case["QEp"], rtol=RTOL) + # logp(Q) exists so that a p-value past the underflow point is still usable. + # None of these designs is anywhere near it, which is exactly why they can + # check that the log path agrees with the direct one against a third party. + assert np.allclose(np.ravel(stats["logp(Q)"]), np.log(case["QEp"]), rtol=RTOL) + + +@pytest.mark.parametrize( + "case", + [case for case in CASES if case["method"] in Q_BASED_METHODS], + ids=[case_id(case) for case in CASES if case["method"] in Q_BASED_METHODS], +) +def test_i2_and_h_match_metafor_where_the_definitions_coincide(case, metafor_dataset): + """``I^2`` and ``H`` must match metafor's for ``FE`` and ``DL``. + + Both bounds are PyMARE's and not metafor's: PyMARE floors ``I^2`` at 0 and + ``H`` at 1, as :footcite:t:`higgins2002quantifying` define them, while + metafor reports ``H^2 = Q / df`` unfloored and so can print a value below + one. Applying the floors to the reference rather than dropping the cases + where they bite keeps the whole grid compared. + + References + ---------- + .. footbibliography:: + """ + stats = fit(metafor_dataset, case).get_heterogeneity_stats() + assert np.allclose(np.ravel(stats["I^2"]), max(0.0, case["I2"]), rtol=RTOL) + assert np.allclose(np.ravel(stats["H"]), max(1.0, np.sqrt(case["H2"])), rtol=RTOL) + + +@pytest.mark.parametrize( + "case", + [case for case in CASES if case["method"] not in Q_BASED_METHODS], + ids=[case_id(case) for case in CASES if case["method"] not in Q_BASED_METHODS], +) +def test_i2_and_h_are_the_q_based_definition(case, metafor_dataset): + """Where the two disagree, PyMARE's pair must be the Q-based one. + + metafor switches to ``I^2 = 100 tau^2 / (tau^2 + v_t)`` and + ``H^2 = tau^2 / v_t + 1`` for the estimators that are not ``FE`` or ``DL``, + so its ``I^2`` depends on which tau^2 estimator was asked for and PyMARE's + does not. Rather than pin the size of that gap -- which says nothing and + would have to be re-pinned whenever an optimizer tolerance moved -- this + asserts the shape of it: PyMARE's ``I^2`` and ``H`` for this case equal + PyMARE's for the *fixed-effects* fit of the same design and model, because + both are functions of the same Q. + """ + fixed = dict(case, method="FE") + reported = fit(metafor_dataset, case).get_heterogeneity_stats() + q_based = fit(metafor_dataset, fixed).get_heterogeneity_stats() + for key in ("Q", "I^2", "H"): + assert np.allclose(np.ravel(reported[key]), np.ravel(q_based[key]), rtol=RTOL), key + + +@pytest.mark.parametrize( + "case", RANDOM_EFFECT_CASES, ids=[case_id(case) for case in RANDOM_EFFECT_CASES] +) +def test_tau2_interval_matches_metafor(case, metafor_dataset): + """The Q-profile interval must be metafor's ``confint`` bounds. + + :footcite:t:`viechtbauer2007confidence` inverted, which both implementations + do by solving ``Q(tau^2) = chi^2_{1 - alpha/2}`` and ``= chi^2_{alpha/2}`` + for a root. The interval does not depend on the tau^2 point estimate, which + is why every random-effects method in the grid is expected to produce the + same pair and metafor's ``confint`` reports the same pair for each. + + References + ---------- + .. footbibliography:: + """ + stats = fit(metafor_dataset, case).get_re_stats() + assert np.allclose(np.ravel(stats["ci_l"]), case["tau2_ci_lb"], rtol=RTOL_PROFILE) + assert np.allclose(np.ravel(stats["ci_u"]), case["tau2_ci_ub"], rtol=RTOL_PROFILE) + + +@pytest.mark.parametrize( + "case", + [case for case in HEDGES_CASES if not MODELS[case["model"]]], + ids=[case_id(case) for case in HEDGES_CASES if not MODELS[case["model"]]], +) +def test_hedges_tau2_matches_metafor_without_moderators(case, metafor_dataset): + """``Hedges`` tau^2 must be metafor's ``HE`` for an intercept-only model. + + Exactly, not approximately: the two expressions for the correction term are + algebraically the same one when the intercept is the only predictor, so this + holds to machine precision on all four designs. + :func:`test_hedges_tau2_divergence_is_the_trace_term` covers what happens + when a moderator is added. + """ + assert np.allclose(np.ravel(fit(metafor_dataset, case).tau2), case["tau2"], rtol=RTOL) + + +@pytest.mark.parametrize("case", HEDGES_CASES, ids=[case_id(case) for case in HEDGES_CASES]) +def test_hedges_tau2_divergence_is_the_trace_term(case, metafor_dataset): + """Locate the ``HE`` divergence in the term each implementation subtracts. + + Both estimators are the excess of an unweighted mean squared error over an + estimate of the mean sampling variance. They differ in the second term: + metafor uses ``tr(PV) / (K - P)`` with ``P = I - X (X'X)^-1 X'``, PyMARE uses + ``sum(v) / K``. Those coincide when ``X`` is just an intercept, since ``P`` + is then ``I - J/K`` and the trace is ``sum(v) (K - 1) / K``; with a moderator + they do not, and PyMARE's tau^2 is out by up to 0.14 relative on this grid. + + Two assertions, which together say that and nothing more: metafor's own + definition reimplemented here reproduces every pinned ``HE`` tau^2, and the + two correction terms agree exactly when and only when there are no + moderators. A test that instead pinned the size of the gap would be a record + of PyMARE's current behaviour rather than a statement about either formula. + """ + y, v, X = design_arrays(metafor_dataset, case) + n_obs, n_preds = X.shape + residual_maker = np.eye(n_obs) - X @ np.linalg.pinv(X.T @ X) @ X.T + + metafor_correction = np.trace(residual_maker @ np.diag(v)) / (n_obs - n_preds) + pymare_correction = v.sum() / n_obs + metafor_tau2 = max( + 0.0, (y @ residual_maker @ y - np.trace(residual_maker @ np.diag(v))) / (n_obs - n_preds) + ) + + assert np.allclose(metafor_tau2, case["tau2"], rtol=RTOL, atol=1e-12) + if n_preds == 1: + assert np.allclose(pymare_correction, metafor_correction, rtol=RTOL) + else: + assert not np.isclose(pymare_correction, metafor_correction, rtol=1e-6) + + +@pytest.mark.parametrize( + "case", + [c for c in PROFILED_CASES if (c["design"], c["model"]) != ML_BOUNDARY_CASE], + ids=[case_id(c) for c in PROFILED_CASES if (c["design"], c["model"]) != ML_BOUNDARY_CASE], +) +def test_profiled_tau2_matches_metafor(case, metafor_dataset): + """``ML`` and ``REML`` tau^2 must agree to the two search tolerances. + + Not to machine precision, and the gap is not evidence of a different + objective: both implementations maximize the same likelihood by numerical + search, and :data:`RTOL_PROFILED` is where the coarser of the two stops. + What this rules out is the thing a loose tolerance might otherwise hide -- a + different likelihood, a different parameterization, or a boundary handled + differently -- since none of those would stay inside 1e-4 across twenty + design-by-model cells. + """ + assert np.allclose( + np.ravel(fit(metafor_dataset, case).tau2), + case["tau2"], + rtol=RTOL_PROFILED, + atol=ATOL_PROFILED, + ) + + +def test_ml_boundary_case_is_the_only_disagreement(metafor_dataset): + """The excluded ``ML`` cell must still be excluded for the stated reason. + + The exclusion is worth having only if it stays a single named cell. This + asserts both halves of the claim: that metafor is at the tau^2 = 0 boundary + there while PyMARE is not, and that no other profiled cell in the grid + misses :data:`RTOL_PROFILED` and :data:`ATOL_PROFILED` together. + """ + boundary = [ + case + for case in PROFILED_CASES + if (case["design"], case["model"]) == ML_BOUNDARY_CASE and case["method"] == "ML" + ] + assert len(boundary) == 1 + assert boundary[0]["tau2"] == 0.0 + assert np.ravel(fit(metafor_dataset, boundary[0]).tau2)[0] > ATOL_PROFILED + + missed = sorted( + case_id(case) + for case in PROFILED_CASES + if not np.allclose( + np.ravel(fit(metafor_dataset, case).tau2), + case["tau2"], + rtol=RTOL_PROFILED, + atol=ATOL_PROFILED, + ) + ) + assert missed == [case_id(boundary[0])] + + +def test_reference_records_the_new_quantities(): + """Every case must carry the fields this module reads. + + The companion of ``test_reference_covers_every_combination`` in + :mod:`pymare.tests.test_metafor_alignment`, which guards the grid. This + guards the columns: a regenerated reference that dropped ``I2``, or emitted + ``null`` for a Q that metafor does report, would otherwise turn these tests + into no-ops rather than failures. + """ + # int, not just float: "%.17g" writes an exact zero as `0`, so a design with + # no excess dispersion records I2 as a JSON integer. + numeric = (int, float) + for case in REFERENCE["cases"]: + for field in ("QE", "QEp", "I2", "H2"): + assert isinstance(case[field], numeric), (case_id(case), field) + # confint() has no tau^2 to bound under a fixed-effects model and is not + # asked; every other method must produce a pair of finite bounds. + bounds = (case["tau2_ci_lb"], case["tau2_ci_ub"]) + if case["method"] == "FE": + assert bounds == (None, None), case_id(case) + else: + assert all(isinstance(bound, numeric) for bound in bounds), case_id(case) + assert bounds[0] <= bounds[1], case_id(case) diff --git a/pymare/tests/test_results.py b/pymare/tests/test_results.py index 1435a68..6c27efe 100644 --- a/pymare/tests/test_results.py +++ b/pymare/tests/test_results.py @@ -150,8 +150,11 @@ def test_mrr_get_re_stats(results_2d): assert stats["tau^2"].shape == (1, 3) assert stats["ci_u"].shape == (3,) assert round(stats["tau^2"][0, 2], 4) == 7.7649 - assert round(stats["ci_l"][2], 4) == 3.8076 - assert round(stats["ci_u"][2], 2) == 59.61 + assert round(stats["ci_l"][2], 8) == 3.80759937 + # Pinned at eight decimals against metafor's confint.rma.uni, which is the + # same inversion; see test_stats.test_q_profile for why this used to be + # asserted only to two. + assert round(stats["ci_u"][2], 8) == 59.61602529 def test_mrr_get_heterogeneity_stats(results_2d): diff --git a/pymare/tests/test_stats.py b/pymare/tests/test_stats.py index 31f09ca..e52d2d7 100644 --- a/pymare/tests/test_stats.py +++ b/pymare/tests/test_stats.py @@ -187,11 +187,45 @@ def test_q_gen(vars_with_intercept): def test_q_profile(vars_with_intercept): - """Test pymare.stats.q_profile.""" + """Test pymare.stats.q_profile. + + Both bounds are pinned at eight decimals against ``metafor``'s + ``confint.rma.uni`` on the same design, which is the same inversion. The + upper one used to be asserted only to two decimals because the bound was + computed by minimizing ``(Q - crit)**2`` and landed at 59.6127 rather than + 59.6160; see :func:`pymare.stats._invert_q`. + """ bounds = stats.q_profile(*vars_with_intercept, 0.05) assert set(bounds.keys()) == {"ci_l", "ci_u"} - assert round(bounds["ci_l"], 4) == 3.8076 - assert round(bounds["ci_u"], 2) == 59.61 + assert round(bounds["ci_l"], 8) == 3.80759937 + assert round(bounds["ci_u"], 8) == 59.61602529 + + +def test_q_profile_inverts_q(vars_with_intercept): + """Each bound must put Q exactly on its critical value. + + The property the interval is defined by, checked without a reference + implementation: ``ci_l`` and ``ci_u`` are the tau^2 at which Q equals the + upper and lower chi-squared quantiles on K - P degrees of freedom. + """ + y, v, X = vars_with_intercept + bounds = stats.q_profile(y, v, X, 0.05) + df = X.shape[0] - X.shape[1] + for key, crit in (("ci_l", ss.chi2.ppf(0.975, df)), ("ci_u", ss.chi2.ppf(0.025, df))): + assert np.allclose(stats.q_gen(y, v, X, bounds[key]), crit, rtol=1e-12), key + + +def test_q_profile_returns_zero_when_q_never_crosses(): + """A bound the profile cannot reach is reported as the boundary, not a root. + + With no excess dispersion, Q at tau^2 = 0 already sits below both critical + values, so neither bound exists as a positive root. metafor reports zero + here too. + """ + y = np.array([[1.0, 1.0, 1.0, 1.0, 1.0]]).T + v = np.ones((5, 1)) + X = np.ones((5, 1)) + assert stats.q_profile(y, v, X, 0.05) == {"ci_l": 0.0, "ci_u": 0.0} def test_var_to_ci(): diff --git a/validation/metafor/run_metafor.R b/validation/metafor/run_metafor.R index f8679fc..52b49bc 100644 --- a/validation/metafor/run_metafor.R +++ b/validation/metafor/run_metafor.R @@ -1,4 +1,9 @@ -# Reference values for PyMARE's Knapp-Hartung adjustment. +# Reference values for PyMARE's rma.uni-equivalent output. +# +# Covers the whole of what rma.uni reports and PyMARE also computes: the +# fixed-effect inference path under each of metafor's three `test` settings +# (which is PyMARE's small-sample correction), the tau^2 estimate itself, the +# heterogeneity statistics, and the Q-profile confidence interval for tau^2. # # Writes pymare/tests/data/metafor_reference.json, which # pymare/tests/test_metafor_alignment.py reads. Run it through the harness in @@ -59,6 +64,10 @@ lines <- c( ' "call": "rma.uni(y, v, mods = , data = , ', 'method = , test = )",' ), + paste0( + ' "tau2_ci_call": "confint(, control = list(tol = 1e-12, ', + 'maxiter = 1000))$random, row tau^2",' + ), sprintf(' "metafor_version": "%s",', as.character(packageVersion("metafor"))), sprintf(' "r_version": "%s"', paste(R.version$major, R.version$minor, sep = ".")), " },", @@ -88,6 +97,29 @@ for (i in seq_len(nrow(cases))) { ) }) + # The Q-profile interval for tau^2, which PyMARE spells + # MetaRegressionResults.get_re_stats(method="QP"). It inverts the same Q the + # heterogeneity block reports, so it does not depend on `method` or on + # `test`; recording it per case rather than per design lets the alignment + # test assert that invariance instead of assuming it. confint() rejects a + # fixed-effects fit, which has no tau^2 to bound. + # Asked for to convergence rather than at confint()'s default tolerance. + # That default is uniroot's, .Machine$double.eps^0.25 or about 1.2e-4 + # relative, which is far coarser than the quantity itself: pinning it would + # make the alignment test agree with metafor's *display* precision instead of + # with the bound metafor is solving for. PyMARE solves the same root to a few + # multiples of machine epsilon, so the reference has to as well. + ci <- if (case$method == "FE") { + NULL + } else { + suppressWarnings(try(confint(fit, control = list(tol = 1e-12, maxiter = 1000)), silent = TRUE)) + } + tau2_ci <- if (is.null(ci) || inherits(ci, "try-error")) { + c(NA_real_, NA_real_) + } else { + c(ci$random["tau^2", "ci.lb"], ci$random["tau^2", "ci.ub"]) + } + lines <- c( lines, sprintf( @@ -100,6 +132,18 @@ for (i in seq_len(nrow(cases))) { sprintf(' "pval": [%s],', vector_json(fit$pval)), sprintf(' "ci_lb": [%s],', vector_json(fit$ci.lb)), sprintf(' "ci_ub": [%s],', vector_json(fit$ci.ub)), + # Heterogeneity. QE and its p-value are the fixed-effects Q whatever + # `method` is, so they too are recorded per case to be checked rather than + # assumed. I2 and H2 are *not* method-independent: metafor reports the + # Q-based Higgins-Thompson pair only for method="FE" and "DL", and switches + # to tau^2 / (tau^2 + vt) for the others. PyMARE always reports the Q-based + # pair, so those are the two methods it can be compared on. + sprintf(' "QE": %s,', scalar_json(fit$QE)), + sprintf(' "QEp": %s,', scalar_json(fit$QEp)), + sprintf(' "I2": %s,', scalar_json(fit$I2)), + sprintf(' "H2": %s,', scalar_json(fit$H2)), + sprintf(' "tau2_ci_lb": %s,', scalar_json(tau2_ci[[1]])), + sprintf(' "tau2_ci_ub": %s,', scalar_json(tau2_ci[[2]])), sprintf( ' "dof": %s}%s', scalar_json(fit$ddf), From f3b6d459dcc38bf943a412cb74ce5fd9f6afeab2 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 20:30:18 +0000 Subject: [PATCH 02/13] [TST] verify the effect-size converters against metafor's escalc PyMARE's effect-size converters had no reference implementation behind them. escalc computes the same conversions, so this pins its output on an eight-design grid of summary statistics -- balanced and unbalanced groups, n from 5 to 400, zero to very large effects, correlations from 0 to 0.99 -- and compares measure by measure. Six comparisons agree and are now asserted: RM <-> MN estimate and variance, exact R <-> COR estimate, exact ZR <-> ZCOR estimate and variance, exact RMD <-> MD estimate, exact sdp <-> escalc's pooled SD, recovered as MD.yi / (SMD.yi / c(m)), exact SM, SMD estimates to the bias-correction bound below metafor has no single-group standardized mean, so SM's reference is escalc(measure="SMCC") with the second measurement set to zero and uncorrelated with the first, which reduces algebraically to m / sd with the exact correction on n - 1 degrees of freedom. PyMARE corrects for bias with 1 - 3/(4m - 1) where metafor uses the exact gamma-function c(m). Rather than pin a tolerance, the tests bound the error at 0.05 / m**2 -- the approximation's actual second order, measured at 0.043 / m**2 over the grid, worst 2.7e-3 at m = 4 and 2.0e-7 at m = 398. A first-order error would break that bound. Measuring the variances turned up three defects beyond the one PR #144 fixes, all of the same kind -- a missing pair of parentheses in expressions.json changing what the expression solves to: v_rmd solves to sd1**2/n1 - sd2**2/n2 instead of the sum, so the variance of a raw mean difference is negative whenever the second group is the more variable one, and exactly zero for two equally sized equally variable groups, which gives that study infinite weight. Four of the eight grid rows are affected. v_sm solves to A + d**2 where the noncentral-t variance is A - d**2, so the single-group Hedges' g variance is 9x to 110x too large and grows with the effect rather than being dominated by 1/n. v_d (one-sample) adds n * d**2 / j**2 where the variance subtracts d**2 / j**2, so the reported variance grows with the sample size. Each is recorded as xfail(strict=True) naming the expression and what it should be, together with the two-sample v_d that PR #144 fixes. strict is the point: correcting an expression turns the test green, pytest reports XPASS as a failure, and the marker has to go in the same change. All four were confirmed to flip to XPASS under the corresponding one-line fix, and under all four fixes together the rest of the suite -- 991 tests -- still passes, so nothing currently pins the wrong values. Two variances are divergences rather than defects and are recorded as such, with a test asserting which formula each side uses so the shape of the divergence cannot change silently: - the raw correlation variance: metafor's (1 - r**2)**2 / (n - 1) is the asymptotic sampling variance, PyMARE's (1 - r**2) / (n - 2) is the squared standard error under the null of no correlation. They are 51x apart at r = 0.99, so no tolerance relates them. - the single-group standardized-mean variances: metafor's are large-sample approximations, PyMARE's are the exact noncentral-t expressions and are the better quantity. They are 2.5x apart at n = 5, so only an order-of-magnitude bound can span them. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011o2j5ZNAdzPFL6LBqDqUsU --- pymare/tests/data/metafor_escalc_inputs.csv | 9 + .../tests/data/metafor_escalc_reference.json | 83 +++ pymare/tests/test_metafor_escalc.py | 474 ++++++++++++++++++ validation/metafor/run_escalc.R | 140 ++++++ 4 files changed, 706 insertions(+) create mode 100644 pymare/tests/data/metafor_escalc_inputs.csv create mode 100644 pymare/tests/data/metafor_escalc_reference.json create mode 100644 pymare/tests/test_metafor_escalc.py create mode 100644 validation/metafor/run_escalc.R diff --git a/pymare/tests/data/metafor_escalc_inputs.csv b/pymare/tests/data/metafor_escalc_inputs.csv new file mode 100644 index 0000000..2d1802f --- /dev/null +++ b/pymare/tests/data/metafor_escalc_inputs.csv @@ -0,0 +1,9 @@ +case,m,sd,n,r,m1,m2,sd1,sd2,n1,n2 +balanced_small,0.8,1.0,10,0.5,0.8,0.2,1.0,1.0,10,10 +balanced_large,0.5,1.2,200,0.05,0.5,0.3,1.2,1.1,200,200 +unbalanced,2.0,0.9,12,-0.4,2.0,1.4,0.9,1.3,12,90 +very_unbalanced,1.0,0.7,5,0.95,1.0,0.1,0.7,1.5,5,400 +zero_effect,0.0,1.0,40,0.0,0.5,0.5,1.0,1.0,40,40 +large_effect,3.0,1.0,25,0.99,3.0,0.5,1.0,1.1,25,25 +heteroscedastic,1.0,0.3,30,-0.9,1.0,0.5,0.3,2.5,30,30 +tiny,1.0,1.0,5,-0.6,1.0,0.0,1.0,1.0,5,5 diff --git a/pymare/tests/data/metafor_escalc_reference.json b/pymare/tests/data/metafor_escalc_reference.json new file mode 100644 index 0000000..486720f --- /dev/null +++ b/pymare/tests/data/metafor_escalc_reference.json @@ -0,0 +1,83 @@ +{ + "source": { + "data": "metafor_escalc_inputs.csv", + "call": "escalc(measure = , ...)", + "sm_call": "escalc(measure = \"SMCC\", m1i = m, m2i = 0, sd1i = sd, sd2i = 0, ni = n, ri = 0)", + "metafor_version": "4.6.0", + "r_version": "4.4.1" + }, + "cases": [ + {"case": "balanced_small", + "MN": {"yi": 0.80000000000000004, "vi": 0.10000000000000001}, + "COR": {"yi": 0.5, "vi": 0.0625}, + "ZCOR": {"yi": 0.54930614433405478, "vi": 0.14285714285714285}, + "SMCC": {"yi": 0.73109991343404213, "vi": 0.12672535417116321}, + "MD": {"yi": 0.60000000000000009, "vi": 0.20000000000000001}, + "SMD": {"yi": 0.57458785621423214, "vi": 0.20825378011272169}, + "cm_one_sample": 0.91387489179255255, + "cm_two_sample": 0.95764642702372005}, + {"case": "balanced_large", + "MN": {"yi": 0.5, "vi": 0.0071999999999999998}, + "COR": {"yi": 0.050000000000000003, "vi": 0.0050000314070351767}, + "ZCOR": {"yi": 0.050041729278491272, "vi": 0.005076142131979695}, + "SMCC": {"yi": 0.41509400959365378, "vi": 0.0054307575920013408}, + "MD": {"yi": 0.20000000000000001, "vi": 0.013250000000000001}, + "SMD": {"yi": 0.1734212407075082, "vi": 0.010037593658410665}, + "cm_one_sample": 0.99622562302476902, + "cm_two_sample": 0.99811419581258609}, + {"case": "unbalanced", + "MN": {"yi": 2, "vi": 0.067500000000000004}, + "COR": {"yi": -0.40000000000000002, "vi": 0.064145454545454533}, + "ZCOR": {"yi": -0.42364893019360184, "vi": 0.1111111111111111}, + "SMCC": {"yi": 2.0665773555017304, "vi": 0.26128091526135522}, + "MD": {"yi": 0.60000000000000009, "vi": 0.086277777777777787}, + "SMD": {"yi": 0.47177727137710784, "vi": 0.095535492453209289}, + "cm_one_sample": 0.92995980997577854, + "cm_two_sample": 0.99247805498141328}, + {"case": "very_unbalanced", + "MN": {"yi": 1, "vi": 0.09799999999999999}, + "COR": {"yi": 0.94999999999999996, "vi": 0.0023765625000000015}, + "ZCOR": {"yi": 1.8317808230648227, "vi": 0.5}, + "SMCC": {"yi": 1.1398350868612361, "vi": 0.32992240252399618}, + "MD": {"yi": 0.90000000000000002, "vi": 0.10362499999999999}, + "SMD": {"yi": 0.60122105154316374, "vi": 0.20294625525039342}, + "cm_one_sample": 0.79788456080286529, + "cm_two_sample": 0.99813760983513533}, + {"case": "zero_effect", + "MN": {"yi": 0, "vi": 0.025000000000000001}, + "COR": {"yi": 0, "vi": 0.02564102564102564}, + "ZCOR": {"yi": 0, "vi": 0.027027027027027029}, + "SMCC": {"yi": 0, "vi": 0.025000000000000001}, + "MD": {"yi": 0, "vi": 0.050000000000000003}, + "SMD": {"yi": 0, "vi": 0.050000000000000003}, + "cm_one_sample": 0.98062423868033832, + "cm_two_sample": 0.99034851305326044}, + {"case": "large_effect", + "MN": {"yi": 3, "vi": 0.040000000000000001}, + "COR": {"yi": 0.98999999999999999, "vi": 1.6500416666666716e-05}, + "ZCOR": {"yi": 2.6466524123622457, "vi": 0.045454545454545456}, + "SMCC": {"yi": 2.9050957003431512, "vi": 0.20879162056304532}, + "MD": {"yi": 2.5, "vi": 0.088400000000000006}, + "SMD": {"yi": 2.3408698989159209, "vi": 0.13479671883650635}, + "cm_one_sample": 0.96836523344771708, + "cm_two_sample": 0.98427942629592347}, + {"case": "heteroscedastic", + "MN": {"yi": 1, "vi": 0.0030000000000000001}, + "COR": {"yi": -0.90000000000000002, "vi": 0.0012448275862068957}, + "ZCOR": {"yi": -1.4722194895832204, "vi": 0.037037037037037035}, + "SMCC": {"yi": 3.2462499486557514, "vi": 0.20896897881912449}, + "MD": {"yi": 0.5, "vi": 0.21133333333333335}, + "SMD": {"yi": 0.27717822008775811, "vi": 0.067306898047425151}, + "cm_one_sample": 0.97387498459672539, + "cm_two_sample": 0.98700358102800445}, + {"case": "tiny", + "MN": {"yi": 1, "vi": 0.20000000000000001}, + "COR": {"yi": -0.59999999999999998, "vi": 0.1024}, + "ZCOR": {"yi": -0.69314718055994529, "vi": 0.5}, + "SMCC": {"yi": 0.79788456080286529, "vi": 0.2636619772367581}, + "MD": {"yi": 1, "vi": 0.40000000000000002}, + "SMD": {"yi": 0.90270333367640987, "vi": 0.44074366543152521}, + "cm_one_sample": 0.79788456080286529, + "cm_two_sample": 0.90270333367640987} + ] +} diff --git a/pymare/tests/test_metafor_escalc.py b/pymare/tests/test_metafor_escalc.py new file mode 100644 index 0000000..7800645 --- /dev/null +++ b/pymare/tests/test_metafor_escalc.py @@ -0,0 +1,474 @@ +"""Alignment between PyMARE's effect-size converters and metafor's ``escalc``. + +:class:`~pymare.effectsize.OneSampleEffectSizeConverter` and +:class:`~pymare.effectsize.TwoSampleEffectSizeConverter` turn study-level summary +statistics into an estimate and its sampling variance, which is what +``metafor::escalc`` does. This module compares the two, measure by measure, and +is deliberately explicit about which of the three relationships each pair is in: + +**The same closed form.** The raw mean and its variance, the raw correlation, the +Fisher z-transformed correlation and its variance, the raw mean difference, and +the pooled standard deviation. These are compared at :data:`RTOL_EXACT` and a +failure means one of the two is wrong. + +**The same quantity, one of them approximated.** PyMARE corrects a standardized +mean for bias with ``1 - 3/(4m - 1)``; metafor uses the exact +``c(m) = gamma(m/2) / (sqrt(m/2) gamma((m-1)/2))``. The approximation is +second-order accurate in ``m``, so rather than pinning a tolerance these tests +assert the error stays under :data:`BIAS_CORRECTION_BOUND` divided by ``m**2``, +which is a statement about the approximation and not about either +implementation's current output. + +**Different formulas for the same thing.** metafor takes the sampling variance of +a correlation to be ``(1 - r**2)**2 / (n - 1)`` and PyMARE takes it to be +``(1 - r**2) / (n - 2)``; metafor's standardized-mean variances are large-sample +approximations where PyMARE's are the exact noncentral-t expressions. These +cannot be verified against each other, so +:func:`test_raw_correlation_variance_is_a_different_formula` and +``validation/metafor/README.md`` record what each one is instead. + +Four comparisons are marked :func:`pytest.mark.xfail` with ``strict=True``, +because measuring these turned up defects rather than divergences: three +sampling-variance expressions in ``pymare/effectsize/expressions.json`` have +misplaced parentheses or a sign error, one of which is the subject of +https://github.com/neurostuff/PyMARE/pull/144. Each marker names the expression +and what it should be. ``strict=True`` is the point: when one is corrected the +test passes, pytest reports XPASS as a failure, and the marker has to be removed +in the same change. + +""" + +import json +import os.path as op + +import numpy as np +import pytest + +from pymare.effectsize import OneSampleEffectSizeConverter, TwoSampleEffectSizeConverter +from pymare.tests.utils import get_test_data_path + +pytestmark = pytest.mark.metafor + +with open(op.join(get_test_data_path(), "metafor_escalc_reference.json")) as _fobj: + REFERENCE = json.load(_fobj) + +CASES = REFERENCE["cases"] + +#: Tolerance for the measures both implementations reach by the same closed +#: form. Observed exact on every row of the input grid; this allows for the +#: reassociation a compiler or a different order of operations can introduce. +RTOL_EXACT = 1e-13 + +#: Absolute floor for the rows where the true value is zero -- the grid includes +#: a zero-effect design so that the bias-corrected measures are checked where +#: the correction has nothing to scale. +ATOL_EXACT = 1e-15 + +#: Numerator of the bound on PyMARE's bias-correction approximation. The error +#: of ``1 - 3/(4m - 1)`` against the exact ``c(m)`` falls as ``m**-2``, with a +#: constant measured at 0.043 over the grid -- worst 2.7e-3 at ``m = 4``, the +#: smallest this grid goes, and 2.0e-7 at ``m = 398``. Rounded up to 0.05, which +#: leaves the bound tight enough that a first-order error would break it. +BIAS_CORRECTION_BOUND = 0.05 + +#: Bound on the relative difference between PyMARE's and metafor's *corrected* +#: two-sample standardized-mean-difference variance, as a multiple of +#: ``1 / (n1 + n2)``. The two use different approximations of the same quantity +#: -- PyMARE ``d**2 / (2 (n1 + n2 - 2))`` scaled by ``j**2``, metafor +#: ``g**2 / (2 (n1 + n2))`` -- which differ at order ``1 / N``. Worst observed +#: 0.72 of this bound, at ``n1 = n2 = 5``. +SMD_VARIANCE_BOUND = 2.0 + +#: Factor within which PyMARE's exact standardized-mean variances must sit +#: relative to metafor's large-sample approximations. Not a tolerance: the two +#: are different expressions and disagree by 2.5x at ``n = 5``, where the +#: ``(n - 1) / (n - 3)`` inflation in the exact form is largest. It is a bound +#: loose enough to be satisfied by any implementation of the right quantity and +#: tight enough to catch the sign error the markers below describe, which puts +#: PyMARE out by one to two orders of magnitude. +SAME_ORDER_FACTOR = 4.0 + + +def case_ids(): + """Name each row of the input grid.""" + return [case["case"] for case in CASES] + + +@pytest.fixture(scope="module") +def escalc_inputs(): + """Load the summary statistics the escalc reference values were computed on.""" + import pandas as pd + + return pd.read_csv(op.join(get_test_data_path(), "metafor_escalc_inputs.csv")) + + +@pytest.fixture(scope="module") +def one_sample(escalc_inputs): + """Build a converter over the single-group columns of the input grid.""" + return OneSampleEffectSizeConverter( + m=escalc_inputs["m"].to_numpy(), + sd=escalc_inputs["sd"].to_numpy(), + n=escalc_inputs["n"].to_numpy(), + ) + + +@pytest.fixture(scope="module") +def correlations(escalc_inputs): + """Build a converter over the correlation columns of the input grid. + + Separate from :func:`one_sample` because the correlation measures are solved + from ``r`` and ``n`` while the mean measures are solved from ``m``, ``sd`` + and ``n``; one converter holding all five would let the solver reach a + measure by a path the documented one does not offer. + """ + return OneSampleEffectSizeConverter( + r=escalc_inputs["r"].to_numpy(), n=escalc_inputs["n"].to_numpy() + ) + + +@pytest.fixture(scope="module") +def two_sample(escalc_inputs): + """Build a converter over the two-group columns of the input grid.""" + return TwoSampleEffectSizeConverter( + m1=escalc_inputs["m1"].to_numpy(), + m2=escalc_inputs["m2"].to_numpy(), + sd1=escalc_inputs["sd1"].to_numpy(), + sd2=escalc_inputs["sd2"].to_numpy(), + n1=escalc_inputs["n1"].to_numpy(), + n2=escalc_inputs["n2"].to_numpy(), + ) + + +def measure(converter, name): + """Return one measure and its variance as a pair of 1-D arrays.""" + dataset = converter.to_dataset(measure=name) + return np.ravel(dataset.y), np.ravel(dataset.v) + + +def expected(field, key): + """Collect one escalc column across the grid, in input order.""" + return np.array([case[field][key] for case in CASES], dtype=float) + + +def reference(field): + """Collect one scalar reference column across the grid, in input order.""" + return np.array([case[field] for case in CASES], dtype=float) + + +def assert_exact(got, want, label): + """Compare a whole column at :data:`RTOL_EXACT`, naming the rows that miss.""" + close = np.isclose(got, want, rtol=RTOL_EXACT, atol=ATOL_EXACT) + missed = [ + f"{case['case']}: {g!r} != {w!r}" + for case, g, w, ok in zip(CASES, got, want, close) + if not ok + ] + assert not missed, f"{label}: " + "; ".join(missed) + + +# ----------------------------------------------------------------------------- +# The same closed form on both sides. +# ----------------------------------------------------------------------------- + + +def test_raw_mean_matches_metafor(one_sample): + """``RM`` must be ``escalc(measure="MN")``, estimate and variance alike.""" + y, v = measure(one_sample, "RM") + assert_exact(y, expected("MN", "yi"), "RM estimate") + assert_exact(v, expected("MN", "vi"), "RM variance") + + +def test_raw_correlation_matches_metafor(correlations): + """``R``'s estimate must be ``escalc(measure="COR")``'s. + + The estimate only. The two variances are different formulas, which + :func:`test_raw_correlation_variance_is_a_different_formula` records. + """ + y, _ = measure(correlations, "R") + assert_exact(y, expected("COR", "yi"), "R estimate") + + +def test_fisher_z_correlation_matches_metafor(correlations): + """``ZR`` must be ``escalc(measure="ZCOR")``, estimate and variance alike. + + The one transformed correlation measure where PyMARE and metafor agree on + both halves: ``atanh(r)`` and ``1 / (n - 3)``. + """ + y, v = measure(correlations, "ZR") + assert_exact(y, expected("ZCOR", "yi"), "ZR estimate") + assert_exact(v, expected("ZCOR", "vi"), "ZR variance") + + +def test_raw_mean_difference_matches_metafor(two_sample): + """``RMD``'s estimate must be ``escalc(measure="MD")``'s.""" + y, _ = measure(two_sample, "RMD") + assert_exact(y, expected("MD", "yi"), "RMD estimate") + + +def test_pooled_standard_deviation_matches_metafor(two_sample, escalc_inputs): + """The pooled SD must be the one metafor divides by. + + ``escalc`` does not report ``sdpi``, but it is recoverable from what it does + report: the bias-corrected estimate divided by the exact correction factor is + the raw Cohen's d, and the raw mean difference over that is the pooled SD. + Checking it separately means a failure in + :func:`test_standardized_mean_difference_matches_metafor` can be read as + being about the correction factor rather than about the pooling. + """ + # The zero-effect row has m1 == m2, so d is zero and the quotient is not + # defined. Its pooled SD is covered by every other row's. + varies = escalc_inputs["m1"].to_numpy() != escalc_inputs["m2"].to_numpy() + metafor_d = expected("SMD", "yi") / reference("cm_two_sample") + metafor_sdp = expected("MD", "yi")[varies] / metafor_d[varies] + got = np.ravel(two_sample.get("sdp"))[varies] + assert np.allclose(got, metafor_sdp, rtol=RTOL_EXACT) + + +# ----------------------------------------------------------------------------- +# The same quantity, PyMARE approximating metafor's exact correction factor. +# ----------------------------------------------------------------------------- + + +def bias_correction_bound(dof): + """Return the allowed relative error of the correction factor at ``m = dof``.""" + return BIAS_CORRECTION_BOUND / np.asarray(dof, dtype=float) ** 2 + + +def test_bias_correction_approximates_the_exact_factor(one_sample, two_sample, escalc_inputs): + """``1 - 3/(4m - 1)`` must approximate ``c(m)`` to second order. + + Checked directly on the factor rather than only through the measures that + use it, so that a change to the approximation is attributed here rather than + showing up as a drifting tolerance on two other tests. Both converters + expose the factor as ``j``; the degrees of freedom are ``n - 1`` for a single + group and ``n1 + n2 - 2`` for two. + """ + single_dof = escalc_inputs["n"].to_numpy() - 1 + pair_dof = escalc_inputs["n1"].to_numpy() + escalc_inputs["n2"].to_numpy() - 2 + for converter, dof, exact in ( + (one_sample, single_dof, reference("cm_one_sample")), + (two_sample, pair_dof, reference("cm_two_sample")), + ): + approximate = np.ravel(converter.get("j")) + error = np.abs(approximate - exact) / exact + assert np.all(error <= bias_correction_bound(dof)), list( + zip(case_ids(), error, bias_correction_bound(dof)) + ) + + +def test_standardized_mean_matches_metafor(one_sample, escalc_inputs): + """``SM`` must be metafor's single-group standardized mean. + + metafor has no single-group standardized mean, so the reference is + ``escalc(measure="SMCC")`` with the second measurement set to zero and + uncorrelated with the first, which reduces algebraically to ``m / sd`` with + the exact correction applied on ``n - 1`` degrees of freedom. See the header + of ``validation/metafor/run_escalc.R``. + + The estimate only, and only to the bias-correction bound: the two agree + exactly on ``m / sd`` and differ only in the factor multiplying it. + """ + y, _ = measure(one_sample, "SM") + want = expected("SMCC", "yi") + error = np.abs(y - want) / np.where(want == 0, 1.0, np.abs(want)) + assert np.all(error <= bias_correction_bound(escalc_inputs["n"].to_numpy() - 1)) + + +def test_standardized_mean_difference_matches_metafor(two_sample, escalc_inputs): + """``SMD``'s estimate must be ``escalc(measure="SMD")``'s. + + To the bias-correction bound, for the same reason as + :func:`test_standardized_mean_matches_metafor`: the raw ``d`` and the pooled + SD agree exactly, as the two tests above establish, so the whole of the + difference here is the correction factor. + """ + y, _ = measure(two_sample, "SMD") + want = expected("SMD", "yi") + error = np.abs(y - want) / np.where(want == 0, 1.0, np.abs(want)) + dof = escalc_inputs["n1"].to_numpy() + escalc_inputs["n2"].to_numpy() - 2 + assert np.all(error <= bias_correction_bound(dof)) + + +# ----------------------------------------------------------------------------- +# Different formulas for the same thing, recorded rather than compared. +# ----------------------------------------------------------------------------- + + +def test_raw_correlation_variance_is_a_different_formula(correlations, escalc_inputs): + """Record that ``R``'s variance is not metafor's, and which is which. + + metafor uses the asymptotic sampling variance of a correlation, + ``(1 - r**2)**2 / (n - 1)``. PyMARE uses ``(1 - r**2) / (n - 2)``, which is + the squared standard error of ``r`` under the null hypothesis of no + correlation rather than its sampling variance at the observed value. The two + diverge without limit as ``|r|`` approaches one -- 51x apart at ``r = 0.99`` + on this grid -- so no tolerance relates them. + + This asserts that each side is the formula named above and nothing more. It + exists so that the divergence cannot quietly change shape: if either + expression were replaced, this test would say so rather than a tolerance + somewhere else drifting. + """ + _, v = measure(correlations, "R") + r = escalc_inputs["r"].to_numpy() + n = escalc_inputs["n"].to_numpy() + assert np.allclose(v, (1 - r**2) / (n - 2), rtol=RTOL_EXACT) + assert np.allclose(expected("COR", "vi"), (1 - r**2) ** 2 / (n - 1), rtol=RTOL_EXACT) + + +def test_standardized_mean_variances_are_exact_not_asymptotic(one_sample, escalc_inputs): + """Record that the single-group standardized-mean variances are not metafor's. + + metafor reports ``1 / n + y**2 / (2n)``, the large-sample approximation. + PyMARE's expressions are the exact noncentral-t variance, + ``(n - 1)/(n - 3) (1/n + d**2) - d**2 / c**2``, which is the better quantity + but not the same one: the two are 2.5x apart at ``n = 5``. So the single- + group standardized-mean variance has no usable metafor reference, and + :func:`test_standardized_mean_variance_is_the_same_order_as_metafor` is the + most that can be asserted across the two. + + This test pins metafor's side of that statement, which is the half that can + be checked exactly, so that the factor-of-four bound in the other test is + known to be comparing against the asymptotic formula and not against + something else metafor might report in a future release. + """ + n = escalc_inputs["n"].to_numpy() + y = expected("SMCC", "yi") + assert np.allclose(expected("SMCC", "vi"), 1 / n + y**2 / (2 * n), rtol=RTOL_EXACT) + + +# ----------------------------------------------------------------------------- +# Defects, recorded as strict xfails until the expressions are corrected. +# ----------------------------------------------------------------------------- + + +@pytest.mark.xfail( + strict=True, + reason=( + "v_rmd in pymare/effectsize/expressions.json reads " + "'v_rmd - (sd1**2 / n1) + (sd2**2 / n2)', which solves to " + "sd1**2/n1 - sd2**2/n2 rather than the sum. The variance comes out " + "negative whenever the second group is the more variable one, and " + "exactly zero for two equally sized, equally variable groups -- which " + "gives that study infinite weight. The fix is to parenthesize the " + "denominator, as PR #144 does for the two-sample Cohen's d" + ), +) +def test_raw_mean_difference_variance_matches_metafor(two_sample): + """``RMD``'s variance must be ``escalc(measure="MD")``'s, exactly. + + ``sd1**2 / n1 + sd2**2 / n2`` on both sides, with nothing approximated, so + this is an exact comparison once the expression is corrected: it was + verified to agree on every row of the grid to zero relative error with the + parentheses in place. + """ + _, v = measure(two_sample, "RMD") + assert_exact(v, expected("MD", "vi"), "RMD variance") + + +@pytest.mark.xfail( + strict=True, + reason=( + "v_d in pymare/effectsize/expressions.json reads " + "'v_d - ((n1 + n2)/(n1 * n2) + d**2 / 2 * (n1 + n2 - 2))', so the " + "squared-effect term is multiplied by the residual degrees of freedom " + "instead of divided by twice them. Fixed by PR #144; remove this " + "marker with it. Issue #143" + ), +) +def test_standardized_mean_difference_variance_matches_metafor(two_sample, escalc_inputs): + """``SMD``'s variance must agree with metafor's to order ``1 / N``. + + Not exactly, even with the expression corrected: PyMARE scales + ``d**2 / (2 (n1 + n2 - 2))`` by ``j**2`` and metafor adds + ``g**2 / (2 (n1 + n2))``, two approximations of the same variance that + differ at order ``1 / N``. :data:`SMD_VARIANCE_BOUND` is that order, + measured at 0.72 of the bound in the worst cell of the grid. + """ + _, v = measure(two_sample, "SMD") + want = expected("SMD", "vi") + total = escalc_inputs["n1"].to_numpy() + escalc_inputs["n2"].to_numpy() + error = np.abs(v - want) / want + assert np.all(error <= SMD_VARIANCE_BOUND / total), list(zip(case_ids(), error)) + + +@pytest.mark.xfail( + strict=True, + reason=( + "v_sm in pymare/effectsize/expressions.json reads " + "'v_sm - ((n - 1)/(n - 3)) * j**2 * (1 / n + d**2) - d**2', which " + "solves to A + d**2 where the noncentral-t variance is A - d**2. The " + "reported variance is two orders of magnitude too large for any " + "appreciable effect, and grows with the effect instead of being " + "dominated by 1/n" + ), +) +def test_standardized_mean_variance_is_the_same_order_as_metafor(one_sample): + """``SM``'s variance must be within a factor of metafor's approximation. + + A factor and not a tolerance, for the reason + :func:`test_standardized_mean_variances_are_exact_not_asymptotic` gives: the + exact and asymptotic expressions are 2.5x apart at ``n = 5``. Anything + outside :data:`SAME_ORDER_FACTOR` is not two approximations of one variance + disagreeing, and that is what this is here to catch. + """ + _, v = measure(one_sample, "SM") + ratio = v / expected("SMCC", "vi") + assert np.all(ratio <= SAME_ORDER_FACTOR), list(zip(case_ids(), ratio)) + assert np.all(ratio >= 1 / SAME_ORDER_FACTOR), list(zip(case_ids(), ratio)) + + +@pytest.mark.xfail( + strict=True, + reason=( + "v_d (one-sample) in pymare/effectsize/expressions.json reads " + "'v_d - ((n - 1)/(n - 3)) * (1 / n + d**2) - d**2 / j**2 * n', so the " + "last term is added and scaled by n where the noncentral-t variance " + "subtracts d**2 / j**2. The reported variance therefore grows with the " + "sample size, which is backwards" + ), +) +def test_one_sample_d_variance_is_the_same_order_as_metafor(one_sample): + """``D``'s variance must be within a factor of metafor's approximation. + + metafor reports no uncorrected d, so its variance is recovered by undoing + the exact correction: ``Var(d) = Var(g) / c(m)**2``. Bounded by a factor for + the same reason as :func:`test_standardized_mean_variance_is_the_same_order_as_metafor`. + """ + _, v = measure(one_sample, "D") + metafor_variance = expected("SMCC", "vi") / reference("cm_one_sample") ** 2 + ratio = v / metafor_variance + assert np.all(ratio <= SAME_ORDER_FACTOR), list(zip(case_ids(), ratio)) + assert np.all(ratio >= 1 / SAME_ORDER_FACTOR), list(zip(case_ids(), ratio)) + + +# ----------------------------------------------------------------------------- +# Guards on the reference itself. +# ----------------------------------------------------------------------------- + + +def test_reference_covers_every_input_row(escalc_inputs): + """The pinned cases must be the input grid, in order and complete. + + Every test above lines a PyMARE column up against a reference column by + position, so a reference that dropped or reordered a row would compare the + wrong pairs rather than fail. The designs are named here as well, so that + dropping one from the CSV is a failure and not a quieter check. + """ + assert case_ids() == list(escalc_inputs["case"]) + assert case_ids() == [ + "balanced_small", + "balanced_large", + "unbalanced", + "very_unbalanced", + "zero_effect", + "large_effect", + "heteroscedastic", + "tiny", + ] + for case in CASES: + for field in ("MN", "COR", "ZCOR", "SMCC", "MD", "SMD"): + assert set(case[field]) == {"yi", "vi"}, (case["case"], field) + assert all(isinstance(value, (int, float)) for value in case[field].values()) + for field in ("cm_one_sample", "cm_two_sample"): + assert 0 < case[field] < 1, (case["case"], field) diff --git a/validation/metafor/run_escalc.R b/validation/metafor/run_escalc.R new file mode 100644 index 0000000..86a3232 --- /dev/null +++ b/validation/metafor/run_escalc.R @@ -0,0 +1,140 @@ +# Reference values for PyMARE's effect-size converters. +# +# Writes pymare/tests/data/metafor_escalc_reference.json, which +# pymare/tests/test_metafor_escalc.py reads. Run it through the harness in this +# directory rather than directly, so the R and metafor versions are the pinned +# ones: +# +# validation/metafor/regenerate.sh +# +# metafor is not a test dependency, so the numbers are pinned rather than +# recomputed on every test run. The alignment workflow regenerates them and +# fails on any difference beyond the tolerances in +# validation/compare_reference.py, which is what keeps the pin honest. +# +# The output is written by hand rather than with jsonlite so the formatting is +# byte-stable: every number goes through "%.17g", which round-trips a double +# exactly. +# +# Which escalc measure corresponds to which PyMARE measure is not always +# obvious, so each one is named here with the PyMARE spelling it is the +# reference for: +# +# escalc measure PyMARE measure converter +# -------------- --------------------------------- ---------- +# MN RM (raw mean) one-sample +# COR R (raw correlation) one-sample +# ZCOR ZR (Fisher z correlation) one-sample +# SMCC, see below SM (standardized mean) one-sample +# MD RMD (raw mean difference) two-sample +# SMD SMD (standardized mean difference) two-sample +# +# metafor has no single-group standardized mean, so SM is referenced against +# measure="SMCC" -- the standardized mean *change* -- with the second +# measurement set to zero and uncorrelated with the first. SMCC's change score +# standard deviation is then sd1 and its numerator is m1, so the measure reduces +# algebraically to the single-group m / sd with metafor's exact bias correction +# applied on n - 1 degrees of freedom. That is precisely the quantity PyMARE +# calls SM, which is why the substitution is a reference and not an +# approximation. +# +# PyMARE's "D" (two-sample) and one-sample Cohen's d have no escalc counterpart: +# every standardized measure metafor offers is bias-corrected. They are +# recoverable as yi / cm, and the exact correction factors are recorded below +# for that reason as well as to bound PyMARE's approximation of them. +library(metafor) + +args <- commandArgs(trailingOnly = TRUE) +csv_path <- if (length(args) >= 1) args[[1]] else "/data/metafor_escalc_inputs.csv" +out_path <- if (length(args) >= 2) args[[2]] else "/data/metafor_escalc_reference.json" + +d <- read.csv(csv_path) + +# metafor's exact bias correction, c(m) = gamma(m/2) / (sqrt(m/2) gamma((m-1)/2)), +# through lgamma so that it does not overflow at the large sample sizes in the +# input. PyMARE approximates this with 1 - 3/(4m - 1); recording the exact value +# is what lets the alignment test bound that approximation rather than merely +# observe it. +cm <- function(m) exp(lgamma(m / 2) - log(sqrt(m / 2)) - lgamma((m - 1) / 2)) + +one <- escalc(measure = "MN", mi = d$m, sdi = d$sd, ni = d$n) +cor_raw <- escalc(measure = "COR", ri = d$r, ni = d$n) +cor_z <- escalc(measure = "ZCOR", ri = d$r, ni = d$n) + +# The SMCC reduction described in the header comment. m2i and sd2i are zero and +# ri is zero, so sddi = sqrt(sd1^2 + 0 - 0) = sd1. +std_mean <- escalc( + measure = "SMCC", + m1i = d$m, m2i = rep(0, nrow(d)), + sd1i = d$sd, sd2i = rep(0, nrow(d)), + ni = d$n, ri = rep(0, nrow(d)) +) + +mean_diff <- escalc( + measure = "MD", + m1i = d$m1, m2i = d$m2, sd1i = d$sd1, sd2i = d$sd2, n1i = d$n1, n2i = d$n2 +) +std_mean_diff <- escalc( + measure = "SMD", + m1i = d$m1, m2i = d$m2, sd1i = d$sd1, sd2i = d$sd2, n1i = d$n1, n2i = d$n2 +) + +scalar_json <- function(x) { + if (length(x) == 0 || is.null(x) || all(is.na(x))) "null" else sprintf("%.17g", x[[1]]) +} + +lines <- c( + "{", + ' "source": {', + sprintf(' "data": "%s",', basename(csv_path)), + ' "call": "escalc(measure = , ...)",', + paste0( + ' "sm_call": "escalc(measure = \\"SMCC\\", m1i = m, m2i = 0, sd1i = sd, ', + 'sd2i = 0, ni = n, ri = 0)",' + ), + sprintf(' "metafor_version": "%s",', as.character(packageVersion("metafor"))), + sprintf(' "r_version": "%s"', paste(R.version$major, R.version$minor, sep = ".")), + " },", + ' "cases": [' +) + +for (i in seq_len(nrow(d))) { + lines <- c( + lines, + sprintf(' {"case": "%s",', d$case[[i]]), + sprintf(' "MN": {"yi": %s, "vi": %s},', scalar_json(one$yi[i]), scalar_json(one$vi[i])), + sprintf( + ' "COR": {"yi": %s, "vi": %s},', + scalar_json(cor_raw$yi[i]), scalar_json(cor_raw$vi[i]) + ), + sprintf( + ' "ZCOR": {"yi": %s, "vi": %s},', + scalar_json(cor_z$yi[i]), scalar_json(cor_z$vi[i]) + ), + sprintf( + ' "SMCC": {"yi": %s, "vi": %s},', + scalar_json(std_mean$yi[i]), scalar_json(std_mean$vi[i]) + ), + sprintf( + ' "MD": {"yi": %s, "vi": %s},', + scalar_json(mean_diff$yi[i]), scalar_json(mean_diff$vi[i]) + ), + sprintf( + ' "SMD": {"yi": %s, "vi": %s},', + scalar_json(std_mean_diff$yi[i]), scalar_json(std_mean_diff$vi[i]) + ), + # The exact correction factors, on the degrees of freedom each measure uses: + # n - 1 for the single-group standardized mean, n1 + n2 - 2 for the + # two-group one. + sprintf(' "cm_one_sample": %s,', scalar_json(cm(d$n[[i]] - 1))), + sprintf( + ' "cm_two_sample": %s}%s', + scalar_json(cm(d$n1[[i]] + d$n2[[i]] - 2)), + if (i < nrow(d)) "," else "" + ) + ) +} + +lines <- c(lines, " ]", "}") +writeLines(lines, out_path) +cat(sprintf("wrote %d cases to %s\n", nrow(d), out_path)) From 01e88edff535f84d1527b6ba99330be699926492 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 20:35:38 +0000 Subject: [PATCH 03/13] [TST] verify the permutation test against metafor's permutest PyMARE's permutation test had no reference implementation behind it. permutest does the same thing, and with exact = TRUE its p-value is a property of the data rather than of a generator, so it can be pinned: ten cases, being the designs small enough to enumerate -- 2^K sign flips for the intercept-only models up to K = 10, and K! orderings for the moderator ones at K = 5. The two disagree, for two reasons that together account for every counted permutation. The statistic. permutest counts |beta / se|; PyMARE counts |beta|. A permutation test needs a statistic whose null distribution does not move with what is being permuted away, and |beta| does move: refitting a permuted dataset re-estimates tau^2, which changes the weights and so the standard error. The two coincide only where the standard error happens to be invariant -- a fixed-effects intercept-only model under sign flipping, which is why four of the ten cases agree anyway. On unequal_k5 under DerSimonianLaird, PyMARE reports 0.5625 against permutest's 0.5. The tie. The observed estimate is computed by a different code path from the permuted ones and the two differ by one unit in the last place, so the inclusive comparison can drop the identity permutation -- the one that reproduces the observed data and must therefore count -- along with its sign-flipped mirror. That understates the p-value by 2/2^K whenever it bites: 0.033203125 against permutest's 0.03515625 on extreme_k10. metafor avoids this by comparing against |zval| - sqrt(eps). test_metafor_permutest_is_reproduced_by_the_z_statistic is the load-bearing test and it passes. It counts the same permutations of the same PyMARE fits, changing only the statistic to |beta / se| and reading the observed value out of the identity permutation in the same batch, and reproduces permutest exactly in all ten cases -- both coefficients of the moderator models included. So the permutation sets agree, the batched refits agree, and the inclusive comparison agrees; the disagreement is entirely in the statistic and the tie, which is also what the fix would be. What PyMARE currently reports is asserted by test_permutation_p_value_matches_metafor, a strict xfail naming both causes. It covers all ten cases in one test rather than parametrizing, because one of the two causes is a last-place rounding difference that need not reproduce on every platform in the test matrix while the other is structural: asserting them together keeps the xfail driven by the structural cause and out of reach of an unexpected pass on some runner. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011o2j5ZNAdzPFL6LBqDqUsU --- .../data/metafor_permutest_reference.json | 50 ++++ pymare/tests/test_metafor_permutest.py | 226 ++++++++++++++++++ validation/metafor/run_permutest.R | 101 ++++++++ 3 files changed, 377 insertions(+) create mode 100644 pymare/tests/data/metafor_permutest_reference.json create mode 100644 pymare/tests/test_metafor_permutest.py create mode 100644 validation/metafor/run_permutest.R diff --git a/pymare/tests/data/metafor_permutest_reference.json b/pymare/tests/data/metafor_permutest_reference.json new file mode 100644 index 0000000..06f5fc9 --- /dev/null +++ b/pymare/tests/data/metafor_permutest_reference.json @@ -0,0 +1,50 @@ +{ + "source": { + "data": "metafor_small_sample.csv", + "call": "permutest(rma.uni(...), exact = TRUE)", + "metafor_version": "4.6.0", + "r_version": "4.4.1" + }, + "cases": [ + {"design": "equal_k5", "model": "intercept", "method": "FE", + "pval": [0.625], + "zval": [1.3356827983210124], + "n_perm": 32}, + {"design": "unequal_k5", "model": "intercept", "method": "FE", + "pval": [0.1875], + "zval": [-3.0007229965677955], + "n_perm": 32}, + {"design": "extreme_k10", "model": "intercept", "method": "FE", + "pval": [0.03515625], + "zval": [8.8737260033050234], + "n_perm": 1024}, + {"design": "equal_k5", "model": "intercept", "method": "DL", + "pval": [0.625], + "zval": [0.93970359834505923], + "n_perm": 32}, + {"design": "unequal_k5", "model": "intercept", "method": "DL", + "pval": [0.5], + "zval": [-0.81179775818184374], + "n_perm": 32}, + {"design": "extreme_k10", "model": "intercept", "method": "DL", + "pval": [0.048828125], + "zval": [1.7772563219105519], + "n_perm": 1024}, + {"design": "equal_k5", "model": "one", "method": "FE", + "pval": [0.10000000000000001, 0.10833333333333334], + "zval": [2.39865053253143, 2.3195750310357406], + "n_perm": 120}, + {"design": "unequal_k5", "model": "one", "method": "FE", + "pval": [0.58333333333333337, 0.82499999999999996], + "zval": [-2.717537699900995, -0.92951285711430942], + "n_perm": 120}, + {"design": "equal_k5", "model": "one", "method": "DL", + "pval": [0.10000000000000001, 0.10833333333333334], + "zval": [2.3913154032604025, 2.3121648582948993], + "n_perm": 120}, + {"design": "unequal_k5", "model": "one", "method": "DL", + "pval": [0.76666666666666672, 0.20000000000000001], + "zval": [-0.23228023114542842, -2.0263014923699285], + "n_perm": 120} + ] +} diff --git a/pymare/tests/test_metafor_permutest.py b/pymare/tests/test_metafor_permutest.py new file mode 100644 index 0000000..fc84d83 --- /dev/null +++ b/pymare/tests/test_metafor_permutest.py @@ -0,0 +1,226 @@ +"""Alignment between PyMARE's permutation test and metafor's ``permutest``. + +:meth:`~pymare.results.MetaRegressionResults.permutation_test` and +``metafor::permutest`` do the same thing: enumerate or sample a permutation set, +refit the model on each member, and report the proportion of permuted statistics +at least as extreme as the observed one. Only the exact mode can be compared -- +the approximate one draws from each package's own generator -- so the reference +is ``permutest(exact = TRUE)`` on the designs small enough to enumerate, and +PyMARE is asked for the same enumeration. + +The two disagree. This module locates the disagreement rather than tolerating +it, in two tests that between them say where every counted permutation goes: + +- :func:`test_metafor_permutest_is_reproduced_by_the_z_statistic` counts the + *same* permutations of the *same* PyMARE fits, but on ``|beta / se|`` + instead of ``|beta|``, and reproduces metafor exactly in all ten cases. So + the permutation sets agree, the refits agree, and the tie handling of an + inclusive comparison agrees; what differs is which statistic gets counted. +- :func:`test_permutation_p_value_matches_metafor` is what PyMARE currently + reports, and is a strict xfail. Two things break it, and the test's + docstring and marker name both. + +Why the statistic matters rather than being a convention: a permutation test +needs a statistic whose null distribution does not move with the parameters +being permuted away. ``|beta|`` does move, because refitting a permuted dataset +changes tau^2 and so changes the weights and the standard error. The two +coincide exactly when the standard error happens to be invariant -- a +fixed-effects, intercept-only model under sign flipping, which is why four of +the ten cases here agree anyway. + +""" + +import copy +import itertools +import json +import math +import os.path as op + +import numpy as np +import pytest + +from pymare import Dataset +from pymare.estimators import DerSimonianLaird, WeightedLeastSquares +from pymare.tests.utils import get_test_data_path + +pytestmark = pytest.mark.metafor + +with open(op.join(get_test_data_path(), "metafor_permutest_reference.json")) as _fobj: + REFERENCE = json.load(_fobj) + +CASES = REFERENCE["cases"] + +#: A permutation p-value is a count over a known denominator, so agreement is +#: exact or it is not agreement. This is here only to absorb the division. +RTOL = 1e-12 + +#: Moderator columns each model adds beside the intercept, matching ``mods`` in +#: ``validation/metafor/run_permutest.R``. +MODELS = {"intercept": [], "one": ["mod1"]} + +#: The estimators compared. ``small_sample_correction="wald"`` because +#: ``permutest`` was asked with ``test="z"``: the correction rescales the +#: standard error, and comparing a corrected statistic against an uncorrected +#: reference would confound the two things this module is trying to separate. +ESTIMATORS = { + "FE": lambda: WeightedLeastSquares(tau2=0.0, small_sample_correction="wald"), + "DL": lambda: DerSimonianLaird(small_sample_correction="wald"), +} + + +def case_id(case): + """Name a case by the three knobs that distinguish it.""" + return f"{case['design']}-{case['model']}-{case['method']}" + + +def design_arrays(frame, case): + """Return ``(y, v, moderators)`` for one case, without the intercept.""" + rows = frame[frame["case"] == case["design"]] + columns = MODELS[case["model"]] + moderators = rows[columns].to_numpy() if columns else None + return rows["y"].to_numpy(), rows["v"].to_numpy(), moderators + + +def enumerate_permutations(case, n_obs): + """Return the permuted-column indices, and which column is the observed data. + + The observed statistic is read out of the enumeration rather than computed + separately, which is what makes the tie exact: the identity permutation *is* + the observed dataset, so its statistic is the observed statistic to the last + bit, however the batched refit happens to round. + + Returns + ------- + :obj:`tuple` of (:obj:`numpy.ndarray`, :obj:`numpy.ndarray`, :obj:`int`) + Sign multipliers of shape ``(K, n_perm)`` (all ones for a model with + moderators), row indices of shape ``(K, n_perm)``, and the column + holding the identity permutation. + """ + if MODELS[case["model"]]: + orders = list(itertools.permutations(range(n_obs))) + rows = np.array(orders).T + signs = np.ones_like(rows) + identity = orders.index(tuple(range(n_obs))) + else: + signs = np.array(list(itertools.product([-1, 1], repeat=n_obs))).T + rows = np.repeat(np.arange(n_obs)[:, None], signs.shape[1], axis=1) + identity = int(np.flatnonzero((signs == 1).all(axis=0))[0]) + return signs, rows, identity + + +def permuted_statistics(case, frame): + """Refit every permutation of one case and return ``|beta|`` and ``|beta / se|``. + + One batched call per case: PyMARE's closed-form estimators accept a column + per dataset, which is the same vectorization + :meth:`~pymare.results.MetaRegressionResults.permutation_test` uses + internally, so this exercises the estimator on exactly the inputs the + production path would hand it. + """ + y, v, moderators = design_arrays(frame, case) + n_obs = y.shape[0] + signs, rows, identity = enumerate_permutations(case, n_obs) + + design = np.column_stack([np.ones(n_obs)] + ([moderators] if moderators is not None else [])) + params = ( + copy.copy(ESTIMATORS[case["method"]]()).fit(y=y[rows] * signs, v=v[rows], X=design).params_ + ) + beta = np.atleast_2d(params["fe_params"]) + cov = np.asarray(params["inv_cov"]) + se = np.sqrt(np.stack([cov[i, i, :] for i in range(beta.shape[0])])) + return np.abs(beta), np.abs(beta / se), identity + + +@pytest.mark.parametrize("case", CASES, ids=[case_id(case) for case in CASES]) +def test_metafor_permutest_is_reproduced_by_the_z_statistic(case, metafor_dataset): + """Counting the same permutations on ``|z|`` must reproduce metafor exactly. + + This is the load-bearing test of the module. It uses PyMARE's own estimator, + PyMARE's own enumeration of the permutation set, and the same inclusive + comparison PyMARE uses -- changing only the statistic counted, from the + coefficient to the coefficient over its standard error. That it then matches + ``permutest`` in every case, to the last bit of a rational number, is what + establishes that the rest of PyMARE's permutation machinery is right and + that :func:`test_permutation_p_value_matches_metafor` fails for the two + reasons its marker names and not for some third one. + + The observed statistic is taken from the identity permutation inside the + same batch, so the permutation that reproduces the data ties with it + exactly. metafor gets the same effect by comparing against + ``|zval| - sqrt(eps)``. + """ + _, z_statistic, identity = permuted_statistics(case, metafor_dataset) + observed = z_statistic[:, [identity]] + p_values = (z_statistic >= observed).mean(axis=1) + assert z_statistic.shape[1] == case["n_perm"] + assert np.allclose(p_values, case["pval"], rtol=RTOL) + + +@pytest.mark.xfail( + strict=True, + reason=( + "MetaRegressionResults.permutation_test counts |beta| where permutest " + "counts |beta / se|, so the two agree only where the standard error is " + "invariant under the permutation -- a fixed-effects intercept-only " + "model under sign flipping. Refitting a permuted dataset re-estimates " + "tau^2, which moves the weights and hence the standard error, so under " + "DerSimonianLaird the statistic being permuted is not pivotal: " + "unequal_k5 comes out at 0.5625 against permutest's 0.5. Separately, " + "the observed estimate is computed by a different code path from the " + "permuted ones, and the two disagree by one unit in the last place, so " + "the inclusive comparison can drop the identity permutation and its " + "mirror -- always understating the p-value, by 2/1024 on extreme_k10. " + "test_metafor_permutest_is_reproduced_by_the_z_statistic shows both go " + "away when the statistic is |z| and the observed value is read out of " + "the same batch" + ), +) +@pytest.mark.filterwarnings("ignore:Cluster-robust") +def test_permutation_p_value_matches_metafor(metafor_dataset): + """Report the exact permutation p-values PyMARE gives, against metafor's. + + Asserted over the whole grid in one test rather than parametrized, on + purpose. One of the two causes is a one-unit-in-the-last-place difference, + which need not reproduce on every platform and BLAS the test matrix covers; + the other is structural and does. Asserting all ten cases together means + the xfail is driven by the structural cause and cannot flip to an + unexpected pass because a rounding difference went the other way on some + runner. + """ + mismatched = [] + for case in CASES: + y, v, moderators = design_arrays(metafor_dataset, case) + dataset = Dataset(y=y, v=v, X=moderators, add_intercept=True) + results = ESTIMATORS[case["method"]]().fit_dataset(dataset).summary() + permuted = results.permutation_test(n_perm=int(case["n_perm"])) + assert permuted.exact and permuted.n_perm == case["n_perm"], case_id(case) + + reported = np.ravel(permuted.perm_p["fe_p"]) + if not np.allclose(reported, case["pval"], rtol=RTOL): + mismatched.append((case_id(case), list(reported), case["pval"])) + + assert not mismatched, mismatched + + +def test_reference_covers_the_enumerable_designs(metafor_dataset): + """The pinned grid must be the designs small enough to enumerate exactly. + + ``2**K`` sign flips for the intercept-only models and ``K!`` orderings for + the moderator ones, which is why the grid stops where it does: the recorded + ``n_perm`` is asserted against the design's own size so that a reference + regenerated against a different CSV, or with ``exact`` dropped, cannot pass + for the pinned one. + """ + assert {(case["design"], case["model"], case["method"]) for case in CASES} == { + (design, "intercept", method) + for design in ("equal_k5", "unequal_k5", "extreme_k10") + for method in ("FE", "DL") + } | { + (design, "one", method) for design in ("equal_k5", "unequal_k5") for method in ("FE", "DL") + } + + for case in CASES: + n_obs = (metafor_dataset["case"] == case["design"]).sum() + expected = 2 ** n_obs if not MODELS[case["model"]] else math.factorial(n_obs) + assert case["n_perm"] == expected, case_id(case) + assert len(case["pval"]) == 1 + len(MODELS[case["model"]]), case_id(case) diff --git a/validation/metafor/run_permutest.R b/validation/metafor/run_permutest.R new file mode 100644 index 0000000..6286e9b --- /dev/null +++ b/validation/metafor/run_permutest.R @@ -0,0 +1,101 @@ +# Reference values for PyMARE's permutation test. +# +# Writes pymare/tests/data/metafor_permutest_reference.json, which +# pymare/tests/test_metafor_permutest.py reads. Run it through the harness in +# this directory rather than directly, so the R and metafor versions are the +# pinned ones: +# +# validation/metafor/regenerate.sh +# +# Only exact tests are recorded. permutest's approximate mode draws random +# permutations, so its p-values are not reproducible across runs and could not +# be pinned; with exact = TRUE it enumerates the whole permutation set and the +# p-value is a property of the data. That is also the only mode in which a +# comparison against PyMARE means anything, since PyMARE draws its own +# permutations from its own generator. +# +# Which permutation set gets enumerated depends on the model, in both +# implementations: +# +# - Intercept-only: the 2^K assignments of sign to the K estimates. +# - With moderators: the K! orderings. metafor permutes the rows of the +# moderator matrix and PyMARE permutes the (y, v) pairs, which enumerate +# the same set -- one is the other under the inverse permutation, and the +# whole set is covered either way. +# +# So the grid stops at K = 10 for the intercept-only models and K = 5 for the +# moderator ones: 10! refits would make regenerating this file take longer than +# everything else in validation/ put together, for no coverage that 5! does not +# already give. +library(metafor) + +args <- commandArgs(trailingOnly = TRUE) +csv_path <- if (length(args) >= 1) args[[1]] else "/data/metafor_small_sample.csv" +out_path <- if (length(args) >= 2) args[[2]] else "/data/metafor_permutest_reference.json" + +d <- read.csv(csv_path) + +# test = "z" throughout. permutest refers the permuted statistics to their own +# distribution rather than to a reference one, so the small-sample correction +# has nothing to do here, and PyMARE's permutation_test likewise reports a +# p-value that does not depend on it. +cases <- rbind( + expand.grid( + design = c("equal_k5", "unequal_k5", "extreme_k10"), model = "intercept", + method = c("FE", "DL"), stringsAsFactors = FALSE + ), + expand.grid( + design = c("equal_k5", "unequal_k5"), model = "one", + method = c("FE", "DL"), stringsAsFactors = FALSE + ) +) + +mods <- list(intercept = character(0), one = "mod1") + +vector_json <- function(x) paste(sprintf("%.17g", x), collapse = ", ") + +lines <- c( + "{", + ' "source": {', + sprintf(' "data": "%s",', basename(csv_path)), + ' "call": "permutest(rma.uni(...), exact = TRUE)",', + sprintf(' "metafor_version": "%s",', as.character(packageVersion("metafor"))), + sprintf(' "r_version": "%s"', paste(R.version$major, R.version$minor, sep = ".")), + " },", + ' "cases": [' +) + +for (i in seq_len(nrow(cases))) { + case <- cases[i, ] + sub <- d[d$case == case$design, ] + columns <- mods[[case$model]] + + fit <- suppressWarnings(if (length(columns) == 0) { + rma.uni(yi = sub$y, vi = sub$v, method = case$method, test = "z") + } else { + rma.uni( + yi = sub$y, vi = sub$v, mods = as.matrix(sub[, columns, drop = FALSE]), + method = case$method, test = "z" + ) + }) + perm <- suppressWarnings(permutest(fit, exact = TRUE, progbar = FALSE)) + + # The size of the permutation set, recorded so the alignment test can assert + # that PyMARE enumerated the same one rather than falling back to sampling. + n_perm <- if (length(columns) == 0) 2^nrow(sub) else factorial(nrow(sub)) + + lines <- c( + lines, + sprintf( + ' {"design": "%s", "model": "%s", "method": "%s",', + case$design, case$model, case$method + ), + sprintf(' "pval": [%s],', vector_json(perm$pval)), + sprintf(' "zval": [%s],', vector_json(fit$zval)), + sprintf(' "n_perm": %.17g}%s', n_perm, if (i < nrow(cases)) "," else "") + ) +} + +lines <- c(lines, " ]", "}") +writeLines(lines, out_path) +cat(sprintf("wrote %d cases to %s\n", nrow(cases), out_path)) From f276c07a96d143c21110a5e32f2af6f48065a708 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 20:40:23 +0000 Subject: [PATCH 04/13] [TST] verify CR2 and its Satterthwaite dof against clubSandwich pymare.stats.cluster_robust_cov takes the CRn names from clubSandwich and satterthwaite_dof implements the degrees of freedom that package pairs with CR2, but neither had been checked against it. validation/robumeta cannot: robumeta is the reference for the correlated-effects working model -- how weight is spread across a study's rows -- and its model has constant within-study weights by construction, which is precisely the condition under which the two CR2 adjustments below coincide. 12 cases, the grid of variance column x model x closed-form estimator, on the CSV the robumeta check already uses. That CSV carries two variance columns for the same effects, one constant within each study and one varying, and the pair turns out to be exactly what separates agreement from divergence. With the variances constant within a cluster, everything reported agrees to machine precision over all six cases: coefficients 2.9e-15, the full CR2 covariance including its off-diagonal entries 5.1e-15, standard errors 1.8e-15, Satterthwaite dof 2.4e-15, p-values 5.4e-15. Coefficients and tau^2 agree in all twelve, which is what makes the divergence attributable to the sandwich rather than to the fit. With them varying, the covariance is out by up to 5.8e-2, the standard errors 1.0e-2, the dof 3.9e-3 and the p-values 3.4e-2. The cause is exact. Both implementations build the same Bell-McCaffrey matrix B_j = W_j^-1 - X_j (X'WX)^-1 X_j' and take its inverse square root, but in different metrics. With Psi = W^-1 the assumed target, clubSandwich forms A_j = Psi_j^(1/2) (Psi_j^(1/2) B_j Psi_j^(1/2))^(-1/2) Psi_j^(1/2) and _cr2_scores forms A_j = W_j^(1/2) (W_j^(1/2) B_j W_j^(1/2))^(-1/2) W_j^(1/2) A matrix square root does not commute with an asymmetric congruence, so the two are equal if and only if W_j is a multiple of the identity -- when the weights are constant within the cluster. test_cr2_divergence_is_the_whitening_metric pins that rather than the size of the gap: it writes both forms out from their definitions in one function differing only in that congruence, then asserts the clubSandwich form reproduces the pinned standard errors on every case, the PyMARE form reproduces what PyMARE reports on every case, and the two agree when and only when the within-cluster weights are constant. So the divergence is not a different target covariance, a different cluster set, or a bug in either sandwich. PyMARE's form is Fisher & Tipton's A_j^C, which _cr2_scores' docstring says, so it is answering a different question rather than answering this one wrongly. But it is not clubSandwich's CR2 once the variances vary inside a cluster, which is the common case in practice, and cluster_robust_cov's docstring does say the CRn naming follows clubSandwich. Which form method="CR2" should mean is a maintainer's decision; these tests only make the choice visible and keep it from changing by accident. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011o2j5ZNAdzPFL6LBqDqUsU --- pymare/tests/conftest.py | 12 + pymare/tests/data/clubsandwich_reference.json | 107 ++++++ pymare/tests/test_clubsandwich_alignment.py | 317 ++++++++++++++++++ pyproject.toml | 1 + validation/clubsandwich/Dockerfile | 15 + validation/clubsandwich/README.md | 140 ++++++++ validation/clubsandwich/regenerate.sh | 21 ++ validation/clubsandwich/run_clubsandwich.R | 134 ++++++++ 8 files changed, 747 insertions(+) create mode 100644 pymare/tests/data/clubsandwich_reference.json create mode 100644 pymare/tests/test_clubsandwich_alignment.py create mode 100644 validation/clubsandwich/Dockerfile create mode 100644 validation/clubsandwich/README.md create mode 100755 validation/clubsandwich/regenerate.sh create mode 100644 validation/clubsandwich/run_clubsandwich.R diff --git a/pymare/tests/conftest.py b/pymare/tests/conftest.py index 4798281..b6520c7 100644 --- a/pymare/tests/conftest.py +++ b/pymare/tests/conftest.py @@ -383,6 +383,18 @@ def robumeta_dataset(): return frame, designs +@pytest.fixture(scope="package") +def clubsandwich_dataset(): + """Load the dataset the clubSandwich reference values were computed on. + + The same CSV :func:`robumeta_dataset` reads, and returning only the frame: + the clubSandwich alignment builds its designs through + :class:`~pymare.core.Dataset` so that the group labels travel with them, + rather than assembling a bare design matrix as the robumeta alignment does. + """ + return pd.read_csv(op.join(get_test_data_path(), "robumeta_correlated_effects.csv")) + + @pytest.fixture(scope="package") def metafor_dataset(): """Load the designs the metafor reference values were computed on.""" diff --git a/pymare/tests/data/clubsandwich_reference.json b/pymare/tests/data/clubsandwich_reference.json new file mode 100644 index 0000000..d811706 --- /dev/null +++ b/pymare/tests/data/clubsandwich_reference.json @@ -0,0 +1,107 @@ +{ + "source": { + "data": "robumeta_correlated_effects.csv", + "call": "coef_test(rma.uni(effect, , mods = , method = ), vcov = \"CR2\", cluster = study)", + "metafor_version": "4.6.0", + "clubSandwich_version": "0.5.11", + "r_version": "4.4.1" + }, + "cases": [ + {"variances": "var_constant_within_study", "model": "intercept", "method": "FE", + "tau2": 0, + "beta": [0.53753218061181063], + "se": [0.259239381734923], + "dof": [6.2235205324956793], + "pval": [0.081805688003590601], + "cov": [0.067205057042305144], + "n_groups": 10}, + {"variances": "var_within_study", "model": "intercept", "method": "FE", + "tau2": 0, + "beta": [0.49084321493242206], + "se": [0.25749684226447872], + "dof": [6.1344887032545525], + "pval": [0.10419385814171717], + "cov": [0.066304623776177823], + "n_groups": 10}, + {"variances": "var_constant_within_study", "model": "within", "method": "FE", + "tau2": 0, + "beta": [0.68633516793392213, 0.582738074385725], + "se": [0.28065599823373383, 0.12111033741211959], + "dof": [6.9096550829395724, 4.1724225338639247], + "pval": [0.044843596354010619, 0.0077123070105041235], + "cov": [0.07876778934457361, -0.0095255470422427341, -0.0095255470422427341, 0.014667713828077454], + "n_groups": 10}, + {"variances": "var_within_study", "model": "within", "method": "FE", + "tau2": 0, + "beta": [0.68164767020383554, 0.56855668957947647], + "se": [0.25546141053709426, 0.12720698858577267], + "dof": [6.287472801335956, 4.4898764740051895], + "pval": [0.035479915373499184, 0.0084845254718426601], + "cov": [0.065260532273601804, -0.0068993322895347653, -0.0068993322895347636, 0.016181617945060896], + "n_groups": 10}, + {"variances": "var_constant_within_study", "model": "both", "method": "FE", + "tau2": 0, + "beta": [0.17196632888318647, 0.51897290532546325, -0.82522160789383092], + "se": [0.33492272967249914, 0.1665475485776011, 0.28829709120047597], + "dof": [2.8405811635263905, 4.1822109634869236, 2.0287125209396089], + "pval": [0.64488710819193007, 0.033569344809408395, 0.10178828600597814], + "cov": [0.11217323485127793, -0.027810766195209828, 0.063372487582780107, -0.027810766195209828, 0.027738085937208397, -0.010739039173599562, 0.063372487582780079, -0.010739039173599564, 0.083115212794655571], + "n_groups": 10}, + {"variances": "var_within_study", "model": "both", "method": "FE", + "tau2": 0, + "beta": [0.12198382627798961, 0.47021642792942853, -0.83579586307419762], + "se": [0.29841310032614599, 0.15724352866872468, 0.2880654403974352], + "dof": [2.8779791064424156, 4.5970171865112004, 2.0207493796442142], + "pval": [0.71120881620834242, 0.03381612923002273, 0.0998950631235518], + "cov": [0.089050378446262457, -0.021583894559654322, 0.058734651351821238, -0.021583894559654322, 0.024725527308192042, -0.0092938207933039169, 0.058734651351821224, -0.0092938207933039151, 0.082981697951368283], + "n_groups": 10}, + {"variances": "var_constant_within_study", "model": "intercept", "method": "DL", + "tau2": 1.0992037625926645, + "beta": [0.41651754907144251], + "se": [0.2584364868178306], + "dof": [7.1627445477464704], + "pval": [0.15009332784880092], + "cov": [0.066789417718742736], + "n_groups": 10}, + {"variances": "var_within_study", "model": "intercept", "method": "DL", + "tau2": 0.9438104700677189, + "beta": [0.41954714222469219], + "se": [0.25615201396186832], + "dof": [7.0575059955862436], + "pval": [0.14510757764092549], + "cov": [0.065613854256721171], + "n_groups": 10}, + {"variances": "var_constant_within_study", "model": "within", "method": "DL", + "tau2": 0.67397928541642682, + "beta": [0.52773621746611332, 0.60815801481049436], + "se": [0.25830374761824926, 0.11645378630268853], + "dof": [6.9728804979784726, 3.8595336983004676], + "pval": [0.080503179388532506, 0.0070767198506476508], + "cov": [0.066720826033632219, -0.014885881213864419, -0.014885881213864423, 0.013561484344232246], + "n_groups": 10}, + {"variances": "var_within_study", "model": "within", "method": "DL", + "tau2": 0.60101056194151981, + "beta": [0.54627365604819766, 0.60518477606418386], + "se": [0.26035696396728125, 0.12201332708632466], + "dof": [6.8244419815254522, 4.023734976209135], + "pval": [0.075089460686826176, 0.0075888280820123281], + "cov": [0.06778574868626018, -0.016420333180503816, -0.016420333180503816, 0.014887251986674448], + "n_groups": 10}, + {"variances": "var_constant_within_study", "model": "both", "method": "DL", + "tau2": 0.35266890099319592, + "beta": [0.11352838149761001, 0.56921583508871654, -0.75921882194357349], + "se": [0.33867711582498067, 0.15639036813887491, 0.32972353985874264], + "dof": [2.5496880739616605, 3.9317127556782672, 2.1206038918830084], + "pval": [0.76310400524882382, 0.022631470292669459, 0.14067188354304361], + "cov": [0.11470218878352736, -0.023753501083897642, 0.076035763249923408, -0.023753501083897639, 0.024457947246612818, -0.0084020138794114255, 0.076035763249923408, -0.0084020138794114289, 0.10871761273697986], + "n_groups": 10}, + {"variances": "var_within_study", "model": "both", "method": "DL", + "tau2": 0.27174525692509427, + "beta": [0.11115584365338886, 0.54099973767352472, -0.7971880314319657], + "se": [0.31813938703017364, 0.15692193533725682, 0.3101374172023138], + "dof": [2.611154172132041, 4.2207352680476076, 2.0555366368637942], + "pval": [0.75303322820044449, 0.023971531916840962, 0.12052945604062308], + "cov": [0.10121266957993462, -0.023059548157033444, 0.067943669669125045, -0.023059548157033444, 0.024624493789990206, -0.0073803725964035659, 0.067943669669125087, -0.0073803725964035668, 0.09618521754892205], + "n_groups": 10} + ] +} diff --git a/pymare/tests/test_clubsandwich_alignment.py b/pymare/tests/test_clubsandwich_alignment.py new file mode 100644 index 0000000..8331de4 --- /dev/null +++ b/pymare/tests/test_clubsandwich_alignment.py @@ -0,0 +1,317 @@ +"""Alignment between PyMARE and the R package clubSandwich on CR2 and its dof. + +:func:`~pymare.stats.cluster_robust_cov` takes the ``CRn`` names from +``clubSandwich`` :footcite:p:`pustejovsky2018small`, and +:func:`~pymare.stats.satterthwaite_dof` implements the degrees of freedom that +package pairs with ``CR2``. This module checks that against the package itself. + +It is not a duplicate of :mod:`pymare.tests.test_robumeta_alignment`, which +covers a different question about the same estimator. ``robumeta`` is the +reference for the correlated-effects *working model* -- how weight is spread +across a study's rows, which PyMARE spells ``weight_scheme="rescale"`` -- and it +assumes the sampling variances are constant within a study. ``clubSandwich`` is +the reference for the CR2 residual adjustment itself, under whatever weights, +which is what PyMARE does with ``weight_scheme="individual"`` and group labels. + +**What agrees.** When the sampling variances are constant within each cluster, +PyMARE's coefficients, CR2 covariance, Satterthwaite degrees of freedom and +p-values are ``clubSandwich``'s to machine precision, over every model and both +closed-form estimators. + +**What does not.** When they vary within a cluster, the standard errors diverge +by up to 1e-2 relative and the degrees of freedom by 4e-3. The cause is exact and +:func:`test_cr2_divergence_is_the_whitening_metric` pins it: both implementations +build the same Bell-McCaffrey matrix :footcite:p:`bell2002bias` +``B_j = W_j^-1 - X_j (X'WX)^-1 X_j'`` and invert its square root, but they do it +in different metrics. ``clubSandwich`` forms ``A_j = Psi_j^(1/2) (Psi_j^(1/2) B_j +Psi_j^(1/2))^(-1/2) Psi_j^(1/2)`` with ``Psi = W^-1`` the assumed target; +PyMARE's ``_cr2_scores`` forms ``A_j = W_j^(1/2) (W_j^(1/2) B_j W_j^(1/2))^(-1/2) +W_j^(1/2)``. A matrix square root does not commute with an asymmetric +congruence, so the two coincide if and only if ``W_j`` is a multiple of the +identity -- that is, when the weights are constant within the cluster. PyMARE's +form is the one :footcite:t:`fisher2015robumeta` give as ``A_j^C``, which its +docstring says, and the correlated-effects model it belongs to has constant +within-study weights by construction. It is not ``clubSandwich``'s ``CR2`` once +they vary, which is worth knowing given where the name came from. + +``validation/clubsandwich/README.md`` records the measurements. + +References +---------- +.. footbibliography:: + +""" + +import json +import os.path as op + +import numpy as np +import pytest + +from pymare import Dataset +from pymare.estimators import DerSimonianLaird, WeightedLeastSquares +from pymare.tests.utils import get_test_data_path + +# Both warnings are expected over this grid and say nothing about alignment. +# "Cluster-robust" fires because ten studies is few for a sandwich, which is the +# point of comparing the small-sample correction at all; the Satterthwaite one +# fires on the three-predictor model, where `between` is constant within a study +# and so is carried by very few clusters. clubSandwich reports the same low +# degrees of freedom, which is what these tests check. +pytestmark = [ + pytest.mark.clubsandwich, + pytest.mark.filterwarnings("ignore:Cluster-robust"), + pytest.mark.filterwarnings("ignore:Satterthwaite degrees of freedom below"), +] + +with open(op.join(get_test_data_path(), "clubsandwich_reference.json")) as _fobj: + REFERENCE = json.load(_fobj) + +CASES = REFERENCE["cases"] + +#: Tolerance where the two implementations compute the same thing. Worst +#: observed 7.1e-15, on a p-value under the three-predictor model; the +#: coefficients and the covariance agree to a few multiples of machine epsilon. +RTOL = 1e-12 + +#: Absolute floor, for the covariance's off-diagonal entries which pass through +#: zero. +ATOL = 1e-14 + +#: Moderator columns each model adds beside the intercept, matching ``mods`` in +#: ``validation/clubsandwich/run_clubsandwich.R``. +MODELS = {"intercept": [], "within": ["within"], "both": ["within", "between"]} + +#: The estimators compared, in metafor's names. Both reach tau^2 in closed form, +#: so nothing but the covariance sits between PyMARE and clubSandwich. +#: ``weight_scheme="individual"`` throughout: it is the scheme that uses +#: ``1 / (v + tau^2)`` row by row, which is the weight matrix clubSandwich takes +#: from the ``rma.uni`` fit. The other two schemes are robumeta's question. +ESTIMATORS = { + "FE": lambda: WeightedLeastSquares(tau2=0.0, weight_scheme="individual"), + "DL": lambda: DerSimonianLaird(weight_scheme="individual"), +} + +#: The variance column whose values are constant within each study, and so the +#: condition under which the two CR2 adjustments coincide. Named rather than +#: inferred so that the split between the two tests below is explicit. +CONSTANT_WITHIN_CLUSTER = "var_constant_within_study" + +AGREEING_CASES = [case for case in CASES if case["variances"] == CONSTANT_WITHIN_CLUSTER] + +DIVERGING_CASES = [case for case in CASES if case["variances"] != CONSTANT_WITHIN_CLUSTER] + + +def case_id(case): + """Name a case by the three knobs that distinguish it.""" + columns = "shared" if case["variances"] == CONSTANT_WITHIN_CLUSTER else "varying" + return f"{columns}-v-{case['model']}-{case['method']}" + + +def build_dataset(frame, case): + """Assemble the model PyMARE should fit, with the study labels attached.""" + columns = MODELS[case["model"]] + return Dataset( + y=frame["effect"].to_numpy(), + v=frame[case["variances"]].to_numpy(), + X=frame[columns].to_numpy() if columns else None, + add_intercept=True, + g=frame["study"].to_numpy(), + ) + + +def fit(frame, case): + """Fit one case and return its results object.""" + return ESTIMATORS[case["method"]]().fit_dataset(build_dataset(frame, case)).summary() + + +def inverse_sqrt(matrix): + """Return the symmetric inverse square root of a positive semidefinite matrix. + + Restricted to the range, as ``clubSandwich``'s ``matrix_power(g, -1/2)`` is: + a cluster whose rows are fitted away exactly leaves a singular ``B_j``, and + the pseudo-inverse drops that direction rather than dividing by zero. + """ + values, vectors = np.linalg.eigh(matrix) + keep = values > values.max() * 1e-12 + return (vectors[:, keep] * values[keep] ** -0.5) @ vectors[:, keep].T + + +def cr2_standard_errors(y, v, X, groups, tau2, metric): + """Compute CR2 standard errors under one of the two whitening metrics. + + Parameters + ---------- + y, v, X, groups : :obj:`numpy.ndarray` + One case's data, with the intercept already in ``X``. + tau2 : :obj:`float` + The variance component the weights use. + metric : {"clubSandwich", "pymare"} + Which congruence to apply to the Bell-McCaffrey matrix before taking its + inverse square root: the target ``Psi_j^(1/2) = W_j^(-1/2)``, or the + weights ``W_j^(1/2)``. + + Returns + ------- + :obj:`numpy.ndarray` + The square roots of the sandwich's diagonal. + + Notes + ----- + Written out from the definition rather than called from + :mod:`pymare.stats`, which is the point: the two forms differ only in the + line selected by ``metric``, so a test that reproduces each implementation's + output from this one function has located the difference between them and + not merely measured it. + """ + weights = 1.0 / (v + tau2) + W = np.diag(weights) + bread = np.linalg.inv(X.T @ W @ X) + resid = y - X @ (bread @ X.T @ W @ y) + + meat = np.zeros((X.shape[1], X.shape[1])) + for label in np.unique(groups): + rows = np.flatnonzero(groups == label) + X_j = X[rows] + root = np.diag(np.sqrt(weights[rows])) + inverse_root = np.diag(1.0 / np.sqrt(weights[rows])) + bell_mccaffrey = np.diag(1.0 / weights[rows]) - X_j @ bread @ X_j.T + if metric == "clubSandwich": + adjustment = ( + inverse_root + @ inverse_sqrt(inverse_root @ bell_mccaffrey @ inverse_root) + @ inverse_root + ) + score = X_j.T @ np.diag(weights[rows]) @ adjustment @ resid[rows] + else: + adjustment = inverse_sqrt(root @ bell_mccaffrey @ root) + score = X_j.T @ root @ adjustment @ root @ resid[rows] + meat += np.outer(score, score) + + return np.sqrt(np.diag(bread @ meat @ bread)) + + +def design_arrays(frame, case): + """Return ``(y, v, X, groups)`` for one case, with the intercept in ``X``.""" + y = frame["effect"].to_numpy() + columns = MODELS[case["model"]] + X = np.column_stack([np.ones_like(y)] + [frame[name].to_numpy() for name in columns]) + return y, frame[case["variances"]].to_numpy(), X, frame["study"].to_numpy() + + +@pytest.mark.parametrize("case", AGREEING_CASES, ids=[case_id(case) for case in AGREEING_CASES]) +def test_cr2_matches_clubsandwich_with_constant_within_cluster_weights(case, clubsandwich_dataset): + """With constant within-cluster weights, everything reported must match. + + The whole inference path, not only the standard errors: the coefficients, the + full CR2 covariance including its off-diagonal entries, the Satterthwaite + degrees of freedom, and the p-values that follow from the two together. A + failure here is a failure of PyMARE's cluster-robust estimator, since there + is no approximation on either side -- tau^2 is closed form for both + estimators compared, so the only thing between the two implementations is + the sandwich. + """ + results = fit(clubsandwich_dataset, case) + stats = results.get_fe_stats() + n_preds = 1 + len(MODELS[case["model"]]) + + assert np.allclose(np.ravel(results.tau2), case["tau2"], rtol=RTOL, atol=ATOL) + assert np.allclose(np.ravel(stats["est"]), case["beta"], rtol=RTOL) + assert np.allclose(np.ravel(stats["se"]), case["se"], rtol=RTOL) + assert np.allclose(np.ravel(results.fe_dof), case["dof"], rtol=RTOL) + assert np.allclose(np.ravel(stats["p"]), case["pval"], rtol=RTOL) + + covariance = np.asarray(results.estimator.params_["inv_cov"]).reshape(n_preds, n_preds) + assert np.allclose( + covariance, np.reshape(case["cov"], (n_preds, n_preds)), rtol=RTOL, atol=ATOL + ) + + +@pytest.mark.parametrize("case", CASES, ids=[case_id(case) for case in CASES]) +def test_coefficients_and_tau2_match_clubsandwich_everywhere(case, clubsandwich_dataset): + """The point estimates must match whether or not the weights vary. + + CR2 changes the covariance and nothing else, so the coefficients and tau^2 + have to agree on every case in the grid. Separating this from the test above + is what makes the divergence attributable: if the coefficients also moved, + the standard errors would be disagreeing for a reason that had nothing to do + with the residual adjustment. + """ + results = fit(clubsandwich_dataset, case) + assert np.allclose(np.ravel(results.tau2), case["tau2"], rtol=RTOL, atol=ATOL) + assert np.allclose(np.ravel(results.get_fe_stats()["est"]), case["beta"], rtol=RTOL) + + +@pytest.mark.parametrize("case", CASES, ids=[case_id(case) for case in CASES]) +def test_cr2_divergence_is_the_whitening_metric(case, clubsandwich_dataset): + """Locate the CR2 divergence in the congruence each implementation applies. + + Three assertions, which together say exactly where the two part company and + leave nothing to a tolerance: + + 1. The ``clubSandwich`` metric, written out from the definition in + :func:`cr2_standard_errors`, reproduces the pinned standard errors on + every case in the grid. + 2. The PyMARE metric, from the same function, reproduces what PyMARE + reports on every case in the grid. + 3. The two metrics give the same answer when the within-cluster weights are + constant, and different answers when they are not. + + So the divergence is not a bug in either sandwich, a different target + covariance, or a different set of clusters. It is the choice of metric in + which the Bell-McCaffrey matrix's inverse square root is taken, and it is + invisible until the sampling variances vary inside a cluster. + """ + y, v, X, groups = design_arrays(clubsandwich_dataset, case) + tau2 = float(np.ravel(fit(clubsandwich_dataset, case).tau2)[0]) + + from_clubsandwich = cr2_standard_errors(y, v, X, groups, tau2, "clubSandwich") + from_pymare = cr2_standard_errors(y, v, X, groups, tau2, "pymare") + reported = np.ravel(fit(clubsandwich_dataset, case).get_fe_stats()["se"]) + + assert np.allclose(from_clubsandwich, case["se"], rtol=RTOL) + assert np.allclose(from_pymare, reported, rtol=RTOL) + + if case["variances"] == CONSTANT_WITHIN_CLUSTER: + assert np.allclose(from_pymare, from_clubsandwich, rtol=RTOL) + else: + assert not np.allclose(from_pymare, from_clubsandwich, rtol=1e-6) + + +def test_reference_covers_every_combination(clubsandwich_dataset): + """The pinned grid must stay the full grid, on the data it claims. + + Without this the alignment check could quietly shrink to the cases that + happen to pass -- and in particular to the constant-variance column, which + is the half that agrees. + """ + assert {(case["variances"], case["model"], case["method"]) for case in CASES} == { + (variances, model, method) + for variances in ("var_constant_within_study", "var_within_study") + for model in MODELS + for method in ESTIMATORS + } + + n_groups = clubsandwich_dataset["study"].nunique() + for case in CASES: + n_preds = 1 + len(MODELS[case["model"]]) + assert case["n_groups"] == n_groups, case_id(case) + assert len(case["beta"]) == n_preds, case_id(case) + assert len(case["cov"]) == n_preds**2, case_id(case) + # The Satterthwaite degrees of freedom cannot exceed the number of + # clusters, and fall well below it when a predictor is thinly supported. + assert all(0 < dof <= n_groups for dof in case["dof"]), case_id(case) + + +def test_the_two_variance_columns_differ_in_the_way_the_split_assumes(clubsandwich_dataset): + """One variance column must be constant within study and the other not. + + The split between the agreeing and diverging halves of this module rests on + that property of the input, so it is asserted rather than assumed: a CSV + edited to make both columns constant would turn + :func:`test_cr2_divergence_is_the_whitening_metric`'s third assertion into a + statement about nothing. + """ + grouped = clubsandwich_dataset.groupby("study") + assert (grouped["var_constant_within_study"].nunique() == 1).all() + assert (grouped["var_within_study"].nunique() > 1).any() diff --git a/pyproject.toml b/pyproject.toml index b3605dc..0fc3969 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -8,6 +8,7 @@ markers = [ "stan: tests that sample, needing cmdstanpy and a CmdStan installation (slow: the model is compiled)", "robumeta: tests that pin PyMARE against the R package robumeta", "metafor: tests that pin PyMARE against the R package metafor", + "clubsandwich: tests that pin PyMARE against the R package clubSandwich", ] [tool.black] diff --git a/validation/clubsandwich/Dockerfile b/validation/clubsandwich/Dockerfile new file mode 100644 index 0000000..42633a2 --- /dev/null +++ b/validation/clubsandwich/Dockerfile @@ -0,0 +1,15 @@ +# Pinned so the reference values are reproducible. +# +# Three things are pinned, because any one moving would move the numbers: the R +# version (the base image) and the metafor and clubSandwich versions (a dated +# CRAN snapshot, since ``install.packages`` on its own tracks whatever CRAN +# happens to be serving that day). metafor is here because clubSandwich has no +# model of its own: ``vcovCR`` and ``coef_test`` take a fitted ``rma.uni`` and +# replace its covariance, so the fit has to come from metafor. +# +# The same R pin as validation/metafor and validation/robumeta, so the three +# reference images share a base layer. +FROM r-base:4.4.1 +RUN R -e 'install.packages(c("metafor", "clubSandwich"), repos="https://packagemanager.posit.co/cran/2024-07-01")' +COPY run_clubsandwich.R /opt/run_clubsandwich.R +ENTRYPOINT ["Rscript", "/opt/run_clubsandwich.R"] diff --git a/validation/clubsandwich/README.md b/validation/clubsandwich/README.md new file mode 100644 index 0000000..b772c1d --- /dev/null +++ b/validation/clubsandwich/README.md @@ -0,0 +1,140 @@ +# clubSandwich reference values + +PyMARE's cluster-robust covariance takes the `CRn` names from the R package +[clubSandwich](https://cran.r-project.org/package=clubSandwich) +(Pustejovsky & Tipton, 2018, *Journal of Business & Economic Statistics* 36(4), +672-683), and `pymare.stats.satterthwaite_dof` implements the degrees of freedom +that package pairs with `CR2`. This directory regenerates the reference values +`pymare/tests/test_clubsandwich_alignment.py` pins. + +```bash +make check_clubsandwich_alignment # needs Docker; rewrites the pinned file in place +make test_clubsandwich # check PyMARE against the pinned values +``` + +Regeneration goes through a Docker image with pinned R, metafor and clubSandwich +versions. metafor is in the image because clubSandwich has no model of its own: +`vcovCR` and `coef_test` take a fitted `rma.uni` and replace its covariance, so +the fit has to come from metafor. Mirrors `validation/robumeta` and +`validation/metafor` otherwise, including the numeric rather than byte +comparison the workflow makes -- see `validation/compare_reference.py`. + +## Why this is not the robumeta check again + +`validation/robumeta` and this directory answer different questions about the +same estimator, which is why both exist. + +| | robumeta | clubSandwich | +| --- | --- | --- | +| Reference for | the correlated-effects **working model** -- how weight is spread across a study's rows | the **CR2 residual adjustment** and its Satterthwaite degrees of freedom | +| PyMARE spelling | `weight_scheme="rescale"` | `weight_scheme="individual"` with group labels | +| Assumes | sampling variances constant within a study | nothing about them | + +robumeta cannot distinguish the two CR2 adjustments below, because its model has +constant within-study weights by construction. That is exactly the condition +under which they coincide. + +## What is compared + +12 cases, the full grid of variance column x model x tau^2 estimator, on the +same `pymare/tests/data/robumeta_correlated_effects.csv` the robumeta check +uses. `test_reference_covers_every_combination` asserts that grid. + +| Knob | Values | +| --- | --- | +| variance column | `var_constant_within_study`, `var_within_study` | +| model | `effect ~ 1`, `effect ~ within`, `effect ~ within + between` | +| tau^2 estimator | `FE`, `DL` | + +Only the two estimators that reach tau^2 in closed form, so that nothing but the +sandwich sits between the two implementations. The CSV's two variance columns are +the point of the grid: one is constant inside every study and the other is not, +and `test_the_two_variance_columns_differ_in_the_way_the_split_assumes` asserts +that property of the input rather than trusting the column names. + +Everything both implementations report is compared -- coefficients, the full CR2 +covariance including its off-diagonal entries, the Satterthwaite degrees of +freedom, the standard errors and the p-values that follow from them. + +## What agrees + +**With the sampling variances constant within each cluster, everything agrees to +machine precision, in all six cases:** + +| Quantity | Worst relative deviation | +| --- | --- | +| coefficients | 2.9e-15 | +| CR2 covariance | 5.1e-15 | +| standard errors | 1.8e-15 | +| Satterthwaite dof | 2.4e-15 | +| p-values | 5.4e-15 | + +The coefficients and tau^2 agree in all twelve cases, varying variances +included, to 3.5e-15. CR2 changes the covariance and nothing else, so that is +what makes the divergence below attributable to the sandwich. + +## What does not, and why + +**With the sampling variances varying inside a cluster, the covariance +diverges:** + +| Quantity | Worst relative deviation | +| --- | --- | +| CR2 covariance | 5.8e-2 | +| standard errors | 1.0e-2 | +| Satterthwaite dof | 3.9e-3 | +| p-values | 3.4e-2 | + +The cause is exact, and `test_cr2_divergence_is_the_whitening_metric` pins it by +writing both forms out from their definitions and showing each reproduces its +own implementation's output. Both build the same Bell-McCaffrey matrix +(Bell & McCaffrey, 2002, *Survey Methodology* 28(2), 169-181) + +``` +B_j = W_j^-1 - X_j (X'WX)^-1 X_j' +``` + +and take its inverse square root, but in different metrics. With `Psi = W^-1` +the assumed target covariance, clubSandwich forms + +``` +A_j = Psi_j^(1/2) (Psi_j^(1/2) B_j Psi_j^(1/2))^(-1/2) Psi_j^(1/2) +``` + +and PyMARE's `_cr2_scores` forms + +``` +A_j = W_j^(1/2) (W_j^(1/2) B_j W_j^(1/2))^(-1/2) W_j^(1/2) +``` + +A matrix square root does not commute with an asymmetric congruence, so the two +are equal if and only if `W_j` is a multiple of the identity -- that is, when the +weights, and hence the sampling variances, are constant within the cluster. + +PyMARE's form is the one Fisher & Tipton (2015, arXiv:1503.02220) give as +`A_j^C`, which `pymare.stats._cr2_scores` says in its docstring, and the +correlated-effects model it belongs to has constant within-study weights by +construction. So it is not wrong so much as answering a different question. But +it is not clubSandwich's `CR2` once the variances vary inside a cluster, which is +worth knowing given that `cluster_robust_cov`'s docstring says the `CRn` naming +follows clubSandwich, and given that variances varying within a study is the +common case in practice rather than the corner one. + +Deciding which form `method="CR2"` should mean is a question for PyMARE's +maintainers, not something these tests settle. What they do is make the choice +visible and keep it from changing by accident. + +## What is not compared + +`CR0`, which PyMARE also offers. clubSandwich's `CR0` is the unadjusted +sandwich, and PyMARE's applies an `m / (m - p)` scaling that +`cluster_robust_cov`'s docstring already records as being neither clubSandwich's +`CR1` nor Stata's `CR1S` -- it is the original adjustment of Hedges, Tipton & +Johnson (2010), kept for reproducing analyses that predate the leverage-based +corrections. There is nothing in clubSandwich to compare it against. + +`CR1`, `CR3` and `CR4` have no PyMARE counterpart. + +`weight_scheme="rescale"` and `"collapse"` change the weights away from +`1 / (v + tau^2)`, which is the weight matrix clubSandwich reads off the +`rma.uni` fit. `validation/robumeta` is the reference for those. diff --git a/validation/clubsandwich/regenerate.sh b/validation/clubsandwich/regenerate.sh new file mode 100755 index 0000000..c17da6e --- /dev/null +++ b/validation/clubsandwich/regenerate.sh @@ -0,0 +1,21 @@ +#!/usr/bin/env bash +# Regenerate the clubSandwich reference values PyMARE's alignment test reads. +# +# Run from the repository root: +# +# validation/clubsandwich/regenerate.sh +# +# Rewrites pymare/tests/data/clubsandwich_reference.json in place. The +# alignment workflow runs this script and then compares the result numerically +# against the pinned file, so CI and a local run cannot drift apart. +set -euo pipefail + +repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +data_dir="${repo_root}/pymare/tests/data" + +docker build --quiet -t pymare-clubsandwich "${repo_root}/validation/clubsandwich" >/dev/null +docker run --rm \ + --user "$(id -u):$(id -g)" \ + -v "${data_dir}:/data" \ + pymare-clubsandwich \ + /data/robumeta_correlated_effects.csv /data/clubsandwich_reference.json diff --git a/validation/clubsandwich/run_clubsandwich.R b/validation/clubsandwich/run_clubsandwich.R new file mode 100644 index 0000000..bd6857f --- /dev/null +++ b/validation/clubsandwich/run_clubsandwich.R @@ -0,0 +1,134 @@ +# Reference values for PyMARE's cluster-robust covariance and its degrees of freedom. +# +# Writes pymare/tests/data/clubsandwich_reference.json, which +# pymare/tests/test_clubsandwich_alignment.py reads. Run it through the harness +# in this directory rather than directly, so the R, metafor and clubSandwich +# versions are the pinned ones: +# +# validation/clubsandwich/regenerate.sh +# +# clubSandwich is not a test dependency, so the numbers are pinned rather than +# recomputed on every test run. The alignment workflow regenerates them and +# fails on any difference beyond the tolerances in +# validation/compare_reference.py, which is what keeps the pin honest. +# +# Why clubSandwich and not robumeta, which validation/robumeta already covers: +# the two answer different questions about the same estimator. robumeta is the +# reference for the correlated-effects *working model* -- how weight is spread +# across a study's rows -- which PyMARE spells weight_scheme="rescale", and it +# assumes the sampling variances are constant within a study. clubSandwich is +# the reference for the CR2 *residual adjustment* and the Satterthwaite degrees +# of freedom it needs, under whatever weights, and is the implementation the +# CRn names in pymare.stats.cluster_robust_cov are taken from. +# +# The input is the robumeta CSV because it carries two variance columns for the +# same effects: one constant within each study and one varying. That pair is +# what separates the case where PyMARE and clubSandwich agree exactly from the +# case where they do not -- see the README in this directory. +# +# The output is written by hand rather than with jsonlite so the formatting is +# byte-stable: every number goes through "%.17g", which round-trips a double +# exactly. +library(metafor) +library(clubSandwich) + +args <- commandArgs(trailingOnly = TRUE) +csv_path <- if (length(args) >= 1) args[[1]] else "/data/robumeta_correlated_effects.csv" +out_path <- if (length(args) >= 2) args[[2]] else "/data/clubsandwich_reference.json" + +d <- read.csv(csv_path) + +# The moderator columns each model adds beside the intercept, named rather than +# passed as a formula so the column order is explicit: coef_test reports the +# intercept first and so does pymare.core.Dataset, so the vectors line up +# position by position. The same three models validation/robumeta uses. +mods <- list(intercept = character(0), within = "within", both = c("within", "between")) + +# Only the estimators whose tau^2 PyMARE reaches in closed form, so that nothing +# but the covariance sits between the two implementations. "FE" is the +# fixed-effects model, which PyMARE spells WeightedLeastSquares(tau2=0). +methods <- c("FE", "DL") + +variance_columns <- c("var_constant_within_study", "var_within_study") + +cases <- expand.grid( + variances = variance_columns, model = names(mods), method = methods, + stringsAsFactors = FALSE +) + +vector_json <- function(x) paste(sprintf("%.17g", x), collapse = ", ") + +scalar_json <- function(x) { + if (length(x) == 0 || is.null(x) || all(is.na(x))) "null" else sprintf("%.17g", x[[1]]) +} + +lines <- c( + "{", + ' "source": {', + sprintf(' "data": "%s",', basename(csv_path)), + paste0( + ' "call": "coef_test(rma.uni(effect, , mods = , ', + 'method = ), vcov = \\"CR2\\", cluster = study)",' + ), + sprintf(' "metafor_version": "%s",', as.character(packageVersion("metafor"))), + sprintf( + ' "clubSandwich_version": "%s",', + as.character(packageVersion("clubSandwich")) + ), + sprintf(' "r_version": "%s"', paste(R.version$major, R.version$minor, sep = ".")), + " },", + ' "cases": [' +) + +for (i in seq_len(nrow(cases))) { + case <- cases[i, ] + columns <- mods[[case$model]] + + # rma.uni rejects a NULL passed through a variable, so the intercept-only + # model has to omit the argument rather than pass nothing to it. + fit <- suppressWarnings(if (length(columns) == 0) { + rma.uni(yi = d$effect, vi = d[[case$variances]], method = case$method) + } else { + rma.uni( + yi = d$effect, vi = d[[case$variances]], + mods = as.matrix(d[, columns, drop = FALSE]), method = case$method + ) + }) + + # vcov = "CR2" and the Satterthwaite degrees of freedom together, which is + # what PyMARE reports when group labels are supplied: coef_test's p_Satt is + # the two-sided t p-value on df_Satt, and its SE is the square root of the + # CR2 covariance's diagonal. + test <- coef_test(fit, vcov = "CR2", cluster = d$study) + + # The full CR2 covariance, not only its diagonal: PyMARE returns a matrix and + # the off-diagonal entries are what the Satterthwaite degrees of freedom of a + # linear combination would rest on, so pinning them costs nothing and closes + # off a way for the two to agree on every standard error while disagreeing + # about the covariance. + covariance <- as.matrix(vcovCR(fit, cluster = d$study, type = "CR2")) + + lines <- c( + lines, + sprintf( + ' {"variances": "%s", "model": "%s", "method": "%s",', + case$variances, case$model, case$method + ), + sprintf(' "tau2": %s,', scalar_json(fit$tau2)), + sprintf(' "beta": [%s],', vector_json(test$beta)), + sprintf(' "se": [%s],', vector_json(test$SE)), + sprintf(' "dof": [%s],', vector_json(test$df_Satt)), + sprintf(' "pval": [%s],', vector_json(test$p_Satt)), + # Row-major, which is how numpy.reshape will read it back. + sprintf(' "cov": [%s],', vector_json(as.vector(t(covariance)))), + sprintf( + ' "n_groups": %.17g}%s', + length(unique(d$study)), + if (i < nrow(cases)) "," else "" + ) + ) +} + +lines <- c(lines, " ]", "}") +writeLines(lines, out_path) +cat(sprintf("wrote %d cases to %s\n", nrow(cases), out_path)) From 1636e5f21d3b2b0aeb4438d8bdedf2bdc113df97 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 20:48:00 +0000 Subject: [PATCH 05/13] [CI] run the alignment checks against R on every pull request The metafor reference values had no workflow re-verifying them against metafor -- validation/metafor/README.md named adding one, and generalising the robumeta comparator it would need, as a follow-up. This does that, and brings the two new reference sets in with it. validation/compare_reference.py replaces validation/robumeta/compare_reference.py and serves all five reference files. It walks any {"source": ..., "cases": [...]} document without being told its shape, folding the keys that name a case into the label and comparing everything else as numbers -- and raising on a value that is neither, so a generator that stopped recording a quantity cannot pass the check by having nothing left to compare. The comparison has to be numeric, which is a change for metafor: its Makefile target used `git diff --exit-code`, and that can no longer hold. Regenerating metafor_reference.json under the same R 4.4.1 and metafor 4.6-0 the file already named, on a different BLAS, moves 1326 of its lines -- by at most 9.5e-14 relative on the inference path and 3.3e-9 on an ML tau^2, where an optimizer stops a step earlier or later. Hence the per-quantity tolerance for tau^2, still three orders tighter than the 1e-4 the profiled-tau^2 test itself allows. Verified both directions: the comparator accepts that real cross-BLAS regeneration on all 1260 shared quantities, and rejects a single standard error perturbed by 1e-9, a dropped field, and a non-numeric value. .github/workflows/alignment.yml covers metafor and clubSandwich as two matrix legs, each regenerating from its pinned image, comparing every reference it owns, and then running its marker's tests against the values just regenerated rather than only against the pin. robumeta keeps its own workflow rather than becoming a third leg: its check is named "Check robumeta alignment / PyMARE vs robumeta" and may be a required check on the repository, which folding it in would silently rename. It now shares the comparator, so the duplication is a workflow header and not logic. Both R images take Rscript as their entrypoint instead of a single generator, so one built image can run the three metafor scripts in turn. Also here: a test_clubsandwich Makefile target and marker, a rewritten validation/metafor/README.md recording what agrees to what tolerance and what does not and why across all three metafor checks, and the correction of two claims in it that measurement contradicted -- the Hedges tau^2 divergence is specific to meta-regression rather than general, and it is metafor, not PyMARE, that stops at tau^2 = 0 on the ML boundary cell. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011o2j5ZNAdzPFL6LBqDqUsU --- .github/workflows/alignment.yml | 106 +++++++++ .github/workflows/robumeta-alignment.yml | 11 +- Makefile | 45 +++- pymare/tests/utils.py | 9 +- validation/clubsandwich/Dockerfile | 6 +- validation/clubsandwich/regenerate.sh | 1 + validation/compare_reference.py | 189 ++++++++++++++++ validation/metafor/Dockerfile | 10 +- validation/metafor/README.md | 270 +++++++++++++++++++---- validation/metafor/regenerate.sh | 36 ++- validation/robumeta/README.md | 20 +- validation/robumeta/compare_reference.py | 102 --------- validation/robumeta/regenerate.sh | 8 +- 13 files changed, 638 insertions(+), 175 deletions(-) create mode 100644 .github/workflows/alignment.yml create mode 100755 validation/compare_reference.py delete mode 100755 validation/robumeta/compare_reference.py diff --git a/.github/workflows/alignment.yml b/.github/workflows/alignment.yml new file mode 100644 index 0000000..aa6a69c --- /dev/null +++ b/.github/workflows/alignment.yml @@ -0,0 +1,106 @@ +name: "Check alignment with R packages" + +# metafor and clubSandwich, one matrix leg each. robumeta has its own workflow +# rather than a third leg here: its check is named "Check robumeta alignment / +# PyMARE vs robumeta" and may be a required check on the repository, which +# folding it in would silently rename. The two share everything that matters -- +# validation/compare_reference.py and the regenerate.sh contract -- so the +# duplication is a workflow header, not logic. + +on: + push: + branches: + - "master" + pull_request: + branches: + - "*" + schedule: + # Monthly, to catch the harness rotting rather than PyMARE moving: the R + # version and the package versions are pinned, so a failure here on an + # unchanged tree means a pinned image can no longer be built. + - cron: "0 0 1 * *" + # GitHub turns off a workflow with a schedule trigger once the repository has + # been quiet for 60 days, and a disabled workflow stops answering push and + # pull_request too. Being able to dispatch it means re-enabling is enough to + # get a run without pushing a commit to prove it works. + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: alignment-${{ github.ref }} + cancel-in-progress: true + +jobs: + alignment: + name: PyMARE vs ${{ matrix.label }} + runs-on: ubuntu-latest + strategy: + # Independent references, so a failure in one should still report the + # other rather than being masked by it. + fail-fast: false + matrix: + include: + - package: metafor + label: metafor + marker: metafor + # rma.uni, escalc and permutest. validation/metafor/regenerate.sh + # rewrites all three; each is compared on its own so a failure + # names which reference moved. + references: >- + metafor_reference.json + metafor_escalc_reference.json + metafor_permutest_reference.json + - package: clubsandwich + label: clubSandwich + marker: clubsandwich + references: clubsandwich_reference.json + defaults: + run: + shell: bash + steps: + - uses: actions/checkout@v4 + - name: "Set up python" + uses: actions/setup-python@v5 + with: + python-version: "3.11" + - name: "Install PyMARE" + run: | + python -m pip install --progress-bar off --upgrade pip setuptools wheel + python -m pip install -e .[tests] + + # The pinned reference values are only worth anything if they still match + # what the R package prints, so regenerate them from the pinned image. + - name: "Regenerate the reference values from ${{ matrix.label }}" + run: validation/${{ matrix.package }}/regenerate.sh + + # Compared numerically rather than with git diff: the values are written + # at full double precision, and which BLAS kernel R's image picks depends + # on the runner's CPU, so two runners disagree in the last bits with the R + # and package versions identical. See the tolerances in + # validation/compare_reference.py. + - name: "Require the reference values to still match ${{ matrix.label }}" + run: | + status=0 + for reference in ${{ matrix.references }}; do + git show "HEAD:pymare/tests/data/${reference}" > "${RUNNER_TEMP}/${reference}" + python validation/compare_reference.py \ + "${RUNNER_TEMP}/${reference}" \ + "pymare/tests/data/${reference}" \ + | tee -a "$GITHUB_STEP_SUMMARY" || status=1 + done + exit "$status" + + # Run the alignment tests against the values just regenerated, so this job + # checks PyMARE against the R package itself and not only against the pin. + - name: "Check PyMARE against the regenerated values" + if: always() + run: python -m pytest -m ${{ matrix.marker }} -v + + - name: "Upload the regenerated reference values" + if: failure() + uses: actions/upload-artifact@v4 + with: + name: ${{ matrix.package }}-reference + path: pymare/tests/data/*_reference.json diff --git a/.github/workflows/robumeta-alignment.yml b/.github/workflows/robumeta-alignment.yml index aaa6b47..aa36199 100644 --- a/.github/workflows/robumeta-alignment.yml +++ b/.github/workflows/robumeta-alignment.yml @@ -1,5 +1,10 @@ name: "Check robumeta alignment" +# Separate from the "Check alignment with R packages" workflow, which covers +# metafor and clubSandwich, only so that this job's check name does not change: +# it predates that workflow and may be a required check on the repository. The +# steps are the same ones, against the same shared comparator. + on: push: branches: @@ -54,11 +59,13 @@ jobs: # Compared numerically rather than with git diff: the values are written at # full double precision, and which BLAS kernel R's image picks depends on # the runner's CPU, so two runners disagree in the last bits with the R and - # robumeta versions identical. See the tolerances in compare_reference.py. + # robumeta versions identical. See the tolerances in + # validation/compare_reference.py, which the metafor and clubSandwich legs + # of the "Check alignment with R packages" workflow share. - name: "Require the reference values to still match robumeta" run: | git show HEAD:pymare/tests/data/robumeta_reference.json > "${RUNNER_TEMP}/pinned.json" - python validation/robumeta/compare_reference.py \ + python validation/compare_reference.py \ "${RUNNER_TEMP}/pinned.json" \ pymare/tests/data/robumeta_reference.json \ | tee -a "$GITHUB_STEP_SUMMARY" diff --git a/Makefile b/Makefile index a63a464..ccb07a5 100644 --- a/Makefile +++ b/Makefile @@ -6,7 +6,7 @@ # reports their combined coverage rather than only the last one's. PYTEST_COV := --cov-append --cov-report=xml --cov=pymare -all_tests: lint unittest test_stan test_robumeta test_metafor +all_tests: lint unittest test_stan test_robumeta test_metafor test_clubsandwich help: @echo "Please use 'make ' where is one of:" @@ -16,8 +16,10 @@ help: @echo " test_stan to run the Stan sampling tests (needs the stan extra and CmdStan)" @echo " test_robumeta to run the robumeta alignment tests" @echo " test_metafor to run the metafor alignment tests" + @echo " test_clubsandwich to run the clubSandwich alignment tests" @echo " check_robumeta_alignment to regenerate the robumeta reference values (needs Docker)" @echo " check_metafor_alignment to regenerate the metafor reference values (needs Docker)" + @echo " check_clubsandwich_alignment to regenerate the clubSandwich reference values (needs Docker)" @echo " validate_stan to re-measure the Stan model's bias and coverage (~10 min)" @echo " validate_knapp_hartung to re-measure the small-sample corrections (~20 min)" @echo " benchmark to run the asv suite once in the current environment" @@ -45,6 +47,9 @@ test_robumeta: test_metafor: @python -m pytest -m "metafor" $(PYTEST_COV) +test_clubsandwich: + @python -m pytest -m "clubsandwich" $(PYTEST_COV) + # Re-measures the Type I error of the small-sample corrections and fails if any cell # misses the thresholds the default rests on. Not wired into CI and nothing is # pinned from it: these are Monte Carlo estimates, so re-measuring is the honest @@ -59,19 +64,41 @@ validate_knapp_hartung: validate_stan: @python validation/stan/simulate.py --check -# What the "Check robumeta alignment" workflow runs. Needs Docker, because the -# reference values come from R. +# What the alignment workflows run. Needs Docker, because the reference values +# come from R. +# +# Each target regenerates its package's reference files in place and then +# compares them against the pinned copies from git. Numerically, through +# validation/compare_reference.py, rather than with `git diff --exit-code`, +# which the first two of these used to use: the numbers are written at full +# double precision and R reaches them through linear algebra whose last bits +# depend on which BLAS kernel its image picks for the CPU it runs on, so two +# machines with identical R and package versions produce files that differ in +# the 16th digit. A byte comparison reads that as drift. +# +# Note that these rewrite the working tree, so the comparison is against HEAD +# rather than against the files on disk. +COMPARE_REFERENCE = validation/compare_reference.py + +define compare_against_head + pinned="$$(mktemp)"; git show HEAD:pymare/tests/data/$(1) > "$$pinned"; \ + python $(COMPARE_REFERENCE) "$$pinned" pymare/tests/data/$(1); \ + status=$$?; rm -f "$$pinned"; exit $$status + +endef + check_robumeta_alignment: @validation/robumeta/regenerate.sh - @git diff --exit-code -- pymare/tests/data/robumeta_reference.json \ - && echo "The pinned robumeta reference values still match robumeta." + @$(call compare_against_head,robumeta_reference.json) -# Regenerates the metafor reference values that pin the Knapp-Hartung adjustment. -# Needs Docker, for the same reason as the robumeta target. check_metafor_alignment: @validation/metafor/regenerate.sh - @git diff --exit-code -- pymare/tests/data/metafor_reference.json \ - && echo "The pinned metafor reference values still match metafor." + @$(foreach reference,metafor_reference.json metafor_escalc_reference.json \ + metafor_permutest_reference.json,$(call compare_against_head,$(reference))) + +check_clubsandwich_alignment: + @validation/clubsandwich/regenerate.sh + @$(call compare_against_head,clubsandwich_reference.json) # A smoke test of the benchmark suite, not a measurement: --quick takes one # sample per benchmark. The Benchmark workflow is what measures, by timing a diff --git a/pymare/tests/utils.py b/pymare/tests/utils.py index af0003d..6b04c70 100644 --- a/pymare/tests/utils.py +++ b/pymare/tests/utils.py @@ -60,8 +60,13 @@ def load_metafor_reference(): metafor is not a test dependency, so these numbers are pinned rather than recomputed on every run, in the same arrangement as :func:`load_robumeta_reference`. ``validation/metafor/regenerate.sh`` rewrites - the file from the pinned R image, and the ``Check metafor alignment`` workflow - runs that script on every pull request and fails if anything moved. + this file and the escalc and permutest references beside it from the pinned R + image, and the ``Check alignment with R packages`` workflow runs that script + on every pull request and fails if anything moved. + + The escalc, permutest and clubSandwich references are read directly by the + test modules that use them rather than through a loader here, since none of + them is needed by a fixture. """ with open(op.join(get_test_data_path(), "metafor_reference.json")) as fobj: return json.load(fobj) diff --git a/validation/clubsandwich/Dockerfile b/validation/clubsandwich/Dockerfile index 42633a2..1a919fa 100644 --- a/validation/clubsandwich/Dockerfile +++ b/validation/clubsandwich/Dockerfile @@ -11,5 +11,7 @@ # reference images share a base layer. FROM r-base:4.4.1 RUN R -e 'install.packages(c("metafor", "clubSandwich"), repos="https://packagemanager.posit.co/cran/2024-07-01")' -COPY run_clubsandwich.R /opt/run_clubsandwich.R -ENTRYPOINT ["Rscript", "/opt/run_clubsandwich.R"] +# Rscript as the entrypoint rather than the generator, matching +# validation/metafor, so regenerate.sh names the script it wants. +COPY run_clubsandwich.R /opt/ +ENTRYPOINT ["Rscript"] diff --git a/validation/clubsandwich/regenerate.sh b/validation/clubsandwich/regenerate.sh index c17da6e..ff737d4 100755 --- a/validation/clubsandwich/regenerate.sh +++ b/validation/clubsandwich/regenerate.sh @@ -18,4 +18,5 @@ docker run --rm \ --user "$(id -u):$(id -g)" \ -v "${data_dir}:/data" \ pymare-clubsandwich \ + /opt/run_clubsandwich.R \ /data/robumeta_correlated_effects.csv /data/clubsandwich_reference.json diff --git a/validation/compare_reference.py b/validation/compare_reference.py new file mode 100755 index 0000000..e99afd9 --- /dev/null +++ b/validation/compare_reference.py @@ -0,0 +1,189 @@ +#!/usr/bin/env python +"""Compare a regenerated R reference file against the pinned one. + +Usage +----- +:: + + validation/compare_reference.py PINNED REGENERATED + +Exits non-zero, naming the values that moved, if the two disagree by more than +:data:`RTOL`. + +Works on any of the reference files under ``pymare/tests/data`` that +``validation/*/run_*.R`` writes, without being told which: every one of them is +a ``{"source": {...}, "cases": [...]}`` document whose cases hold numbers, lists +of numbers, nested objects of them, or the strings that name the case. See +:func:`numbers`. + +Why not ``git diff`` +-------------------- +The reference values are written at full double precision, and R reaches them +through linear algebra whose last bits depend on which BLAS kernel its image +picks for the CPU it runs on. Two GitHub runners therefore produce files that +differ in the 16th significant digit -- on one observed run of the robumeta +reference, 84 of 168 values, by at most 2.6e-16 absolute -- with the R and +package versions identical. A byte comparison reads that as drift and fails a +tree that never touched the harness. + +This generalizes the per-package script ``validation/robumeta`` used to carry, +which the metafor README named as a follow-up. That was worth doing rather than +copying, because the metafor reference has grown past what a byte comparison can +police: regenerating it under the same R and metafor versions on a different BLAS +moves numbers by up to 9.5e-14 relative on the inference path and 3.3e-9 on an ML +tau^2, the latter being an optimizer landing a step earlier or later. +""" + +import json +import sys + +import numpy as np + +#: Relative tolerance for the numbers. Set at or below the tolerance the +#: alignment tests hold PyMARE to for the same quantities, so that a pin this +#: check accepts is still good to the precision those tests rely on. +RTOL = 1e-11 + +#: Absolute floor, for any value near zero where a relative tolerance says +#: little. +ATOL = 1e-12 + +#: Keys whose values are searched by their own rules rather than compared as +#: numbers: strings naming the case, which :func:`numbers` folds into the label +#: instead. Anything not listed here and not a number is a mistake in the +#: generator, and :func:`numbers` raises rather than skipping it -- a reference +#: file that stopped recording a quantity would otherwise pass this check by +#: having nothing left to compare. +LABEL_KEYS = ("case", "design", "model", "method", "test", "variances", "rho") + +#: Quantities allowed to move further than :data:`RTOL`, with the tolerance each +#: gets instead and why. Keyed by the trailing component of the label, so it +#: applies to that quantity in every case of every file. +LOOSE = { + # Both metafor and PyMARE reach an ML or REML tau^2 by numerical search, and + # a different BLAS moves the objective enough for the search to stop a step + # earlier or later. Observed 3.3e-9 relative between two runs that agreed on + # every closed-form quantity to 1e-14. Still far tighter than the 1e-4 that + # pymare/tests/test_metafor_random_effects.py holds a profiled tau^2 to. + "tau2": 1e-7, +} + + +def numbers(document, path=()): + """Return the document's numbers, labelled by where each one came from. + + Parameters + ---------- + document : :obj:`dict` or :obj:`list` or :obj:`float` + A parsed reference file, or any part of one. + path : :obj:`tuple` of :obj:`str`, optional + The labels of the enclosing cases and keys, used to build the label. + + Returns + ------- + :obj:`dict` + Label to array. Comparing two of these compares the values, the number + of them, and which cases are present, all at once. + + Raises + ------ + TypeError + If a value is neither a number, nor a list of them, nor a nested object + of them, nor one of :data:`LABEL_KEYS`. Raising rather than skipping is + deliberate: a quantity this function silently ignored would be a + quantity the check does not police. + """ + if isinstance(document, dict): + label = tuple(str(document[key]) for key in LABEL_KEYS if key in document) + collected = {} + for key, value in document.items(): + if key in LABEL_KEYS: + continue + collected.update(numbers(value, path + label + (key,))) + return collected + + if isinstance(document, list) and document and isinstance(document[0], dict): + return { + label: value for entry in document for label, value in numbers(entry, path).items() + } + + if document is None: + # A quantity the R side reported as NA, such as the degrees of freedom + # of an uncorrected fit. Compared as a value so that becoming a number, + # or a number becoming null, is a difference rather than a silent skip. + return {" ".join(path): np.array([np.nan])} + + try: + values = np.atleast_1d(np.asarray(document, dtype=float)) + except (TypeError, ValueError) as error: + raise TypeError(f"{' '.join(path)}: not numeric ({document!r})") from error + return {" ".join(path): values} + + +def tolerance(label): + """Return the relative tolerance for one label, per :data:`LOOSE`.""" + return LOOSE.get(label.rsplit(" ", 1)[-1], RTOL) + + +def compare(pinned, regenerated): + """Return a list of human-readable differences, empty when the two agree.""" + problems = [] + + # Exactly, because this is where a rotted image shows up: the source block + # records the R and package versions the numbers came from, and those either + # match or the pin is stale rather than wobbly. + if pinned["source"] != regenerated["source"]: + problems.append(f"source block moved: {pinned['source']} -> {regenerated['source']}") + + old, new = numbers(pinned["cases"]), numbers(regenerated["cases"]) + if old.keys() != new.keys(): + problems.append(f"cases moved: {sorted(old.keys() ^ new.keys())}") + + for label in sorted(old.keys() & new.keys()): + rtol = tolerance(label) + if old[label].shape != new[label].shape: + problems.append(f"{label}: {old[label].size} values -> {new[label].size}") + elif not np.allclose(old[label], new[label], rtol=rtol, atol=ATOL, equal_nan=True): + problems.append( + f"{label} (rtol {rtol:g}): {old[label].tolist()} -> {new[label].tolist()}" + ) + + return problems + + +def main(argv): + """Compare the two files named on the command line.""" + if len(argv) != 3: + print(__doc__) + return 2 + + # Named after the regenerated file rather than the pinned one, because the + # pinned copy is usually a temporary file `git show` was piped into. + name = argv[2].rsplit("/", 1)[-1] + with open(argv[1]) as fobj: + pinned = json.load(fobj) + with open(argv[2]) as fobj: + regenerated = json.load(fobj) + + problems = compare(pinned, regenerated) + if not problems: + print(f"The pinned values in {name} still match R, to {RTOL:g}.") + return 0 + + print(f"## Reference values moved in {name}") + print() + print( + "Regenerating produced numbers further from those pinned in" + f" `pymare/tests/data/{name}` than the tolerances in" + " `validation/compare_reference.py` allow. Either the pinned image no" + " longer computes what it did, or it is no longer the image the pin came" + " from." + ) + print() + for problem in problems: + print(f"- {problem}") + return 1 + + +if __name__ == "__main__": + sys.exit(main(sys.argv)) diff --git a/validation/metafor/Dockerfile b/validation/metafor/Dockerfile index b7e1072..10f4c46 100644 --- a/validation/metafor/Dockerfile +++ b/validation/metafor/Dockerfile @@ -3,9 +3,11 @@ # Two things are pinned, because either one moving would move the numbers: the R # version (the base image) and the metafor version (a dated CRAN snapshot, since # ``install.packages`` on its own tracks whatever CRAN happens to be serving -# that day). The same pins as validation/robumeta, so the two reference images -# share a base layer. +# that day). The same R pin as validation/robumeta and validation/clubsandwich, +# so the three reference images share a base layer. FROM r-base:4.4.1 RUN R -e 'install.packages("metafor", repos="https://packagemanager.posit.co/cran/2024-07-01")' -COPY run_metafor.R /opt/run_metafor.R -ENTRYPOINT ["Rscript", "/opt/run_metafor.R"] +# All three generators, with Rscript as the entrypoint rather than one of them, +# so regenerate.sh can run each in turn against one built image. +COPY run_metafor.R run_escalc.R run_permutest.R /opt/ +ENTRYPOINT ["Rscript"] diff --git a/validation/metafor/README.md b/validation/metafor/README.md index 9e4f46b..78e1c59 100644 --- a/validation/metafor/README.md +++ b/validation/metafor/README.md @@ -1,60 +1,116 @@ # metafor reference values -PyMARE's `small_sample_correction` parameter implements the Knapp-Hartung adjustment (Knapp & -Hartung, 2003, *Statistics in Medicine* 22(17), 2693-2710) in the form `metafor`'s -`rma.uni` applies it, including the modification metafor spells `test="adhoc"` and PyMARE -`"knapp-hartung-conservative"`. This -directory regenerates the reference values `pymare/tests/test_metafor_alignment.py` -pins. +[metafor](https://cran.r-project.org/package=metafor) (Viechtbauer, 2010, +*Journal of Statistical Software* 36(3), 1-48) is the reference implementation +for most of what PyMARE computes. This directory regenerates the values three +test modules pin against it: + +| Reference file | metafor call | Test module | +| --- | --- | --- | +| `metafor_reference.json` | `rma.uni`, `confint` | `test_metafor_alignment.py`, `test_metafor_random_effects.py` | +| `metafor_escalc_reference.json` | `escalc` | `test_metafor_escalc.py` | +| `metafor_permutest_reference.json` | `permutest(exact = TRUE)` | `test_metafor_permutest.py` | ```bash -make check_metafor_alignment # needs Docker; rewrites the pinned file in place +make check_metafor_alignment # needs Docker; rewrites the pinned files in place make test_metafor # check PyMARE against the pinned values ``` -Regeneration goes through a Docker image with pinned R and metafor versions, so -unchanged numbers produce an unchanged file and the diff is the answer. metafor is -an R package and cannot be a test dependency, which is why the numbers are pinned -rather than recomputed on every test run. Mirrors `validation/robumeta`, except -that there is no scheduled workflow re-verifying the pin against metafor; adding -one means copying `robumeta-alignment.yml` and generalising the numeric comparison -in `validation/robumeta/compare_reference.py`, which is a follow-up rather than -part of this change. +Regeneration goes through a Docker image with pinned R and metafor versions. +metafor is an R package and cannot be a test dependency, which is why the numbers +are pinned rather than recomputed on every test run. The `Check alignment with R +packages` workflow regenerates them on every pull request and fails if they move, +which is what keeps a pinned file from quietly becoming a stale one. The +comparison is numeric, through the shared `validation/compare_reference.py`: the +numbers are written at full double precision and R reaches them through linear +algebra whose last bits depend on which BLAS kernel its image picks for the CPU +it runs on, so regenerating an unchanged tree on a different machine moves +numbers by up to 9.5e-14 relative on the inference path, and 3.3e-9 on an `ML` +tau^2 where an optimizer stops a step earlier or later. + +The pinned files record metafor's own vocabulary -- its `test=` spellings, its +`measure=` names -- because they record what metafor was *asked*. Each test +module translates to PyMARE's names in the open rather than hiding the mapping in +the generator. + +--- -The pinned file records metafor's own `test=` spellings, because it records what -metafor was asked; `CORRECTIONS` in the alignment module translates them to -PyMARE's `small_sample_correction` values. +# `rma.uni` ## What agrees -**The adjustment itself agrees to machine precision, in all 180 cases:** 1.8e-15 -on the coefficients, 5.6e-16 on the standard errors, 3.0e-15 on the p-values, and -5.9e-11 absolute on the interval bounds -- the last bounded not by the adjustment -but by `scipy.stats.t.ppf` and R's `qt` disagreeing in their final bits. The -degrees of freedom agree exactly, being a count. +**The Knapp-Hartung adjustment agrees to machine precision, in all 180 cases:** +1.8e-15 on the coefficients, 5.6e-16 on the standard errors, 3.0e-15 on the +p-values, and 5.9e-11 absolute on the interval bounds -- the last bounded not by +the adjustment but by `scipy.stats.t.ppf` and R's `qt` disagreeing in their final +bits. The degrees of freedom agree exactly, being a count. That comparison supplies PyMARE with metafor's own tau^2, which is what isolates the adjustment from the tau^2 estimators. `FE` and `DL` also agree end to end, tau^2 included, because both reach it in closed form. -## What does not, and why +**Everything else `rma.uni` reports and PyMARE also computes agrees too**, over +the 60 distinct design x model x estimator cells: + +| Quantity | PyMARE | Worst relative deviation | +| --- | --- | --- | +| `QE`, `QEp` | `get_heterogeneity_stats()["Q"]`, `["p(Q)"]` | 2.9e-14, 1.5e-13 | +| `QEp` in logs | `["logp(Q)"]` | 4.7e-14 | +| `I2`, `H2` for `FE` and `DL` | `["I^2"]`, `["H"]` | 4.9e-14, 1.5e-14 | +| `confint` tau^2 bounds | `get_re_stats()["ci_l"]`, `["ci_u"]` | 1.3e-13 | +| `HE` tau^2, intercept-only models | `Hedges().fit(...)` | 2.2e-16 | +| `ML`, `REML` tau^2 | `VarianceBasedLikelihoodEstimator` | 2.7e-5 | + +Two notes on that table. + +`I2` and `H2` are floored by PyMARE and not by metafor -- at 0 and 1 +respectively, as Higgins & Thompson (2002) define them -- so the comparison +applies the floors to metafor's values rather than dropping the cells where they +bite. metafor can print an `H2` below one. + +The `confint` reference is generated with `control = list(tol = 1e-12)`. Its +default is `uniroot`'s `.Machine$double.eps^0.25`, about 1.2e-4 relative, which +is far coarser than the bound itself; pinning the default would check PyMARE +against metafor's display precision instead of against the bound it solves for. +Finding this is also what turned up the defect below. + +## What this check found + +`pymare.stats.q_profile` computed both tau^2 bounds by minimizing +`(Q(tau^2) - crit)**2` with `scipy.optimize.minimize`. Squaring turns a +transversal crossing into a tangential minimum, so the gradient the minimizer +follows vanishes as the critical value is approached and it stops early in the +flat upper tail. The upper bound was out by up to 3.6e-2 relative. It now solves +for the root with `scipy.optimize.brentq` and agrees with metafor to 1.3e-13. + +Both tests that pinned the old value asserted it to two decimals, which is why +nothing failed. + +## What does not agree, and why -Three divergences, all in tau^2 and all visible with no correction applied, so none of them is -caused by the adjustment. They are why the alignment tests compare `ML`, `REML` -and `HE` only through metafor's own tau^2 -- folding an optimizer's tolerance into -a check on a closed-form scale factor would blunt it. +Three divergences, all in tau^2 or in quantities derived from it, and all +visible with no correction applied -- so none of them is caused by the +adjustment. They are why `test_metafor_alignment` compares `ML`, `REML` and `HE` +only through metafor's own tau^2: folding an optimizer's tolerance into a check +on a closed-form scale factor would blunt it. +`test_metafor_random_effects` covers each of them directly instead, by asserting +the *cause* rather than the size of the gap. | Divergence | Size | Cause | | --- | --- | --- | -| `ML`, `REML` tau^2 | ~3e-5 relative | PyMARE profiles tau^2 at `xtol=1e-6`; metafor runs its own optimizer to its own tolerance | -| `ML` on `extreme_k10` with one moderator | 0 vs 0.011 | the two optimizers land on opposite sides of the tau^2 = 0 boundary, where a profile likelihood is flattest because the weights are most unequal | -| `HE` tau^2 | up to 0.08 absolute | PyMARE and metafor differ on whether the mean sampling variance is taken over raw or weighted rows; predates this work by years, and `test_hedges_estimator` has recorded it since PyMARE's Hedges estimator was written | +| `I2`, `H2` for `HE`, `ML`, `REML` | unbounded | PyMARE always reports the Q-based Higgins-Thompson pair. metafor reports that pair only for `FE` and `DL`, where it coincides with `tau^2 / (tau^2 + v_t)`, and switches to the tau^2-based pair otherwise -- so metafor's `I2` depends on which tau^2 estimator was asked for and PyMARE's does not. Both are defensible; they are not the same number. | +| `HE` tau^2, models with moderators | up to 0.14 relative | metafor subtracts `tr(PV) / (K - P)`, PyMARE subtracts `sum(v) / K`. With an intercept as the only predictor `P = I - J/K`, the trace is `sum(v)(K - 1)/K`, and the two are algebraically the same; with a moderator they are not. So the divergence is specific to meta-regression. A previous version of this README recorded it as general. | +| `ML`, `REML` tau^2 | ~3e-5 relative | PyMARE profiles tau^2 at `xtol=1e-6`; metafor runs its own optimizer to its own tolerance. | +| `ML` on `extreme_k10` with one moderator | 0 vs 0.011 | The two searches land on opposite sides of the tau^2 = 0 boundary, where a profile likelihood is flattest because the weights are most unequal. metafor is the one that stops at zero. A previous version of this README had the direction backwards. | ## What is compared 180 cases, the full grid of design x model x tau^2 estimator x `test`. `test_reference_covers_every_combination` asserts that grid, so the check cannot -quietly shrink to the cases that happen to pass. +quietly shrink to the cases that happen to pass. The heterogeneity statistics and +the tau^2 interval do not depend on `test`, which +`test_heterogeneity_does_not_depend_on_the_correction` asserts from the reference +before the other tests drop to the 60 distinct cells. | Knob | Values | | --- | --- | @@ -63,9 +119,9 @@ quietly shrink to the cases that happen to pass. | tau^2 estimator | `FE`, `DL`, `HE`, `ML`, `REML` | | `test` (metafor's spellings) | `z`, `knha`, `adhoc` | -The designs are in `pymare/tests/data/metafor_small_sample.csv`, chosen to bracket -the condition that decides whether the adjustment behaves -- how unequal the -weights are, and how few observations there are: +The designs are in `metafor_small_sample.csv`, chosen to bracket the condition +that decides whether the adjustment behaves -- how unequal the weights are, and +how few observations there are: | Design | K | max(v) / min(v) | | --- | --- | --- | @@ -74,8 +130,8 @@ weights are, and how few observations there are: | `extreme_k10` | 10 | 10,000 | | `moderate_k20` | 20 | 30 | -`extreme_k10` is there because IntHout, Ioannidis & Borm (2014) and Röver, Knapp & -Friede (2015) both report the adjustment exceeding its nominal level for few +`extreme_k10` is there because IntHout, Ioannidis & Borm (2014) and Röver, Knapp +& Friede (2015) both report the adjustment exceeding its nominal level for few observations of very unequal precision. Comparing against metafor in that cell checks that PyMARE reproduces the reference implementation there too, including its `test="adhoc"` remedy, which PyMARE spells @@ -89,3 +145,143 @@ under `test="knha"`) has no PyMARE counterpart -- PyMARE reports per-coefficient statistics and has no joint test. `test="t"`, the t reference without the covariance scaling, is not exposed by PyMARE either: nothing recommends it as a default and it is not needed to reproduce `knha`. + +`rma.uni`'s prediction interval (`predict`) has no PyMARE counterpart. Neither do +its other tau^2 estimators (`EB`, `PM`, `SJ`, `GENQ`) or its other confidence +interval methods for tau^2 (`confint(type = "PL")`), and PyMARE's +`SampleSizeBasedLikelihoodEstimator` has no metafor counterpart, since metafor +has no estimator that takes sample sizes in place of variances. + +--- + +# `escalc` + +`test_metafor_escalc.py` compares PyMARE's effect-size converters against +`escalc` on an eight-row grid of summary statistics in +`metafor_escalc_inputs.csv` -- balanced and unbalanced groups, n from 5 to 400, +zero to very large effects, correlations from 0 to 0.99. + +Each comparison is in one of three relationships, and the test module is +explicit about which. + +## The same closed form + +Exact, to 1e-13: + +| PyMARE | `escalc` measure | Compared | +| --- | --- | --- | +| `RM` | `MN` | estimate and variance | +| `R` | `COR` | estimate | +| `ZR` | `ZCOR` | estimate and variance | +| `RMD` | `MD` | estimate | +| `sdp` | the pooled SD `escalc` divides by, recovered as `MD$yi / (SMD$yi / c(m))` | value | + +## The same quantity, one side approximated + +PyMARE corrects a standardized mean for bias with `1 - 3/(4m - 1)`; metafor uses +the exact `c(m) = gamma(m/2) / (sqrt(m/2) gamma((m-1)/2))`. Rather than pin a +tolerance, the tests bound the error at `0.05 / m**2` -- the approximation's +actual second order, measured at `0.043 / m**2` over the grid, worst 2.7e-3 at +`m = 4` and 2.0e-7 at `m = 398`. A first-order error would break that bound. +This covers the `SM` and `SMD` estimates and the factor itself. + +metafor has no single-group standardized mean, so `SM`'s reference is +`escalc(measure = "SMCC")` with the second measurement set to zero and +uncorrelated with the first. SMCC's change-score SD is then `sd1` and its +numerator is `m1`, so the measure reduces algebraically to `m / sd` with the +exact correction on `n - 1` degrees of freedom -- which is a reference and not an +approximation. The header of `run_escalc.R` spells this out. + +## Different formulas for the same thing + +Two, neither of which can be verified against the other. Each is recorded by a +test asserting which formula each side uses, so the divergence cannot change +shape unnoticed: + +| Quantity | metafor | PyMARE | +| --- | --- | --- | +| raw correlation variance | `(1 - r**2)**2 / (n - 1)`, the asymptotic sampling variance | `(1 - r**2) / (n - 2)`, the squared standard error under the null of no correlation. 51x apart at `r = 0.99` | +| single-group standardized-mean variances | `1/n + y**2 / (2n)`, the large-sample approximation | the exact noncentral-t expressions, which are the better quantity. 2.5x apart at `n = 5` | + +The second is why `test_standardized_mean_variance_is_the_same_order_as_metafor` +bounds by a factor of four rather than a tolerance. + +## What this check found + +Three defects in `pymare/effectsize/expressions.json`, beyond the one +[PR #144](https://github.com/neurostuff/PyMARE/pull/144) fixes, all of the same +kind: a missing pair of parentheses changing what the expression solves to. + +| Expression | Reads | Solves to | Should be | Effect | +| --- | --- | --- | --- | --- | +| `v_rmd` | `v_rmd - (sd1**2 / n1) + (sd2**2 / n2)` | `sd1**2/n1 - sd2**2/n2` | the sum | negative variance whenever the second group is the more variable one, and exactly zero for two equally sized equally variable groups -- which gives that study infinite weight | +| `v_sm` | `... * (1/n + d**2) - d**2` | `A + d**2` | `A - d**2` | the single-group Hedges' g variance is 9x to 110x too large, and grows with the effect instead of being dominated by `1/n` | +| `v_d` (one-sample) | `... - d**2 / j**2 * n` | `A + n d**2 / j**2` | `A - d**2 / j**2` | the variance grows with the sample size | + +Each is recorded as an `xfail(strict=True)` naming the expression and what it +should be, alongside the two-sample `v_d` that PR #144 fixes. `strict` is the +point: correcting an expression turns the test green, pytest reports XPASS as a +failure, and the marker has to go in the same change. All four were confirmed to +flip to XPASS under the corresponding one-line fix, and under all four together +the rest of the suite still passes -- so nothing currently pins the wrong values. + +## What is not compared + +`escalc`'s binary-outcome measures (`RR`, `OR`, `RD`, `PETO`, ...), its +proportion and rate measures, and its other standardized measures (`SMDH`, +`SMD1`, `SMCR`, `ROM`, ...) have no PyMARE counterpart. PyMARE's two-sample `D` +and one-sample `D` have no `escalc` counterpart either, every standardized +measure metafor offers being bias-corrected; they are recoverable as `yi / c(m)`, +which is how the one-sample `D` variance above is compared at all. + +--- + +# `permutest` + +`test_metafor_permutest.py` compares PyMARE's permutation test against +`permutest(exact = TRUE)` on ten cases: the designs small enough to enumerate -- +`2**K` sign flips for the intercept-only models up to K = 10, and `K!` orderings +for the moderator ones at K = 5. Only the exact mode, since the approximate one +draws from each package's own generator. + +## What agrees + +Counting the same permutations of the same PyMARE fits on `|beta / se|` instead +of `|beta|`, and reading the observed statistic out of the identity permutation +in the same batch, reproduces `permutest` **exactly in all ten cases**, both +coefficients of the moderator models included. So the permutation sets agree, the +batched refits agree, and the inclusive comparison agrees. + +## What does not, and why + +What PyMARE actually reports does not, for two reasons. + +**The statistic.** `permutest` counts `|beta / se|`; PyMARE counts `|beta|`. A +permutation test needs a statistic whose null distribution does not move with +what is being permuted away, and `|beta|` does: refitting a permuted dataset +re-estimates tau^2, which changes the weights and so the standard error. The two +coincide only where the standard error is invariant under the permutation -- a +fixed-effects intercept-only model under sign flipping -- which is why four of +the ten cases agree anyway. On `unequal_k5` under `DL`, PyMARE reports 0.5625 +against `permutest`'s 0.5. + +**The tie.** The observed estimate is computed by a different code path from the +permuted ones and the two differ by one unit in the last place, so the inclusive +comparison can drop the identity permutation -- the one that reproduces the +observed data and must therefore count -- along with its sign-flipped mirror. +That understates the p-value by `2 / 2**K` whenever it bites: 0.033203125 against +`permutest`'s 0.03515625 on `extreme_k10`. metafor avoids this by comparing +against `|zval| - sqrt(eps)`. + +Both are recorded by one strict `xfail` covering the whole grid rather than a +parametrized one, because the second cause is a last-place rounding difference +that need not reproduce on every platform in the test matrix while the first is +structural. + +## What is not compared + +`permutest`'s `permci = TRUE`, which inverts the permutation test for a +confidence interval, has no PyMARE counterpart. Neither does its omnibus +`QM` permutation p-value, for the same reason `rma.uni`'s `QM` is not compared. +PyMARE's tau^2 permutation p-value (`perm_p["tau2_p"]`) has no `permutest` +counterpart. diff --git a/validation/metafor/regenerate.sh b/validation/metafor/regenerate.sh index 1894c2b..bdbe2db 100755 --- a/validation/metafor/regenerate.sh +++ b/validation/metafor/regenerate.sh @@ -1,22 +1,38 @@ #!/usr/bin/env bash -# Regenerate the metafor reference values PyMARE's alignment test reads. +# Regenerate the metafor reference values PyMARE's alignment tests read. # # Run from the repository root: # # validation/metafor/regenerate.sh # -# Rewrites pymare/tests/data/metafor_reference.json in place. If the file -# changes, either metafor or PyMARE's reference moved, and the diff says which -# numbers. The alignment workflow runs this script and fails on a non-empty -# diff, so CI and a local run cannot drift apart. +# Rewrites all three files in place: +# +# pymare/tests/data/metafor_reference.json rma.uni +# pymare/tests/data/metafor_escalc_reference.json escalc +# pymare/tests/data/metafor_permutest_reference.json permutest +# +# The alignment workflow runs this script and then compares each result +# numerically against the pinned file with validation/compare_reference.py, so +# CI and a local run cannot drift apart. Numerically rather than with +# `git diff --exit-code`, which this script used to rely on: the numbers are +# written at full double precision and R reaches them through linear algebra +# whose last bits depend on which BLAS kernel its image picks for the CPU it +# runs on, so two machines with identical R and metafor versions produce files +# that differ in the 16th digit. set -euo pipefail repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" data_dir="${repo_root}/pymare/tests/data" docker build --quiet -t pymare-metafor "${repo_root}/validation/metafor" >/dev/null -docker run --rm \ - --user "$(id -u):$(id -g)" \ - -v "${data_dir}:/data" \ - pymare-metafor \ - /data/metafor_small_sample.csv /data/metafor_reference.json + +run() { + docker run --rm \ + --user "$(id -u):$(id -g)" \ + -v "${data_dir}:/data" \ + pymare-metafor "$@" +} + +run /opt/run_metafor.R /data/metafor_small_sample.csv /data/metafor_reference.json +run /opt/run_escalc.R /data/metafor_escalc_inputs.csv /data/metafor_escalc_reference.json +run /opt/run_permutest.R /data/metafor_small_sample.csv /data/metafor_permutest_reference.json diff --git a/validation/robumeta/README.md b/validation/robumeta/README.md index cf8a0a1..7b0960b 100644 --- a/validation/robumeta/README.md +++ b/validation/robumeta/README.md @@ -11,8 +11,7 @@ validation/robumeta/regenerate.sh ``` That rewrites `pymare/tests/data/robumeta_reference.json` in place, from a Docker -image with pinned R and robumeta versions. Unchanged numbers produce an -unchanged file, so the diff is the answer: +image with pinned R and robumeta versions: ```bash make check_robumeta_alignment @@ -21,9 +20,24 @@ make check_robumeta_alignment robumeta is an R package and cannot be a test dependency, so the numbers are pinned rather than recomputed on every test run. What keeps a pinned file from becoming a stale one is the `Check robumeta alignment` workflow, which runs the -script above on every pull request and fails on any difference. Rerun it and +script above on every pull request and fails if the result moves. Rerun it and commit the result when you change the estimator on purpose. +The comparison is numeric rather than a `git diff`, through +`validation/compare_reference.py` -- shared with `validation/metafor` and +`validation/clubsandwich`, which is why it lives one directory up. The numbers +are written at full double precision and R reaches them through linear algebra +whose last bits depend on which BLAS kernel its image picks for the CPU it runs +on, so two runners with identical R and robumeta versions produce files that +differ in the 16th digit. + +## What this check does *not* cover + +The CR2 residual adjustment in general. robumeta's working model has constant +within-study weights by construction, and that is exactly the condition under +which PyMARE's CR2 and `clubSandwich`'s coincide -- so this check cannot tell +them apart. `validation/clubsandwich` is the one that can, and does not. + ## What agrees PyMARE and robumeta agree to ~1e-14 on tau^2, the coefficients, the diff --git a/validation/robumeta/compare_reference.py b/validation/robumeta/compare_reference.py deleted file mode 100755 index 6cecdcd..0000000 --- a/validation/robumeta/compare_reference.py +++ /dev/null @@ -1,102 +0,0 @@ -#!/usr/bin/env python -"""Compare a regenerated robumeta reference file against the pinned one. - -Usage ------ -:: - - validation/robumeta/compare_reference.py PINNED REGENERATED - -Exits non-zero, naming the values that moved, if the two disagree by more than -:data:`RTOL`. - -Why not ``git diff`` --------------------- -The reference values are written at full double precision, and ``robu()`` reaches -them through linear algebra whose last bits depend on which BLAS kernel R's image -picks for the CPU it runs on. Two GitHub runners therefore produce files that -differ in the 16th significant digit -- on one observed run, 84 of 168 values, by -at most 2.6e-16 absolute -- with the R and robumeta versions identical. A byte -comparison reads that as drift and fails a tree that never touched the harness. -""" - -import json -import sys - -import numpy as np - -#: Relative tolerance for the numbers. Set below the ``RTOL`` that -#: ``pymare/tests/test_robumeta_alignment.py`` holds PyMARE to, so that a pin this -#: check accepts is still good to the precision that test relies on, and far above -#: the 4e-14 that runner-to-runner rounding has been seen to produce. -RTOL = 1e-11 - -#: Absolute floor, for any value near zero where a relative tolerance says little. -ATOL = 1e-12 - - -def numbers(document): - """Return the file's numbers, labelled by the case and quantity they belong to. - - Parameters - ---------- - document : :obj:`dict` - A parsed reference file. - - Returns - ------- - :obj:`dict` - Label to array. Comparing two of these compares the values, the number of - them, and which cases are present, all at once. - """ - return { - f"{case['model']} rho={case['rho']} {case['variances']} {key}": np.atleast_1d(case[key]) - for case in document["cases"] - for key in ("tau2", "beta", "se", "dof") - } - - -def main(argv): - """Compare the two files named on the command line.""" - if len(argv) != 3: - print(__doc__) - return 2 - - pinned, regenerated = (json.load(open(path)) for path in argv[1:3]) - problems = [] - - # Exactly, because this is where a rotted image shows up: it records the R and - # robumeta versions the numbers came from, and those either match or the pin - # is stale rather than wobbly. - if pinned["source"] != regenerated["source"]: - problems.append(f"source block moved: {pinned['source']} -> {regenerated['source']}") - - old, new = numbers(pinned), numbers(regenerated) - if old.keys() != new.keys(): - problems.append(f"cases moved: {sorted(old.keys() ^ new.keys())}") - for label in sorted(old.keys() & new.keys()): - if old[label].shape != new[label].shape: - problems.append(f"{label}: {old[label].size} values -> {new[label].size}") - elif not np.allclose(old[label], new[label], rtol=RTOL, atol=ATOL): - problems.append(f"{label}: {old[label].tolist()} -> {new[label].tolist()}") - - if not problems: - print(f"The pinned robumeta reference values still match robumeta, to {RTOL:g}.") - return 0 - - print("## robumeta reference values moved") - print() - print( - "`validation/robumeta/regenerate.sh` produced numbers more than" - f" {RTOL:g} relative away from those pinned in" - " `pymare/tests/data/robumeta_reference.json`. Either the pinned image no" - " longer computes what it did, or it is no longer the image the pin came from." - ) - print() - for problem in problems: - print(f"- {problem}") - return 1 - - -if __name__ == "__main__": - sys.exit(main(sys.argv)) diff --git a/validation/robumeta/regenerate.sh b/validation/robumeta/regenerate.sh index efdcfb3..67804ea 100755 --- a/validation/robumeta/regenerate.sh +++ b/validation/robumeta/regenerate.sh @@ -5,10 +5,10 @@ # # validation/robumeta/regenerate.sh # -# Rewrites pymare/tests/data/robumeta_reference.json in place. If the file -# changes, either robumeta or PyMARE's reference moved, and the diff says -# which numbers. The alignment workflow runs this script and fails on a -# non-empty diff, so CI and a local run cannot drift apart. +# Rewrites pymare/tests/data/robumeta_reference.json in place. The alignment +# workflow runs this script and then compares the result numerically against +# the pinned file with validation/compare_reference.py, so CI and a local run +# cannot drift apart. set -euo pipefail repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" From f63eee7cb68cc09341aa153e672bb6f68c96bb92 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 20:49:49 +0000 Subject: [PATCH 06/13] [DOC] document the R alignment checks in CONTRIBUTING The contributing guide described only the robumeta check; the metafor targets predate this branch and were never added to it, and this branch adds two more reference sets and a shared comparator. Replaces "Alignment with robumeta" with a section covering all six alignment modules, what each pins against, and where the divergences are written down -- with a pointer to read the validation README before changing an estimator, since several of the divergences are deliberate and the READMEs are the only place that says which. Also documents the strict-xfail convention the new modules use, because it has a trap in it: a contributor who fixes one of the recorded defects will see pytest report XPASS as a failure, and needs to know the marker is what to delete rather than something to work around. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011o2j5ZNAdzPFL6LBqDqUsU --- CONTRIBUTING.md | 70 +++++++++++++++++++++++++++++++++++++++---------- 1 file changed, 56 insertions(+), 14 deletions(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index ff6361f..e5dd05d 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -93,18 +93,27 @@ and so on. Fixtures live in `pymare/tests/conftest.py` and helpers that are neither fixtures nor tests live in `pymare/tests/utils.py`, so a test file holds only tests. -Two groups of tests are marked, because they need something the default -environment does not have: +Some tests are marked, because they need something the default environment does +not have, or because they cost more than a pull request should: | Target | What it runs | Needs | | --- | --- | --- | | `make unittest` | everything except the Stan sampling tests | nothing extra | | `make test_stan` | the Stan sampling tests | `pip install -e .[stan]`, then `make install_cmdstan` | | `make test_robumeta` | the robumeta alignment tests | nothing extra | +| `make test_metafor` | the metafor alignment tests | nothing extra | +| `make test_clubsandwich` | the clubSandwich alignment tests | nothing extra | | `make check_robumeta_alignment` | regenerates the robumeta reference values | Docker | +| `make check_metafor_alignment` | regenerates the metafor reference values | Docker | +| `make check_clubsandwich_alignment` | regenerates the clubSandwich reference values | Docker | | `make validate_stan` | re-measures the Stan model's bias and coverage (~10 min) | the same as `test_stan` | | `make lint` | flake8 over `pymare` and `benchmarks` | nothing extra | +The three `test_*` alignment targets need nothing extra because they read pinned +numbers; only regenerating those numbers needs R, and that is what the +`check_*_alignment` targets do. `make unittest` runs the alignment tests too, so +you do not have to remember them. + Each of these has a GitHub Actions job behind it, so a target that passes locally is the same check that runs on your pull request. @@ -134,20 +143,51 @@ file to the same thresholds on every run, and the `Validate the Stan model` workflow re-measures on a schedule. See `validation/stan/README.md` for the arrangement and the measurements. -### Alignment with robumeta +### Alignment with R packages -`pymare/tests/test_robumeta_alignment.py` pins PyMARE's correlated-effects model -against the R package [robumeta][link_robumeta], over every combination of model, -rho and variance column that both implementations can express. robumeta cannot be -a test dependency, so its output is pinned in -`pymare/tests/data/robumeta_reference.json`. +Most of what PyMARE computes has a reference implementation in R, and six test +modules pin PyMARE against one: -`make check_robumeta_alignment` regenerates that file inside a Docker image with -pinned R and robumeta versions, and fails if any number moved. The -`Check robumeta alignment` workflow runs the same script on every pull request, -so a change to the estimator that breaks agreement shows up as a failing check -rather than as a stale pin. If you changed the estimator on purpose, rerun the -script and commit the regenerated file. +| Module | Pins against | Covers | +| --- | --- | --- | +| `test_robumeta_alignment.py` | [robumeta][link_robumeta] | the correlated-effects working model (`weight_scheme="rescale"`) | +| `test_metafor_alignment.py` | [metafor][link_metafor] `rma.uni` | the Knapp-Hartung adjustment, over the whole inference path | +| `test_metafor_random_effects.py` | metafor `rma.uni`, `confint` | tau^2, Cochran's Q, `I^2`, `H`, and the Q-profile interval | +| `test_metafor_escalc.py` | metafor `escalc` | the effect-size converters | +| `test_metafor_permutest.py` | metafor `permutest` | the exact permutation test | +| `test_clubsandwich_alignment.py` | [clubSandwich][link_clubsandwich] | the CR2 covariance and its Satterthwaite degrees of freedom | + +None of those packages can be a test dependency, so their output is pinned under +`pymare/tests/data/*_reference.json`. Each `validation//` directory +holds the R script that produced its file, a Dockerfile pinning the R and package +versions, a `regenerate.sh` that runs one against the other, and a README +recording what agrees, to what tolerance, and what does not and why. **Read the +README before changing an estimator**: it is where the known divergences are +written down, and several of them are deliberate. + +`make check__alignment` regenerates the files and fails if any number +moved. The `Check alignment with R packages` and `Check robumeta alignment` +workflows run the same scripts on every pull request, so a change that breaks +agreement shows up as a failing check rather than as a stale pin. If you changed +an estimator on purpose, rerun the script and commit the regenerated file. + +The regenerated file is compared numerically, by +`validation/compare_reference.py`, rather than with `git diff`. The numbers are +written at full double precision and R reaches them through linear algebra whose +last bits depend on which BLAS kernel its image picks for the CPU it runs on, so +two machines with identical R and package versions produce files that differ in +the last digits. The tolerances are in that script, set below the ones the +alignment tests themselves rely on. + +#### Divergences recorded as strict xfails + +Some alignment tests are marked `xfail(strict=True)`, because measuring a +quantity against its reference implementation turned up a defect rather than a +divergence. The marker names the expression or code path and what it should be. +`strict` is load-bearing: correcting the defect turns the test green, pytest +reports XPASS as a *failure*, and the marker has to be removed in the same +change. So if a fix of yours makes one of these pass, delete its marker -- do not +work around it. ### Benchmarks @@ -203,4 +243,6 @@ You're awesome. [link_stemmrolemodels]: https://github.com/KirstieJane/STEMMRoleModels [link_zenodo]: https://github.com/neurostuff/PyMARE/blob/master/.zenodo.json [link_robumeta]: https://cran.r-project.org/package=robumeta +[link_metafor]: https://cran.r-project.org/package=metafor +[link_clubsandwich]: https://cran.r-project.org/package=clubSandwich [link_asv]: https://asv.readthedocs.io/en/stable/ From 400cf1a422b8f7070e43795fe7fcc92571994aa9 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 22:35:12 +0000 Subject: [PATCH 07/13] [TST] retire the SMD variance xfail, which PR #144 satisfied Merging master brings in #144, which corrects the two-sample v_d expression. test_standardized_mean_difference_variance_matches_metafor was a strict xfail on exactly that, so the merge turned it green and pytest reported XPASS as a failure -- which is what strict is for. The marker comes out here, and the test becomes an ordinary passing one holding PyMARE's SMD variance to within 2 / (n1 + n2) of metafor's, the order at which the two approximations legitimately differ. Three markers remain, on v_rmd and on the two one-sample standardized-mean variances. Nothing else in the merge conflicts. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011o2j5ZNAdzPFL6LBqDqUsU --- CONTRIBUTING.md | 4 ++++ pymare/tests/test_metafor_escalc.py | 28 ++++++++++++---------------- validation/metafor/README.md | 21 +++++++++++++++------ 3 files changed, 31 insertions(+), 22 deletions(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index e5dd05d..7e5580d 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -189,6 +189,9 @@ reports XPASS as a *failure*, and the marker has to be removed in the same change. So if a fix of yours makes one of these pass, delete its marker -- do not work around it. +That is not hypothetical: it is how the marker covering the two-sample Cohen's d +variance was retired when [PR #144][link_pr144] landed. + ### Benchmarks Performance is guarded by [asv][link_asv]. The suite lives in `benchmarks/`, and @@ -245,4 +248,5 @@ You're awesome. [link_robumeta]: https://cran.r-project.org/package=robumeta [link_metafor]: https://cran.r-project.org/package=metafor [link_clubsandwich]: https://cran.r-project.org/package=clubSandwich +[link_pr144]: https://github.com/neurostuff/PyMARE/pull/144 [link_asv]: https://asv.readthedocs.io/en/stable/ diff --git a/pymare/tests/test_metafor_escalc.py b/pymare/tests/test_metafor_escalc.py index 7800645..8bd73aa 100644 --- a/pymare/tests/test_metafor_escalc.py +++ b/pymare/tests/test_metafor_escalc.py @@ -351,7 +351,7 @@ def test_standardized_mean_variances_are_exact_not_asymptotic(one_sample, escalc "negative whenever the second group is the more variable one, and " "exactly zero for two equally sized, equally variable groups -- which " "gives that study infinite weight. The fix is to parenthesize the " - "denominator, as PR #144 does for the two-sample Cohen's d" + "denominator, as PR #144 did for the two-sample Cohen's d" ), ) def test_raw_mean_difference_variance_matches_metafor(two_sample): @@ -366,24 +366,20 @@ def test_raw_mean_difference_variance_matches_metafor(two_sample): assert_exact(v, expected("MD", "vi"), "RMD variance") -@pytest.mark.xfail( - strict=True, - reason=( - "v_d in pymare/effectsize/expressions.json reads " - "'v_d - ((n1 + n2)/(n1 * n2) + d**2 / 2 * (n1 + n2 - 2))', so the " - "squared-effect term is multiplied by the residual degrees of freedom " - "instead of divided by twice them. Fixed by PR #144; remove this " - "marker with it. Issue #143" - ), -) def test_standardized_mean_difference_variance_matches_metafor(two_sample, escalc_inputs): """``SMD``'s variance must agree with metafor's to order ``1 / N``. - Not exactly, even with the expression corrected: PyMARE scales - ``d**2 / (2 (n1 + n2 - 2))`` by ``j**2`` and metafor adds - ``g**2 / (2 (n1 + n2))``, two approximations of the same variance that - differ at order ``1 / N``. :data:`SMD_VARIANCE_BOUND` is that order, - measured at 0.72 of the bound in the worst cell of the grid. + Not exactly: PyMARE scales ``d**2 / (2 (n1 + n2 - 2))`` by ``j**2`` and + metafor adds ``g**2 / (2 (n1 + n2))``, two approximations of the same + variance that differ at order ``1 / N``. :data:`SMD_VARIANCE_BOUND` is that + order, measured at 0.72 of the bound in the worst cell of the grid. + + This was a strict xfail until + https://github.com/neurostuff/PyMARE/pull/144 corrected the expression: + it used to read ``d**2 / 2 * (n1 + n2 - 2)``, multiplying the + squared-effect term by the residual degrees of freedom instead of dividing + by twice them, which put the variance out by up to three orders of + magnitude and made it *grow* with the sample size. """ _, v = measure(two_sample, "SMD") want = expected("SMD", "vi") diff --git a/validation/metafor/README.md b/validation/metafor/README.md index 78e1c59..87794fc 100644 --- a/validation/metafor/README.md +++ b/validation/metafor/README.md @@ -209,7 +209,7 @@ bounds by a factor of four rather than a tolerance. ## What this check found Three defects in `pymare/effectsize/expressions.json`, beyond the one -[PR #144](https://github.com/neurostuff/PyMARE/pull/144) fixes, all of the same +[PR #144](https://github.com/neurostuff/PyMARE/pull/144) fixed, all of the same kind: a missing pair of parentheses changing what the expression solves to. | Expression | Reads | Solves to | Should be | Effect | @@ -219,11 +219,20 @@ kind: a missing pair of parentheses changing what the expression solves to. | `v_d` (one-sample) | `... - d**2 / j**2 * n` | `A + n d**2 / j**2` | `A - d**2 / j**2` | the variance grows with the sample size | Each is recorded as an `xfail(strict=True)` naming the expression and what it -should be, alongside the two-sample `v_d` that PR #144 fixes. `strict` is the -point: correcting an expression turns the test green, pytest reports XPASS as a -failure, and the marker has to go in the same change. All four were confirmed to -flip to XPASS under the corresponding one-line fix, and under all four together -the rest of the suite still passes -- so nothing currently pins the wrong values. +should be. `strict` is the point: correcting an expression turns the test green, +pytest reports XPASS as a failure, and the marker has to go in the same change. +All three were confirmed to flip to XPASS under the corresponding one-line fix, +and under all three together the rest of the suite still passes -- so nothing +currently pins the wrong values. + +That mechanism has already been exercised once. A fourth marker covered the +two-sample `v_d`, which read `d**2 / 2 * (n1 + n2 - 2)` and so multiplied the +squared-effect term by the residual degrees of freedom instead of dividing by +twice them. Merging PR #144 turned its test green, the strict marker reported +XPASS as a failure, and the marker came out with the merge. +`test_standardized_mean_difference_variance_matches_metafor` is now an ordinary +passing test, holding PyMARE's SMD variance to within `2 / (n1 + n2)` of +metafor's -- the order at which the two approximations legitimately differ. ## What is not compared From 8d15b4d79bc27f1b5f0677f9bdbec253bb50dfeb Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 23:43:27 +0000 Subject: [PATCH 08/13] [FIX] correct three sampling-variance expressions All three are the defect PR #144 fixed for the two-sample Cohen's d, in three other expressions: a missing pair of parentheses, so the expression solves to something other than the variance it is named for. The escalc alignment check found them, and each has been recorded as a strict xfail naming the expression and its fix since that check was added. v_rmd read "v_rmd - (sd1**2 / n1) + (sd2**2 / n2)" and so solved to the *difference* of the two terms. The variance of a raw mean difference came out negative whenever the second group was the more variable one, and exactly zero for two equally sized equally variable groups -- which gives that study infinite weight in any inverse-variance meta-analysis. Non-positive on five of the eight rows of the reference grid. Now the sum, which matches escalc(measure="MD") exactly, to zero relative error on every row. v_sm read "... * j**2 * (1 / n + d**2) - d**2" -> A + d**2 v_d read "... * (1 / n + d**2) - d**2 / j**2 * n" -> A + n d**2/j**2 Both are the exact noncentral-t variances, and both had the final term added rather than subtracted; v_d additionally scaled it by n. With t noncentral-t on nu = n - 1 degrees of freedom and noncentrality lambda = delta sqrt(n), and d = t/sqrt(n): Var(d) = (n-1)/(n-3) (1/n + delta^2) - delta^2 / c^2 Var(g) = c^2 Var(d) = c^2 (n-1)/(n-3) (1/n + delta^2) - delta^2 The released expressions gave a single-group Hedges' g variance up to 93x too large and a single-group Cohen's d variance up to 6,399x too large, both growing with the effect, and v_d growing with the sample size -- backwards, and the same signature as the bug in #143. metafor has no exact counterpart to compare these two against: its single-group standardized-mean variance is the large-sample 1/n + y^2/(2n), which differs from the exact form by 2.5x at n = 5 and cannot settle an exact expression. So they are checked three ways instead -- against that approximation within a factor, against each other through the Var(g) = j^2 Var(d) identity, and for falling rather than rising with n. No existing test pinned the old values: the suite passes unchanged apart from the three markers, which come out here. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011o2j5ZNAdzPFL6LBqDqUsU --- pymare/effectsize/expressions.json | 6 ++-- pymare/tests/test_metafor_escalc.py | 50 +++++----------------------- validation/metafor/README.md | 51 ++++++++++++++++------------- 3 files changed, 40 insertions(+), 67 deletions(-) diff --git a/pymare/effectsize/expressions.json b/pymare/effectsize/expressions.json index 47208c1..fbbdf83 100644 --- a/pymare/effectsize/expressions.json +++ b/pymare/effectsize/expressions.json @@ -15,7 +15,7 @@ "description": "Cohen's d (one-sample)" }, { - "expression": "v_d - ((n - 1)/(n - 3)) * (1 / n + d**2) - d**2 / j**2 * n", + "expression": "v_d - (((n - 1)/(n - 3)) * (1 / n + d**2) - d**2 / j**2)", "type": 1, "description": "Variance of Cohen's d" }, @@ -30,7 +30,7 @@ "description": "Standardized mean (Hedges's g)" }, { - "expression": "v_sm - ((n - 1)/(n - 3)) * j**2 * (1 / n + d**2) - d**2", + "expression": "v_sm - (((n - 1)/(n - 3)) * j**2 * (1 / n + d**2) - d**2)", "type": 1, "description": "Variance of standardized mean" }, @@ -60,7 +60,7 @@ "description": "Raw mean difference" }, { - "expression": "v_rmd - (sd1**2 / n1) + (sd2**2 / n2)", + "expression": "v_rmd - ((sd1**2 / n1) + (sd2**2 / n2))", "type": 2, "description": "Variance of raw mean difference" }, diff --git a/pymare/tests/test_metafor_escalc.py b/pymare/tests/test_metafor_escalc.py index 8bd73aa..1d6f492 100644 --- a/pymare/tests/test_metafor_escalc.py +++ b/pymare/tests/test_metafor_escalc.py @@ -27,14 +27,15 @@ :func:`test_raw_correlation_variance_is_a_different_formula` and ``validation/metafor/README.md`` record what each one is instead. -Four comparisons are marked :func:`pytest.mark.xfail` with ``strict=True``, -because measuring these turned up defects rather than divergences: three -sampling-variance expressions in ``pymare/effectsize/expressions.json`` have -misplaced parentheses or a sign error, one of which is the subject of -https://github.com/neurostuff/PyMARE/pull/144. Each marker names the expression -and what it should be. ``strict=True`` is the point: when one is corrected the -test passes, pytest reports XPASS as a failure, and the marker has to be removed -in the same change. +Four of these comparisons were strict xfails when this module was written, +because measuring them turned up defects rather than divergences: four +sampling-variance expressions in ``pymare/effectsize/expressions.json`` had +misplaced parentheses or a sign error, which between them made the variance of a +raw mean difference come out negative, made two standardized-mean variances two +orders of magnitude too large, and made two of the four *grow* with the sample +size. All four are now corrected -- the two-sample Cohen's d by +https://github.com/neurostuff/PyMARE/pull/144 and the rest here -- and every +comparison in this module passes. """ @@ -342,18 +343,6 @@ def test_standardized_mean_variances_are_exact_not_asymptotic(one_sample, escalc # ----------------------------------------------------------------------------- -@pytest.mark.xfail( - strict=True, - reason=( - "v_rmd in pymare/effectsize/expressions.json reads " - "'v_rmd - (sd1**2 / n1) + (sd2**2 / n2)', which solves to " - "sd1**2/n1 - sd2**2/n2 rather than the sum. The variance comes out " - "negative whenever the second group is the more variable one, and " - "exactly zero for two equally sized, equally variable groups -- which " - "gives that study infinite weight. The fix is to parenthesize the " - "denominator, as PR #144 did for the two-sample Cohen's d" - ), -) def test_raw_mean_difference_variance_matches_metafor(two_sample): """``RMD``'s variance must be ``escalc(measure="MD")``'s, exactly. @@ -388,17 +377,6 @@ def test_standardized_mean_difference_variance_matches_metafor(two_sample, escal assert np.all(error <= SMD_VARIANCE_BOUND / total), list(zip(case_ids(), error)) -@pytest.mark.xfail( - strict=True, - reason=( - "v_sm in pymare/effectsize/expressions.json reads " - "'v_sm - ((n - 1)/(n - 3)) * j**2 * (1 / n + d**2) - d**2', which " - "solves to A + d**2 where the noncentral-t variance is A - d**2. The " - "reported variance is two orders of magnitude too large for any " - "appreciable effect, and grows with the effect instead of being " - "dominated by 1/n" - ), -) def test_standardized_mean_variance_is_the_same_order_as_metafor(one_sample): """``SM``'s variance must be within a factor of metafor's approximation. @@ -414,16 +392,6 @@ def test_standardized_mean_variance_is_the_same_order_as_metafor(one_sample): assert np.all(ratio >= 1 / SAME_ORDER_FACTOR), list(zip(case_ids(), ratio)) -@pytest.mark.xfail( - strict=True, - reason=( - "v_d (one-sample) in pymare/effectsize/expressions.json reads " - "'v_d - ((n - 1)/(n - 3)) * (1 / n + d**2) - d**2 / j**2 * n', so the " - "last term is added and scaled by n where the noncentral-t variance " - "subtracts d**2 / j**2. The reported variance therefore grows with the " - "sample size, which is backwards" - ), -) def test_one_sample_d_variance_is_the_same_order_as_metafor(one_sample): """``D``'s variance must be within a factor of metafor's approximation. diff --git a/validation/metafor/README.md b/validation/metafor/README.md index 87794fc..742da6d 100644 --- a/validation/metafor/README.md +++ b/validation/metafor/README.md @@ -208,31 +208,36 @@ bounds by a factor of four rather than a tolerance. ## What this check found -Three defects in `pymare/effectsize/expressions.json`, beyond the one -[PR #144](https://github.com/neurostuff/PyMARE/pull/144) fixed, all of the same -kind: a missing pair of parentheses changing what the expression solves to. +Four defects in `pymare/effectsize/expressions.json`, all of the same kind: a +missing pair of parentheses changing what the expression solves to. One is the +two-sample Cohen's d that +[PR #144](https://github.com/neurostuff/PyMARE/pull/144) fixed; the other three +were found by this check and are fixed here. -| Expression | Reads | Solves to | Should be | Effect | +| Expression | Read | Solved to | Should be | Effect | | --- | --- | --- | --- | --- | -| `v_rmd` | `v_rmd - (sd1**2 / n1) + (sd2**2 / n2)` | `sd1**2/n1 - sd2**2/n2` | the sum | negative variance whenever the second group is the more variable one, and exactly zero for two equally sized equally variable groups -- which gives that study infinite weight | -| `v_sm` | `... * (1/n + d**2) - d**2` | `A + d**2` | `A - d**2` | the single-group Hedges' g variance is 9x to 110x too large, and grows with the effect instead of being dominated by `1/n` | -| `v_d` (one-sample) | `... - d**2 / j**2 * n` | `A + n d**2 / j**2` | `A - d**2 / j**2` | the variance grows with the sample size | - -Each is recorded as an `xfail(strict=True)` naming the expression and what it -should be. `strict` is the point: correcting an expression turns the test green, -pytest reports XPASS as a failure, and the marker has to go in the same change. -All three were confirmed to flip to XPASS under the corresponding one-line fix, -and under all three together the rest of the suite still passes -- so nothing -currently pins the wrong values. - -That mechanism has already been exercised once. A fourth marker covered the -two-sample `v_d`, which read `d**2 / 2 * (n1 + n2 - 2)` and so multiplied the -squared-effect term by the residual degrees of freedom instead of dividing by -twice them. Merging PR #144 turned its test green, the strict marker reported -XPASS as a failure, and the marker came out with the merge. -`test_standardized_mean_difference_variance_matches_metafor` is now an ordinary -passing test, holding PyMARE's SMD variance to within `2 / (n1 + n2)` of -metafor's -- the order at which the two approximations legitimately differ. +| `v_rmd` | `v_rmd - (sd1**2 / n1) + (sd2**2 / n2)` | `sd1**2/n1 - sd2**2/n2` | the sum | **negative** variance whenever the second group is the more variable one, and exactly **zero** for two equally sized equally variable groups -- which gives that study infinite weight. Non-positive on five of the eight rows of this grid | +| `v_sm` | `... * (1/n + d**2) - d**2` | `A + d**2` | `A - d**2` | the single-group Hedges' g variance 1x to 93x too large, growing with the effect instead of being dominated by `1/n` | +| `v_d` (one-sample) | `... - d**2 / j**2 * n` | `A + n d**2 / j**2` | `A - d**2 / j**2` | up to 6,399x too large on this grid, and *growing* with the sample size | + +`v_sm` and `v_d` are the exact noncentral-t variances. With `t` noncentral-t on +`nu = n - 1` degrees of freedom and noncentrality `lambda = delta sqrt(n)`, and +`d = t / sqrt(n)`, `Var(t) = nu(1 + lambda^2)/(nu - 2) - lambda^2/c^2` gives + +``` +Var(d) = (n-1)/(n-3) (1/n + delta^2) - delta^2 / c^2 +Var(g) = c^2 Var(d) = c^2 (n-1)/(n-3) (1/n + delta^2) - delta^2 +``` + +which is what the two expressions now read. `test_metafor_escalc.py` checks the +`Var(g) = j^2 Var(d)` identity between them as well as both against metafor. + +Each defect was recorded as an `xfail(strict=True)` naming the expression and +what it should be before being fixed. `strict` is what made that safe: correcting +an expression turns the test green, pytest reports XPASS as a failure, and the +marker has to go in the same change. The mechanism was exercised for real when +PR #144 landed -- the merge turned its test green and the strict marker reported +the XPASS, so the marker came out with the merge rather than being forgotten. ## What is not compared From 0821d3e59b905ecc679aacaa62a405e134d90434 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 23:52:13 +0000 Subject: [PATCH 09/13] [FIX] permute the test statistic, and generalize the Hedges correction The two misalignments with metafor that were defects rather than conventions. Both were recorded as measurements by the alignment suite before being fixed here; both change released numbers. permutation_test counted |beta| where metafor::permutest counts |beta / se|. A permutation test needs a statistic whose null distribution does not move with what is being permuted away, and |beta| does: refitting a permuted dataset re-estimates tau^2, which moves the weights and so the standard error. The two coincide only where that standard error is invariant under the permutation -- a fixed-effects intercept-only model under sign flipping -- so four of the ten pinned cases agreed anyway and the rest did not, unequal_k5 under DL coming out at 0.5625 against permutest's 0.5. Separately, the observed statistic is computed by a different code path from the permuted ones, which refit in one batched call, and the two could disagree by a unit in the last place. An exactly inclusive comparison then dropped the identity permutation -- the one that reproduces the observed data, and so must count -- and with sign flipping its mirror too, understating the p-value by 2/2**K whenever it bit: 0.033203125 against permutest's 0.03515625 on extreme_k10. The comparison now allows a relative slack of sqrt(eps), which is the same constant permutest uses for this, applied relatively rather than absolutely so it does not depend on the scale of the statistic. It sits seven orders of magnitude below the closest genuine near-tie on these designs. With both fixed, PyMARE reproduces permutest exactly in all ten pinned cases, both coefficients of the moderator models included, and test_permutation_p_value_matches_metafor stops being a strict xfail. Hedges tau^2 subtracted the mean sampling variance sum(v) / K. The term that belongs there is tr(PV) / (K - P), with P the OLS residual maker and V = diag(v); since only P's diagonal is needed that is sum_i (1 - h_i) v_i / (K - P) over the OLS leverages. The two are the same quantity when the intercept is the only predictor -- P is then I - J/K, every h_i is 1/K, and the sum collapses -- so PyMARE was applying a special case unconditionally, and tau^2 was out by up to 0.14 relative in a meta-regression. It now agrees with metafor's HE to 1.9e-15 across all twelve design-by-model cells, and intercept-only models are unchanged. test_hedges_estimator carried a comment saying metafor "always gives negligibly different values for tau2, likely due to algorithmic differences", and pinned PyMARE's 11.3881 rather than metafor's 11.3594. It was neither algorithmic nor negligible. tau^2 is now pinned at metafor's value there like every other quantity in that test. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011o2j5ZNAdzPFL6LBqDqUsU --- pymare/estimators/estimators.py | 32 +++++++- pymare/results.py | 91 +++++++++++++++++++-- pymare/tests/test_estimators.py | 24 ++++-- pymare/tests/test_metafor_alignment.py | 12 +-- pymare/tests/test_metafor_permutest.py | 79 ++++++++---------- pymare/tests/test_metafor_random_effects.py | 82 ++++++++++--------- validation/metafor/README.md | 11 ++- 7 files changed, 220 insertions(+), 111 deletions(-) diff --git a/pymare/estimators/estimators.py b/pymare/estimators/estimators.py index 4b29fbb..f5ea582 100644 --- a/pymare/estimators/estimators.py +++ b/pymare/estimators/estimators.py @@ -1134,9 +1134,19 @@ class Hedges(BaseEstimator): The ``X`` matrix must be identical for all iterates. Unlike the coefficients, tau^2 is derived from an *unweighted* fit: it is the excess of - the ordinary mean squared error over the mean sampling variance. The coefficients are - then refitted with ``1 / (v + tau^2)`` weights, and the reported covariance comes from - that second fit. + the ordinary mean squared error over what that error is expected to be when tau^2 is + zero, namely ``tr(PV) / (K - P)`` with ``P`` the ordinary-least-squares residual maker + and ``V`` the diagonal matrix of sampling variances. The coefficients are then refitted + with ``1 / (v + tau^2)`` weights, and the reported covariance comes from that second + fit. + + .. versionchanged:: 0.0.13 + + The subtracted term was previously the mean sampling variance ``sum(v) / K``. That + is the same quantity when the intercept is the only predictor, but not otherwise, + so tau^2 was out by up to 0.14 relative in a meta-regression -- the divergence from + ``metafor``'s ``HE`` that ``validation/metafor/README.md`` recorded. Intercept-only + models are unaffected. .. versionchanged:: 0.0.11 @@ -1217,7 +1227,21 @@ def fit(self, y, v, X, g=None): # inverse-variance weights below. tau_beta = weighted_least_squares(tau_y, np.ones_like(tau_y), tau_X) mse = ((tau_y - tau_X.dot(tau_beta)) ** 2).sum(0) / (tau_k - tau_p) - tau_ho = np.maximum(0, mse - tau_v.sum(0) / tau_k) + + # What that unweighted mean squared error is expected to be when tau^2 + # is zero, which is what has to be subtracted off. With P the OLS + # residual maker I - X (X'X)^-1 X' and V = diag(v), it is + # tr(PV) / (K - P) -- and since only P's diagonal is needed, that is + # sum_i (1 - h_i) v_i / (K - P) for the OLS leverages h_i. + # + # Not the mean sampling variance sum(v) / K, which is the same quantity + # only when the intercept is the only predictor: P is then I - J/K, every + # h_i is 1/K, and the sum collapses to sum(v) (K - 1) / K / (K - 1). With + # a moderator the two part company, and using the intercept-only form + # put tau^2 out by up to 0.14 relative against metafor's HE. + leverage = np.einsum("ij,jk,ik->i", tau_X, np.linalg.pinv(tau_X.T @ tau_X), tau_X) + expected_mse = ((1.0 - leverage)[:, None] * tau_v).sum(0) / (tau_k - tau_p) + tau_ho = np.maximum(0, mse - expected_mse) # Estimate beta with tau^2 estimate. The covariance has to come from # this fit rather than the OLS one above: (X'WX)^-1 is only the diff --git a/pymare/results.py b/pymare/results.py index b96767a..85b9471 100644 --- a/pymare/results.py +++ b/pymare/results.py @@ -121,6 +121,46 @@ def _random_unit_order(classes, n_units): return order +#: Relative slack allowed when deciding whether a permuted statistic is at least +#: as extreme as the observed one. The permutation that reproduces the observed +#: data -- the identity, always a member of the set -- must tie with it and so +#: must count; but the observed statistic is computed by a different code path +#: from the permuted ones, which refit every dataset in one batched call, and +#: the two can disagree by a unit in the last place. Without slack that drops +#: the identity, and with sign flipping its mirror too, understating the p-value +#: by 2 / 2**m. The square root of machine epsilon is the same constant +#: ``metafor::permutest`` uses for this, applied relatively rather than +#: absolutely so that it does not depend on the scale of the statistic. It is +#: seven orders of magnitude below the closest genuine near-tie observed on the +#: designs in ``pymare/tests/data/metafor_small_sample.csv``. +_PERMUTATION_TIE_RTOL = np.sqrt(np.finfo(np.float64).eps) + + +def _at_least_as_extreme(permuted, observed): + """Return which permuted statistics are at least as extreme as the observed one. + + Parameters + ---------- + permuted : :obj:`numpy.ndarray` + Absolute permuted statistics, permutations along the last axis. + observed : :obj:`numpy.ndarray` + The absolute observed statistic, broadcastable against ``permuted``. + + Returns + ------- + :obj:`numpy.ndarray` of :obj:`bool` + Elementwise ``permuted >= observed``, to within + :data:`_PERMUTATION_TIE_RTOL`. + + Notes + ----- + ``>=`` and not ``>``: a permuted statistic that ties with the observed one is + at least as extreme, and excluding ties makes the test anti-conservative on + discrete or degenerate data. + """ + return permuted >= observed - np.abs(observed) * _PERMUTATION_TIE_RTOL + + class MetaRegressionResults: """Container for results generated by PyMARE meta-regression estimators. @@ -626,6 +666,24 @@ def permutation_test(self, n_perm=1000): Notes ----- + The statistic permuted is the coefficient divided by its standard error, + not the coefficient itself, and the same for tau^2. Refitting a permuted + dataset re-estimates tau^2, which moves the weights and so the standard + error, so ``|beta|`` is not pivotal and its permutation distribution is + not the null this test needs. The two coincide only where the standard + error is invariant under the permutation -- a fixed-effects, + intercept-only model under sign flipping. This matches + ``metafor::permutest``. + + .. versionchanged:: 0.0.13 + Counted ``|beta|`` in earlier releases, and compared it against the + observed value exactly. Both are fixed: the statistic is now + ``|beta / se|``, and the comparison allows a relative slack of the + square root of machine epsilon so that the identity permutation -- + which reproduces the observed data and must therefore count -- is not + dropped over a unit in the last place. Every permutation p-value this + method reports is affected. + If the number of possible permutations is smaller than n_perm, an exact test will be conducted. Otherwise an approximate test will be conducted by randomly shuffling the outcomes n_perm @@ -745,16 +803,35 @@ def permutation_test(self, n_perm=1000): # freedom (or raising on reshape). params = copy.copy(self.estimator).fit(**kwargs).params_ + # The statistic is the coefficient over its standard error, not + # the coefficient: refitting a permuted dataset re-estimates tau^2, + # which moves the weights and hence the standard error, so |beta| + # is not pivotal and its permutation distribution is not the null + # the test needs. The two coincide only where the standard error + # happens to be invariant under the permutation -- a fixed-effects + # intercept-only model under sign flipping. This is what + # metafor::permutest counts. fe_obs = fe_stats["est"][:, i] + se_obs = fe_stats["se"][:, i] if fe_obs.ndim == 1: - fe_obs = fe_obs[:, None] - # <=, not <: a permuted statistic that ties with the observed one - # is at least as extreme, and excluding ties makes the test - # anti-conservative on discrete or degenerate data. - fe_p[:, i] = (np.abs(fe_obs) <= np.abs(params["fe_params"])).mean(1) + fe_obs, se_obs = fe_obs[:, None], se_obs[:, None] + + # The permuted standard errors, taken from the covariance the + # estimator just reported, exactly as `fe_se` takes the observed + # one from `fe_cov` -- including whatever small-sample correction + # the estimator applies, which the observed statistic carries too. + perm_cov = np.asarray(params["inv_cov"]) + if perm_cov.ndim == 2: + perm_cov = perm_cov[:, :, None] + # A zero standard error divides to +-inf; the comparison below + # still orders those correctly, and a NaN counts as not extreme. + with np.errstate(invalid="ignore", divide="ignore"): + perm_se = np.sqrt(np.diagonal(perm_cov)).T + fe_p[:, i] = _at_least_as_extreme( + np.abs(params["fe_params"] / perm_se), np.abs(fe_obs / se_obs) + ).mean(1) if rfx: - abs_obs = np.abs(tau2[i]) - tau_p[i] = (abs_obs <= np.abs(params["tau2"])).mean() + tau_p[i] = _at_least_as_extreme(np.abs(params["tau2"]), np.abs(tau2[i])).mean() # p-values can't be smaller than 1/n_perm params = {"fe_p": np.maximum(1 / n_perm, fe_p)} diff --git a/pymare/tests/test_estimators.py b/pymare/tests/test_estimators.py index e86ac47..312c299 100644 --- a/pymare/tests/test_estimators.py +++ b/pymare/tests/test_estimators.py @@ -109,9 +109,15 @@ def test_2d_DL_estimator(dataset_2d): def test_hedges_estimator(dataset): """Test Hedges estimator.""" - # ground truth values are from metafor package in R, except that metafor - # always gives negligibly different values for tau2, likely due to - # algorithmic differences in the computation. + # ground truth values are from metafor package in R. + # + # tau^2 used to be excluded from that, with a comment saying metafor + # "always gives negligibly different values ... likely due to algorithmic + # differences". It was not algorithmic and not negligible: PyMARE subtracted + # the mean sampling variance where metafor subtracts tr(PV) / (K - P), which + # are the same only without moderators. With the trace term the two agree to + # every digit metafor prints, so tau^2 is now pinned at metafor's value like + # everything else here. # # "wald" because metafor's own default is test="z", so that is the # configuration the reference values were read off. PyMARE's default is @@ -135,8 +141,8 @@ def test_hedges_estimator(dataset): # Check output values assert np.allclose(beta.ravel(), [-0.1066, 0.7704], atol=1e-4) - assert np.allclose(tau2, 11.3881, atol=1e-4) - assert np.allclose(fe_stats["se"].ravel(), [3.0479, 1.1335], atol=1e-4) + assert np.allclose(tau2, 11.3594, atol=1e-4) + assert np.allclose(fe_stats["se"].ravel(), [3.0444, 1.1322], atol=1e-4) # The unweighted fit that produces tau^2 would have given these instead. assert not np.allclose(fe_stats["se"].ravel(), [0.8639, 0.3217], atol=1e-4) @@ -146,7 +152,7 @@ def test_hedges_estimator(dataset): default = Hedges().fit_dataset(dataset).summary() assert np.allclose(np.ravel(default.tau2), tau2, rtol=0, atol=0) assert np.allclose(default.fe_params, beta, rtol=0, atol=0) - assert np.allclose(default.get_fe_stats()["se"].ravel(), [3.0213, 1.1236], atol=1e-4) + assert np.allclose(default.get_fe_stats()["se"].ravel(), [3.0212, 1.1236], atol=1e-4) assert np.all(default.fe_dof == 6) @@ -169,11 +175,11 @@ def test_2d_hedges(dataset_2d): # First and third sets are identical to single dim test; second set is # randomly different. assert np.allclose(beta[:, 0], [-0.1066, 0.7704], atol=1e-4) - assert np.allclose(tau2[0], 11.3881, atol=1e-4) + assert np.allclose(tau2[0], 11.3594, atol=1e-4) assert not np.allclose(beta[:, 1], [-0.1070, 0.7664], atol=1e-4) - assert not np.allclose(tau2[1], 11.3881, atol=1e-4) + assert not np.allclose(tau2[1], 11.3594, atol=1e-4) assert np.allclose(beta[:, 2], [-0.1066, 0.7704], atol=1e-4) - assert np.allclose(tau2[2], 11.3881, atol=1e-4) + assert np.allclose(tau2[2], 11.3594, atol=1e-4) def test_variance_based_maximum_likelihood_estimator(dataset): diff --git a/pymare/tests/test_metafor_alignment.py b/pymare/tests/test_metafor_alignment.py index 760f386..d2f5a1d 100644 --- a/pymare/tests/test_metafor_alignment.py +++ b/pymare/tests/test_metafor_alignment.py @@ -16,11 +16,13 @@ that reach tau^2 in closed form, so nothing but the adjustment sits between the two implementations. -Whether PyMARE's ML, REML and Hedges tau^2 match ``metafor``'s is a separate -question that predates this work -- they differ by up to the tau^2 search -tolerance, and Hedges by a definitional choice. ``validation/metafor/README.md`` -records the measurements; comparing them here would mix an optimizer's tolerance -into a check on a closed-form scale factor. +Whether PyMARE's ML and REML tau^2 match ``metafor``'s is a separate question: +they differ by up to the tau^2 search tolerance, since both sides reach it by +numerical search. ``validation/metafor/README.md`` records the measurements, and +:mod:`pymare.tests.test_metafor_random_effects` compares them directly; +comparing them here would mix an optimizer's tolerance into a check on a +closed-form scale factor. Hedges is closed form and does agree exactly, so +``HE`` is compared end to end there. The reference values are pinned in ``data/metafor_reference.json`` because metafor is an R package and cannot be a test dependency. The pin is kept honest by diff --git a/pymare/tests/test_metafor_permutest.py b/pymare/tests/test_metafor_permutest.py index fc84d83..507f1dd 100644 --- a/pymare/tests/test_metafor_permutest.py +++ b/pymare/tests/test_metafor_permutest.py @@ -8,25 +8,37 @@ is ``permutest(exact = TRUE)`` on the designs small enough to enumerate, and PyMARE is asked for the same enumeration. -The two disagree. This module locates the disagreement rather than tolerating -it, in two tests that between them say where every counted permutation goes: +The two agree exactly, in all ten cases and on both coefficients of the +moderator models. They did not when this module was written, for two reasons +that it still isolates rather than merely asserting the totals: - :func:`test_metafor_permutest_is_reproduced_by_the_z_statistic` counts the - *same* permutations of the *same* PyMARE fits, but on ``|beta / se|`` - instead of ``|beta|``, and reproduces metafor exactly in all ten cases. So - the permutation sets agree, the refits agree, and the tie handling of an - inclusive comparison agrees; what differs is which statistic gets counted. -- :func:`test_permutation_p_value_matches_metafor` is what PyMARE currently - reports, and is a strict xfail. Two things break it, and the test's - docstring and marker name both. - -Why the statistic matters rather than being a convention: a permutation test -needs a statistic whose null distribution does not move with the parameters -being permuted away. ``|beta|`` does move, because refitting a permuted dataset -changes tau^2 and so changes the weights and the standard error. The two -coincide exactly when the standard error happens to be invariant -- a -fixed-effects, intercept-only model under sign flipping, which is why four of -the ten cases here agree anyway. + permutations independently of the production path, on ``|beta / se|`` and + with the observed statistic read out of the identity permutation in the same + batch. It is what established that the permutation sets and the batched + refits were already right, and it remains the check that says *why* the + totals agree rather than only that they do. +- :func:`test_permutation_p_value_matches_metafor` is what PyMARE reports. + +The two defects it used to record, both now fixed in +:meth:`~pymare.results.MetaRegressionResults.permutation_test`: + +**The statistic.** PyMARE counted ``|beta|`` where ``permutest`` counts +``|beta / se|``. A permutation test needs a statistic whose null distribution +does not move with the parameters being permuted away, and ``|beta|`` does: +refitting a permuted dataset re-estimates tau^2, which changes the weights and +so the standard error. The two coincide only where the standard error is +invariant under the permutation -- a fixed-effects, intercept-only model under +sign flipping -- which is why four of the ten cases agreed anyway, and why +``unequal_k5`` under ``DL`` came out at 0.5625 against ``permutest``'s 0.5. + +**The tie.** The observed statistic is computed by a different code path from +the permuted ones, and the two could disagree by a unit in the last place, so an +exactly inclusive comparison dropped the identity permutation -- the one that +reproduces the observed data, and must therefore count -- along with its +sign-flipped mirror. That understated the p-value by ``2 / 2**K``: 0.033203125 +against ``permutest``'s 0.03515625 on ``extreme_k10``. The comparison now allows +the same square-root-of-epsilon slack ``permutest`` does. """ @@ -156,36 +168,13 @@ def test_metafor_permutest_is_reproduced_by_the_z_statistic(case, metafor_datase assert np.allclose(p_values, case["pval"], rtol=RTOL) -@pytest.mark.xfail( - strict=True, - reason=( - "MetaRegressionResults.permutation_test counts |beta| where permutest " - "counts |beta / se|, so the two agree only where the standard error is " - "invariant under the permutation -- a fixed-effects intercept-only " - "model under sign flipping. Refitting a permuted dataset re-estimates " - "tau^2, which moves the weights and hence the standard error, so under " - "DerSimonianLaird the statistic being permuted is not pivotal: " - "unequal_k5 comes out at 0.5625 against permutest's 0.5. Separately, " - "the observed estimate is computed by a different code path from the " - "permuted ones, and the two disagree by one unit in the last place, so " - "the inclusive comparison can drop the identity permutation and its " - "mirror -- always understating the p-value, by 2/1024 on extreme_k10. " - "test_metafor_permutest_is_reproduced_by_the_z_statistic shows both go " - "away when the statistic is |z| and the observed value is read out of " - "the same batch" - ), -) @pytest.mark.filterwarnings("ignore:Cluster-robust") def test_permutation_p_value_matches_metafor(metafor_dataset): - """Report the exact permutation p-values PyMARE gives, against metafor's. - - Asserted over the whole grid in one test rather than parametrized, on - purpose. One of the two causes is a one-unit-in-the-last-place difference, - which need not reproduce on every platform and BLAS the test matrix covers; - the other is structural and does. Asserting all ten cases together means - the xfail is driven by the structural cause and cannot flip to an - unexpected pass because a rounding difference went the other way on some - runner. + """The exact permutation p-values PyMARE reports must be metafor's. + + Asserted over the whole grid in one test rather than parametrized, so that a + failure reports every case that moved rather than the first. This was a + strict xfail until the two defects in the module docstring were fixed. """ mismatched = [] for case in CASES: diff --git a/pymare/tests/test_metafor_random_effects.py b/pymare/tests/test_metafor_random_effects.py index 4514a18..69fad9d 100644 --- a/pymare/tests/test_metafor_random_effects.py +++ b/pymare/tests/test_metafor_random_effects.py @@ -19,18 +19,21 @@ distinct cases. What agrees, and how exactly, is recorded in ``validation/metafor/README.md``. -Three divergences are pinned down here rather than merely tolerated, each by a -test that asserts *why* the two differ instead of how much: +Two divergences are pinned down here rather than merely tolerated, each by a +test that asserts *why* the two differ instead of how much, plus one that used +to be: - ``I^2`` and ``H``: PyMARE always reports the Q-based definition of :footcite:t:`higgins2002quantifying`. metafor reports that pair only for ``FE`` and ``DL`` -- where it coincides with tau^2 / (tau^2 + v_t) -- and the tau^2-based pair otherwise. See :func:`test_i2_and_h_are_the_q_based_definition`. -- :class:`~pymare.estimators.Hedges` tau^2: PyMARE subtracts the mean sampling - variance, metafor subtracts ``tr(PV) / (K - P)``. The two are equal when the - only predictor is the intercept and differ otherwise. See - :func:`test_hedges_tau2_divergence_is_the_trace_term`. +- :class:`~pymare.estimators.Hedges` tau^2 used to diverge with moderators, + PyMARE subtracting the mean sampling variance where metafor subtracts + ``tr(PV) / (K - P)``. PyMARE now subtracts the trace form too and the two + agree everywhere; see + :func:`test_hedges_correction_reduces_to_the_mean_variance_without_moderators` + for why the intercept-only cells never differed. - ``ML`` and ``REML`` tau^2: both profile the likelihood numerically, to different tolerances, and on one cell of the grid they stop on opposite sides of the tau^2 = 0 boundary. See :data:`RTOL_PROFILED` and @@ -246,55 +249,56 @@ def test_tau2_interval_matches_metafor(case, metafor_dataset): assert np.allclose(np.ravel(stats["ci_u"]), case["tau2_ci_ub"], rtol=RTOL_PROFILE) -@pytest.mark.parametrize( - "case", - [case for case in HEDGES_CASES if not MODELS[case["model"]]], - ids=[case_id(case) for case in HEDGES_CASES if not MODELS[case["model"]]], -) -def test_hedges_tau2_matches_metafor_without_moderators(case, metafor_dataset): - """``Hedges`` tau^2 must be metafor's ``HE`` for an intercept-only model. - - Exactly, not approximately: the two expressions for the correction term are - algebraically the same one when the intercept is the only predictor, so this - holds to machine precision on all four designs. - :func:`test_hedges_tau2_divergence_is_the_trace_term` covers what happens - when a moderator is added. +@pytest.mark.parametrize("case", HEDGES_CASES, ids=[case_id(case) for case in HEDGES_CASES]) +def test_hedges_tau2_matches_metafor(case, metafor_dataset): + """``Hedges`` tau^2 must be metafor's ``HE``, on every model. + + Exactly, not approximately: both are the excess of an unweighted mean + squared error over what that error is expected to be at tau^2 = 0, and both + now compute the second term the same way, so this holds to machine precision + on all twelve design-by-model cells. + + It used to hold only for the four intercept-only cells, because PyMARE + subtracted the mean sampling variance rather than ``tr(PV) / (K - P)``; + :func:`test_hedges_correction_reduces_to_the_mean_variance_without_moderators` + is why those four were unaffected. """ - assert np.allclose(np.ravel(fit(metafor_dataset, case).tau2), case["tau2"], rtol=RTOL) + assert np.allclose( + np.ravel(fit(metafor_dataset, case).tau2), case["tau2"], rtol=RTOL, atol=1e-12 + ) @pytest.mark.parametrize("case", HEDGES_CASES, ids=[case_id(case) for case in HEDGES_CASES]) -def test_hedges_tau2_divergence_is_the_trace_term(case, metafor_dataset): - """Locate the ``HE`` divergence in the term each implementation subtracts. - - Both estimators are the excess of an unweighted mean squared error over an - estimate of the mean sampling variance. They differ in the second term: - metafor uses ``tr(PV) / (K - P)`` with ``P = I - X (X'X)^-1 X'``, PyMARE uses - ``sum(v) / K``. Those coincide when ``X`` is just an intercept, since ``P`` - is then ``I - J/K`` and the trace is ``sum(v) (K - 1) / K``; with a moderator - they do not, and PyMARE's tau^2 is out by up to 0.14 relative on this grid. - - Two assertions, which together say that and nothing more: metafor's own - definition reimplemented here reproduces every pinned ``HE`` tau^2, and the - two correction terms agree exactly when and only when there are no - moderators. A test that instead pinned the size of the gap would be a record - of PyMARE's current behaviour rather than a statement about either formula. +def test_hedges_correction_reduces_to_the_mean_variance_without_moderators(case, metafor_dataset): + """The two forms of the ``HE`` correction term must agree iff ``P`` is one. + + The term subtracted from the unweighted mean squared error is + ``tr(PV) / (K - P)``, with ``P = I - X (X'X)^-1 X'`` and ``V = diag(v)``. + When the intercept is the only predictor ``P`` is ``I - J/K``, every leverage + is ``1/K``, and the sum collapses to ``sum(v) / K`` -- the mean sampling + variance, which is what PyMARE used to subtract unconditionally. + + Asserting that identity, and its failure with a moderator, is what says the + correction PyMARE now applies is a generalization of the old one rather than + a different quantity, and so why intercept-only results did not move when it + changed. The reimplementation of metafor's whole definition alongside it + keeps this anchored to metafor rather than to PyMARE's own arithmetic. """ y, v, X = design_arrays(metafor_dataset, case) n_obs, n_preds = X.shape residual_maker = np.eye(n_obs) - X @ np.linalg.pinv(X.T @ X) @ X.T - metafor_correction = np.trace(residual_maker @ np.diag(v)) / (n_obs - n_preds) - pymare_correction = v.sum() / n_obs + trace_form = np.trace(residual_maker @ np.diag(v)) / (n_obs - n_preds) + mean_variance = v.sum() / n_obs metafor_tau2 = max( 0.0, (y @ residual_maker @ y - np.trace(residual_maker @ np.diag(v))) / (n_obs - n_preds) ) assert np.allclose(metafor_tau2, case["tau2"], rtol=RTOL, atol=1e-12) if n_preds == 1: - assert np.allclose(pymare_correction, metafor_correction, rtol=RTOL) + assert np.allclose(mean_variance, trace_form, rtol=RTOL) else: - assert not np.isclose(pymare_correction, metafor_correction, rtol=1e-6) + assert not np.isclose(mean_variance, trace_form, rtol=1e-6) @pytest.mark.parametrize( diff --git a/validation/metafor/README.md b/validation/metafor/README.md index 742da6d..0719cd6 100644 --- a/validation/metafor/README.md +++ b/validation/metafor/README.md @@ -58,7 +58,7 @@ the 60 distinct design x model x estimator cells: | `QEp` in logs | `["logp(Q)"]` | 4.7e-14 | | `I2`, `H2` for `FE` and `DL` | `["I^2"]`, `["H"]` | 4.9e-14, 1.5e-14 | | `confint` tau^2 bounds | `get_re_stats()["ci_l"]`, `["ci_u"]` | 1.3e-13 | -| `HE` tau^2, intercept-only models | `Hedges().fit(...)` | 2.2e-16 | +| `HE` tau^2, every model | `Hedges().fit(...)` | 1.9e-15 | | `ML`, `REML` tau^2 | `VarianceBasedLikelihoodEstimator` | 2.7e-5 | Two notes on that table. @@ -99,8 +99,15 @@ the *cause* rather than the size of the gap. | Divergence | Size | Cause | | --- | --- | --- | | `I2`, `H2` for `HE`, `ML`, `REML` | unbounded | PyMARE always reports the Q-based Higgins-Thompson pair. metafor reports that pair only for `FE` and `DL`, where it coincides with `tau^2 / (tau^2 + v_t)`, and switches to the tau^2-based pair otherwise -- so metafor's `I2` depends on which tau^2 estimator was asked for and PyMARE's does not. Both are defensible; they are not the same number. | -| `HE` tau^2, models with moderators | up to 0.14 relative | metafor subtracts `tr(PV) / (K - P)`, PyMARE subtracts `sum(v) / K`. With an intercept as the only predictor `P = I - J/K`, the trace is `sum(v)(K - 1)/K`, and the two are algebraically the same; with a moderator they are not. So the divergence is specific to meta-regression. A previous version of this README recorded it as general. | | `ML`, `REML` tau^2 | ~3e-5 relative | PyMARE profiles tau^2 at `xtol=1e-6`; metafor runs its own optimizer to its own tolerance. | + +`HE` tau^2 used to be a third row here, out by up to 0.14 relative on models with +moderators: metafor subtracts `tr(PV) / (K - P)` where PyMARE subtracted +`sum(v) / K`. With an intercept as the only predictor `P = I - J/K`, the trace is +`sum(v)(K - 1)/K` and the two are algebraically the same, so the divergence was +specific to meta-regression -- an earlier version of this README recorded it as +general. PyMARE now subtracts the trace form and the two agree to 1.9e-15 across +all twelve design-by-model cells. | `ML` on `extreme_k10` with one moderator | 0 vs 0.011 | The two searches land on opposite sides of the tau^2 = 0 boundary, where a profile likelihood is flattest because the weights are most unequal. metafor is the one that stops at zero. A previous version of this README had the direction backwards. | ## What is compared From 087555314a4b459d2213e2f73401c1bccb757496 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 00:10:36 +0000 Subject: [PATCH 10/13] [FIX] use the sampling variance of a correlation, and record why the rest stay v_r read (1 - r**2) / (n - 2), which is the squared standard error of r under the null hypothesis of no correlation rather than its sampling variance at the observed value. It is now (1 - r**2)**2 / (n - 1), which matches escalc(measure="COR") exactly, so R joins the measures compared on both halves rather than only the estimate. The tempting defence of the old expression was that it is conservative, being larger by (n - 1) / ((n - 2)(1 - r**2)). That does not survive contact with what a meta-analysis does with a variance: the inflation factor depends on the data, so the expression did not widen intervals uniformly, it reweighted studies against one another -- down-weighting those with strong correlations by up to 51x at r = 0.99 and pulling the pooled estimate toward the weak ones. ZR is unaffected, always agreed with metafor, and remains the measure to prefer for pooling correlations; the converter's docstring now says so. The four remaining divergences are deliberate, and the reasoning for each is now written down in the validation READMEs rather than living in a review thread. Each argument was checked rather than asserted: CR2 whitening metric. Both PyMARE's form and clubSandwich's satisfy the Bell-McCaffrey condition A_j B_j A_j' = Psi_j exactly -- 1.3e-15 and 1.7e-15 on the varying-variance grid -- and both give E[V_R] = (X'WX)^-1 under the working model, so both are exactly unbiased and PyMARE's is a CR2 in the defining sense. The condition does not determine A_j uniquely; clubSandwich takes it symmetric, PyMARE symmetric in the whitened metric (max |A - A'| of 2.8e-17 against 9.1e-2). What the whitened choice buys is that I - H_j is then identity-minus-rank-p, so its spectrum collapses to p non-unit eigenvalues at any group size and _cr2_low_rank_factors works in p x p; clubSandwich's is Psi^2 minus rank p, whose n_j distinct eigenvalues need the full decomposition -- 416x more work at n_j = 200 and 5,901x at 800. method="CR2" is kept, and the parameter and _cr2_scores now both say which CR2 it is and when it coincides with clubSandwich's. I^2 and H stay Q-based. It is the Higgins & Thompson definition the docstring cites, and estimator-independence is a feature: on unequal_k5 metafor reports I^2 of 80.31, 80.31, 98.92, 50.54 and 97.77 for FE, DL, HE, ML and REML on one dataset and one Q, where PyMARE reports 80.31 throughout. The ML/REML search tolerance stays. 2.7e-5 relative on a tau^2 whose own Q-profile interval spans a factor of 167 on that design, and bounded_scalar_min is built to fit 10^5 or more datasets in one vectorized search, so iterations multiply through the whole analysis. The bias correction stays approximate. Not only because the error is 0.043/m^2 -- a thousandth of SE(g) at m = 4 -- but because the converters are a symbolic system: substituting the exact gamma factor keeps the forward solve working and makes the reverse one raise NotImplementedError, sympy being unable to invert a ratio of gammas. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011o2j5ZNAdzPFL6LBqDqUsU --- pymare/effectsize/base.py | 20 ++++- pymare/effectsize/expressions.json | 2 +- pymare/stats.py | 37 ++++++++ pymare/tests/test_effectsize_base.py | 7 +- pymare/tests/test_metafor_escalc.py | 62 ++++++------- validation/clubsandwich/README.md | 83 +++++++++++++++--- validation/metafor/README.md | 126 +++++++++++++++++++++++---- 7 files changed, 272 insertions(+), 65 deletions(-) diff --git a/pymare/effectsize/base.py b/pymare/effectsize/base.py index 4c2ce6b..e47372b 100644 --- a/pymare/effectsize/base.py +++ b/pymare/effectsize/base.py @@ -250,6 +250,19 @@ class OneSampleEffectSizeConverter(EffectSizeConverter): summaries, and are _not_ individual data points. E.g., do not pass in a vector of point estimates as `m` and a scalar for the SDs `sd`. The lengths of all inputs must match. + + .. versionchanged:: 0.0.13 + + The sampling variance of a raw correlation (``'R'``) is now + ``(1 - r**2)**2 / (n - 1)``, matching ``metafor::escalc(measure="COR")``. + It was ``(1 - r**2) / (n - 2)``, which is the squared standard error of + ``r`` under the null hypothesis of *no* correlation rather than its + sampling variance at the observed value. The two agree near ``r = 0`` + and diverge by a factor of ``(n - 1) / ((n - 2)(1 - r**2))`` -- 51x at + ``r = 0.99``. Since that factor depends on the data, the old expression + did not simply inflate variances: it reweighted studies against one + another, pulling a pooled estimate toward those with the weakest + correlations. ``'ZR'`` is unaffected and remains the measure to prefer. """ _type = 1 @@ -272,7 +285,12 @@ def to_dataset(self, measure="RM", **kwargs): a bias correction applied. - 'D': Cohen's d. Note that no bias correction is applied (use 'SM' instead). - - 'R': Raw correlation coefficient. + - 'R': Raw correlation coefficient. Prefer 'ZR' for + meta-analysis: the sampling variance of a raw correlation + depends strongly on the correlation itself, so studies are + weighted very unequally by how large their correlations + happen to be, which is the problem the Fisher transform + exists to remove. - 'ZR': Fisher z-transformed correlation coefficient. **kwargs Optional keyword arguments to pass onto the Dataset diff --git a/pymare/effectsize/expressions.json b/pymare/effectsize/expressions.json index fbbdf83..f5bf18a 100644 --- a/pymare/effectsize/expressions.json +++ b/pymare/effectsize/expressions.json @@ -40,7 +40,7 @@ "description": "Raw correlation coefficient" }, { - "expression": "v_r - (1 - r**2) / (n - 2)", + "expression": "v_r - (1 - r**2)**2 / (n - 1)", "type": 1, "description": "Variance of raw correlation coefficient" }, diff --git a/pymare/stats.py b/pymare/stats.py index 06f46e8..88f7b71 100644 --- a/pymare/stats.py +++ b/pymare/stats.py @@ -1515,6 +1515,27 @@ def _cr2_scores(X, w, resid, group_members, bread): above is the solution for :math:`\Phi = I` in the whitened metric, i.e. under the assumption that the weights are correct and the observations independent -- the same assumption the sandwich exists to avoid relying on. + + That condition has many solutions, because it constrains :math:`A_j` only + through :math:`A_j B_j A_j'`. ``clubSandwich`` selects the symmetric one, + :math:`A_j = \Psi_j^{1/2} (\Psi_j^{1/2} B_j \Psi_j^{1/2})^{-1/2} + \Psi_j^{1/2}`; the form here is symmetric in the whitened metric instead, + and the two agree exactly when :math:`W_j` is a multiple of the identity -- + that is, when the sampling variances are constant within the group. Both + satisfy the condition and both are exactly unbiased under the working model, + so the choice is not between a right and a wrong one. + + It is made this way for the reason the next paragraph gives: in the whitened + metric :math:`I_j - H_j` is the identity minus a rank-:math:`p` term, so its + spectrum collapses to :math:`p` non-unit eigenvalues however large the group + is, and :func:`_cr2_low_rank_factors` can take the inverse square root in + :math:`p \times p` work. ``clubSandwich``'s matrix is + :math:`\Psi_j^2` minus a rank-:math:`p` term, whose diagonal part is not a + multiple of the identity, so it has :math:`n_j` distinct eigenvalues and + needs the full :math:`n_j \times n_j` eigendecomposition -- a factor of 400 + more work at :math:`n_j = 200` and 5,900 at :math:`n_j = 800`. The square + root not commuting with an asymmetric congruence is at once why the two + forms differ and why only one of them factors. That is pragmatic rather than circular: simulation shows the correction helps substantially even when the working model is wrong :footcite:p:`tipton2015small,imbens2016robust`, and its influence fades as @@ -1945,6 +1966,22 @@ def cluster_robust_cov( - ``"CR2"`` (default) inflates each group's residuals by :math:`(I_j - H_j)^{-1/2}` to undo the shrinkage caused by fitting :math:`\beta` with that group included :footcite:p:`bell2002bias`. + + .. note:: + + This is a CR2 in the defining sense -- it satisfies + :math:`A_j B_j A_j' = \Psi_j` exactly, and the resulting + sandwich is exactly unbiased for the model-based covariance + under the working model -- but it is not bit-for-bit + ``clubSandwich``'s ``CR2``. That condition does not pin + :math:`A_j` down uniquely: ``clubSandwich`` closes it by taking + :math:`A_j` symmetric, and this takes it symmetric in the + whitened metric instead. The two coincide exactly when the + weights are constant within a group, and differ otherwise -- + by up to 1e-2 relative on the standard errors of the designs in + ``validation/clubsandwich``, which measures it. See + :func:`_cr2_scores` for why the whitened form is the one + implemented. - ``"CR0"`` uses the raw residuals with the blunt ``m / (m - p)`` scaling. This is the historical behaviour. diff --git a/pymare/tests/test_effectsize_base.py b/pymare/tests/test_effectsize_base.py index 3410e50..9ab1380 100644 --- a/pymare/tests/test_effectsize_base.py +++ b/pymare/tests/test_effectsize_base.py @@ -124,7 +124,12 @@ def test_convert_r_to_itself(): esc.get_v_r() esc = OneSampleEffectSizeConverter(r=r, n=n) v_r = esc.get("V_R") - assert np.allclose(v_r, (1 - r**2) / (n - 2)) + # The asymptotic sampling variance of a correlation, which is what + # escalc(measure="COR") reports; pinned against it in + # test_metafor_escalc.py. Earlier releases used (1 - r**2) / (n - 2), the + # squared standard error of r under the null of no correlation, which is a + # different quantity -- see that module. + assert np.allclose(v_r, (1 - r**2) ** 2 / (n - 1)) ds = esc.to_dataset(measure="R") assert np.allclose(ds.y.ravel(), r) assert np.allclose(ds.v.ravel(), v_r) diff --git a/pymare/tests/test_metafor_escalc.py b/pymare/tests/test_metafor_escalc.py index 1d6f492..09068eb 100644 --- a/pymare/tests/test_metafor_escalc.py +++ b/pymare/tests/test_metafor_escalc.py @@ -19,13 +19,13 @@ which is a statement about the approximation and not about either implementation's current output. -**Different formulas for the same thing.** metafor takes the sampling variance of -a correlation to be ``(1 - r**2)**2 / (n - 1)`` and PyMARE takes it to be -``(1 - r**2) / (n - 2)``; metafor's standardized-mean variances are large-sample -approximations where PyMARE's are the exact noncentral-t expressions. These -cannot be verified against each other, so -:func:`test_raw_correlation_variance_is_a_different_formula` and -``validation/metafor/README.md`` record what each one is instead. +**Different formulas for the same thing.** metafor's single-group +standardized-mean variances are large-sample approximations where PyMARE's are +the exact noncentral-t expressions, which is the better quantity but not the same +one. They cannot be verified against each other, so +:func:`test_standardized_mean_variances_are_exact_not_asymptotic` pins metafor's +side of the statement and a factor bound spans the two. +``validation/metafor/README.md`` records the reasoning. Four of these comparisons were strict xfails when this module was written, because measuring them turned up defects rather than divergences: four @@ -179,14 +179,30 @@ def test_raw_mean_matches_metafor(one_sample): assert_exact(v, expected("MN", "vi"), "RM variance") -def test_raw_correlation_matches_metafor(correlations): - """``R``'s estimate must be ``escalc(measure="COR")``'s. +def test_raw_correlation_matches_metafor(correlations, escalc_inputs): + """``R`` must be ``escalc(measure="COR")``, estimate and variance alike. - The estimate only. The two variances are different formulas, which - :func:`test_raw_correlation_variance_is_a_different_formula` records. + The variance is the asymptotic sampling variance of a correlation, + ``(1 - r**2)**2 / (n - 1)``, on both sides. + + .. note:: + + PyMARE used to report ``(1 - r**2) / (n - 2)`` here, which is the squared + standard error of ``r`` under the null hypothesis of *no* correlation + rather than its sampling variance at the observed value. The two agree + near ``r = 0`` and diverge as ``|r|`` approaches one -- by a factor of + ``(n - 1) / ((n - 2)(1 - r**2))``, which is 51x at ``r = 0.99``. Because + that factor depends on the data, the old expression did not merely + inflate variances but reweighted studies against each other, pulling a + pooled estimate toward the ones with the weakest correlations. """ - y, _ = measure(correlations, "R") + y, v = measure(correlations, "R") assert_exact(y, expected("COR", "yi"), "R estimate") + assert_exact(v, expected("COR", "vi"), "R variance") + + r = escalc_inputs["r"].to_numpy() + n = escalc_inputs["n"].to_numpy() + assert np.allclose(v, (1 - r**2) ** 2 / (n - 1), rtol=RTOL_EXACT) def test_fisher_z_correlation_matches_metafor(correlations): @@ -295,28 +311,6 @@ def test_standardized_mean_difference_matches_metafor(two_sample, escalc_inputs) # ----------------------------------------------------------------------------- -def test_raw_correlation_variance_is_a_different_formula(correlations, escalc_inputs): - """Record that ``R``'s variance is not metafor's, and which is which. - - metafor uses the asymptotic sampling variance of a correlation, - ``(1 - r**2)**2 / (n - 1)``. PyMARE uses ``(1 - r**2) / (n - 2)``, which is - the squared standard error of ``r`` under the null hypothesis of no - correlation rather than its sampling variance at the observed value. The two - diverge without limit as ``|r|`` approaches one -- 51x apart at ``r = 0.99`` - on this grid -- so no tolerance relates them. - - This asserts that each side is the formula named above and nothing more. It - exists so that the divergence cannot quietly change shape: if either - expression were replaced, this test would say so rather than a tolerance - somewhere else drifting. - """ - _, v = measure(correlations, "R") - r = escalc_inputs["r"].to_numpy() - n = escalc_inputs["n"].to_numpy() - assert np.allclose(v, (1 - r**2) / (n - 2), rtol=RTOL_EXACT) - assert np.allclose(expected("COR", "vi"), (1 - r**2) ** 2 / (n - 1), rtol=RTOL_EXACT) - - def test_standardized_mean_variances_are_exact_not_asymptotic(one_sample, escalc_inputs): """Record that the single-group standardized-mean variances are not metafor's. diff --git a/validation/clubsandwich/README.md b/validation/clubsandwich/README.md index b772c1d..183c75d 100644 --- a/validation/clubsandwich/README.md +++ b/validation/clubsandwich/README.md @@ -111,18 +111,81 @@ A matrix square root does not commute with an asymmetric congruence, so the two are equal if and only if `W_j` is a multiple of the identity -- that is, when the weights, and hence the sampling variances, are constant within the cluster. -PyMARE's form is the one Fisher & Tipton (2015, arXiv:1503.02220) give as +## Why PyMARE keeps its form + +Neither implementation is approximating the other, and the difference is not a +defect in either. Bell and McCaffrey define `A_j` by the condition that the +adjusted residuals carry the working-model covariance, + +``` +A_j B_j A_j' = Psi_j +``` + +and **both forms satisfy it exactly** -- measured at 1.7e-15 (clubSandwich) and +1.3e-15 (PyMARE) on the varying-variance column of this directory's grid. +Feeding each through to the sandwich, both give `E[V_R] = (X'WX)^-1` to every +digit under the working model, so both are *exactly unbiased* in the sense CR2 +exists to provide. + +The condition does not determine `A_j` uniquely: it constrains it only through +`A_j B_j A_j'`. clubSandwich closes that freedom by requiring `A_j` symmetric; +PyMARE's is symmetric in the whitened metric instead. On the same grid, +`max |A - A'|` is 2.8e-17 for clubSandwich's and 9.1e-2 for PyMARE's. That is +the entire difference between them. + +**What PyMARE's choice buys is an algorithm.** In the whitened metric the matrix +whose inverse square root is needed is + +``` +W_j^(1/2) B_j W_j^(1/2) = I - X~_j M X~_j' +``` + +the identity minus a rank-`p` term, so its spectrum collapses to `p` non-unit +eigenvalues *whatever the group size* and `pymare.stats._cr2_low_rank_factors` +can take the inverse square root in `p x p` work. clubSandwich's matrix is +`Psi_j^2` minus a rank-`p` term, and because that diagonal part is not a multiple +of the identity its spectrum does not collapse -- measured on random designs with +`p = 2`, PyMARE's matrix has 2 non-repeated eigenvalues at every group size while +clubSandwich's has `n_j`: + +| group size | eigenvalues off the repeated one, PyMARE | clubSandwich | +| --- | --- | --- | +| 6 | 2 | 6 | +| 40 | 2 | 40 | +| 200 | 2 | 200 | + +So the symmetric form needs the full `n_j x n_j` eigendecomposition. Timed +against the `p x p` one it replaces: + +| `n_j` | full `n_j x n_j` | `p x p` | ratio | +| --- | --- | --- | --- | +| 40 | 146 us | 7.7 us | 19x | +| 200 | 3,108 us | 7.5 us | 416x | +| 800 | 52,293 us | 8.9 us | 5,901x | + +The square root not commuting with an asymmetric congruence is at once why the +two forms differ at all and why only one of them factors. + +PyMARE's form is also the one Fisher & Tipton (2015, arXiv:1503.02220) give as `A_j^C`, which `pymare.stats._cr2_scores` says in its docstring, and the correlated-effects model it belongs to has constant within-study weights by -construction. So it is not wrong so much as answering a different question. But -it is not clubSandwich's `CR2` once the variances vary inside a cluster, which is -worth knowing given that `cluster_robust_cov`'s docstring says the `CRn` naming -follows clubSandwich, and given that variances varying within a study is the -common case in practice rather than the corner one. - -Deciding which form `method="CR2"` should mean is a question for PyMARE's -maintainers, not something these tests settle. What they do is make the choice -visible and keep it from changing by accident. +construction -- which is why `validation/robumeta` cannot tell the two apart. + +### What is genuinely against it + +The published small-sample simulation evidence (Tipton 2015; Imbens & Kolesar +2016; Pustejovsky & Tipton 2018) is for the symmetric form. Exact unbiasedness +holds for both *under the working model*; how the two behave when that model is +wrong is studied for one of them and not the other. A user comparing against +`clubSandwich` or `metafor::robust(..., clubSandwich = TRUE)` will see different +standard errors whenever sampling variances vary inside a cluster, which is the +common case rather than the corner one. + +`method="CR2"` is kept, because it is a CR2 by the defining condition and +because of the complexity argument above. What was missing was saying so: the +`method` parameter and `_cr2_scores` now both record that this is the +whitened-metric solution, that it coincides with clubSandwich exactly when +within-cluster weights are constant, and that it differs otherwise. ## What is not compared diff --git a/validation/metafor/README.md b/validation/metafor/README.md index 0719cd6..d4916ed 100644 --- a/validation/metafor/README.md +++ b/validation/metafor/README.md @@ -110,6 +110,58 @@ general. PyMARE now subtracts the trace form and the two agree to 1.9e-15 across all twelve design-by-model cells. | `ML` on `extreme_k10` with one moderator | 0 vs 0.011 | The two searches land on opposite sides of the tau^2 = 0 boundary, where a profile likelihood is flattest because the weights are most unequal. metafor is the one that stops at zero. A previous version of this README had the direction backwards. | +## Why PyMARE keeps its form + +Both remaining divergences are deliberate, and both have an argument behind them +worth writing down so the next person does not have to rediscover it. + +### `I^2` and `H` + +Two reasons to stay Q-based. + +**It is the definition PyMARE cites.** Higgins & Thompson (2002), the reference +in `get_heterogeneity_stats`' docstring, define `I^2 = (Q - df)/Q` and +`H^2 = Q/df`. That is what PyMARE computes. metafor's `tau^2 / (tau^2 + v_t)` is +a defensible generalization -- it extends to models where Q is not the right +summary, such as `rma.mv` -- but it is not the cited definition. + +**Estimator-independence is a feature.** metafor's form makes `I^2` a function of +which tau^2 estimator was asked for. On `unequal_k5`, intercept-only, one dataset +and one Q, metafor reports: + +| method | FE | DL | HE | ML | REML | +| --- | --- | --- | --- | --- | --- | +| metafor `I^2` | 80.31 | 80.31 | 98.92 | 50.54 | 97.77 | +| PyMARE `I^2` | 80.31 | 80.31 | 80.31 | 80.31 | 80.31 | + +`I^2` between 50% and 99% for the same data, depending on a nuisance-parameter +estimator. For a descriptive statistic that gets quoted in abstracts and compared +across papers, invariance to that choice is worth having. It also keeps PyMARE's +heterogeneity block internally coherent: `Q`, `p(Q)`, `I^2` and `H` all describe +the same statistic, where mixing a Q-based p-value with a tau^2-based `I^2` would +not. And the two coincide for `FE` and `DL`, so with the common default most +users never see a difference. + +### The `ML` and `REML` search tolerance + +**The gap is negligible against what is being estimated.** 2.7e-5 relative on +tau^2, when the Q-profile interval for tau^2 on `unequal_k5` runs from 0.37 to +62.6 -- a factor of 167. Tightening chases the fifth decimal place of a quantity +known to within two orders of magnitude. + +**And it is not free.** `bounded_scalar_min` exists because PyMARE fits many +datasets at once: it costs "a few dozen vectorized evaluations of `f` no matter +how many datasets there are, where a per-dataset `scipy.optimize.minimize` costs +a Python-level optimization each". For the voxelwise workload that is 10^5 to +10^6 datasets in one call, so extra iterations multiply through the whole +analysis. + +It is also not a disagreement so much as two stopping rules: tightening PyMARE's +would not produce exact agreement, because metafor has its own tolerance. And it +would not touch the `extreme_k10` boundary cell, where the likelihood is flat +near zero and the two answers differ in tau^2 while barely differing in the +objective. + ## What is compared 180 cases, the full grid of design x model x tau^2 estimator x `test`. @@ -178,9 +230,9 @@ Exact, to 1e-13: | PyMARE | `escalc` measure | Compared | | --- | --- | --- | | `RM` | `MN` | estimate and variance | -| `R` | `COR` | estimate | +| `R` | `COR` | estimate and variance | | `ZR` | `ZCOR` | estimate and variance | -| `RMD` | `MD` | estimate | +| `RMD` | `MD` | estimate and variance | | `sdp` | the pooled SD `escalc` divides by, recovered as `MD$yi / (SMD$yi / c(m))` | value | ## The same quantity, one side approximated @@ -192,6 +244,26 @@ actual second order, measured at `0.043 / m**2` over the grid, worst 2.7e-3 at `m = 4` and 2.0e-7 at `m = 398`. A first-order error would break that bound. This covers the `SM` and `SMD` estimates and the factor itself. +PyMARE keeps the approximation rather than adopting the exact factor, and not +only because 2.7e-3 at `m = 4` is negligible beside the sampling error of `g` +itself (`SE(g)` is about 0.5 there, so the correction error is a thousandth of +it). The converters are a *symbolic* system: `solve_system` inverts the +expression set for whichever variable the caller did not supply. Substituting +the exact factor and asking for each direction: + +``` +forward (m, sd, n -> SM): works, and matches metafor exactly (1.47690327) +reverse (sm, d -> n): + 1 - 3/(4m - 1) -> n = 49.958 + exact c(m) -> NotImplementedError: could not solve + -sqrt(2)*gamma(n/2 - 1/2) + _sm*sqrt(n - 1)*gamma(n/2 - 1)/_d +``` + +sympy cannot invert a ratio of gamma functions, so the exact factor would buy +exact agreement on the forward path at the cost of the bidirectional solving the +module is built around. The approximation is Hedges' own, which is also what +`hedges2014statistical` -- the reference the estimator cites -- uses. + metafor has no single-group standardized mean, so `SM`'s reference is `escalc(measure = "SMCC")` with the second measurement set to zero and uncorrelated with the first. SMCC's change-score SD is then `sd1` and its @@ -201,25 +273,43 @@ approximation. The header of `run_escalc.R` spells this out. ## Different formulas for the same thing -Two, neither of which can be verified against the other. Each is recorded by a -test asserting which formula each side uses, so the divergence cannot change -shape unnoticed: - -| Quantity | metafor | PyMARE | -| --- | --- | --- | -| raw correlation variance | `(1 - r**2)**2 / (n - 1)`, the asymptotic sampling variance | `(1 - r**2) / (n - 2)`, the squared standard error under the null of no correlation. 51x apart at `r = 0.99` | -| single-group standardized-mean variances | `1/n + y**2 / (2n)`, the large-sample approximation | the exact noncentral-t expressions, which are the better quantity. 2.5x apart at `n = 5` | - -The second is why `test_standardized_mean_variance_is_the_same_order_as_metafor` -bounds by a factor of four rather than a tolerance. +One, and it is the one case in this directory where PyMARE's expression is the +*better* quantity rather than the divergent one. + +metafor's single-group standardized-mean variances are the large-sample +`1/n + y**2 / (2n)`. PyMARE's are the exact noncentral-t expressions, which is +why they cannot be verified against metafor: the two are 2.5x apart at `n = 5`, +where the approximation is poor, and converge as `n` grows. So +`test_standardized_mean_variance_is_the_same_order_as_metafor` bounds by a +factor of four rather than a tolerance, and +`test_standardized_mean_variances_are_exact_not_asymptotic` pins metafor's side +of the statement exactly so the factor is known to be spanning the asymptotic +formula and not something else a future release might report. + +The raw correlation variance used to be a second row here, PyMARE reporting +`(1 - r**2) / (n - 2)` against metafor's `(1 - r**2)**2 / (n - 1)`. That one was +not a defensible divergence: PyMARE's is the squared standard error of `r` under +the null hypothesis of *no* correlation, not its sampling variance at the +observed value. The tempting defence -- that it is conservative, being larger by +`(n - 1) / ((n - 2)(1 - r**2))` -- does not survive contact with what a +meta-analysis does with a variance. The inflation factor depends on the data, so +the expression did not merely widen intervals: it reweighted studies against one +another, down-weighting those with strong correlations (51x at `r = 0.99`) and +pulling the pooled estimate toward the weak ones. PyMARE now reports the +asymptotic sampling variance and matches `escalc(measure = "COR")` exactly. + +`ZR` remains the measure to prefer for pooling correlations, and always agreed +with metafor on both the estimate and the variance. ## What this check found -Four defects in `pymare/effectsize/expressions.json`, all of the same kind: a -missing pair of parentheses changing what the expression solves to. One is the -two-sample Cohen's d that -[PR #144](https://github.com/neurostuff/PyMARE/pull/144) fixed; the other three -were found by this check and are fixed here. +Five defective expressions in `pymare/effectsize/expressions.json`, all now +fixed. Four are the same kind of mistake -- a missing pair of parentheses +changing what the expression solves to -- of which one is the two-sample Cohen's +d that [PR #144](https://github.com/neurostuff/PyMARE/pull/144) fixed and the +other three were found here. The fifth, the raw correlation variance, was a +different kind: a correct formula for the wrong quantity, described under +"Different formulas for the same thing" above. | Expression | Read | Solved to | Should be | Effect | | --- | --- | --- | --- | --- | From 4e0174ec2c1d4e9a76200285109eccca77a91c9d Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 00:20:47 +0000 Subject: [PATCH 11/13] [FIX] do not let rounding decide the Hedges tau^2 of a saturated design Generalizing the HE correction term to tr(PV) / (K - P) introduced a second division by K - P, and at K == P both it and the residual sum of squares are zero in exact arithmetic. Dividing anyway let the sign of the rounding noise pick the answer: every leverage comes back as 1 +- 1e-16, so the subtracted term lands either side of zero and tau^2 came out +inf on this machine and NaN on the CI runners, where inf - inf replaced inf - (-inf). Only the NaN branch lost the Knapp-Hartung fallback warning that test_estimator_warns_and_falls_back_without_residual_dof asserts, so the suite passed locally and failed on all five CI platforms. The correction is now taken as a single division, which is also how metafor writes it -- (RSS - tr(PV)) / (K - P) -- and the saturated case is guarded explicitly. It reports no excess dispersion, which is what DerSimonianLaird and the likelihood estimators already report for these designs, and what Hedges itself already reported for K < P. Agreement with metafor's HE is unchanged at 1.9e-15 over the twelve cells, none of which is saturated. test_tau2_is_finite_without_residual_dof asserts finiteness rather than a value, across every variance estimator and both K < P and K == P. That is the property that was missing: a test pinning a number would have been just as platform-dependent as the bug. Verified to fail on the unguarded code here, where the bug presents as inf rather than as the NaN CI saw. Also installs arviz and cmdstanpy in the development environment used for this branch, so a local run collects the same tests as CI rather than skipping seventeen of them -- 1070 passed, 5 skipped, 5 deselected, which now matches the runners. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011o2j5ZNAdzPFL6LBqDqUsU --- pymare/estimators/estimators.py | 41 ++++++++++++++++++++++----------- pymare/tests/test_estimators.py | 30 ++++++++++++++++++++++++ 2 files changed, 58 insertions(+), 13 deletions(-) diff --git a/pymare/estimators/estimators.py b/pymare/estimators/estimators.py index f5ea582..f6a0c0f 100644 --- a/pymare/estimators/estimators.py +++ b/pymare/estimators/estimators.py @@ -1226,22 +1226,37 @@ def fit(self, y, v, X, g=None): # feeds the variance component only; the coefficients are refitted with # inverse-variance weights below. tau_beta = weighted_least_squares(tau_y, np.ones_like(tau_y), tau_X) - mse = ((tau_y - tau_X.dot(tau_beta)) ** 2).sum(0) / (tau_k - tau_p) + residual_ss = ((tau_y - tau_X.dot(tau_beta)) ** 2).sum(0) - # What that unweighted mean squared error is expected to be when tau^2 - # is zero, which is what has to be subtracted off. With P the OLS - # residual maker I - X (X'X)^-1 X' and V = diag(v), it is - # tr(PV) / (K - P) -- and since only P's diagonal is needed, that is - # sum_i (1 - h_i) v_i / (K - P) for the OLS leverages h_i. + # What that residual sum of squares is expected to be when tau^2 is + # zero, which is what has to be subtracted off. With P the OLS residual + # maker I - X (X'X)^-1 X' and V = diag(v), it is tr(PV) -- and since + # only P's diagonal is needed, that is sum_i (1 - h_i) v_i for the OLS + # leverages h_i. # - # Not the mean sampling variance sum(v) / K, which is the same quantity - # only when the intercept is the only predictor: P is then I - J/K, every - # h_i is 1/K, and the sum collapses to sum(v) (K - 1) / K / (K - 1). With - # a moderator the two part company, and using the intercept-only form - # put tau^2 out by up to 0.14 relative against metafor's HE. + # Not the mean sampling variance sum(v) / K, which is tr(PV) / (K - P) + # only when the intercept is the only predictor: P is then I - J/K, + # every h_i is 1/K, and the sum collapses to sum(v) (K - 1) / K. With a + # moderator the two part company, and using the intercept-only form put + # tau^2 out by up to 0.14 relative against metafor's HE. leverage = np.einsum("ij,jk,ik->i", tau_X, np.linalg.pinv(tau_X.T @ tau_X), tau_X) - expected_mse = ((1.0 - leverage)[:, None] * tau_v).sum(0) / (tau_k - tau_p) - tau_ho = np.maximum(0, mse - expected_mse) + expected_ss = ((1.0 - leverage)[:, None] * tau_v).sum(0) + + residual_dof = tau_k - tau_p + if residual_dof > 0: + # One division rather than two, which is also how metafor writes it. + tau_ho = np.maximum(0, (residual_ss - expected_ss) / residual_dof) + else: + # A saturated design fits every observation exactly, so there is no + # residual left to measure dispersion with and both terms above are + # zero to rounding. Dividing anyway lets the sign of that rounding + # decide the answer: the leverages come back as 1 +- 1e-16, so + # expected_ss lands either side of zero and tau^2 comes out +inf on + # one machine and NaN on the next, which is how this reached CI. + # Report no excess dispersion instead, which is what + # DerSimonianLaird and the likelihood estimators already do here, + # and what this estimator already did for K < P. + tau_ho = np.zeros_like(residual_ss) # Estimate beta with tau^2 estimate. The covariance has to come from # this fit rather than the OLS one above: (X'WX)^-1 is only the diff --git a/pymare/tests/test_estimators.py b/pymare/tests/test_estimators.py index 312c299..ddcc0c1 100644 --- a/pymare/tests/test_estimators.py +++ b/pymare/tests/test_estimators.py @@ -708,6 +708,36 @@ def test_weighted_least_squares_defaults_to_no_correction(dataset): assert not np.allclose(adjusted.fe_se, default.fe_se) +@pytest.mark.parametrize("n_estimates", [2, 3], ids=["K= 0), tau2 + + @pytest.mark.parametrize("n_estimates", [2, 3], ids=["K Date: Tue, 29 Sep 2026 00:24:52 +0000 Subject: [PATCH 12/13] [FIX] guard DerSimonianLaird's saturated design too test_tau2_is_finite_without_residual_dof, added in the previous commit to stop rounding deciding the Hedges tau^2, immediately found the same defect in DerSimonianLaird -- failing on the 3.10 runner while passing here, which is precisely the platform-dependence it exists to catch. This one is pre-existing rather than introduced by this branch. With K == P the weighted fit is exact, so both halves of tau^2 = max(0, (Q - (K - P)) / A) are zero in exact arithmetic and the quotient is one rounding residue over another. Measured here: Q = 2.5e-31 over A = -8.9e-16, giving -2.8e-16 which the floor turns into 0. On the runner the residues fall the other way -- A underflows to exactly zero against a positive numerator -- and tau^2 comes back +inf. Guarded the same way as Hedges: a saturated design has no residual dispersion to measure, so report none. That also covers K < P, where the quotient was equally undefined and DL previously returned 2.6 on this design. VarianceBasedLikelihoodEstimator seeds its search scale from this function, so ML and REML move on these designs as well -- from 0 to 1.4e-4 and from 4.0e9 to 7.4e5 at K < P. All of those are meaningless numbers for a design with nothing left to fit; what changes is that they are now reached from a deterministic starting point instead of a random one. Nothing outside the degenerate case moves: DL is untouched for K > P, and tau^2 against metafor is unchanged at 5.7e-14 (DL), 1.9e-15 (HE) and 2.7e-5 (REML) over the alignment grid, whose smallest design is K = 5 with P at most 3. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011o2j5ZNAdzPFL6LBqDqUsU --- pymare/estimators/estimators.py | 13 ++++++++++++- 1 file changed, 12 insertions(+), 1 deletion(-) diff --git a/pymare/estimators/estimators.py b/pymare/estimators/estimators.py index f6a0c0f..0a67986 100644 --- a/pymare/estimators/estimators.py +++ b/pymare/estimators/estimators.py @@ -570,7 +570,9 @@ def _dersimonian_laird_tau2(y, v, X): Returns ------- :obj:`numpy.ndarray` of shape (D,) - The tau^2 estimate per parallel dataset, floored at zero. + The tau^2 estimate per parallel dataset, floored at zero. Zero when the + design is saturated (``K <= P``), where there is no residual dispersion + to measure. Notes ----- @@ -585,6 +587,15 @@ def _dersimonian_laird_tau2(y, v, X): # Estimate initial betas with WLS, assuming tau^2=0 beta_wls, model_cov = weighted_least_squares(y, v, X, return_cov=True) + if k <= p: + # A saturated design fits every observation exactly, so Q and A are both + # zero in exact arithmetic and the quotient below is one rounding + # residue over another -- 2.5e-31 / -8.9e-16 on one machine, and a + # positive numerator over an A that underflowed to zero on the next, + # which is +inf. There is no dispersion left to measure either way, so + # report none rather than let the residues decide. + return np.zeros(np.atleast_2d(y).shape[1]) + # Cochran's Q w = 1.0 / v w_sum = w.sum(0) From 28d6485fa1acbbf8985825ec7903ac8fd43af1e6 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 00:32:58 +0000 Subject: [PATCH 13/13] [TST] cover the two lines this branch added without exercising Codecov flagged 2 uncovered lines in the patch. Both were defensive branches, and looking at them found one of each kind worth having. pymare/results.py reshaped a 2-D permuted covariance to 3-D before taking its diagonal. inv_cov is (P, P, D) on every path permutation_test can drive -- checked against WeightedLeastSquares, DerSimonianLaird, Hedges, both likelihood estimators, the sample size-based one and the cluster-robust branch -- so the branch could not be taken. Removed, with a comment recording the invariant that makes it unnecessary. pymare/stats.py fell back to NaN when the bracket search for a Q-profile bound ran out of doublings. That one is reachable, and returns the right answer: a saturated design has K - P = 0, scipy.stats.chi2.ppf returns NaN for both critical values there, every comparison against NaN is False, so the loop runs out and both bounds come back NaN -- correct for a design with no residual left to profile. It was simply untested. test_q_profile_is_undefined_without_residual_dof pins it, and the comment above the loop now says what reaches the fallback rather than leaving it looking unreachable. That is the same saturated design that produced inf from a quotient of rounding residues in DerSimonianLaird and Hedges earlier on this branch. q_profile was already handling it correctly; this records that it does. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011o2j5ZNAdzPFL6LBqDqUsU --- pymare/results.py | 6 ++++-- pymare/stats.py | 6 +++++- pymare/tests/test_stats.py | 23 +++++++++++++++++++++++ 3 files changed, 32 insertions(+), 3 deletions(-) diff --git a/pymare/results.py b/pymare/results.py index 85b9471..71a05c1 100644 --- a/pymare/results.py +++ b/pymare/results.py @@ -820,9 +820,11 @@ def permutation_test(self, n_perm=1000): # estimator just reported, exactly as `fe_se` takes the observed # one from `fe_cov` -- including whatever small-sample correction # the estimator applies, which the observed statistic carries too. + # (P, P, n_perm) on every path this method can drive -- the + # closed-form estimators, the two likelihood ones, the sample + # size-based one and the cluster-robust branch all report a + # covariance per parallel dataset. perm_cov = np.asarray(params["inv_cov"]) - if perm_cov.ndim == 2: - perm_cov = perm_cov[:, :, None] # A zero standard error divides to +-inf; the comparison below # still orders those correctly, and a NaN counts as not extreme. with np.errstate(invalid="ignore", divide="ignore"): diff --git a/pymare/stats.py b/pymare/stats.py index 88f7b71..054d240 100644 --- a/pymare/stats.py +++ b/pymare/stats.py @@ -2382,7 +2382,11 @@ def _invert_q(excess, crit, scale): # Q(tau^2) -> 0 as tau^2 -> infinity, since the weights approach a common # 1 / tau^2 that scales the residual sum of squares away. So a root exists - # for any positive crit, and doubling finds it. + # for any positive crit, and doubling finds it. What reaches the fallback + # below is a crit that is not a number at all: a saturated design has + # K - P = 0 degrees of freedom, `scipy.stats.chi2.ppf` returns NaN there, + # and every comparison against NaN is False, so the loop runs out. NaN + # bounds are the right answer for a design with no residual to profile. upper = max(abs(scale), 1.0) for _ in range(_Q_PROFILE_MAX_DOUBLINGS): if excess(upper, crit) <= 0: diff --git a/pymare/tests/test_stats.py b/pymare/tests/test_stats.py index e52d2d7..ac49bd5 100644 --- a/pymare/tests/test_stats.py +++ b/pymare/tests/test_stats.py @@ -215,6 +215,29 @@ def test_q_profile_inverts_q(vars_with_intercept): assert np.allclose(stats.q_gen(y, v, X, bounds[key]), crit, rtol=1e-12), key +def test_q_profile_is_undefined_without_residual_dof(): + """A saturated design has no residual to profile, so both bounds are NaN. + + With ``K == P`` the degrees of freedom are zero and + :func:`scipy.stats.chi2.ppf` returns NaN for both critical values. Every + comparison against NaN is False, so the bracket search runs out and + :func:`pymare.stats._invert_q` falls back to NaN -- which is the right + answer rather than an accident: there is no dispersion left to bound. + + Pinned because it is the one input that reaches that fallback, and because + a saturated design is where several of this module's neighbours have + returned ``inf`` from a quotient of rounding residues. + """ + y = np.array([[-1.0, 0.5, 2.0]]).T + v = np.array([[1.0, 1.0, 1.5]]).T + X = np.array([np.ones(3), [1.0, 2.0, 4.0], [0.5, -1.0, 3.0]]).T + assert X.shape[0] == X.shape[1] + + bounds = stats.q_profile(y, v, X, 0.05) + assert set(bounds.keys()) == {"ci_l", "ci_u"} + assert np.isnan(bounds["ci_l"]) and np.isnan(bounds["ci_u"]), bounds + + def test_q_profile_returns_zero_when_q_never_crosses(): """A bound the profile cannot reach is reported as the boundary, not a root.