外观
单因素方差分析
比较多个独立组的均值,并在适用时分析因素交互。
计算口径
本入口为独立组方差分析,不是重复测量设计。需关注组内残差、方差齐性、各组合样本量与空单元格。
选择一个分组变量和连续结局。经典ANOVA配Tukey或Bonferroni;方差不齐时可选Welch并配Games–Howell。不匹配的组合会要求修正。
总体F检验显著不代表每两组都不同。需要定位差异时选择相匹配的事后比较,并解释校正后的p值。
同源实现
以下片段来自 core/statistical/anova.py 的 oneway,由镜像脚本按语法树提取。共享辅助函数和分发逻辑包含在完整下载包中。
py
def oneway(data, options):
selected = columns(data, options.get('variables'))
group_name = columns(data, [options.get('group')])[0]
alpha = number(options, 'alpha', .05, .0001, .25)
method = choice(options, 'anova_method', 'classic', ('classic', 'welch'))
posthoc = choice(options, 'posthoc', 'none', ('none', 'tukey', 'bonferroni', 'games_howell'))
if method == 'welch' and posthoc in ('tukey', 'bonferroni'):
raise ValueError(_('Welch方差分析请配合Games–Howell比较,或不进行事后比较'))
rows, comparisons, details = [], [], {}
all_groups = list(dict.fromkeys(str(v) for v in data[group_name].dropna()))
head = [Column('variable', _('变量'), 'text')]
for i, label in enumerate(all_groups):
head.extend([Column('n'+str(i), 'N', 'integer', label), Column('s'+str(i), _('均值±标准差'), 'mean_sd', label)])
head += [Column('f', 'F'), Column('df1', _('组间df')), Column('df2', _('误差df')), Column('p', 'p', 'p'), Column('eta2', 'η²')]
for variable in selected:
names, groups = grouped_samples(data, variable, group_name)
groups = [enough(x, 2) for x in groups]
n = np.array([len(x) for x in groups], dtype=float)
means = np.array([np.mean(x) for x in groups])
variances = np.array([np.var(x, ddof=1) for x in groups])
k = len(groups)
grand_mean = np.sum(n*means)/sum(n)
between = np.sum(n*(means-grand_mean)**2)
within = np.sum((n-1)*variances)
if within <= 0:
raise ValueError(_('各组均缺乏组内变异,无法进行方差分析'))
if method == 'classic':
df1, df2 = k-1, sum(n)-k
f = between/df1/(within/df2)
else:
if np.any(variances <= 0):
raise ValueError(_('Welch方差分析要求各组存在有效组内变异'))
weights = n/variances
weight_sum = sum(weights)
weighted_mean = sum(weights*means)/weight_sum
correction = sum((1-weights/weight_sum)**2/(n-1))
df1, df2 = k-1, (k*k-1)/(3*correction)
f = sum(weights*(means-weighted_mean)**2)/(k-1)/(1+2*(k-2)*correction/(k*k-1))
p = stats.f.sf(f, df1, df2)
eta2 = between/(between+within)
row = {'variable': variable, 'f': f, 'df1': df1, 'df2': df2, 'p': p, 'eta2': eta2}
for label, values in zip(names, groups):
position = all_groups.index(label)
row['n'+str(position)], row['s'+str(position)] = len(values), mean_sd(values)
rows.append(row)
levene = stats.levene(*groups, center='median')
details[variable] = {'ss_between': between, 'ss_within': within, 'method': method,
'levene': {'statistic': levene.statistic, 'p': levene.pvalue}, 'n': n}
if posthoc != 'none':
comparisons.extend({'variable': variable, **row} for row in posthoc_comparisons(names, groups, posthoc, alpha))
tables = [PaperTable(_('单因素方差分析'), head, rows)]
if comparisons:
tables.append(PaperTable(_('事后多重比较'), [Column('variable', _('变量'), 'text'),
Column('first', _('组1'), 'text'), Column('second', _('组2'), 'text'),
Column('difference', _('均值差')), Column('se', _('标准误')), Column('p', _('校正p值'), 'p'),
Column('ci', _('%(level)s%%置信区间', level='%g' % (100*(1-alpha))), 'interval')], comparisons,
_('比较方法:%(method)s。', method=posthoc)))
details['posthoc'] = comparisons
return StatisticalResult(tables, details)复现本方法
在解压目录安装 requirements.txt 后执行。样例为固定种子的模拟数据,仅供验证;输出不得冒充真实研究结果。
python
import examples._bootstrap
from examples.statistics_cases import build_data, method_cases
from core.statistical.runner import analyze
method = "stat_anova_oneway"
options = dict(method_cases())[method]
result = analyze(build_data(), method, options)
print(result.tables[0].html())