Skip to content

Kruskal–Wallis检验 ​

在有序测量或不适合常规均值模型时,采用秩信息比较样本。

计算口径 ​

秩检验仍须满足研究设计与独立性等条件。只有额外的分布形状条件满足时,部分检验才可直接解释为中位数差异。

指定一个含多个独立组的分组变量。开启事后比较后使用Dunn秩比较,并按所选Holm或Bonferroni方法调整p值。

报告各组中位数及四分位数、检验统计量和p值。多组事后比较另列校正结果,不能把每个未经校正的p值独立解释。

同源实现 ​

以下片段来自 core/statistical/comparisons.py 的 nonparametric,由镜像脚本按语法树提取。共享辅助函数和分发逻辑包含在完整下载包中。

py
def nonparametric(data, options, method):
    alternative = choice(options, 'alternative', 'two-sided', ('two-sided', 'less', 'greater'))
    followup = boolean(options, 'posthoc', False)
    correction = choice(options, 'adjustment', 'holm', ('holm', 'bonferroni'))
    rows, diagnostics, followups = [], {}, []
    if method == 'stat_wilcoxon':
        for first, second in paired_columns(data, options):
            sample = enough(numeric_frame(data, [first, second]).dropna(), 3)
            outcome = _wilcoxon(sample[first].values, sample[second].values, alternative)
            rows.append({'variable': first+' − '+second, 'n': len(sample), 'group': _('配对样本'),
                         'summary': median_iqr(sample[first]-sample[second]), **outcome,
                         'effect': outcome['rank_biserial']})
            diagnostics[first+' − '+second] = outcome
        headers = [Column('variable', _('变量对'), 'text'), Column('n', _('配对数'), 'integer'),
                   Column('summary', _('差值中位数(四分位数)'), 'median_iqr'), Column('statistic', 'W'),
                   Column('p', 'p', 'p'), Column('effect', _('秩二列相关'))]
    elif method == 'stat_friedman':
        selected = columns(data, options.get('variables'), 3, 50)
        sample = enough(numeric_frame(data, selected).dropna(), 3)
        outcome = stats.friedmanchisquare(*[sample[c] for c in selected])
        if not np.isfinite(outcome.statistic):
            raise ValueError(_('数据缺乏有效的组内秩差异,无法进行Friedman检验'))
        kendall = outcome.statistic/(len(sample)*(len(selected)-1))
        for i, name in enumerate(selected):
            rows.append({'variable': name, 'n': len(sample), 'summary': median_iqr(sample[name]),
                         'statistic': outcome.statistic if i == 0 else None,
                         'p': outcome.pvalue if i == 0 else None, 'effect': kendall if i == 0 else None})
        if followup:
            for i, first in enumerate(selected):
                for second in selected[i+1:]:
                    result = _wilcoxon(sample[first].values, sample[second].values, 'two-sided')
                    followups.append({'variable': _('相关样本'), 'first': first, 'second': second, **result})
        headers = [Column('variable', _('测量变量'), 'text'), Column('n', 'N', 'integer'),
                   Column('summary', _('中位数(四分位数)'), 'median_iqr'), Column('statistic', 'χ²'),
                   Column('p', 'p', 'p'), Column('effect', "Kendall's W")]
        diagnostics['df'] = len(selected)-1
    else:
        selected = columns(data, options.get('variables'))
        for name in selected:
            names, groups = grouped_samples(data, name, options.get('group'))
            if method == 'stat_mann_whitney' and len(groups) != 2:
                raise ValueError(_('Mann–Whitney U检验需要两个有效组'))
            if method == 'stat_mann_whitney':
                a, b = groups
                outcome = stats.mannwhitneyu(a, b, alternative=alternative, method='auto')
                effect = 2*outcome.statistic/(len(a)*len(b))-1
            else:
                if len(np.unique(np.concatenate(groups))) < 2:
                    raise ValueError(_('所有观测值相同,无法进行Kruskal–Wallis检验'))
                outcome = stats.kruskal(*groups)
                effect = max(0., (outcome.statistic-len(groups)+1)/(sum(map(len, groups))-len(groups))) if sum(map(len, groups)) > len(groups) else None
                if followup:
                    for comparison in _dunn(groups):
                        followups.append({'variable': name, 'first': names[comparison['i']], 'second': names[comparison['j']],
                                          'statistic': comparison['statistic'], 'p': comparison['p']})
            for i, (label, group) in enumerate(zip(names, groups)):
                rows.append({'variable': name, 'group': label, 'n': len(group), 'summary': median_iqr(group),
                             'statistic': outcome.statistic if i == 0 else None, 'p': outcome.pvalue if i == 0 else None,
                             'effect': effect if i == 0 else None})
            diagnostics[name] = {'statistic': outcome.statistic, 'p': outcome.pvalue, 'groups': names,
                                 'n': list(map(len, groups)), 'effect': effect}
        headers = [Column('variable', _('变量'), 'text', merge=True), Column('group', _('组别'), 'text'),
                   Column('n', 'N', 'integer'), Column('summary', _('中位数(四分位数)'), 'median_iqr'),
                   Column('statistic', 'U' if method == 'stat_mann_whitney' else 'H'), Column('p', 'p', 'p'),
                   Column('effect', _('秩二列相关') if method == 'stat_mann_whitney' else 'ε²')]
    tables = [PaperTable(_('非参数检验结果'), headers, rows)]
    if followups:
        # 每个结局的比较构成一个校正家族,保持不同分析变量相互独立。
        for name in dict.fromkeys(row['variable'] for row in followups):
            family = [row for row in followups if row['variable'] == name]
            adjusted = multipletests([row['p'] for row in family], method=correction)[1]
            for row, value in zip(family, adjusted):
                row['adjusted_p'] = value
        tables.append(PaperTable(_('事后比较'), [Column('variable', _('变量'), 'text'),
                       Column('first', _('比较组1'), 'text'), Column('second', _('比较组2'), 'text'),
                       Column('statistic', _('统计量')), Column('adjusted_p', _('校正p值'), 'p')], followups,
                       _('多重比较采用%(method)s校正。', method=correction)))
    diagnostics.update(alternative=alternative, posthoc=followups)
    return StatisticalResult(tables, diagnostics)

查看完整模块 · 下载算法包 · 复现指南

复现本方法 ​

在解压目录安装 requirements.txt 后执行。样例为固定种子的模拟数据,仅供验证;输出不得冒充真实研究结果。

python
import examples._bootstrap
from examples.statistics_cases import build_data, method_cases
from core.statistical.runner import analyze

method = "stat_kruskal_wallis"
options = dict(method_cases())[method]
result = analyze(build_data(), method, options)
print(result.tables[0].html())

页面操作与解读教程

Released under the AGPL-3.0 License.