外观
题项质量分析
联合检查题项分布、题项与量表的一致性以及高低分组鉴别能力。
计算口径
先统一计分方向。总分分组采用所选题项之和;边界并列分数全部纳入,实际组人数可能超过设定比例。
高低分组默认各取27%,可在0.10至0.49之间设置。边界并列导致两组重叠或总分没有变化时不能进行有效鉴别检验,应先核对数据。
综合均值、标准差、CITC、高低组Welch检验和删题α评估题目。结果是筛查依据,保留或删题仍须有理论理由。
同源实现
以下片段来自 core/statistical/measurement.py 的 item_analysis,由镜像脚本按语法树提取。共享辅助函数和分发逻辑包含在完整下载包中。
py
def item_analysis(data, options):
fraction = number(options, 'extreme_fraction', .27, .1, .49)
rows, details = [], {}
for label, selected in dimension_sets(data, options):
frame = enough(numeric_frame(data, selected).dropna(), 8)
values = frame.values
scores = values.sum(axis=1)
low_cut = np.quantile(scores, fraction)
high_cut = np.quantile(scores, 1-fraction)
low, high = scores <= low_cut, scores >= high_cut
if low_cut >= high_cut or np.any(low & high):
raise ValueError(_('维度“%(name)s”的高低分组因并列分数重叠,请调整比例或题项', name=label))
if min(low.sum(), high.sum()) < 2:
raise ValueError(_('高低分组各需要至少两条有效记录'))
for i, item in enumerate(item_statistics(values, selected)):
a, b = values[high, i], values[low, i]
if np.var(a)+np.var(b) == 0:
# 常数题项保留在项目分析表中,避免自动删除掩盖问题。
t, p = (0., 1.) if np.mean(a) == np.mean(b) else (None, None)
else:
outcome = stats.ttest_ind(a, b, equal_var=False)
t, p = outcome.statistic, outcome.pvalue
rows.append({'dimension': label, **item, 'low': (np.mean(b), np.std(b, ddof=1)),
'high': (np.mean(a), np.std(a, ddof=1)), 't': t, 'p': p})
details[label] = {'n': len(values), 'high_n': int(high.sum()), 'low_n': int(low.sum()),
'high_cutoff': high_cut, 'low_cutoff': low_cut, 'tie_policy': 'include_boundary_ties',
'test': 'Welch', 'item_order': selected}
table = PaperTable(_('项目分析'), [Column('dimension', _('维度'), 'text', merge=True), Column('item', _('题项'), 'text'),
Column('mean', _('均值')), Column('sd', _('标准差')), Column('low', _('低分组'), 'mean_sd'),
Column('high', _('高分组'), 'mean_sd'), Column('t', 't'), Column('p', 'p', 'p'),
Column('citc', 'CITC'), Column('alpha_deleted', _('删除后α'))], rows)
return StatisticalResult([table], details)复现本方法
在解压目录安装 requirements.txt 后执行。样例为固定种子的模拟数据,仅供验证;输出不得冒充真实研究结果。
python
import examples._bootstrap
from examples.statistics_cases import build_data, method_cases
from core.statistical.runner import analyze
method = "stat_item_analysis"
options = dict(method_cases())[method]
result = analyze(build_data(), method, options)
print(result.tables[0].html())