外观
探索性因子分析
概括多个相关指标的共同结构或进行降维。
计算口径
使用数值题项的完整案例,排除常量与线性依赖。因子/成分数可手动指定;0使用特征值大于1的辅助规则,应结合理论和碎石图复核。
提取法选择主轴因子法PAF或最大似然ML。Varimax保持因子正交;Promax允许因子相关。理论预期相关的维度不应仅为方便而强制正交。
主表展示载荷与共同度。原始输出包括KMO、Bartlett、方差解释、旋转矩阵等。Promax下模式载荷与结构载荷含义不同,相关因子的解释方差不宜简单相加。
同源实现
以下片段来自 core/statistical/measurement.py 的 factor_analysis,由镜像脚本按语法树提取。共享辅助函数和分发逻辑包含在完整下载包中。
py
def factor_analysis(data, options, pca=False):
selected = columns(data, options.get('variables'), 3, 80)
frame = enough(numeric_frame(data, selected).dropna(), len(selected)+2)
for name in selected:
varying(frame[name].values, name)
correlation = frame.corr().to_numpy()
adequacy = sampling_adequacy(correlation, len(frame))
eigenvalues, eigenvectors = np.linalg.eigh(correlation)
order = np.argsort(eigenvalues)[::-1]
eigenvalues, eigenvectors = eigenvalues[order], eigenvectors[:, order]
factors = number(options, 'factors', 0, 0, len(selected), integer=True)
selection = 'manual' if factors else 'eigenvalue_greater_than_one'
factors = factors or max(1, int(np.sum(eigenvalues > 1)))
extraction = 'pca' if pca else choice(options, 'extraction', 'paf', ('paf', 'ml'))
if not pca and ((len(selected)-factors)**2-(len(selected)+factors))/2 < 0:
raise ValueError(_('指定公共因子数导致模型欠识别,请减少因子数'))
iterations = None
if pca:
loadings = eigenvectors[:, :factors]*np.sqrt(eigenvalues[:factors])
elif extraction == 'paf':
loadings, iterations = principal_axis(correlation, factors)
else:
model = FactorAnalyzer(n_factors=factors, rotation=None, method='ml', is_corr_matrix=True, bounds=(.005, 1))
model.fit(correlation)
loadings = model.loadings_
if np.any(model.get_uniquenesses() <= .00501):
raise ValueError(_('最大似然因子解触及独特性边界,请检查模型与题项'))
unrotated = loadings.copy()
rotation = choice(options, 'rotation', 'varimax', ('none', 'varimax', 'promax'))
phi = np.eye(factors)
if rotation != 'none' and factors > 1:
loadings, phi = rotate_loadings(loadings, rotation)
# 统一因子方向,斜交因子相关矩阵随载荷同时变号。
signs = np.where(loadings[np.argmax(abs(loadings), axis=0), np.arange(factors)] < 0, -1, 1)
loadings = loadings*signs
phi = phi*np.outer(signs, signs)
communalities = np.diag(loadings @ phi @ loadings.T)
headers = [Column('item', _('题项'), 'text')]
headers += [Column('factor'+str(i), _('成分%(n)s', n=i+1) if pca else _('因子%(n)s', n=i+1)) for i in range(factors)]
headers.append(Column('communality', _('共同度')))
rows = [{'item': name, 'communality': communalities[i],
**{'factor'+str(j): loadings[i, j] for j in range(factors)}} for i, name in enumerate(selected)]
raw = {'n': len(frame), 'variables': selected, 'correlation': correlation, **adequacy,
'eigenvalues': eigenvalues, 'factor_count': factors, 'factor_selection': selection,
'extraction': extraction, 'rotation': rotation, 'unrotated_loadings': unrotated,
'pattern_matrix': loadings, 'structure_matrix': loadings @ phi, 'factor_correlation': phi,
'communalities': communalities, 'unrotated_explained_variance': np.sum(unrotated**2, axis=0)/len(selected),
'iterations': iterations}
note = _('N=%(n)s;提取方法:%(method)s;旋转:%(rotation)s。', n=len(frame), method=extraction.upper(), rotation=rotation)
table = PaperTable(_('主成分载荷矩阵') if pca else _('探索性因子载荷矩阵'), headers, rows, note)
from .plots import scree_plot
return StatisticalResult([table], raw, [scree_plot(eigenvalues, factors)])复现本方法
在解压目录安装 requirements.txt 后执行。样例为固定种子的模拟数据,仅供验证;输出不得冒充真实研究结果。
python
import examples._bootstrap
from examples.statistics_cases import build_data, method_cases
from core.statistical.runner import analyze
method = "stat_efa"
options = dict(method_cases())[method]
result = analyze(build_data(), method, options)
print(result.tables[0].html())