Skip to content

探索性因子分析 ​

概括多个相关指标的共同结构或进行降维。

计算口径 ​

使用数值题项的完整案例,排除常量与线性依赖。因子/成分数可手动指定;0使用特征值大于1的辅助规则,应结合理论和碎石图复核。

提取法选择主轴因子法PAF或最大似然ML。Varimax保持因子正交;Promax允许因子相关。理论预期相关的维度不应仅为方便而强制正交。

主表展示载荷与共同度。原始输出包括KMO、Bartlett、方差解释、旋转矩阵等。Promax下模式载荷与结构载荷含义不同,相关因子的解释方差不宜简单相加。

同源实现 ​

以下片段来自 core/statistical/measurement.py 的 factor_analysis,由镜像脚本按语法树提取。共享辅助函数和分发逻辑包含在完整下载包中。

py
def factor_analysis(data, options, pca=False):
    selected = columns(data, options.get('variables'), 3, 80)
    frame = enough(numeric_frame(data, selected).dropna(), len(selected)+2)
    for name in selected:
        varying(frame[name].values, name)
    correlation = frame.corr().to_numpy()
    adequacy = sampling_adequacy(correlation, len(frame))
    eigenvalues, eigenvectors = np.linalg.eigh(correlation)
    order = np.argsort(eigenvalues)[::-1]
    eigenvalues, eigenvectors = eigenvalues[order], eigenvectors[:, order]
    factors = number(options, 'factors', 0, 0, len(selected), integer=True)
    selection = 'manual' if factors else 'eigenvalue_greater_than_one'
    factors = factors or max(1, int(np.sum(eigenvalues > 1)))
    extraction = 'pca' if pca else choice(options, 'extraction', 'paf', ('paf', 'ml'))
    if not pca and ((len(selected)-factors)**2-(len(selected)+factors))/2 < 0:
        raise ValueError(_('指定公共因子数导致模型欠识别,请减少因子数'))
    iterations = None
    if pca:
        loadings = eigenvectors[:, :factors]*np.sqrt(eigenvalues[:factors])
    elif extraction == 'paf':
        loadings, iterations = principal_axis(correlation, factors)
    else:
        model = FactorAnalyzer(n_factors=factors, rotation=None, method='ml', is_corr_matrix=True, bounds=(.005, 1))
        model.fit(correlation)
        loadings = model.loadings_
        if np.any(model.get_uniquenesses() <= .00501):
            raise ValueError(_('最大似然因子解触及独特性边界,请检查模型与题项'))
    unrotated = loadings.copy()
    rotation = choice(options, 'rotation', 'varimax', ('none', 'varimax', 'promax'))
    phi = np.eye(factors)
    if rotation != 'none' and factors > 1:
        loadings, phi = rotate_loadings(loadings, rotation)
    # 统一因子方向,斜交因子相关矩阵随载荷同时变号。
    signs = np.where(loadings[np.argmax(abs(loadings), axis=0), np.arange(factors)] < 0, -1, 1)
    loadings = loadings*signs
    phi = phi*np.outer(signs, signs)
    communalities = np.diag(loadings @ phi @ loadings.T)
    headers = [Column('item', _('题项'), 'text')]
    headers += [Column('factor'+str(i), _('成分%(n)s', n=i+1) if pca else _('因子%(n)s', n=i+1)) for i in range(factors)]
    headers.append(Column('communality', _('共同度')))
    rows = [{'item': name, 'communality': communalities[i],
             **{'factor'+str(j): loadings[i, j] for j in range(factors)}} for i, name in enumerate(selected)]
    raw = {'n': len(frame), 'variables': selected, 'correlation': correlation, **adequacy,
           'eigenvalues': eigenvalues, 'factor_count': factors, 'factor_selection': selection,
           'extraction': extraction, 'rotation': rotation, 'unrotated_loadings': unrotated,
           'pattern_matrix': loadings, 'structure_matrix': loadings @ phi, 'factor_correlation': phi,
           'communalities': communalities, 'unrotated_explained_variance': np.sum(unrotated**2, axis=0)/len(selected),
           'iterations': iterations}
    note = _('N=%(n)s;提取方法:%(method)s;旋转:%(rotation)s。', n=len(frame), method=extraction.upper(), rotation=rotation)
    table = PaperTable(_('主成分载荷矩阵') if pca else _('探索性因子载荷矩阵'), headers, rows, note)
    from .plots import scree_plot
    return StatisticalResult([table], raw, [scree_plot(eigenvalues, factors)])

查看完整模块 · 下载算法包 · 复现指南

复现本方法 ​

在解压目录安装 requirements.txt 后执行。样例为固定种子的模拟数据,仅供验证;输出不得冒充真实研究结果。

python
import examples._bootstrap
from examples.statistics_cases import build_data, method_cases
from core.statistical.runner import analyze

method = "stat_efa"
options = dict(method_cases())[method]
result = analyze(build_data(), method, options)
print(result.tables[0].html())

页面操作与解读教程

Released under the AGPL-3.0 License.