1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155
| """ 熵基多类逻辑回归分类器 检测清醒/低BAC/高BAC三分类 """
import numpy as np from sklearn.linear_model import LogisticRegression from sklearn.model_selection import LeaveOneOut, cross_val_score from sklearn.metrics import roc_auc_score, classification_report from sklearn.preprocessing import StandardScaler from sklearn.pipeline import Pipeline
class AlcoholImpairmentClassifier: """ 酒精损伤运动学分类器 特征:7个关键运动学特征的 PE 和 SD 分类器:多类逻辑回归 """ def __init__(self): self.key_features = [ 'ax_pe', 'ax_sd', 'ay_pe', 'ay_sd', 'gx_pe', 'gx_sd', 'throttle_pe', ] self.pipeline = Pipeline([ ('scaler', StandardScaler()), ('clf', LogisticRegression( multi_class='multinomial', solver='lbfgs', C=1.0, max_iter=1000 )) ]) def extract_features(self, imu_data: np.ndarray, throttle_data: np.ndarray, window_size: int = 100) -> np.ndarray: """ 提取特征向量 Args: imu_data: (N, 6) IMU数据 throttle_data: (N,) 油门位置 window_size: 窗口大小 """ features = compute_kinematic_features(imu_data, window_size) num_windows = len(throttle_data) // window_size pe_values = [] sd_values = [] for w in range(num_windows): window = throttle_data[w * window_size:(w + 1) * window_size] pe_values.append(permutation_entropy(window, order=3)) sd_values.append(np.std(window)) features['throttle_pe'] = np.mean(pe_values) features['throttle_sd'] = np.mean(sd_values) feat_vector = np.array([ features.get(f, 0.0) for f in self.key_features ]) return feat_vector def train(self, X: np.ndarray, y: np.ndarray) -> dict: """ 训练分类器(留一被试交叉验证) Args: X: (n_samples, n_features) y: (n_samples,) 0=清醒, 1=低BAC, 2=高BAC Returns: metrics: 准确率和AuROC """ loo = LeaveOneOut() y_pred = np.zeros(len(y)) y_proba = np.zeros((len(y), 3)) for train_idx, test_idx in loo.split(X): X_train, X_test = X[train_idx], X[test_idx] y_train, y_test = y[train_idx], y[test_idx] from sklearn.base import clone clf = clone(self.pipeline) clf.fit(X_train, y_train) y_pred[test_idx] = clf.predict(X_test) y_proba[test_idx] = clf.predict_proba(X_test) accuracy = np.mean(y_pred == y) from sklearn.preprocessing import label_binarize y_bin = label_binarize(y, classes=[0, 1, 2]) auc = roc_auc_score(y_bin, y_proba, multi_class='ovr', average='weighted') mask = (y == 0) | (y == 2) if np.sum(mask) > 1: auc_sober_high = roc_auc_score( y[mask] // 2, y_proba[mask, 2] ) else: auc_sober_high = 0.0 self.pipeline.fit(X, y) return { 'accuracy': accuracy, 'auc_weighted': auc, 'auc_sober_vs_high': auc_sober_high, 'classification_report': classification_report(y, y_pred) }
if __name__ == "__main__": classifier = AlcoholImpairmentClassifier() np.random.seed(42) n_samples = 75 n_features = len(classifier.key_features) sober_X = np.random.randn(25, n_features) * 0.5 + np.array([0.8, 0.3] * 3 + [0.7]) low_X = np.random.randn(25, n_features) * 0.5 + np.array([0.65, 0.5] * 3 + [0.6]) high_X = np.random.randn(25, n_features) * 0.5 + np.array([0.45, 0.8] * 3 + [0.45]) X = np.vstack([sober_X, low_X, high_X]) y = np.array([0]*25 + [1]*25 + [2]*25) results = classifier.train(X, y) print("=== 分类结果 ===") print(f"准确率: {results['accuracy']:.1%}") print(f"加权AuROC: {results['auc_weighted']:.2f}") print(f"清醒vs高BAC AuROC: {results['auc_sober_vs_high']:.2f}") print(f"\n{results['classification_report']}")
|