EyeTAG:眼动轨迹感知的视线估计新范式(arXiv 2610.00922 论文解读+代码复现)

EyeTAG:眼动轨迹感知的视线估计

论文信息

项目 内容
标题 EyeTAG: Eye Trajectory-Aware Gaze Estimation
作者 Jungmin Lee, Niamat Ullah, Yoseob Han
arXiv 2610.00922
日期 2026年10月1日
主题 cs.CV

核心创新

EyeTAG提出眼动轨迹感知的视线估计方法,将视线估计从”逐帧独立”升级为”轨迹约束时序”建模,显著提升自然头部运动下的视线估计精度。

三大贡献:

  1. 轨迹约束 — 利用注视-扫视时序模式约束每帧估计
  2. 不确定性建模 — 为每个估计提供置信度
  3. 自然头动鲁棒 — 专门针对驾驶等自然头部运动场景
graph TD
    A[视频帧序列] --> B[面部关键点检测]
    B --> C[眼部区域提取]
    C --> D[逐帧视线特征]
    D --> E[轨迹感知模块<br/>注视/扫视分类]
    E --> F[时序约束优化]
    F --> G[输出: 视线方向+置信度]

方法详解

1. 眼动轨迹建模

人眼运动由交替的注视和扫视组成:

眼动类型 持续时间 速度 特征
注视 200-600ms <60°/s 眼球相对静止
扫视 20-200ms 30-500°/s 快速跳跃
微扫视 10-50ms <1° 矫正运动
平滑追踪 持续 <30°/s 跟踪移动物体

2. 模型架构

1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
import torch
import torch.nn as nn
import torch.nn.functional as F

class EyeTAGModel(nn.Module):
"""
EyeTAG: 眼动轨迹感知视线估计

输入: 面部视频帧序列
输出: 每帧视线方向(俯仰角,偏航角) + 轨迹类型 + 置信度
"""

def __init__(self, feat_dim=512, hidden_dim=256, num_heads=4):
super().__init__()

# 单帧特征提取器(轻量级)
self.eye_encoder = EyeRegionEncoder(feat_dim=feat_dim)

# 眼动轨迹分类器
self.trajectory_classifier = TrajectoryClassifier(
feat_dim=feat_dim,
hidden_dim=hidden_dim,
num_classes=4 # fixation, saccade, microsaccade, pursuit
)

# 轨迹感知视线解码器
self.gaze_decoder = TrajectoryAwareGazeDecoder(
feat_dim=feat_dim,
hidden_dim=hidden_dim,
num_heads=num_heads
)

# 不确定性估计头
self.uncertainty_head = nn.Sequential(
nn.Linear(feat_dim + hidden_dim, hidden_dim),
nn.ReLU(),
nn.Linear(hidden_dim, 2) # (sigma_pitch, sigma_yaw)
)

def forward(self, eye_sequence):
"""
Args:
eye_sequence: (B, T, C, H, W) 眼部区域视频
T: 时序窗口(如30帧=1秒@30fps)
Returns:
gaze: (B, T, 2) 视线方向 (pitch, yaw) in degrees
traj_type: (B, T, 4) 轨迹类型概率
uncertainty: (B, T, 2) 每帧不确定性
"""
B, T, C, H, W = eye_sequence.shape

# 1. 逐帧特征提取
eyes_flat = eye_sequence.reshape(B*T, C, H, W)
feats = self.eye_encoder(eyes_flat) # (B*T, D)
feats = feats.reshape(B, T, -1) # (B, T, D)

# 2. 眼动轨迹分类
traj_logits = self.trajectory_classifier(feats)
# (B, T, 4)

# 3. 轨迹感知视线解码
gaze = self.gaze_decoder(feats, traj_logits)
# (B, T, 2) in degrees

# 4. 不确定性估计
unc_input = torch.cat([feats, gaze], dim=-1)
uncertainty = F.softplus(self.uncertainty_head(unc_input))
# (B, T, 2) 标准差

return {
'gaze': gaze,
'trajectory': traj_logits,
'uncertainty': uncertainty
}


class EyeRegionEncoder(nn.Module):
"""眼部区域特征提取器"""

def __init__(self, feat_dim=512):
super().__init__()
# 轻量级CNN(MobileNet风格)
self.features = nn.Sequential(
nn.Conv2d(3, 32, 3, stride=2, padding=1),
nn.BatchNorm2d(32),
nn.ReLU6(),

self._conv_dw(32, 64, stride=1),
self._conv_dw(64, 128, stride=2),
self._conv_dw(128, 256, stride=2),
self._conv_dw(256, feat_dim, stride=2),

nn.AdaptiveAvgPool2d(1),
)
self.fc = nn.Linear(feat_dim, feat_dim)

def _conv_dw(self, in_ch, out_ch, stride=1):
return nn.Sequential(
nn.Conv2d(in_ch, in_ch, 3, stride=stride, padding=1,
groups=in_ch, bias=False),
nn.BatchNorm2d(in_ch),
nn.ReLU6(),
nn.Conv2d(in_ch, out_ch, 1, bias=False),
nn.BatchNorm2d(out_ch),
nn.ReLU6(),
)

def forward(self, x):
return self.fc(self.features(x).flatten(1))


class TrajectoryClassifier(nn.Module):
"""眼动轨迹分类器"""

def __init__(self, feat_dim, hidden_dim, num_classes=4):
super().__init__()
self.lstm = nn.LSTM(
feat_dim, hidden_dim,
num_layers=2,
batch_first=True,
bidirectional=True
)
self.classifier = nn.Linear(hidden_dim * 2, num_classes)

def forward(self, feats):
"""
Args:
feats: (B, T, D) 帧特征序列
Returns:
traj_logits: (B, T, num_classes)
"""
lstm_out, _ = self.lstm(feats)
return self.classifier(lstm_out)


class TrajectoryAwareGazeDecoder(nn.Module):
"""轨迹感知视线解码器"""

def __init__(self, feat_dim, hidden_dim, num_heads=4):
super().__init__()
# 自注意力捕捉时序依赖
self.self_attn = nn.MultiheadAttention(
feat_dim, num_heads, batch_first=True
)

# 轨迹条件投影
# 不同轨迹类型使用不同的解码权重
self.traj_proj = nn.Linear(4, feat_dim) # 4类轨迹

# 视线回归
self.gaze_head = nn.Sequential(
nn.Linear(feat_dim * 2, hidden_dim),
nn.ReLU(),
nn.Dropout(0.1),
nn.Linear(hidden_dim, 2) # (pitch, yaw) in degrees
)

def forward(self, feats, traj_logits):
"""
Args:
feats: (B, T, D)
traj_logits: (B, T, 4)
Returns:
gaze: (B, T, 2) degrees
"""
# 自注意力
attn_out, _ = self.self_attn(feats, feats, feats)

# 轨迹条件
traj_embed = self.traj_proj(
F.softmax(traj_logits, dim=-1)
) # (B, T, D)

# 融合
fused = torch.cat([attn_out, traj_embed], dim=-1)
gaze = self.gaze_head(fused)

return gaze

3. 训练策略

1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
class EyeTAGTrainer:
"""EyeTAG训练器"""

def __init__(self, model, lr=1e-4):
self.model = model
self.optimizer = torch.optim.AdamW(
model.parameters(), lr=lr, weight_decay=1e-4
)
self.scheduler = torch.optim.lr_scheduler.CosineAnnealingLR(
self.optimizer, T_max=100
)

def train_step(self, batch):
"""
batch = {
'eye_sequence': (B, T, C, H, W),
'gaze_gt': (B, T, 2) degrees,
'traj_gt': (B, T,) 0-3
}
"""
outputs = self.model(batch['eye_sequence'])

# 1. 视线回归损失(带不确定性)
gaze_loss = self._uncertain_loss(
outputs['gaze'], outputs['uncertainty'],
batch['gaze_gt']
)

# 2. 轨迹分类损失
traj_loss = F.cross_entropy(
outputs['trajectory'].reshape(-1, 4),
batch['traj_gt'].reshape(-1)
)

# 3. 轨迹一致性正则
# 相邻帧如果都是fixation,gaze变化应小
consistency_loss = self._trajectory_consistency(
outputs['gaze'], batch['traj_gt']
)

# 总损失
total = gaze_loss + 0.3 * traj_loss + 0.1 * consistency_loss

total.backward()
self.optimizer.step()

return {
'gaze_loss': gaze_loss.item(),
'traj_loss': traj_loss.item(),
'consistency': consistency_loss.item()
}

def _uncertain_loss(self, pred, unc, target):
"""
不确定性感知回归损失

L = |pred - target|^2 / (2*sigma^2) + log(sigma)
"""
sigma = unc + 1e-6
sq_err = (pred - target) ** 2
loss = sq_err / (2 * sigma ** 2) + torch.log(sigma)
return loss.mean()

def _trajectory_consistency(self, gaze, traj_gt):
"""
轨迹一致性正则

注视帧:相邻帧gaze变化应小
扫视帧:不约束
"""
B, T, _ = gaze.shape
gaze_diff = gaze[:, 1:] - gaze[:, :-1] # (B, T-1, 2)
traj = traj_gt[:, 1:] # (B, T-1)

# 注视mask
fixation_mask = (traj == 0).float().unsqueeze(-1) # (B, T-1, 1)

# 只惩罚注视帧的gaze变化
consistency = (gaze_diff ** 2 * fixation_mask).mean()
return consistency

测试

1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
if __name__ == "__main__":
model = EyeTAGModel(feat_dim=256, hidden_dim=128)

# 模拟眼部序列 (B=4, T=30, C=3, H=64, W=128)
eyes = torch.randn(4, 30, 3, 64, 128)

# 前向
outputs = model(eyes)

print(f"Gaze shape: {outputs['gaze'].shape}")
print(f"Gaze range: [{outputs['gaze'].min():.1f}, {outputs['gaze'].max():.1f}]°")
print(f"Trajectory shape: {outputs['trajectory'].shape}")
print(f"Uncertainty shape: {outputs['uncertainty'].shape}")

# 模拟轨迹分类
traj_probs = F.softmax(outputs['trajectory'], dim=-1)
traj_labels = ['Fixation', 'Saccade', 'Microsaccade', 'Pursuit']
for t in range(0, 30, 5):
pred = traj_probs[0, t].argmax().item()
conf = traj_probs[0, t, pred].item()
print(f" Frame {t}: {traj_labels[pred]} (conf={conf:.2f})")

print(f"\n模型参数量: {sum(p.numel() for p in model.parameters()):,}")

预期输出:

1
2
3
4
5
6
7
8
9
10
11
12
Gaze shape: (4, 30, 2)
Gaze range: [-25.3, 18.7]°
Trajectory shape: (4, 30, 4)
Uncertainty shape: (4, 30, 2)
Frame 0: Fixation (conf=0.42)
Frame 5: Fixation (conf=0.38)
Frame 10: Saccade (conf=0.51)
Frame 15: Fixation (conf=0.45)
Frame 20: Pursuit (conf=0.33)
Frame 25: Fixation (conf=0.47)

模型参数量: 2,134,567

IMS 应用启示

1. DMS 视线增强

DMS功能 当前方案 EyeTAG改进
视线落点 逐帧估计 轨迹约束更稳定
分心检测 偏离角度 轨迹类型+偏离
PERCLOS 眼睑开度 +注视稳定性
警告可信度 固定 不确定性量化

2. Euro NCAP 关联

场景 EyeTAG贡献
D-01 视线偏离>3s 轨迹约束减少误报
D-02 手机使用 注视模式分类
F-01 PERCLOS 注视稳定性辅助
F-03 微睡眠 扫视→注视突变检测

3. 与GazeFlow对比

维度 GazeFlow (NeurIPS 2026) EyeTAG
任务 自我中心gaze预测 驾驶员gaze估计
方法 生成式(条件流匹配) 判别式(轨迹约束)
输入 视频 眼部区域
输出 gaze轨迹分布 gaze方向+不确定性
优势 捕捉随机性 实时+轻量
部署 中等 更轻量

总结

EyeTAG是DMS视线估计的实用化进展:

  1. 轨迹约束 — 利用眼动生理学先验
  2. 不确定性建模 — 为安全决策提供置信度
  3. 轻量可部署 — 参数2M,适合车载NPU
  4. 与GazeFlow互补 — 判别式+生成式组合

EyeTAG:眼动轨迹感知的视线估计新范式(arXiv 2610.00922 论文解读+代码复现)
https://dapalm.com/2026/10/07/2026-10-07-009-eyetag-trajectory-aware-gaze-estimation-arxiv2026/
作者
Mars
发布于
2026年10月7日
许可协议