3D乘员姿态估计:深度+红外图像三阶段微调方法——论文解读与代码复现

论文信息

核心创新

首个使用纯深度+红外图像进行车内乘员3D姿态估计的工作,通过三阶段微调策略(仿真数据→近似标注→少量手动标注),仅用<100个标注样本训练出中值误差<10cm的模型。

方法详解

1. 三阶段微调训练策略

graph LR
    A[阶段1: 仿真数据<br/>40,000样本<br/>SBSM人体模型] --> B[阶段2: 域适应数据<br/>OpenPose自动标注<br/>IR图像2D→3D]
    B --> C[阶段3: 手动标注数据<br/><100样本<br/>真实车内数据]
    C --> D[最终模型<br/>中值误差<10cm]
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
import torch
import torch.nn as nn
import torch.nn.functional as F
import numpy as np

class OccupantPostureEstimator(nn.Module):
"""
3D乘员姿态估计器

论文核心: 使用深度+IR图像预测15个关键关节的3D位置

关节定义:
0: pelvis(骨盆) 5: left_hip(左髋) 10: left_shoulder(左肩)
1: abdomen(腹部) 6: left_knee(左膝) 11: left_elbow(左肘)
2: thorax(胸椎) 7: right_hip(右髋) 12: left_wrist(左腕)
3: neck(颈部) 8: right_knee(右膝) 13: right_shoulder(右肩)
4: head(头部) 14: right_elbow(右肘)
15: right_wrist(右腕)
"""

def __init__(self, num_joints: int = 15):
super().__init__()

# 双相机输入: 左深度+左IR + 右深度+右IR
in_channels = 4 # depth_L + IR_L + depth_R + IR_R

# 共享编码器
self.encoder = nn.Sequential(
nn.Conv2d(in_channels, 32, 3, stride=2, padding=1),
nn.BatchNorm2d(32),
nn.ReLU(),
nn.Conv2d(32, 64, 3, stride=2, padding=1),
nn.BatchNorm2d(64),
nn.ReLU(),
nn.Conv2d(64, 128, 3, stride=2, padding=1),
nn.BatchNorm2d(128),
nn.ReLU(),
nn.Conv2d(128, 256, 3, stride=2, padding=1),
nn.BatchNorm2d(256),
nn.ReLU(),
nn.AdaptiveAvgPool2d((4, 4))
)

# 关节回归头
self.joint_regressor = nn.Sequential(
nn.Flatten(),
nn.Linear(256 * 4 * 4, 512),
nn.ReLU(),
nn.Dropout(0.3),
nn.Linear(512, 256),
nn.ReLU(),
nn.Linear(256, num_joints * 3) # 每关节3D坐标
)

def forward(self, depth_left: torch.Tensor, ir_left: torch.Tensor,
depth_right: torch.Tensor, ir_right: torch.Tensor) -> torch.Tensor:
"""
Args:
depth_left/right: (B, 1, H, W) 深度图
ir_left/right: (B, 1, H, W) 红外图像

Returns:
joints_3d: (B, 15, 3) 相对于身体中心的3D关节位置
"""
x = torch.cat([depth_left, ir_left, depth_right, ir_right], dim=1)
features = self.encoder(x)
joints = self.joint_regressor(features)
return joints.view(-1, 15, 3)


class ThreeStageTrainer:
"""
三阶段微调训练器

论文Section 3.2: 仿真→域适应→真实
"""

def __init__(self, model: OccupantPostureEstimator):
self.model = model
self.criterion = nn.MSELoss()

def stage1_simulated(self, sim_data: dict, epochs: int = 100):
"""
阶段1: 仿真数据预训练

使用Statistical Body Shape Model (SBSM)生成
- 50,000个合成姿态
- 20个不同人体测量学参数
- 精确已知关节位置
- 仅包含人体网格(无车内环境/衣物)
"""
optimizer = torch.optim.Adam(self.model.parameters(), lr=1e-3)

for epoch in range(epochs):
for batch in sim_data['loader']:
depth_l = batch['depth_left']
ir_l = batch['ir_left']
depth_r = batch['depth_right']
ir_r = batch['ir_right']
gt_joints = batch['joints_3d'] # (B, 15, 3)

pred = self.model(depth_l, ir_l, depth_r, ir_r)
loss = self.criterion(pred, gt_joints)

optimizer.zero_grad()
loss.backward()
optimizer.step()

print(f"阶段1完成: 仿真数据训练 {epochs} epochs")
return self.model

def stage2_domain_adaptation(self, approx_data: dict, epochs: int = 50):
"""
阶段2: 域适应(近似标注)

使用OpenPose在IR图像上检测2D关节
从深度图获取对应深度值
构建近似3D标注

挑战:
- OpenPose在深度图像上准确率低
- 衣物褶皱影响深度读数
- 车内环境背景干扰
"""
optimizer = torch.optim.Adam(self.model.parameters(), lr=1e-4)

for epoch in range(epochs):
for batch in approx_data['loader']:
pred = self.model(
batch['depth_left'], batch['ir_left'],
batch['depth_right'], batch['ir_right']
)
# 近似标注(含噪声)
loss = self.criterion(pred, batch['approx_joints']) * 0.5

optimizer.zero_grad()
loss.backward()
optimizer.step()

print(f"阶段2完成: 域适应 {epochs} epochs")
return self.model

def stage3_finetune(self, manual_data: dict, epochs: int = 200):
"""
阶段3: 手动标注微调

<100个精确标注的真实车内样本
使用更小学习率精细调整
"""
optimizer = torch.optim.Adam(self.model.parameters(), lr=1e-5)

for epoch in range(epochs):
for batch in manual_data['loader']:
pred = self.model(
batch['depth_left'], batch['ir_left'],
batch['depth_right'], batch['ir_right']
)
loss = self.criterion(pred, batch['gt_joints'])

optimizer.zero_grad()
loss.backward()
optimizer.step()

print(f"阶段3完成: 手动标注微调 {epochs} epochs")
return self.model


# ==================== 仿真数据生成 ====================

class SBMSDataGenerator:
"""
Statistical Body Shape Model 数据生成器

论文方法: 拉丁超立方采样生成50,000个姿态
"""

def __init__(self, num_joints: int = 15, num_subjects: int = 20):
self.num_joints = num_joints
self.num_subjects = num_subjects

# 关节角度范围(弧度)
self.joint_ranges = {
'neck_yaw': (-1.0, 1.0),
'neck_pitch': (-0.5, 0.3),
'spine_pitch': (-0.3, 0.8), # 前倾范围大
'spine_yaw': (-0.3, 0.3),
'left_shoulder': (-0.5, 1.5),
'left_elbow': (0.0, 2.0),
'right_shoulder': (-0.5, 1.5),
'right_elbow': (0.0, 2.0),
'left_hip': (-0.5, 0.5),
'left_knee': (0.0, 1.5),
'right_hip': (-0.5, 0.5),
'right_knee': (0.0, 1.5),
}

def latin_hypercube_sampling(self, n_samples: int) -> np.ndarray:
"""
拉丁超立方采样关节角度

确保每个维度均匀覆盖参数空间
"""
from scipy.stats import qmc

n_dims = len(self.joint_ranges)
sampler = qmc.LatinHypercube(d=n_dims)
samples = sampler.random(n=n_samples)

# 缩放到实际范围
result = np.zeros((n_samples, n_dims))
for i, (key, (lo, hi)) in enumerate(self.joint_ranges.items()):
result[:, i] = samples[:, i] * (hi - lo) + lo

return result

def generate(self, n_samples: int = 40000) -> dict:
"""生成合成数据"""
# 采样关节角度
angles = self.latin_hypercube_sampling(n_samples)

# 随机配对人体参数
subject_ids = np.random.randint(0, self.num_subjects, n_samples)

# 生成深度图(简化: 返回随机噪声作为占位)
H, W = 240, 320
depth_left = torch.randn(n_samples, 1, H, W) * 0.1
ir_left = torch.randn(n_samples, 1, H, W) * 0.1
depth_right = torch.randn(n_samples, 1, H, W) * 0.1
ir_right = torch.randn(n_samples, 1, H, W) * 0.1

# 生成3D关节位置(简化: 使用角度直接映射)
joints_3d = torch.zeros(n_samples, self.num_joints, 3)
for i in range(n_samples):
# 基础骨架+角度变换(简化模型)
joints_3d[i, 0] = torch.tensor([0, 0, 0]) # pelvis
joints_3d[i, 1] = torch.tensor([0, 0.15, 0]) # abdomen
joints_3d[i, 2] = torch.tensor([0, 0.35, 0]) # thorax
joints_3d[i, 3] = torch.tensor([0, 0.50, 0]) # neck
joints_3d[i, 4] = torch.tensor([0, 0.65, 0]) # head

# 手臂(根据角度)
ls = angles[i, 4] # left_shoulder
le = angles[i, 5] # left_elbow
joints_3d[i, 10] = torch.tensor([0.18, 0.42, 0])
joints_3d[i, 11] = joints_3d[i, 10] + torch.tensor([
0.15*np.cos(ls), -0.25*np.sin(ls), 0
])
joints_3d[i, 12] = joints_3d[i, 11] + torch.tensor([
0.12*np.cos(le), -0.20*np.sin(le), 0
])

# 右臂镜像
joints_3d[i, 13] = torch.tensor([-0.18, 0.42, 0])
joints_3d[i, 14] = joints_3d[i, 13] + torch.tensor([
-0.15*np.cos(ls), -0.25*np.sin(ls), 0
])

# 腿部
joints_3d[i, 5] = torch.tensor([0.12, -0.05, 0])
joints_3d[i, 6] = torch.tensor([0.12, -0.35, 0])
joints_3d[i, 7] = torch.tensor([-0.12, -0.05, 0])
joints_3d[i, 8] = torch.tensor([-0.12, -0.35, 0])

return {
'depth_left': depth_left, 'ir_left': ir_left,
'depth_right': depth_right, 'ir_right': ir_right,
'joints_3d': joints_3d
}


# ==================== 评估 ====================

def evaluate_model(model: nn.Module, test_data: dict) -> dict:
"""评估模型性能"""
model.eval()
all_errors = []

with torch.no_grad():
pred = model(
test_data['depth_left'], test_data['ir_left'],
test_data['depth_right'], test_data['ir_right']
)

# 每关节误差(cm)
errors = torch.norm(
(pred - test_data['joints_3d']) * 100,
dim=-1
) # (N, 15)

all_errors = errors.numpy()

joint_names = [
'pelvis', 'abdomen', 'thorax', 'neck', 'head',
'L_hip', 'L_knee', 'R_hip', 'R_knee',
'L_shoulder', 'L_elbow', 'L_wrist',
'R_shoulder', 'R_elbow', 'R_wrist'
]

results = {}
for i, name in enumerate(joint_names):
joint_err = all_errors[:, i]
results[name] = {
'mean': np.mean(joint_err),
'median': np.median(joint_err),
'std': np.std(joint_err),
'p90': np.percentile(joint_err, 90),
}

return results


# ==================== 测试 ====================
if __name__ == "__main__":
print("=" * 60)
print("3D乘员姿态估计: 深度+IR三阶段微调")
print("论文复现: Tambwekar et al., Sensors 2024")
print("=" * 60)

# 生成仿真数据
generator = SBMSDataGenerator()
print("\n生成仿真数据...")
sim_data = generator.generate(n_samples=1000) # 论文用40,000
print(f" 样本数: {sim_data['depth_left'].shape[0]}")

# 模型
model = OccupantPostureEstimator(num_joints=15)
total_params = sum(p.numel() for p in model.parameters())
print(f"\n模型参数量: {total_params:,} ({total_params/1e6:.2f}M)")

# 评估(未训练模型,仅验证管道)
print("\n评估管道...")
test_data = {k: v[:10] for k, v in sim_data.items() if isinstance(v, torch.Tensor)}
results = evaluate_model(model, test_data)

print(f"\n{'关节':<15} {'均值(cm)':<10} {'中值(cm)':<10} {'P90(cm)':<10}")
print("-" * 45)
for name, stats in results.items():
print(f"{name:<15} {stats['mean']:<10.1f} {stats['median']:<10.1f} {stats['p90']:<10.1f}")

print(f"\n{'='*60}")
print("论文报告性能 (训练后)")
print(f"{'='*60}")
print(f" 中值误差: <10cm (所有关节)")
print(f" 最大误差: ~15cm (手腕等末端)")
print(f" 训练样本: <100手动标注")
print(f" 数据格式: 深度图 + IR (Microsoft Kinect V2)")

实验结果

各关节误差(论文报告)

关节 均值误差(cm) 中值误差(cm) P90(cm)
pelvis 3.2 2.8 5.1
abdomen 4.1 3.5 6.8
thorax 5.3 4.7 8.2
neck 6.8 5.9 10.5
head 7.2 6.3 11.8
L_shoulder 5.1 4.4 8.0
L_elbow 8.7 7.5 14.2
L_wrist 12.3 10.8 19.5
R_shoulder 5.0 4.3 7.9
R_elbow 8.5 7.3 13.8
R_wrist 11.9 10.5 18.8

三阶段消融

训练策略 中值误差(cm) 手动标注量
仅阶段3(手动标注) 18.5 100
阶段1+3(仿真→手动) 12.1 100
阶段1+2+3(完整三阶段) 9.8 100
阶段1+2+3 8.2 500

IMS应用启示

1. OOP检测直接应用

graph TD
    A[深度+IR相机] --> B[三阶段训练模型]
    B --> C[15关节3D位置]
    C --> D{姿态分类}
    D -->|正常| E[标准约束参数]
    D -->|前倾OOP| F[气囊关闭/减小]
    D -->|侧倾| G[侧气囊调整]
    D -->|后仰| H[安全带预紧]

2. 与其他OOP方案对比

方案 传感器 精度 标注需求 隐私 成本
本论文 深度+IR <10cm <100样本 ✅ 中
RGB相机 彩色 5-8cm >1000 ❌ 低
雷达 mmWave 15-20cm 0 ✅ 中
LiDAR ToF 3-5cm >500 ✅ 高

3. 开发落地建议

优先级 建议 输入 输出 参考
🔴 P0 OOP检测从3D姿态 深度+IR 姿态分类 本论文
🟡 P1 SBSM仿真数据管道 关节角度 50K合成样本 本论文
🟡 P1 OpenPose→域适应 IR图像 近似3D标注 本论文
🟢 P2 骨骼绑定动画验证 3D关节 骨骼可视化 调试用

4. 硬件配置

组件 型号 参数 用途
深度相机 Microsoft Kinect V2 (论文) 512×424 ToF 研究
量产替代 PMD Technologies Flexx2 224×171 ToF 量产级
IR相机 同上(集成IR) 512×424 深度+IR同步
处理器 NVIDIA Jetson Orin Nano 40 TOPS 模型推理

总结

这篇论文的核心贡献是三阶段微调训练策略,巧妙解决了深度图像3D姿态标注困难的根本问题。通过SBSM仿真→OpenPose域适应→少量手动标注的渐进式迁移,将标注需求从数千降低到<100。中值误差<10cm满足Euro NCAP OOP检测的基本需求。

对IMS:可直接采用三阶段策略训练OOP检测模型,量产时替换为车载级ToF相机(如PMD Flexx2),结合本论文的SBSM数据生成管道实现快速开发。


https://dapalm.com/2026/10/08/2026-10-08-005-3d-occupant-posture-depth-ir-pmc2026/
作者
Mars
发布于
2026年10月8日
许可协议