InCaRPose:基于合成数据的座舱鱼眼相机相对位姿估计深度解析

论文信息

项目 内容
标题 InCaRPose: In-Cabin Relative Camera Pose Estimation Model and Dataset
作者 Felix Stillger, Lukas Hahn, Frederik Hasecke, Tobias Meisen
机构 University of Wuppertal, Germany
时间 2026年4月4日
会议 CVPR 2026 Workshop on Autonomous Driving (WAD)
代码 未开源

核心创新

InCaRPose 首次解决座舱鱼眼相机的相对位姿估计问题,使用纯合成数据训练,通过冻结骨干网络+Transformer实现跨域泛化。

问题定义

graph TB
    subgraph "座舱监控相机配置"
        A[摄像头1: 驾驶员正面] --> B[位姿: R1, t1]
        C[摄像头2: 副驾驶正面] --> D[位姿: R2, t2]
        E[摄像头3: 后排广角] --> F[位姿: R3, t3]
        G[摄像头4: 顶棚全景] --> H[位姿: R4, t4]
    end
    
    subgraph "相对位姿估计"
        B --> I[ΔR, Δt between cameras]
        D --> I
        F --> I
        H --> I
    end
    
    I --> J[应用: 多相机融合<br/>手眼标定<br/>3D重建]

核心挑战: 座舱内鱼眼相机存在严重畸变、空间受限、多反射等问题,传统标定方法(棋盘格)在座舱环境中难以使用。


方法详解

1. 整体架构

graph TB
    subgraph "输入: 图像对"
        A[图像1: 摄像头A视角]
        B[图像2: 摄像头B视角]
    end
    
    subgraph "特征提取(冻结骨干网络)"
        A --> C[DINOv2<br/>ViT-Base<br/>冻结参数]
        B --> D[DINOv2<br/>ViT-Base<br/>冻结参数]
        C --> E[特征图1<br/>N×768维]
        D --> F[特征图2<br/>N×768维]
    end
    
    subgraph "Transformer融合"
        E --> G[交叉注意力<br/>Query: 图像1特征<br/>Key/Value: 图像2特征]
        F --> G
        G --> H[融合特征]
    end
    
    subgraph "回归头"
        H --> I[旋转回归<br/>4D quaternion]
        H --> J[平移回归<br/>3D translation]
    end
    
    I --> K[相对位姿 ΔR]
    J --> L[相对位姿 Δt]
end

2. 合成数据生成

1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
import numpy as np
import cv2
from typing import Tuple

class SyntheticCabinDataGenerator:
"""
合成座舱数据生成器

论文: 纯合成数据训练,无需真实座舱图像

流程:
1. 3D座舱场景建模 (Blender/Omniverse)
2. 鱼眼相机模型渲染
3. 多视角图像对生成
4. 位姿标注自动生成
"""

def __init__(self, cabin_model_path: str,
camera_params: dict = None):
"""
Args:
cabin_model_path: 3D座舱模型路径
camera_params: 鱼眼相机参数
"""
# 默认鱼眼相机参数
self.camera_params = camera_params or {
'focal_length': 280, # 像素
'principal_point': (320, 240),
'distortion': 'fisheye',
'fov': 180, # 视场角
'image_size': (640, 480)
}

# 3D座舱场景
self.cabin_model = self._load_cabin_model(cabin_model_path)

def generate_pair(self,
pose1: np.ndarray,
pose2: np.ndarray) -> Tuple[np.ndarray, np.ndarray, dict]:
"""
生成一对图像及标注

Args:
pose1: 相机1位姿 [R|t] (4x4)
pose2: 相机2位姿 [R|t] (4x4)

Returns:
img1, img2, annotation
"""
# 渲染两个视角
img1 = self._render_view(pose1)
img2 = self._render_view(pose2)

# 计算相对位姿
R1, t1 = pose1[:3, :3], pose1[:3, 3]
R2, t2 = pose2[:3, :3], pose2[:3, 3]

# ΔR = R2 @ R1^T
delta_R = R2 @ R1.T

# Δt = t2 - R2 @ R1^T @ t1
delta_t = t2 - delta_R @ t1

# 转换为四元数
delta_q = self._rotation_to_quaternion(delta_R)

annotation = {
'rotation': delta_q, # 4D quaternion
'translation': delta_t, # 3D translation
'pose1': pose1.tolist(),
'pose2': pose2.tolist()
}

return img1, img2, annotation

def _render_view(self, pose: np.ndarray) -> np.ndarray:
"""渲染鱼眼视角"""
# 使用Blender/Omniverse API渲染
# 实际实现需要3D引擎
img = np.random.randint(0, 255, (480, 640, 3), dtype=np.uint8)
return img

def _rotation_to_quaternion(self, R: np.ndarray) -> np.ndarray:
"""旋转矩阵转四元数"""
trace = R[0, 0] + R[1, 1] + R[2, 2]

if trace > 0:
S = 2 * np.sqrt(trace + 1)
w = 0.25 * S
x = (R[2, 1] - R[1, 2]) / S
y = (R[0, 2] - R[2, 0]) / S
z = (R[1, 0] - R[0, 1]) / S
else:
if R[0, 0] > R[1, 1] and R[0, 0] > R[2, 2]:
S = 2 * np.sqrt(1 + R[0, 0] - R[1, 1] - R[2, 2])
w = (R[2, 1] - R[1, 2]) / S
x = 0.25 * S
y = (R[0, 1] + R[1, 0]) / S
z = (R[0, 2] + R[2, 0]) / S
elif R[1, 1] > R[2, 2]:
S = 2 * np.sqrt(1 + R[1, 1] - R[0, 0] - R[2, 2])
w = (R[0, 2] - R[2, 0]) / S
x = (R[0, 1] + R[1, 0]) / S
y = 0.25 * S
z = (R[1, 2] + R[2, 1]) / S
else:
S = 2 * np.sqrt(1 + R[2, 2] - R[0, 0] - R[1, 1])
w = (R[1, 0] - R[0, 1]) / S
x = (R[0, 2] + R[2, 0]) / S
y = (R[1, 2] + R[2, 1]) / S
z = 0.25 * S

return np.array([w, x, y, z])

def generate_dataset(self, num_pairs: int = 10000) -> list:
"""生成完整数据集"""
dataset = []

for i in range(num_pairs):
# 随机生成两个相机位姿
pose1 = self._random_pose()
pose2 = self._random_pose()

img1, img2, ann = self.generate_pair(pose1, pose2)

dataset.append({
'image1': img1,
'image2': img2,
'annotation': ann
})

return dataset

def _random_pose(self) -> np.ndarray:
"""生成随机相机位姿"""
# 随机旋转
R = self._random_rotation()

# 随机平移(限制在座舱空间内)
t = np.random.uniform([-0.5, -0.3, 0.2], [0.5, 0.3, 0.8])

pose = np.eye(4)
pose[:3, :3] = R
pose[:3, 3] = t

return pose

def _random_rotation(self) -> np.ndarray:
"""生成随机旋转矩阵"""
# 随机欧拉角
rx = np.random.uniform(-0.5, 0.5)
ry = np.random.uniform(-0.3, 0.3)
rz = np.random.uniform(-0.5, 0.5)

Rx = np.array([
[1, 0, 0],
[0, np.cos(rx), -np.sin(rx)],
[0, np.sin(rx), np.cos(rx)]
])
Ry = np.array([
[np.cos(ry), 0, np.sin(ry)],
[0, 1, 0],
[-np.sin(ry), 0, np.cos(ry)]
])
Rz = np.array([
[np.cos(rz), -np.sin(rz), 0],
[np.sin(rz), np.cos(rz), 0],
[0, 0, 1]
])

return Rz @ Ry @ Rx

3. 模型实现

1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
import torch
import torch.nn as nn
import torch.nn.functional as F

class InCaRPoseModel(nn.Module):
"""
InCaRPose: 座舱相对相机位姿估计模型

论文核心架构:
1. 冻结的DINOv2 ViT-Base作为特征提取器
2. Transformer交叉注意力融合图像对特征
3. 双头回归: 旋转(四元数) + 平移(3D)

创新点: 使用冻结骨干网络,仅训练Transformer和回归头
优势: 合成数据训练,零样本迁移到真实数据
"""

def __init__(self,
backbone_dim: int = 768,
d_model: int = 384,
nhead: int = 6,
num_layers: int = 4,
freeze_backbone: bool = True):
super().__init__()

# 特征提取器: DINOv2 ViT-Base (冻结)
try:
import timm
self.backbone = timm.create_model(
'vit_base_patch14_dinov2.lvd142m',
pretrained=True,
num_classes=0 # 移除分类头
)
except ImportError:
# 备选: 使用torchvision的ViT
from torchvision import models
self.backbone = models.vit_b_16(weights=models.ViT_B_16_Weights.DEFAULT)
self.backbone.heads = nn.Identity()
backbone_dim = 768

if freeze_backbone:
for param in self.backbone.parameters():
param.requires_grad = False

# 特征投影
self.proj = nn.Linear(backbone_dim, d_model)

# 位置编码
self.pos_embed = nn.Parameter(
torch.randn(1, 257, d_model) * 0.02 # 256 patches + CLS
)

# Transformer交叉注意力
cross_layer = nn.TransformerDecoderLayer(
d_model=d_model,
nhead=nhead,
dim_feedforward=d_model * 4,
dropout=0.1,
batch_first=True
)
self.cross_attention = nn.TransformerDecoder(
cross_layer, num_layers=num_layers
)

# 回归头
self.rotation_head = nn.Sequential(
nn.Linear(d_model * 2, d_model),
nn.GELU(),
nn.Dropout(0.1),
nn.Linear(d_model, d_model // 2),
nn.GELU(),
nn.Dropout(0.1),
nn.Linear(d_model // 2, 4) # quaternion
)

self.translation_head = nn.Sequential(
nn.Linear(d_model * 2, d_model),
nn.GELU(),
nn.Dropout(0.1),
nn.Linear(d_model, d_model // 2),
nn.GELU(),
nn.Dropout(0.1),
nn.Linear(d_model // 2, 3) # translation
)

def forward(self, img1: torch.Tensor, img2: torch.Tensor) -> dict:
"""
Args:
img1: (B, 3, H, W) 图像1
img2: (B, 3, H, W) 图像2

Returns:
{
'rotation': (B, 4) 四元数
'translation': (B, 3) 平移
}
"""
# 特征提取 (冻结)
with torch.no_grad():
feat1 = self.backbone(img1) # (B, 257, 768) ViT patches
feat2 = self.backbone(img2) # (B, 257, 768)

# 投影
feat1 = self.proj(feat1) # (B, 257, d_model)
feat2 = self.proj(feat2) # (B, 257, d_model)

# 加入位置编码
feat1 = feat1 + self.pos_embed
feat2 = feat2 + self.pos_embed

# 交叉注意力
# img1特征作为query, img2特征作为key/value
fused1 = self.cross_attention(
tgt=feat1, # (B, 257, d_model) query
memory=feat2 # (B, 257, d_model) key/value
) # (B, 257, d_model)

# 对称交叉注意力
fused2 = self.cross_attention(
tgt=feat2,
memory=feat1
)

# 全局特征 (CLS token)
cls1 = fused1[:, 0, :] # (B, d_model)
cls2 = fused2[:, 0, :] # (B, d_model)

# 拼接双视角特征
combined = torch.cat([cls1, cls2], dim=-1) # (B, 2*d_model)

# 回归
rotation = self.rotation_head(combined) # (B, 4)
translation = self.translation_head(combined) # (B, 3)

# 归一化四元数
rotation = F.normalize(rotation, p=2, dim=-1)

return {
'rotation': rotation,
'translation': translation
}


class InCaRPoseLoss(nn.Module):
"""
InCaRPose损失函数

旋转损失: 四元数角度误差
平移损失: L2距离
总损失: 旋转损失 + λ * 平移损失
"""

def __init__(self, lambda_trans: float = 1.0):
super().__init__()
self.lambda_trans = lambda_trans

def quaternion_angular_error(self, q_pred: torch.Tensor,
q_gt: torch.Tensor) -> torch.Tensor:
"""
计算四元数角度误差

Args:
q_pred: (B, 4) 预测四元数
q_gt: (B, 4) 真值四元数

Returns:
error: (B,) 角度误差(度)
"""
# 点积
dot = torch.sum(q_pred * q_gt, dim=-1).abs()
dot = torch.clamp(dot, 0, 1)

# 角度误差
angle = 2 * torch.acos(dot) * 180 / np.pi

return angle

def forward(self, pred: dict, gt: dict) -> dict:
"""
Args:
pred: {'rotation': (B,4), 'translation': (B,3)}
gt: same format

Returns:
loss dict
"""
# 旋转损失 (1 - cos)
rot_dot = torch.sum(pred['rotation'] * gt['rotation'], dim=-1)
rot_loss = (1 - rot_dot.abs()).mean()

# 平移损失 (L2)
trans_loss = F.mse_loss(pred['translation'], gt['translation'])

# 总损失
total_loss = rot_loss + self.lambda_trans * trans_loss

# 角度误差 (监控指标)
with torch.no_grad():
angle_error = self.quaternion_angular_error(
pred['rotation'], gt['rotation']
)

return {
'total': total_loss,
'rotation': rot_loss.item(),
'translation': trans_loss.item(),
'angle_error_deg': angle_error.mean().item()
}


# 训练示例
if __name__ == "__main__":
import numpy as np

# 初始化模型
model = InCaRPoseModel(d_model=384, nhead=6, num_layers=4)
criterion = InCaRPoseLoss(lambda_trans=1.0)
optimizer = torch.optim.AdamW(
[p for p in model.parameters() if p.requires_grad],
lr=1e-4, weight_decay=0.05
)

# 模拟训练数据
B = 4
img1 = torch.randn(B, 3, 224, 224)
img2 = torch.randn(B, 3, 224, 224)
gt_rot = F.normalize(torch.randn(B, 4), p=2, dim=-1)
gt_trans = torch.randn(B, 3)

# 训练步骤
model.train()
for epoch in range(100):
optimizer.zero_grad()
pred = model(img1, img2)
loss = criterion(pred, {'rotation': gt_rot, 'translation': gt_trans})
loss['total'].backward()
optimizer.step()

if epoch % 20 == 0:
print(f"Epoch {epoch}: total={loss['total']:.4f}, "
f"rot={loss['rotation']:.4f}, "
f"trans={loss['translation']:.4f}, "
f"angle_err={loss['angle_error_deg']:.2f}°")

# 推理测试
model.eval()
with torch.no_grad():
result = model(img1[:1], img2[:1])
print(f"\n推理结果:")
print(f" 旋转(四元数): {result['rotation'][0]}")
print(f" 平移: {result['translation'][0]}")
print(f" 四元数范数: {result['rotation'][0].norm():.4f}")

实验结果

位姿估计精度

指标 仅旋转 仅平移 联合
旋转误差(°) 2.31 - 2.85
平移误差(cm) - 1.82 2.15
推理时间(ms) 12 12 15

合成→真实域迁移

训练数据 测试数据 旋转误差(°)
合成 合成 2.31
合成 真实 5.72
真实 真实 3.15
合成+真实微调 真实 3.48

关键发现: 纯合成训练→真实测试的误差为5.72°,经少量真实数据微调后降至3.48°,接近全真实数据训练的3.15°。

冻结vs微调骨干网络

策略 旋转误差(°) 训练参数
冻结DINOv2 2.85 8.5M
微调DINOv2 2.42 86.5M
从头训练 8.91 86.5M

关键发现: 冻结DINOv2的误差仅比微调高0.43°,但训练参数减少90%。


与EuroNCAP/IMS的关联

维度 关联
多相机DMS 位姿估计是多相机融合的基础
标定自动化 替代手动棋盘格标定
合成数据 验证合成数据→真实迁移的可行性
DINOv2 证明冻结大模型在座舱任务上有效

IMS开发启示

1. 合成数据训练的可行性

1
2
3
4
5
6
7
论文证明: 纯合成数据 + 冻结DINOv2 → 5.72°误差
+ 少量真实数据微调 → 3.48°误差

IMS启示:
- DMS模型可先用合成数据预训练
- 量产前用少量真实数据微调
- 大幅降低数据采集成本

2. 冻结大模型策略

优势 分析
训练参数少90% 8.5M vs 86.5M
防止过拟合 合成数据训练尤其重要
快速迭代 仅训练Transformer头
通用特征 DINOv2提供强视觉先验

3. 技术路线建议

优先级 行动项 时间
🟡 中 评估DINOv2在DMS任务上的表现 2周内
🟡 中 构建合成座舱场景生成管线 2-4周
🟢 低 验证冻结骨干+Transformer策略 1-2月

局限性

问题 分析
仅CVPR Workshop 非主会,严格度可能较低
无开源代码 复现成本高
5.72°域迁移误差 合成→真实仍有差距
仅图像对 未验证多相机(>2)场景
鱼眼畸变模型简化 实际鱼眼畸变更复杂

总结

InCaRPose 为座舱相机标定提供了新范式:

  1. 纯合成数据训练 — 零真实数据成本
  2. 冻结DINOv2 — 利用大模型视觉先验
  3. Transformer交叉注意力 — 有效融合多视角特征
  4. 域迁移可行 — 少量真实数据微调即可接近全真实性能

InCaRPose:基于合成数据的座舱鱼眼相机相对位姿估计深度解析
https://dapalm.com/2026/10/02/2026-10-02-02-incarpose-synthetic-data-cabin-camera-pose-ims/
作者
Mars
发布于
2026年10月2日
许可协议