benchmarks.wiki / Public workspace
LongBench v2 / 66fa50acbb02136c067c6827 / Which realistic factor in collaborative perception does this algorithm model…
Problem
Answer published by the source. Consult the official source to check your work against its answer.
question
Which realistic factor in collaborative perception does this algorithm model mainly address?
context · full text (327,523 characters)
"""Specifies the current version number of v2xvit."""
__version__ = "0.1.0"
import torch
import torch.nn as nn
from v2xvit.models.sub_modules.pillar_vfe import PillarVFE
from v2xvit.models.sub_modules.point_pillar_scatter import PointPillarScatter
from v2xvit.models.sub_modules.base_bev_backbone import BaseBEVBackbone
from v2xvit.models.sub_modules.fuse_utils import regroup
from v2xvit.models.sub_modules.downsample_conv import DownsampleConv
from v2xvit.models.sub_modules.naive_compress import NaiveCompressor
from v2xvit.models.sub_modules.v2xvit_basic import V2XTransformer
class PointPillarTransformer(nn.Module):
def __init__(self, args):
super(PointPillarTransformer, self).__init__()
self.max_cav = args['max_cav']
# PIllar VFE
self.pillar_vfe = PillarVFE(args['pillar_vfe'],
num_point_features=4,
voxel_size=args['voxel_size'],
point_cloud_range=args['lidar_range'])
self.scatter = PointPillarScatter(args['point_pillar_scatter'])
self.backbone = BaseBEVBackbone(args['base_bev_backbone'], 64)
# used to downsample the feature map for efficient computation
self.shrink_flag = False
if 'shrink_header' in args:
self.shrink_flag = True
self.shrink_conv = DownsampleConv(args['shrink_header'])
self.compression = False
if args['compression'] > 0:
self.compression = True
self.naive_compressor = NaiveCompressor(256, args['compression'])
self.fusion_net = V2XTransformer(args['transformer'])
self.cls_head = nn.Conv2d(128 * 2, args['anchor_number'],
kernel_size=1)
self.reg_head = nn.Conv2d(128 * 2, 7 * args['anchor_number'],
kernel_size=1)
if args['backbone_fix']:
self.backbone_fix()
def backbone_fix(self):
"""
Fix the parameters of backbone during finetune on timedelay。
"""
for p in self.pillar_vfe.parameters():
p.requires_grad = False
for p in self.scatter.parameters():
p.requires_grad = False
for p in self.backbone.parameters():
p.requires_grad = False
if self.compression:
for p in self.naive_compressor.parameters():
p.requires_grad = False
if self.shrink_flag:
for p in self.shrink_conv.parameters():
p.requires_grad = False
for p in self.cls_head.parameters():
p.requires_grad = False
for p in self.reg_head.parameters():
p.requires_grad = False
def forward(self, data_dict):
voxel_features = data_dict['processed_lidar']['voxel_features']
voxel_coords = data_dict['processed_lidar']['voxel_coords']
voxel_num_points = data_dict['processed_lidar']['voxel_num_points']
record_len = data_dict['record_len']
spatial_correction_matrix = data_dict['spatial_correction_matrix']
# B, max_cav, 3(dt dv infra), 1, 1
prior_encoding =\
data_dict['prior_encoding'].unsqueeze(-1).unsqueeze(-1)
batch_dict = {'voxel_features': voxel_features,
'voxel_coords': voxel_coords,
'voxel_num_points': voxel_num_points,
'record_len': record_len}
# n, 4 -> n, c
batch_dict = self.pillar_vfe(batch_dict)
# n, c -> N, C, H, W
batch_dict = self.scatter(batch_dict)
batch_dict = self.backbone(batch_dict)
spatial_features_2d = batch_dict['spatial_features_2d']
# downsample feature to reduce memory
if self.shrink_flag:
spatial_features_2d = self.shrink_conv(spatial_features_2d)
# compressor
if self.compression:
spatial_features_2d = self.naive_compressor(spatial_features_2d)
# N, C, H, W -> B, L, C, H, W
regroup_feature, mask = regroup(spatial_features_2d,
record_len,
self.max_cav)
# prior encoding added
prior_encoding = prior_encoding.repeat(1, 1, 1,
regroup_feature.shape[3],
regroup_feature.shape[4])
regroup_feature = torch.cat([regroup_feature, prior_encoding], dim=2)
# b l c h w -> b l h w c
regroup_feature = regroup_feature.permute(0, 1, 3, 4, 2)
# transformer fusion
fused_feature = self.fusion_net(regroup_feature, mask, spatial_correction_matrix)
# b h w c -> b c h w
fused_feature = fused_feature.permute(0, 3, 1, 2)
psm = self.cls_head(fused_feature)
rm = self.reg_head(fused_feature)
output_dict = {'psm': psm,
'rm': rm}
return output_dict
import torch.nn as nn
from v2xvit.models.sub_modules.pillar_vfe import PillarVFE
from v2xvit.models.sub_modules.point_pillar_scatter import PointPillarScatter
from v2xvit.models.sub_modules.base_bev_backbone import BaseBEVBackbone
from v2xvit.models.sub_modules.downsample_conv import DownsampleConv
from v2xvit.models.sub_modules.naive_compress import NaiveCompressor
from v2xvit.models.sub_modules.f_cooper_fuse import SpatialFusion
class PointPillarFCooper(nn.Module):
def __init__(self, args):
super(PointPillarFCooper, self).__init__()
self.max_cav = args['max_cav']
# PIllar VFE
self.pillar_vfe = PillarVFE(args['pillar_vfe'],
num_point_features=4,
voxel_size=args['voxel_size'],
point_cloud_range=args['lidar_range'])
self.scatter = PointPillarScatter(args['point_pillar_scatter'])
self.backbone = BaseBEVBackbone(args['base_bev_backbone'], 64)
# used to downsample the feature map for efficient computation
self.shrink_flag = False
if 'shrink_header' in args:
self.shrink_flag = True
self.shrink_conv = DownsampleConv(args['shrink_header'])
self.compression = False
if args['compression'] > 0:
self.compression = True
self.naive_compressor = NaiveCompressor(256, args['compression'])
self.fusion_net = SpatialFusion()
self.cls_head = nn.Conv2d(128 * 2, args['anchor_number'],
kernel_size=1)
self.reg_head = nn.Conv2d(128 * 2, 7 * args['anchor_number'],
kernel_size=1)
if args['backbone_fix']:
self.backbone_fix()
def backbone_fix(self):
"""
Fix the parameters of backbone during finetune on timedelay。
"""
for p in self.pillar_vfe.parameters():
p.requires_grad = False
for p in self.scatter.parameters():
p.requires_grad = False
for p in self.backbone.parameters():
p.requires_grad = False
if self.compression:
for p in self.naive_compressor.parameters():
p.requires_grad = False
if self.shrink_flag:
for p in self.shrink_conv.parameters():
p.requires_grad = False
for p in self.cls_head.parameters():
p.requires_grad = False
for p in self.reg_head.parameters():
p.requires_grad = False
def forward(self, data_dict):
voxel_features = data_dict['processed_lidar']['voxel_features']
voxel_coords = data_dict['processed_lidar']['voxel_coords']
voxel_num_points = data_dict['processed_lidar']['voxel_num_points']
record_len = data_dict['record_len']
spatial_correction_matrix = data_dict['spatial_correction_matrix']
batch_dict = {'voxel_features': voxel_features,
'voxel_coords': voxel_coords,
'voxel_num_points': voxel_num_points,
'record_len': record_len}
# n, 4 -> n, c
batch_dict = self.pillar_vfe(batch_dict)
# n, c -> N, C, H, W
batch_dict = self.scatter(batch_dict)
batch_dict = self.backbone(batch_dict)
spatial_features_2d = batch_dict['spatial_features_2d']
# downsample feature to reduce memory
if self.shrink_flag:
spatial_features_2d = self.shrink_conv(spatial_features_2d)
# compressor
if self.compression:
spatial_features_2d = self.naive_compressor(spatial_features_2d)
fused_feature = self.fusion_net(spatial_features_2d, record_len)
psm = self.cls_head(fused_feature)
rm = self.reg_head(fused_feature)
output_dict = {'psm': psm,
'rm': rm}
return output_dict
import torch.nn as nn
from v2xvit.models.sub_modules.pillar_vfe import PillarVFE
from v2xvit.models.sub_modules.point_pillar_scatter import PointPillarScatter
from v2xvit.models.sub_modules.base_bev_backbone import BaseBEVBackbone
from v2xvit.models.sub_modules.downsample_conv import DownsampleConv
from v2xvit.models.sub_modules.naive_compress import NaiveCompressor
from v2xvit.models.sub_modules.self_attn import AttFusion
class PointPillarOPV2V(nn.Module):
def __init__(self, args):
super(PointPillarOPV2V, self).__init__()
self.max_cav = args['max_cav']
# PIllar VFE
self.pillar_vfe = PillarVFE(args['pillar_vfe'],
num_point_features=4,
voxel_size=args['voxel_size'],
point_cloud_range=args['lidar_range'])
self.scatter = PointPillarScatter(args['point_pillar_scatter'])
self.backbone = BaseBEVBackbone(args['base_bev_backbone'], 64)
# used to downsample the feature map for efficient computation
self.shrink_flag = False
if 'shrink_header' in args:
self.shrink_flag = True
self.shrink_conv = DownsampleConv(args['shrink_header'])
self.compression = False
if args['compression'] > 0:
self.compression = True
self.naive_compressor = NaiveCompressor(256, args['compression'])
self.fusion_net = AttFusion(256)
self.cls_head = nn.Conv2d(128 * 2, args['anchor_number'],
kernel_size=1)
self.reg_head = nn.Conv2d(128 * 2, 7 * args['anchor_number'],
kernel_size=1)
if args['backbone_fix']:
self.backbone_fix()
def backbone_fix(self):
"""
Fix the parameters of backbone during finetune on timedelay。
"""
for p in self.pillar_vfe.parameters():
p.requires_grad = False
for p in self.scatter.parameters():
p.requires_grad = False
for p in self.backbone.parameters():
p.requires_grad = False
if self.compression:
for p in self.naive_compressor.parameters():
p.requires_grad = False
if self.shrink_flag:
for p in self.shrink_conv.parameters():
p.requires_grad = False
for p in self.cls_head.parameters():
p.requires_grad = False
for p in self.reg_head.parameters():
p.requires_grad = False
def forward(self, data_dict):
voxel_features = data_dict['processed_lidar']['voxel_features']
voxel_coords = data_dict['processed_lidar']['voxel_coords']
voxel_num_points = data_dict['processed_lidar']['voxel_num_points']
record_len = data_dict['record_len']
spatial_correction_matrix = data_dict['spatial_correction_matrix']
# B, max_cav, 3(dt dv infra), 1, 1
prior_encoding =\
data_dict['prior_encoding'].unsqueeze(-1).unsqueeze(-1)
batch_dict = {'voxel_features': voxel_features,
'voxel_coords': voxel_coords,
'voxel_num_points': voxel_num_points,
'record_len': record_len}
# n, 4 -> n, c
batch_dict = self.pillar_vfe(batch_dict)
# n, c -> N, C, H, W
batch_dict = self.scatter(batch_dict)
batch_dict = self.backbone(batch_dict)
spatial_features_2d = batch_dict['spatial_features_2d']
# downsample feature to reduce memory
if self.shrink_flag:
spatial_features_2d = self.shrink_conv(spatial_features_2d)
# compressor
if self.compression:
spatial_features_2d = self.naive_compressor(spatial_features_2d)
fused_feature = self.fusion_net(spatial_features_2d, record_len)
psm = self.cls_head(fused_feature)
rm = self.reg_head(fused_feature)
output_dict = {'psm': psm,
'rm': rm}
return output_dict
"""
Vanilla pointpillar for early and late fusion.
"""
import torch.nn as nn
from v2xvit.models.sub_modules.pillar_vfe import PillarVFE
from v2xvit.models.sub_modules.point_pillar_scatter import PointPillarScatter
from v2xvit.models.sub_modules.base_bev_backbone import BaseBEVBackbone
from v2xvit.models.sub_modules.downsample_conv import DownsampleConv
class PointPillar(nn.Module):
def __init__(self, args):
super(PointPillar, self).__init__()
# PIllar VFE
self.pillar_vfe = PillarVFE(args['pillar_vfe'],
num_point_features=4,
voxel_size=args['voxel_size'],
point_cloud_range=args['lidar_range'])
self.scatter = PointPillarScatter(args['point_pillar_scatter'])
self.backbone = BaseBEVBackbone(args['base_bev_backbone'], 64)
# used to downsample the feature map for efficient computation
self.shrink_flag = False
if 'shrink_header' in args:
self.shrink_flag = True
self.shrink_conv = DownsampleConv(args['shrink_header'])
self.cls_head = nn.Conv2d(args['cls_head_dim'], args['anchor_number'],
kernel_size=1)
self.reg_head = nn.Conv2d(args['cls_head_dim'],
7 * args['anchor_number'],
kernel_size=1)
def forward(self, data_dict):
voxel_features = data_dict['processed_lidar']['voxel_features']
voxel_coords = data_dict['processed_lidar']['voxel_coords']
voxel_num_points = data_dict['processed_lidar']['voxel_num_points']
batch_dict = {'voxel_features': voxel_features,
'voxel_coords': voxel_coords,
'voxel_num_points': voxel_num_points}
batch_dict = self.pillar_vfe(batch_dict)
batch_dict = self.scatter(batch_dict)
batch_dict = self.backbone(batch_dict)
spatial_features_2d = batch_dict['spatial_features_2d']
if self.shrink_flag:
spatial_features_2d = self.shrink_conv(spatial_features_2d)
psm = self.cls_head(spatial_features_2d)
rm = self.reg_head(spatial_features_2d)
output_dict = {'psm': psm,
'rm': rm}
return output_dict
import torch
import torch.nn as nn
from v2xvit.models.sub_modules.pillar_vfe import PillarVFE
from v2xvit.models.sub_modules.point_pillar_scatter import PointPillarScatter
from v2xvit.models.sub_modules.base_bev_backbone import BaseBEVBackbone
from v2xvit.models.sub_modules.downsample_conv import DownsampleConv
from v2xvit.models.sub_modules.naive_compress import NaiveCompressor
from v2xvit.models.sub_modules.v2v_fuse import V2VNetFusion
class PointPillarV2VNet(nn.Module):
def __init__(self, args):
super(PointPillarV2VNet, self).__init__()
self.max_cav = args['max_cav']
# PIllar VFE
self.pillar_vfe = PillarVFE(args['pillar_vfe'],
num_point_features=4,
voxel_size=args['voxel_size'],
point_cloud_range=args['lidar_range'])
self.scatter = PointPillarScatter(args['point_pillar_scatter'])
self.backbone = BaseBEVBackbone(args['base_bev_backbone'], 64)
# used to downsample the feature map for efficient computation
self.shrink_flag = False
if 'shrink_header' in args:
self.shrink_flag = True
self.shrink_conv = DownsampleConv(args['shrink_header'])
self.compression = False
if args['compression'] > 0:
self.compression = True
self.naive_compressor = NaiveCompressor(256, args['compression'])
self.fusion_net = V2VNetFusion(args['v2vfusion'])
self.cls_head = nn.Conv2d(128 * 2, args['anchor_number'],
kernel_size=1)
self.reg_head = nn.Conv2d(128 * 2, 7 * args['anchor_number'],
kernel_size=1)
if args['backbone_fix']:
self.backbone_fix()
def backbone_fix(self):
"""
Fix the parameters of backbone during finetune on timedelay。
"""
for p in self.pillar_vfe.parameters():
p.requires_grad = False
for p in self.scatter.parameters():
p.requires_grad = False
for p in self.backbone.parameters():
p.requires_grad = False
if self.compression:
for p in self.naive_compressor.parameters():
p.requires_grad = False
if self.shrink_flag:
for p in self.shrink_conv.parameters():
p.requires_grad = False
for p in self.cls_head.parameters():
p.requires_grad = False
for p in self.reg_head.parameters():
p.requires_grad = False
def unpad_prior_encoding(self, x, record_len):
# remove padded zeros to form tensor with shape (N, 3)
# x: (B, L, 3); record_len: (B)
B = x.shape[0]
out = []
for i in range(B):
# (valid_len, 3)
out.append(x[i, :record_len[i], :])
out = torch.cat(out, dim=0)
# (N, 3)
return out
def forward(self, data_dict):
voxel_features = data_dict['processed_lidar']['voxel_features']
voxel_coords = data_dict['processed_lidar']['voxel_coords']
voxel_num_points = data_dict['processed_lidar']['voxel_num_points']
record_len = data_dict['record_len']
spatial_correction_matrix = data_dict['spatial_correction_matrix']
pairwise_t_matrix = data_dict['pairwise_t_matrix']
prior_encoding = data_dict['prior_encoding']
prior_encoding = self.unpad_prior_encoding(prior_encoding, record_len)
batch_dict = {'voxel_features': voxel_features,
'voxel_coords': voxel_coords,
'voxel_num_points': voxel_num_points,
'record_len': record_len}
# n, 4 -> n, c
batch_dict = self.pillar_vfe(batch_dict)
# n, c -> N, C, H, W
batch_dict = self.scatter(batch_dict)
batch_dict = self.backbone(batch_dict)
spatial_features_2d = batch_dict['spatial_features_2d']
# downsample feature to reduce memory
if self.shrink_flag:
spatial_features_2d = self.shrink_conv(spatial_features_2d)
# compressor
if self.compression:
spatial_features_2d = self.naive_compressor(spatial_features_2d)
fused_feature = self.fusion_net(spatial_features_2d,
record_len,
pairwise_t_matrix,
prior_encoding)
psm = self.cls_head(fused_feature)
rm = self.reg_head(fused_feature)
output_dict = {'psm': psm,
'rm': rm}
return output_dict
import numpy as np
import torch
import torch.nn as nn
import torch.nn.functional as F
class ScaledDotProductAttention(nn.Module):
"""
Scaled Dot-Product Attention proposed in "Attention Is All You Need"
Compute the dot products of the query with all keys, divide each by sqrt(dim),
and apply a softmax function to obtain the weights on the values
Args: dim, mask
dim (int): dimention of attention
mask (torch.Tensor): tensor containing indices to be masked
Inputs: query, key, value, mask
- **query** (batch, q_len, d_model): tensor containing projection vector for decoder.
- **key** (batch, k_len, d_model): tensor containing projection vector for encoder.
- **value** (batch, v_len, d_model): tensor containing features of the encoded input sequence.
- **mask** (-): tensor containing indices to be masked
Returns: context, attn
- **context**: tensor containing the context vector from attention mechanism.
- **attn**: tensor containing the attention (alignment) from the encoder outputs.
"""
def __init__(self, dim):
super(ScaledDotProductAttention, self).__init__()
self.sqrt_dim = np.sqrt(dim)
def forward(self, query, key, value):
score = torch.bmm(query, key.transpose(1, 2)) / self.sqrt_dim
attn = F.softmax(score, -1)
context = torch.bmm(attn, value)
return context
class AttFusion(nn.Module):
def __init__(self, feature_dim):
super(AttFusion, self).__init__()
self.att = ScaledDotProductAttention(feature_dim)
def forward(self, x, record_len):
split_x = self.regroup(x, record_len)
batch_size = len(record_len)
C, W, H = split_x[0].shape[1:]
out = []
for xx in split_x:
cav_num = xx.shape[0]
xx = xx.view(cav_num, C, -1).permute(2, 0, 1)
h = self.att(xx, xx, xx)
h = h.permute(1, 2, 0).view(cav_num, C, W, H)[0, ...].unsqueeze(0)
out.append(h)
return torch.cat(out, dim=0)
def regroup(self, x, record_len):
cum_sum_len = torch.cumsum(record_len, dim=0)
split_x = torch.tensor_split(x, cum_sum_len[:-1].cpu())
return split_x
import os
import torch
from torch import nn
from torch.autograd import Variable
class ConvGRUCell(nn.Module):
def __init__(self, input_size, input_dim, hidden_dim, kernel_size, bias):
"""
Initialize the ConvLSTM cell
:param input_size: (int, int)
Height and width of input tensor as (height, width).
:param input_dim: int
Number of channels of input tensor.
:param hidden_dim: int
Number of channels of hidden state.
:param kernel_size: (int, int)
Size of the convolutional kernel.
:param bias: bool
Whether or not to add the bias.
:param dtype: torch.cuda.FloatTensor or torch.FloatTensor
Whether or not to use cuda.
"""
super(ConvGRUCell, self).__init__()
self.height, self.width = input_size
self.padding = kernel_size[0] // 2, kernel_size[1] // 2
self.hidden_dim = hidden_dim
self.bias = bias
self.conv_gates = nn.Conv2d(in_channels=input_dim + hidden_dim,
out_channels=2 * self.hidden_dim,
# for update_gate,reset_gate respectively
kernel_size=kernel_size,
padding=self.padding,
bias=self.bias)
self.conv_can = nn.Conv2d(in_channels=input_dim + hidden_dim,
out_channels=self.hidden_dim,
# for candidate neural memory
kernel_size=kernel_size,
padding=self.padding,
bias=self.bias)
def init_hidden(self, batch_size):
return (Variable(
torch.zeros(batch_size, self.hidden_dim, self.height, self.width)))
def forward(self, input_tensor, h_cur):
"""
:param self:
:param input_tensor: (b, c, h, w)
input is actually the target_model
:param h_cur: (b, c_hidden, h, w)
current hidden and cell states respectively
:return: h_next,
next hidden state
"""
combined = torch.cat([input_tensor, h_cur], dim=1)
combined_conv = self.conv_gates(combined)
gamma, beta = torch.split(combined_conv, self.hidden_dim, dim=1)
reset_gate = torch.sigmoid(gamma)
update_gate = torch.sigmoid(beta)
combined = torch.cat([input_tensor, reset_gate * h_cur], dim=1)
cc_cnm = self.conv_can(combined)
cnm = torch.tanh(cc_cnm)
h_next = (1 - update_gate) * h_cur + update_gate * cnm
return h_next
class ConvGRU(nn.Module):
def __init__(self, input_size, input_dim, hidden_dim, kernel_size,
num_layers,
batch_first=False, bias=True, return_all_layers=False):
"""
:param input_size: (int, int)
Height and width of input tensor as (height, width).
:param input_dim: int e.g. 256
Number of channels of input tensor.
:param hidden_dim: int e.g. 1024
Number of channels of hidden state.
:param kernel_size: (int, int)
Size of the convolutional kernel.
:param num_layers: int
Number of ConvLSTM layers
:param dtype: torch.cuda.FloatTensor or torch.FloatTensor
Whether or not to use cuda.
:param alexnet_path: str
pretrained alexnet parameters
:param batch_first: bool
if the first position of array is batch or not
:param bias: bool
Whether or not to add the bias.
:param return_all_layers: bool
if return hidden and cell states for all layers
"""
super(ConvGRU, self).__init__()
# Make sure that both `kernel_size` and
# `hidden_dim` are lists having len == num_layers
kernel_size = self._extend_for_multilayer(kernel_size, num_layers)
hidden_dim = self._extend_for_multilayer(hidden_dim, num_layers)
if not len(kernel_size) == len(hidden_dim) == num_layers:
raise ValueError('Inconsistent list length.')
self.height, self.width = input_size
self.input_dim = input_dim
self.hidden_dim = hidden_dim
self.kernel_size = kernel_size
self.num_layers = num_layers
self.batch_first = batch_first
self.bias = bias
self.return_all_layers = return_all_layers
cell_list = []
for i in range(0, self.num_layers):
cur_input_dim = input_dim if i == 0 else hidden_dim[i - 1]
cell_list.append(ConvGRUCell(input_size=(self.height, self.width),
input_dim=cur_input_dim,
hidden_dim=self.hidden_dim[i],
kernel_size=self.kernel_size[i],
bias=self.bias))
# convert python list to pytorch module
self.cell_list = nn.ModuleList(cell_list)
def forward(self, input_tensor, hidden_state=None):
"""
:param input_tensor: (b, t, c, h, w) or (t,b,c,h,w)
depends on if batch first or not extracted features from alexnet
:param hidden_state:
:return: layer_output_list, last_state_list
"""
if not self.batch_first:
# (t, b, c, h, w) -> (b, t, c, h, w)
input_tensor = input_tensor.permute(1, 0, 2, 3, 4)
# Implement stateful ConvLSTM
if hidden_state is not None:
raise NotImplementedError()
else:
hidden_state = self._init_hidden(batch_size=input_tensor.size(0),
device=input_tensor.device,
dtype=input_tensor.dtype)
layer_output_list = []
last_state_list = []
seq_len = input_tensor.size(1)
cur_layer_input = input_tensor
for layer_idx in range(self.num_layers):
h = hidden_state[layer_idx]
output_inner = []
for t in range(seq_len):
# input current hidden and cell state
# then compute the next hidden
# and cell state through ConvLSTMCell forward function
h = self.cell_list[layer_idx](
input_tensor=cur_layer_input[:, t, :, :, :], # (b,t,c,h,w)
h_cur=h)
output_inner.append(h)
layer_output = torch.stack(output_inner, dim=1)
cur_layer_input = layer_output
layer_output_list.append(layer_output)
last_state_list.append([h])
if not self.return_all_layers:
layer_output_list = layer_output_list[-1:]
last_state_list = last_state_list[-1:]
return layer_output_list, last_state_list
def _init_hidden(self, batch_size, device=None, dtype=None):
init_states = []
for i in range(self.num_layers):
init_states.append(
self.cell_list[i].init_hidden(batch_size).to(device).to(dtype))
return init_states
@staticmethod
def _check_kernel_size_consistency(kernel_size):
if not (isinstance(kernel_size, tuple) or
(isinstance(kernel_size, list) and all(
[isinstance(elem, tuple) for elem in kernel_size]))):
raise ValueError('`kernel_size` must be tuple or list of tuples')
@staticmethod
def _extend_for_multilayer(param, num_layers):
if not isinstance(param, list):
param = [param] * num_layers
return param
if __name__ == '__main__':
# set CUDA device
os.environ["CUDA_VISIBLE_DEVICES"] = "3"
# detect if CUDA is available or not
use_gpu = torch.cuda.is_available()
# if use_gpu:
# dtype = torch.cuda.FloatTensor # computation in GPU
# else:
# dtype = torch.FloatTensor
height = width = 6
channels = 256
hidden_dim = [32, 64]
kernel_size = (3, 3) # kernel size for two stacked hidden layer
num_layers = 2 # number of stacked hidden layer
model = ConvGRU(input_size=(height, width),
input_dim=channels,
hidden_dim=hidden_dim,
kernel_size=kernel_size,
num_layers=num_layers,
batch_first=True,
bias=True,
return_all_layers=False)
batch_size = 1
time_steps = 1
input_tensor = torch.rand(batch_size, time_steps, channels, height,
width) # (b,t,c,h,w)
layer_output_list, last_state_list = model(input_tensor)
"""
Pillar VFE, credits to OpenPCDet.
"""
import torch
import torch.nn as nn
import torch.nn.functional as F
class PFNLayer(nn.Module):
def __init__(self,
in_channels,
out_channels,
use_norm=True,
last_layer=False):
super().__init__()
self.last_vfe = last_layer
self.use_norm = use_norm
if not self.last_vfe:
out_channels = out_channels // 2
if self.use_norm:
self.linear = nn.Linear(in_channels, out_channels, bias=False)
self.norm = nn.BatchNorm1d(out_channels, eps=1e-3, momentum=0.01)
else:
self.linear = nn.Linear(in_channels, out_channels, bias=True)
self.part = 50000
def forward(self, inputs):
if inputs.shape[0] > self.part:
# nn.Linear performs randomly when batch size is too large
num_parts = inputs.shape[0] // self.part
part_linear_out = [self.linear(
inputs[num_part * self.part:(num_part + 1) * self.part])
for num_part in range(num_parts + 1)]
x = torch.cat(part_linear_out, dim=0)
else:
x = self.linear(inputs)
torch.backends.cudnn.enabled = False
x = self.norm(x.permute(0, 2, 1)).permute(0, 2,
1) if self.use_norm else x
torch.backends.cudnn.enabled = True
x = F.relu(x)
x_max = torch.max(x, dim=1, keepdim=True)[0]
if self.last_vfe:
return x_max
else:
x_repeat = x_max.repeat(1, inputs.shape[1], 1)
x_concatenated = torch.cat([x, x_repeat], dim=2)
return x_concatenated
class PillarVFE(nn.Module):
def __init__(self, model_cfg, num_point_features, voxel_size,
point_cloud_range):
super().__init__()
self.model_cfg = model_cfg
self.use_norm = self.model_cfg['use_norm']
self.with_distance = self.model_cfg['with_distance']
self.use_absolute_xyz = self.model_cfg['use_absolute_xyz']
num_point_features += 6 if self.use_absolute_xyz else 3
if self.with_distance:
num_point_features += 1
self.num_filters = self.model_cfg['num_filters']
assert len(self.num_filters) > 0
num_filters = [num_point_features] + list(self.num_filters)
pfn_layers = []
for i in range(len(num_filters) - 1):
in_filters = num_filters[i]
out_filters = num_filters[i + 1]
pfn_layers.append(
PFNLayer(in_filters, out_filters, self.use_norm,
last_layer=(i >= len(num_filters) - 2))
)
self.pfn_layers = nn.ModuleList(pfn_layers)
self.voxel_x = voxel_size[0]
self.voxel_y = voxel_size[1]
self.voxel_z = voxel_size[2]
self.x_offset = self.voxel_x / 2 + point_cloud_range[0]
self.y_offset = self.voxel_y / 2 + point_cloud_range[1]
self.z_offset = self.voxel_z / 2 + point_cloud_range[2]
def get_output_feature_dim(self):
return self.num_filters[-1]
@staticmethod
def get_paddings_indicator(actual_num, max_num, axis=0):
actual_num = torch.unsqueeze(actual_num, axis + 1)
max_num_shape = [1] * len(actual_num.shape)
max_num_shape[axis + 1] = -1
max_num = torch.arange(max_num,
dtype=torch.int,
device=actual_num.device).view(max_num_shape)
paddings_indicator = actual_num.int() > max_num
return paddings_indicator
def forward(self, batch_dict):
voxel_features, voxel_num_points, coords = \
batch_dict['voxel_features'], batch_dict['voxel_num_points'], \
batch_dict['voxel_coords']
points_mean = \
voxel_features[:, :, :3].sum(dim=1, keepdim=True) / \
voxel_num_points.type_as(voxel_features).view(-1, 1, 1)
f_cluster = voxel_features[:, :, :3] - points_mean
f_center = torch.zeros_like(voxel_features[:, :, :3])
f_center[:, :, 0] = voxel_features[:, :, 0] - (
coords[:, 3].to(voxel_features.dtype).unsqueeze(
1) * self.voxel_x + self.x_offset)
f_center[:, :, 1] = voxel_features[:, :, 1] - (
coords[:, 2].to(voxel_features.dtype).unsqueeze(
1) * self.voxel_y + self.y_offset)
f_center[:, :, 2] = voxel_features[:, :, 2] - (
coords[:, 1].to(voxel_features.dtype).unsqueeze(
1) * self.voxel_z + self.z_offset)
if self.use_absolute_xyz:
features = [voxel_features, f_cluster, f_center]
else:
features = [voxel_features[..., 3:], f_cluster, f_center]
if self.with_distance:
points_dist = torch.norm(voxel_features[:, :, :3], 2, 2,
keepdim=True)
features.append(points_dist)
features = torch.cat(features, dim=-1)
voxel_count = features.shape[1]
mask = self.get_paddings_indicator(voxel_num_points, voxel_count,
axis=0)
mask = torch.unsqueeze(mask, -1).type_as(voxel_features)
features *= mask
for pfn in self.pfn_layers:
features = pfn(features)
features = features.squeeze()
batch_dict['pillar_features'] = features
return batch_dict
"""
torch_transformation_utils.py
"""
import os
import torch
import torch.nn.functional as F
import numpy as np
import matplotlib.pyplot as plt
def get_roi_and_cav_mask(shape, cav_mask, spatial_correction_matrix,
discrete_ratio, downsample_rate):
"""
Get mask for the combination of cav_mask and rorated ROI mask.
Parameters
----------
shape : tuple
Shape of (B, L, H, W, C).
cav_mask : torch.Tensor
Shape of (B, L).
spatial_correction_matrix : torch.Tensor
Shape of (B, L, 4, 4)
discrete_ratio : float
Discrete ratio.
downsample_rate : float
Downsample rate.
Returns
-------
com_mask : torch.Tensor
Combined mask with shape (B, H, W, L, 1).
"""
B, L, H, W, C = shape
C = 1
# (B,L,4,4)
dist_correction_matrix = get_discretized_transformation_matrix(
spatial_correction_matrix, discrete_ratio,
downsample_rate)
# (B*L,2,3)
T = get_transformation_matrix(
dist_correction_matrix.reshape(-1, 2, 3), (H, W))
# (B,L,1,H,W)
roi_mask = get_rotated_roi((B, L, C, H, W), T)
# (B,L,1,H,W)
com_mask = combine_roi_and_cav_mask(roi_mask, cav_mask)
# (B,H,W,1,L)
com_mask = com_mask.permute(0, 3, 4, 2, 1)
return com_mask
def combine_roi_and_cav_mask(roi_mask, cav_mask):
"""
Combine ROI mask and CAV mask
Parameters
----------
roi_mask : torch.Tensor
Mask for ROI region after considering the spatial transformation/correction.
cav_mask : torch.Tensor
Mask for CAV to remove padded 0.
Returns
-------
com_mask : torch.Tensor
Combined mask.
"""
# (B, L, 1, 1, 1)
cav_mask = cav_mask.unsqueeze(2).unsqueeze(3).unsqueeze(4)
# (B, L, C, H, W)
cav_mask = cav_mask.expand(roi_mask.shape)
# (B, L, C, H, W)
com_mask = roi_mask * cav_mask
return com_mask
def get_rotated_roi(shape, correction_matrix):
"""
Get rorated ROI mask.
Parameters
----------
shape : tuple
Shape of (B,L,C,H,W).
correction_matrix : torch.Tensor
Correction matrix with shape (N,2,3).
Returns
-------
roi_mask : torch.Tensor
Roated ROI mask with shape (N,2,3).
"""
B, L, C, H, W = shape
# To reduce the computation, we only need to calculate the mask for the first channel.
# (B,L,1,H,W)
x = torch.ones((B, L, 1, H, W)).to(correction_matrix.dtype).to(
correction_matrix.device)
# (B*L,1,H,W)
roi_mask = warp_affine(x.reshape(-1, 1, H, W), correction_matrix,
dsize=(H, W), mode="nearest")
# (B,L,C,H,W)
roi_mask = torch.repeat_interleave(roi_mask, C, dim=1).reshape(B, L, C, H,
W)
return roi_mask
def get_discretized_transformation_matrix(matrix, discrete_ratio,
downsample_rate):
"""
Get disretized transformation matrix.
Parameters
----------
matrix : torch.Tensor
Shape -- (B, L, 4, 4) where B is the batch size, L is the max cav
number.
discrete_ratio : float
Discrete ratio.
downsample_rate : float/int
downsample_rate
Returns
-------
matrix : torch.Tensor
Output transformation matrix in 2D with shape (B, L, 2, 3),
including 2D transformation and 2D rotation.
"""
matrix = matrix[:, :, [0, 1], :][:, :, :, [0, 1, 3]]
# normalize the x,y transformation
matrix[:, :, :, -1] = matrix[:, :, :, -1] \
/ (discrete_ratio * downsample_rate)
return matrix.type(dtype=torch.float)
def _torch_inverse_cast(input):
r"""
Helper function to make torch.inverse work with other than fp32/64.
The function torch.inverse is only implemented for fp32/64 which makes
impossible to be used by fp16 or others. What this function does,
is cast input data type to fp32, apply torch.inverse,
and cast back to the input dtype.
Args:
input : torch.Tensor
Tensor to be inversed.
Returns:
out : torch.Tensor
Inversed Tensor.
"""
dtype = input.dtype
if dtype not in (torch.float32, torch.float64):
dtype = torch.float32
out = torch.inverse(input.to(dtype)).to(input.dtype)
return out
def normal_transform_pixel(
height, width, device, dtype, eps=1e-14):
r"""
Compute the normalization matrix from image size in pixels to [-1, 1].
Args:
height : int
Image height.
width : int
Image width.
device : torch.device
Output tensor devices.
dtype : torch.dtype
Output tensor data type.
eps : float
Epsilon to prevent divide-by-zero errors.
Returns:
tr_mat : torch.Tensor
Normalized transform with shape :math:`(1, 3, 3)`.
"""
tr_mat = torch.tensor(
[[1.0, 0.0, -1.0], [0.0, 1.0, -1.0], [0.0, 0.0, 1.0]], device=device,
dtype=dtype) # 3x3
# prevent divide by zero bugs
width_denom = eps if width == 1 else width - 1.0
height_denom = eps if height == 1 else height - 1.0
tr_mat[0, 0] = tr_mat[0, 0] * 2.0 / width_denom
tr_mat[1, 1] = tr_mat[1, 1] * 2.0 / height_denom
return tr_mat.unsqueeze(0) # 1x3x3
def eye_like(n, B, device, dtype):
r"""
Return a 2-D tensor with ones on the diagonal and
zeros elsewhere with the same batch size as the input.
Args:
n : int
The number of rows :math:`(n)`.
B : int
Btach size.
device : torch.device
Devices of the output tensor.
dtype : torch.dtype
Data type of the output tensor.
Returns:
The identity matrix with the shape :math:`(B, n, n)`.
"""
identity = torch.eye(n, device=device, dtype=dtype)
return identity[None].repeat(B, 1, 1)
def normalize_homography(dst_pix_trans_src_pix, dsize_src, dsize_dst=None):
r"""
Normalize a given homography in pixels to [-1, 1].
Args:
dst_pix_trans_src_pix : torch.Tensor
Homography/ies from source to destination to be normalized with
shape :math:`(B, 3, 3)`.
dsize_src : Tuple[int, int]
Size of the source image (height, width).
dsize_dst : Tuple[int, int]
Size of the destination image (height, width).
Returns:
dst_norm_trans_src_norm : torch.Tensor
The normalized homography of shape :math:`(B, 3, 3)`.
"""
if dsize_dst is None:
dsize_dst = dsize_src
# source and destination sizes
src_h, src_w = dsize_src
dst_h, dst_w = dsize_dst
device = dst_pix_trans_src_pix.device
dtype = dst_pix_trans_src_pix.dtype
# compute the transformation pixel/norm for src/dst
src_norm_trans_src_pix = normal_transform_pixel(src_h, src_w, device,
dtype).to(
dst_pix_trans_src_pix)
src_pix_trans_src_norm = _torch_inverse_cast(src_norm_trans_src_pix)
dst_norm_trans_dst_pix = normal_transform_pixel(dst_h, dst_w, device,
dtype).to(
dst_pix_trans_src_pix)
# compute chain transformations
dst_norm_trans_src_norm: torch.Tensor = dst_norm_trans_dst_pix @ (
dst_pix_trans_src_pix @ src_pix_trans_src_norm)
return dst_norm_trans_src_norm
def get_rotation_matrix2d(M, dsize):
r"""
Return rotation matrix for torch.affine_grid based on transformation matrix.
Args:
M : torch.Tensor
Transformation matrix with shape :math:`(B, 2, 3)`.
dsize : Tuple[int, int]
Size of the source image (height, width).
Returns:
R : torch.Tensor
Rotation matrix with shape :math:`(B, 2, 3)`.
"""
H, W = dsize
B = M.shape[0]
center = torch.Tensor([W / 2, H / 2]).to(M.dtype).to(M.device).unsqueeze(0)
shift_m = eye_like(3, B, M.device, M.dtype)
shift_m[:, :2, 2] = center
shift_m_inv = eye_like(3, B, M.device, M.dtype)
shift_m_inv[:, :2, 2] = -center
rotat_m = eye_like(3, B, M.device, M.dtype)
rotat_m[:, :2, :2] = M[:, :2, :2]
affine_m = shift_m @ rotat_m @ shift_m_inv
return affine_m[:, :2, :] # Bx2x3
def get_transformation_matrix(M, dsize):
r"""
Return transformation matrix for torch.affine_grid.
Args:
M : torch.Tensor
Transformation matrix with shape :math:`(N, 2, 3)`.
dsize : Tuple[int, int]
Size of the source image (height, width).
Returns:
T : torch.Tensor
Transformation matrix with shape :math:`(N, 2, 3)`.
"""
T = get_rotation_matrix2d(M, dsize)
T[..., 2] += M[..., 2]
return T
def convert_affinematrix_to_homography(A):
r"""
Convert to homography coordinates
Args:
A : torch.Tensor
The affine matrix with shape :math:`(B,2,3)`.
Returns:
H : torch.Tensor
The homography matrix with shape of :math:`(B,3,3)`.
"""
H: torch.Tensor = torch.nn.functional.pad(A, [0, 0, 0, 1], "constant",
value=0.0)
H[..., -1, -1] += 1.0
return H
def warp_affine(
src, M, dsize,
mode='bilinear',
padding_mode='zeros',
align_corners=True):
r"""
Transform the src based on transformation matrix M.
Args:
src : torch.Tensor
Input feature map with shape :math:`(B,C,H,W)`.
M : torch.Tensor
Transformation matrix with shape :math:`(B,2,3)`.
dsize : tuple
Tuple of output image H_out and W_out.
mode : str
Interpolation methods for F.grid_sample.
padding_mode : str
Padding methods for F.grid_sample.
align_corners : boolean
Parameter of F.affine_grid.
Returns:
Transformed features with shape :math:`(B,C,H,W)`.
"""
B, C, H, W = src.size()
# we generate a 3x3 transformation matrix from 2x3 affine
M_3x3 = convert_affinematrix_to_homography(M)
dst_norm_trans_src_norm = normalize_homography(M_3x3, (H, W), dsize)
# src_norm_trans_dst_norm = torch.inverse(dst_norm_trans_src_norm)
src_norm_trans_dst_norm = _torch_inverse_cast(dst_norm_trans_src_norm)
grid = F.affine_grid(src_norm_trans_dst_norm[:, :2, :],
[B, C, dsize[0], dsize[1]],
align_corners=align_corners)
return F.grid_sample(src.half() if grid.dtype == torch.half else src, grid,
align_corners=align_corners, mode=mode,
padding_mode=padding_mode)
class Test:
"""
Test the transformation in this file.
The methods in this class are not supposed to be used outside of this file.
"""
def __init__(self):
pass
@staticmethod
def load_img():
torch.manual_seed(0)
x = torch.randn(1, 5, 16, 400, 200) * 100
# x = torch.ones(1, 5, 16, 400, 200)
return x
@staticmethod
def load_raw_transformation_matrix(N):
a = 90 / 180 * np.pi
matrix = torch.Tensor([[np.cos(a), -np.sin(a), 10],
[np.sin(a), np.cos(a), 10]])
matrix = torch.repeat_interleave(matrix.unsqueeze(0).unsqueeze(0), N,
dim=1)
return matrix
@staticmethod
def load_raw_transformation_matrix2(N, alpha):
a = alpha / 180 * np.pi
matrix = torch.Tensor([[np.cos(a), -np.sin(a), 0, 0],
[np.sin(a), np.cos(a), 0, 0]])
matrix = torch.repeat_interleave(matrix.unsqueeze(0).unsqueeze(0), N,
dim=1)
return matrix
@staticmethod
def test():
img = Test.load_img()
B, L, C, H, W = img.shape
raw_T = Test.load_raw_transformation_matrix(5)
T = get_transformation_matrix(raw_T.reshape(-1, 2, 3), (H, W))
img_rot = warp_affine(img.reshape(-1, C, H, W), T, (H, W))
print(img_rot[0, 0, :, :])
plt.matshow(img_rot[0, 0, :, :])
plt.show()
@staticmethod
def test_combine_roi_and_cav_mask():
B = 2
L = 5
C = 16
H = 300
W = 400
# 2, 5
cav_mask = torch.Tensor([[1, 1, 1, 0, 0], [1, 0, 0, 0, 0]])
x = torch.zeros(B, L, C, H, W)
correction_matrix = Test.load_raw_transformation_matrix2(5, 10)
correction_matrix = torch.cat([correction_matrix, correction_matrix],
dim=0)
mask = get_roi_and_cav_mask((B, L, H, W, C), cav_mask,
correction_matrix, 0.4, 4)
plt.matshow(mask[0, :, :, 0, 0])
plt.show()
if __name__ == "__main__":
os.environ['KMP_DUPLICATE_LIB_OK'] = 'True'
Test.test_combine_roi_and_cav_mask()
import torch
import torch.nn as nn
import torch.nn.functional as F
class RadixSoftmax(nn.Module):
def __init__(self, radix, cardinality):
super(RadixSoftmax, self).__init__()
self.radix = radix
self.cardinality = cardinality
def forward(self, x):
# x: (B, L, 1, 1, 3C)
batch = x.size(0)
cav_num = x.size(1)
if self.radix > 1:
# x: (B, L, 1, 3, C)
x = x.view(batch,
cav_num,
self.cardinality, self.radix, -1)
x = F.softmax(x, dim=3)
# B, 3LC
x = x.reshape(batch, -1)
else:
x = torch.sigmoid(x)
return x
class SplitAttn(nn.Module):
def __init__(self, input_dim):
super(SplitAttn, self).__init__()
self.input_dim = input_dim
self.fc1 = nn.Linear(input_dim, input_dim, bias=False)
self.bn1 = nn.LayerNorm(input_dim)
self.act1 = nn.ReLU()
self.fc2 = nn.Linear(input_dim, input_dim * 3, bias=False)
self.rsoftmax = RadixSoftmax(3, 1)
def forward(self, window_list):
# window list: [(B, L, H, W, C) * 3]
assert len(window_list) == 3, 'only 3 windows are supported'
sw, mw, bw = window_list[0], window_list[1], window_list[2]
B, L = sw.shape[0], sw.shape[1]
# global average pooling, B, L, H, W, C
x_gap = sw + mw + bw
# B, L, 1, 1, C
x_gap = x_gap.mean((2, 3), keepdim=True)
x_gap = self.act1(self.bn1(self.fc1(x_gap)))
# B, L, 1, 1, 3C
x_attn = self.fc2(x_gap)
# B L 1 1 3C
x_attn = self.rsoftmax(x_attn).view(B, L, 1, 1, -1)
out = sw * x_attn[:, :, :, :, 0:self.input_dim] + \
mw * x_attn[:, :, :, :, self.input_dim:2*self.input_dim] +\
bw * x_attn[:, :, :, :, self.input_dim*2:]
return out
import numpy as np
import torch
import torch.nn as nn
class BaseBEVBackbone(nn.Module):
def __init__(self, model_cfg, input_channels):
super().__init__()
self.model_cfg = model_cfg
if 'layer_nums' in self.model_cfg:
assert len(self.model_cfg['layer_nums']) == \
len(self.model_cfg['layer_strides']) == \
len(self.model_cfg['num_filters'])
layer_nums = self.model_cfg['layer_nums']
layer_strides = self.model_cfg['layer_strides']
num_filters = self.model_cfg['num_filters']
else:
layer_nums = layer_strides = num_filters = []
if 'upsample_strides' in self.model_cfg:
assert len(self.model_cfg['upsample_strides']) \
== len(self.model_cfg['num_upsample_filter'])
num_upsample_filters = self.model_cfg['num_upsample_filter']
upsample_strides = self.model_cfg['upsample_strides']
else:
upsample_strides = num_upsample_filters = []
num_levels = len(layer_nums)
c_in_list = [input_channels, *num_filters[:-1]]
self.blocks = nn.ModuleList()
self.deblocks = nn.ModuleList()
for idx in range(num_levels):
cur_layers = [
nn.ZeroPad2d(1),
nn.Conv2d(
c_in_list[idx], num_filters[idx], kernel_size=3,
stride=layer_strides[idx], padding=0, bias=False
),
nn.BatchNorm2d(num_filters[idx], eps=1e-3, momentum=0.01),
nn.ReLU()
]
for k in range(layer_nums[idx]):
cur_layers.extend([
nn.Conv2d(num_filters[idx], num_filters[idx],
kernel_size=3, padding=1, bias=False),
nn.BatchNorm2d(num_filters[idx], eps=1e-3, momentum=0.01),
nn.ReLU()
])
self.blocks.append(nn.Sequential(*cur_layers))
if len(upsample_strides) > 0:
stride = upsample_strides[idx]
if stride >= 1:
self.deblocks.append(nn.Sequential(
nn.ConvTranspose2d(
num_filters[idx], num_upsample_filters[idx],
upsample_strides[idx],
stride=upsample_strides[idx], bias=False
),
nn.BatchNorm2d(num_upsample_filters[idx],
eps=1e-3, momentum=0.01),
nn.ReLU()
))
else:
stride = np.round(1 / stride).astype(np.int)
self.deblocks.append(nn.Sequential(
nn.Conv2d(
num_filters[idx], num_upsample_filters[idx],
stride,
stride=stride, bias=False
),
nn.BatchNorm2d(num_upsample_filters[idx], eps=1e-3,
momentum=0.01),
nn.ReLU()
))
c_in = sum(num_upsample_filters)
if len(upsample_strides) > num_levels:
self.deblocks.append(nn.Sequential(
nn.ConvTranspose2d(c_in, c_in, upsample_strides[-1],
stride=upsample_strides[-1], bias=False),
nn.BatchNorm2d(c_in, eps=1e-3, momentum=0.01),
nn.ReLU(),
))
self.num_bev_features = c_in
def forward(self, data_dict):
spatial_features = data_dict['spatial_features']
ups = []
ret_dict = {}
x = spatial_features
for i in range(len(self.blocks)):
x = self.blocks[i](x)
stride = int(spatial_features.shape[2] / x.shape[2])
ret_dict['spatial_features_%dx' % stride] = x
if len(self.deblocks) > 0:
ups.append(self.deblocks[i](x))
else:
ups.append(x)
if len(ups) > 1:
x = torch.cat(ups, dim=1)
elif len(ups) == 1:
x = ups[0]
if len(self.deblocks) > len(self.blocks):
x = self.deblocks[-1](x)
data_dict['spatial_features_2d'] = x
return data_dict
import torch
import numpy as np
from einops import rearrange
from v2xvit.utils.common_utils import torch_tensor_to_numpy
def regroup(dense_feature, record_len, max_len):
"""
Regroup the data based on the record_len.
Parameters
----------
dense_feature : torch.Tensor
N, C, H, W
record_len : list
[sample1_len, sample2_len, ...]
max_len : int
Maximum cav number
Returns
-------
regroup_feature : torch.Tensor
B, L, C, H, W
"""
cum_sum_len = list(np.cumsum(torch_tensor_to_numpy(record_len)))
split_features = torch.tensor_split(dense_feature,
cum_sum_len[:-1])
regroup_features = []
mask = []
for split_feature in split_features:
# M, C, H, W
feature_shape = split_feature.shape
# the maximum M is 5 as most 5 cavs
padding_len = max_len - feature_shape[0]
mask.append([1] * feature_shape[0] + [0] * padding_len)
padding_tensor = torch.zeros(padding_len, feature_shape[1],
feature_shape[2], feature_shape[3])
padding_tensor = padding_tensor.to(split_feature.device)
split_feature = torch.cat([split_feature, padding_tensor],
dim=0)
# 1, 5C, H, W
split_feature = split_feature.view(-1,
feature_shape[2],
feature_shape[3]).unsqueeze(0)
regroup_features.append(split_feature)
# B, 5C, H, W
regroup_features = torch.cat(regroup_features, dim=0)
# B, L, C, H, W
regroup_features = rearrange(regroup_features,
'b (l c) h w -> b l c h w',
l=max_len)
mask = torch.from_numpy(np.array(mask)).to(regroup_features.device)
return regroup_features, mask
import torch
import torch.nn as nn
class PointPillarScatter(nn.Module):
def __init__(self, model_cfg):
super().__init__()
self.model_cfg = model_cfg
self.num_bev_features = self.model_cfg['num_features']
self.nx, self.ny, self.nz = model_cfg['grid_size']
assert self.nz == 1
def forward(self, batch_dict):
pillar_features, coords = batch_dict['pillar_features'], batch_dict[
'voxel_coords']
batch_spatial_features = []
batch_size = coords[:, 0].max().int().item() + 1
for batch_idx in range(batch_size):
spatial_feature = torch.zeros(
self.num_bev_features,
self.nz * self.nx * self.ny,
dtype=pillar_features.dtype,
device=pillar_features.device)
batch_mask = coords[:, 0] == batch_idx
this_coords = coords[batch_mask, :]
indices = this_coords[:, 1] + \
this_coords[:, 2] * self.nx + \
this_coords[:, 3]
indices = indices.type(torch.long)
pillars = pillar_features[batch_mask, :]
pillars = pillars.t()
spatial_feature[:, indices] = pillars
batch_spatial_features.append(spatial_feature)
batch_spatial_features = \
torch.stack(batch_spatial_features, 0)
batch_spatial_features = \
batch_spatial_features.view(batch_size, self.num_bev_features *
self.nz, self.ny, self.nx)
batch_dict['spatial_features'] = batch_spatial_features
return batch_dict
import torch
from torch import nn
from einops import rearrange
class PreNorm(nn.Module):
def __init__(self, dim, fn):
super().__init__()
self.norm = nn.LayerNorm(dim)
self.fn = fn
def forward(self, x, **kwargs):
return self.fn(self.norm(x), **kwargs)
class FeedForward(nn.Module):
def __init__(self, dim, hidden_dim, dropout=0.):
super().__init__()
self.net = nn.Sequential(
nn.Linear(dim, hidden_dim),
nn.GELU(),
nn.Dropout(dropout),
nn.Linear(hidden_dim, dim),
nn.Dropout(dropout)
)
def forward(self, x):
return self.net(x)
class CavAttention(nn.Module):
"""
Vanilla CAV attention.
"""
def __init__(self, dim, heads, dim_head=64, dropout=0.1):
super().__init__()
inner_dim = heads * dim_head
self.heads = heads
self.scale = dim_head ** -0.5
self.attend = nn.Softmax(dim=-1)
self.to_qkv = nn.Linear(dim, inner_dim * 3, bias=False)
self.to_out = nn.Sequential(
nn.Linear(inner_dim, dim),
nn.Dropout(dropout)
)
def forward(self, x, mask, prior_encoding):
# x: (B, L, H, W, C) -> (B, H, W, L, C)
# mask: (B, L)
x = x.permute(0, 2, 3, 1, 4)
# mask: (B, 1, H, W, L, 1)
mask = mask.unsqueeze(1)
# qkv: [(B, H, W, L, C_inner) *3]
qkv = self.to_qkv(x).chunk(3, dim=-1)
# q: (B, M, H, W, L, C)
q, k, v = map(lambda t: rearrange(t, 'b h w l (m c) -> b m h w l c',
m=self.heads), qkv)
# attention, (B, M, H, W, L, L)
att_map = torch.einsum('b m h w i c, b m h w j c -> b m h w i j',
q, k) * self.scale
# add mask
att_map = att_map.masked_fill(mask == 0, -float('inf'))
# softmax
att_map = self.attend(att_map)
# out:(B, M, H, W, L, C_head)
out = torch.einsum('b m h w i j, b m h w j c -> b m h w i c', att_map,
v)
out = rearrange(out, 'b m h w l c -> b h w l (m c)',
m=self.heads)
out = self.to_out(out)
# (B L H W C)
out = out.permute(0, 3, 1, 2, 4)
return out
class BaseEncoder(nn.Module):
def __init__(self, dim, depth, heads, dim_head, mlp_dim, dropout=0.):
super().__init__()
self.layers = nn.ModuleList([])
for _ in range(depth):
self.layers.append(nn.ModuleList([
PreNorm(dim, CavAttention(dim,
heads=heads,
dim_head=dim_head,
dropout=dropout)),
PreNorm(dim, FeedForward(dim, mlp_dim, dropout=dropout))
]))
def forward(self, x, mask):
for attn, ff in self.layers:
x = attn(x, mask=mask) + x
x = ff(x) + x
return x
class BaseTransformer(nn.Module):
def __init__(self, args):
super().__init__()
dim = args['dim']
depth = args['depth']
heads = args['heads']
dim_head = args['dim_head']
mlp_dim = args['mlp_dim']
dropout = args['dropout']
max_cav = args['max_cav']
self.encoder = BaseEncoder(dim, depth, heads, dim_head, mlp_dim,
dropout)
def forward(self, x, mask):
# B, L, H, W, C
output = self.encoder(x, mask)
# B, H, W, C
output = output[:, 0]
return output
"""
Implementation of F-cooper maxout fusing.
"""
import torch
import torch.nn as nn
class SpatialFusion(nn.Module):
def __init__(self):
super(SpatialFusion, self).__init__()
def regroup(self, x, record_len):
cum_sum_len = torch.cumsum(record_len, dim=0)
split_x = torch.tensor_split(x, cum_sum_len[:-1].cpu())
return split_x
def forward(self, x, record_len):
# x: B, C, H, W, split x:[(B1, C, W, H), (B2, C, W, H)]
split_x = self.regroup(x, record_len)
out = []
for xx in split_x:
xx = torch.max(xx, dim=0, keepdim=True)[0]
out.append(xx)
return torch.cat(out, dim=0)
import torch
import torch.nn as nn
class NaiveCompressor(nn.Module):
def __init__(self, input_dim, compress_raito):
super().__init__()
self.encoder = nn.Sequential(
nn.Conv2d(input_dim, input_dim//compress_raito, kernel_size=3,
stride=1, padding=1),
nn.BatchNorm2d(input_dim//compress_raito, eps=1e-3, momentum=0.01),
nn.ReLU()
)
self.decoder = nn.Sequential(
nn.Conv2d(input_dim//compress_raito, input_dim, kernel_size=3,
stride=1, padding=1),
nn.BatchNorm2d(input_dim, eps=1e-3, momentum=0.01),
nn.ReLU(),
nn.Conv2d(input_dim, input_dim, kernel_size=3, stride=1, padding=1),
nn.BatchNorm2d(input_dim, eps=1e-3,
momentum=0.01),
nn.ReLU()
)
def forward(self, x):
x = self.encoder(x)
x = self.decoder(x)
return x
import math
from v2xvit.models.sub_modules.base_transformer import *
from v2xvit.models.sub_modules.hmsa import *
from v2xvit.models.sub_modules.mswin import *
from v2xvit.models.sub_modules.torch_transformation_utils import \
get_transformation_matrix, warp_affine, get_roi_and_cav_mask, \
get_discretized_transformation_matrix
class STTF(nn.Module):
def __init__(self, args):
super(STTF, self).__init__()
self.discrete_ratio = args['voxel_size'][0]
self.downsample_rate = args['downsample_rate']
def forward(self, x, mask, spatial_correction_matrix):
x = x.permute(0, 1, 4, 2, 3)
dist_correction_matrix = get_discretized_transformation_matrix(
spatial_correction_matrix, self.discrete_ratio,
self.downsample_rate)
# Only compensate non-ego vehicles
B, L, C, H, W = x.shape
T = get_transformation_matrix(
dist_correction_matrix[:, 1:, :, :].reshape(-1, 2, 3), (H, W))
cav_features = warp_affine(x[:, 1:, :, :, :].reshape(-1, C, H, W), T,
(H, W))
cav_features = cav_features.reshape(B, -1, C, H, W)
x = torch.cat([x[:, 0, :, :, :].unsqueeze(1), cav_features], dim=1)
x = x.permute(0, 1, 3, 4, 2)
return x
class RelTemporalEncoding(nn.Module):
"""
Implement the Temporal Encoding (Sinusoid) function.
"""
def __init__(self, n_hid, RTE_ratio, max_len=100, dropout=0.2):
super(RelTemporalEncoding, self).__init__()
position = torch.arange(0., max_len).unsqueeze(1)
div_term = torch.exp(torch.arange(0, n_hid, 2) *
-(math.log(10000.0) / n_hid))
emb = nn.Embedding(max_len, n_hid)
emb.weight.data[:, 0::2] = torch.sin(position * div_term) / math.sqrt(
n_hid)
emb.weight.data[:, 1::2] = torch.cos(position * div_term) / math.sqrt(
n_hid)
emb.requires_grad = False
self.RTE_ratio = RTE_ratio
self.emb = emb
self.lin = nn.Linear(n_hid, n_hid)
def forward(self, x, t):
# When t has unit of 50ms, rte_ratio=1.
# So we can train on 100ms but test on 50ms
return x + self.lin(self.emb(t * self.RTE_ratio)).unsqueeze(
0).unsqueeze(1)
class RTE(nn.Module):
def __init__(self, dim, RTE_ratio=2):
super(RTE, self).__init__()
self.RTE_ratio = RTE_ratio
self.emb = RelTemporalEncoding(dim, RTE_ratio=self.RTE_ratio)
def forward(self, x, dts):
# x: (B,L,H,W,C)
# dts: (B,L)
rte_batch = []
for b in range(x.shape[0]):
rte_list = []
for i in range(x.shape[1]):
rte_list.append(
self.emb(x[b, i, :, :, :], dts[b, i]).unsqueeze(0))
rte_batch.append(torch.cat(rte_list, dim=0).unsqueeze(0))
return torch.cat(rte_batch, dim=0)
class V2XFusionBlock(nn.Module):
def __init__(self, num_blocks, cav_att_config, pwindow_config):
super().__init__()
# first multi-agent attention and then multi-window attention
self.layers = nn.ModuleList([])
self.num_blocks = num_blocks
for _ in range(num_blocks):
att = HGTCavAttention(cav_att_config['dim'],
heads=cav_att_config['heads'],
dim_head=cav_att_config['dim_head'],
dropout=cav_att_config['dropout']) if \
cav_att_config['use_hetero'] else \
CavAttention(cav_att_config['dim'],
heads=cav_att_config['heads'],
dim_head=cav_att_config['dim_head'],
dropout=cav_att_config['dropout'])
self.layers.append(nn.ModuleList([
PreNorm(cav_att_config['dim'], att),
PreNorm(cav_att_config['dim'],
PyramidWindowAttention(pwindow_config['dim'],
heads=pwindow_config['heads'],
dim_heads=pwindow_config[
'dim_head'],
drop_out=pwindow_config[
'dropout'],
window_size=pwindow_config[
'window_size'],
relative_pos_embedding=
pwindow_config[
'relative_pos_embedding'],
fuse_method=pwindow_config[
'fusion_method']))]))
def forward(self, x, mask, prior_encoding):
for cav_attn, pwindow_attn in self.layers:
x = cav_attn(x, mask=mask, prior_encoding=prior_encoding) + x
x = pwindow_attn(x) + x
return x
class V2XTEncoder(nn.Module):
def __init__(self, args):
super().__init__()
cav_att_config = args['cav_att_config']
pwindow_att_config = args['pwindow_att_config']
feed_config = args['feed_forward']
num_blocks = args['num_blocks']
depth = args['depth']
mlp_dim = feed_config['mlp_dim']
dropout = feed_config['dropout']
self.downsample_rate = args['sttf']['downsample_rate']
self.discrete_ratio = args['sttf']['voxel_size'][0]
self.use_roi_mask = args['use_roi_mask']
self.use_RTE = cav_att_config['use_RTE']
self.RTE_ratio = cav_att_config['RTE_ratio']
self.sttf = STTF(args['sttf'])
# adjust the channel numbers from 256+3 -> 256
self.prior_feed = nn.Linear(cav_att_config['dim'] + 3,
cav_att_config['dim'])
self.layers = nn.ModuleList([])
if self.use_RTE:
self.rte = RTE(cav_att_config['dim'], self.RTE_ratio)
for _ in range(depth):
self.layers.append(nn.ModuleList([
V2XFusionBlock(num_blocks, cav_att_config, pwindow_att_config),
PreNorm(cav_att_config['dim'],
FeedForward(cav_att_config['dim'], mlp_dim,
dropout=dropout))
]))
def forward(self, x, mask, spatial_correction_matrix):
# transform the features to the current timestamp
# velocity, time_delay, infra
# (B,L,H,W,3)
prior_encoding = x[..., -3:]
# (B,L,H,W,C)
x = x[..., :-3]
if self.use_RTE:
# dt: (B,L)
dt = prior_encoding[:, :, 0, 0, 1].to(torch.int)
x = self.rte(x, dt)
x = self.sttf(x, mask, spatial_correction_matrix)
com_mask = mask.unsqueeze(1).unsqueeze(2).unsqueeze(
3) if not self.use_roi_mask else get_roi_and_cav_mask(x.shape,
mask,
spatial_correction_matrix,
self.discrete_ratio,
self.downsample_rate)
for attn, ff in self.layers:
x = attn(x, mask=com_mask, prior_encoding=prior_encoding)
x = ff(x) + x
return x
class V2XTransformer(nn.Module):
def __init__(self, args):
super(V2XTransformer, self).__init__()
encoder_args = args['encoder']
self.encoder = V2XTEncoder(encoder_args)
def forward(self, x, mask, spatial_correction_matrix):
output = self.encoder(x, mask, spatial_correction_matrix)
output = output[:, 0]
return output
"""
Implementation of V2VNet Fusion
"""
import torch
import torch.nn as nn
from v2xvit.models.sub_modules.torch_transformation_utils import \
get_discretized_transformation_matrix, get_transformation_matrix, \
warp_affine, get_rotated_roi
from v2xvit.models.sub_modules.convgru import ConvGRU
class V2VNetFusion(nn.Module):
def __init__(self, args):
super(V2VNetFusion, self).__init__()
in_channels = args['in_channels']
H, W = args['conv_gru']['H'], args['conv_gru']['W']
kernel_size = args['conv_gru']['kernel_size']
num_gru_layers = args['conv_gru']['num_layers']
self.use_temporal_encoding = args['use_temporal_encoding']
self.discrete_ratio = args['voxel_size'][0]
self.downsample_rate = args['downsample_rate']
self.num_iteration = args['num_iteration']
self.gru_flag = args['gru_flag']
self.agg_operator = args['agg_operator']
self.cnn = nn.Conv2d(in_channels + 1, in_channels, kernel_size=3,
stride=1, padding=1)
self.msg_cnn = nn.Conv2d(in_channels * 2, in_channels, kernel_size=3,
stride=1, padding=1)
self.conv_gru = ConvGRU(input_size=(H, W),
input_dim=in_channels * 2,
hidden_dim=[in_channels],
kernel_size=kernel_size,
num_layers=num_gru_layers,
batch_first=True,
bias=True,
return_all_layers=False)
self.mlp = nn.Linear(in_channels, in_channels)
def regroup(self, x, record_len):
cum_sum_len = torch.cumsum(record_len, dim=0)
split_x = torch.tensor_split(x, cum_sum_len[:-1].cpu())
return split_x
def forward(self, x, record_len, pairwise_t_matrix, prior_encoding):
# x: (B,C,H,W)
# record_len: (B)
# pairwise_t_matrix: (B,L,L,4,4)
# prior_encoding: (B,3)
_, C, H, W = x.shape
B, L = pairwise_t_matrix.shape[:2]
if self.use_temporal_encoding:
# (B,1,1,1)
dt = prior_encoding[:, 1].to(torch.int).unsqueeze(1).unsqueeze(
2).unsqueeze(3)
x = torch.cat([x, dt.repeat(1, 1, H, W)], dim=1)
x = self.cnn(x)
# split x:[(L1, C, H, W), (L2, C, H, W)]
split_x = self.regroup(x, record_len)
# (B,L,L,2,3)
pairwise_t_matrix = get_discretized_transformation_matrix(
pairwise_t_matrix.reshape(-1, L, 4, 4), self.discrete_ratio,
self.downsample_rate).reshape(B, L, L, 2, 3)
# (B*L,L,1,H,W)
roi_mask = get_rotated_roi((B * L, L, 1, H, W),
pairwise_t_matrix.reshape(B * L * L, 2, 3))
roi_mask = roi_mask.reshape(B, L, L, 1, H, W)
batch_node_features = split_x
# iteratively update the features for num_iteration times
for l in range(self.num_iteration):
batch_updated_node_features = []
# iterate each batch
for b in range(B):
# number of valid agent
N = record_len[b]
# (N,N,4,4)
# t_matrix[i, j]-> from i to j
t_matrix = pairwise_t_matrix[b][:N, :N, :, :]
updated_node_features = []
# update each node i
for i in range(N):
# (N,1,H,W)
mask = roi_mask[b, :N, i, ...]
current_t_matrix = t_matrix[:, i, :, :]
current_t_matrix = get_transformation_matrix(
current_t_matrix, (H, W))
# (N,C,H,W)
neighbor_feature = warp_affine(batch_node_features[b],
current_t_matrix,
(H, W))
# (N,C,H,W)
ego_agent_feature = batch_node_features[b][i].unsqueeze(
0).repeat(N, 1, 1, 1)
#(N,2C,H,W)
neighbor_feature = torch.cat(
[neighbor_feature, ego_agent_feature], dim=1)
# (N,C,H,W)
message = self.msg_cnn(neighbor_feature) * mask
# (C,H,W)
if self.agg_operator=="avg":
agg_feature = torch.mean(message, dim=0)
elif self.agg_operator=="max":
agg_feature = torch.max(message, dim=0)[0]
else:
raise ValueError("agg_operator has wrong value")
# (2C, H, W)
cat_feature = torch.cat(
[batch_node_features[b][i, ...], agg_feature], dim=0)
# (C,H,W)
if self.gru_flag:
gru_out = \
self.conv_gru(cat_feature.unsqueeze(0).unsqueeze(0))[
0][
0].squeeze(0).squeeze(0)
else:
gru_out = batch_node_features[b][i, ...] + agg_feature
updated_node_features.append(gru_out.unsqueeze(0))
# (N,C,H,W)
batch_updated_node_features.append(
torch.cat(updated_node_features, dim=0))
batch_node_features = batch_updated_node_features
# (B,C,H,W)
out = torch.cat(
[itm[0, ...].unsqueeze(0) for itm in batch_node_features], dim=0)
# (B,C,H,W)
out = self.mlp(out.permute(0, 2, 3, 1)).permute(0, 3, 1, 2)
return out
"""
Class used to downsample features by 3*3 conv
"""
import torch
import torch.nn as nn
class DoubleConv(nn.Module):
"""
Double convoltuion
Args:
in_channels: input channel num
out_channels: output channel num
"""
def __init__(self, in_channels, out_channels, kernel_size,
stride, padding):
super().__init__()
self.double_conv = nn.Sequential(
nn.Conv2d(in_channels, out_channels, kernel_size=kernel_size,
stride=stride, padding=padding),
nn.ReLU(inplace=True),
nn.Conv2d(out_channels, out_channels, kernel_size=3, padding=1),
nn.ReLU(inplace=True)
)
def forward(self, x):
return self.double_conv(x)
class DownsampleConv(nn.Module):
def __init__(self, config):
super(DownsampleConv, self).__init__()
self.layers = nn.ModuleList([])
input_dim = config['input_dim']
for (ksize, dim, stride, padding) in zip(config['kernal_size'],
config['dim'],
config['stride'],
config['padding']):
self.layers.append(DoubleConv(input_dim,
dim,
kernel_size=ksize,
stride=stride,
padding=padding))
input_dim = dim
def forward(self, x):
for i in range(len(self.layers)):
x = self.layers[i](x)
return x
import torch
from torch import nn
from einops import rearrange
class HGTCavAttention(nn.Module):
def __init__(self, dim, heads, num_types=2,
num_relations=4, dim_head=64, dropout=0.1):
super().__init__()
inner_dim = heads * dim_head
self.heads = heads
self.scale = dim_head ** -0.5
self.num_types = num_types
self.attend = nn.Softmax(dim=-1)
self.drop_out = nn.Dropout(dropout)
self.k_linears = nn.ModuleList()
self.q_linears = nn.ModuleList()
self.v_linears = nn.ModuleList()
self.a_linears = nn.ModuleList()
self.norms = nn.ModuleList()
for t in range(num_types):
self.k_linears.append(nn.Linear(dim, inner_dim))
self.q_linears.append(nn.Linear(dim, inner_dim))
self.v_linears.append(nn.Linear(dim, inner_dim))
self.a_linears.append(nn.Linear(inner_dim, dim))
self.relation_att = nn.Parameter(
torch.Tensor(num_relations, heads, dim_head, dim_head))
self.relation_msg = nn.Parameter(
torch.Tensor(num_relations, heads, dim_head, dim_head))
torch.nn.init.xavier_uniform(self.relation_att)
torch.nn.init.xavier_uniform(self.relation_msg)
def to_qkv(self, x, types):
# x: (B,H,W,L,C)
# types: (B,L)
q_batch = []
k_batch = []
v_batch = []
for b in range(x.shape[0]):
q_list = []
k_list = []
v_list = []
for i in range(x.shape[-2]):
# (H,W,1,C)
q_list.append(
self.q_linears[types[b, i]](x[b, :, :, i, :].unsqueeze(2)))
k_list.append(
self.k_linears[types[b, i]](x[b, :, :, i, :].unsqueeze(2)))
v_list.append(
self.v_linears[types[b, i]](x[b, :, :, i, :].unsqueeze(2)))
# (1,H,W,L,C)
q_batch.append(torch.cat(q_list, dim=2).unsqueeze(0))
k_batch.append(torch.cat(k_list, dim=2).unsqueeze(0))
v_batch.append(torch.cat(v_list, dim=2).unsqueeze(0))
# (B,H,W,L,C)
q = torch.cat(q_batch, dim=0)
k = torch.cat(k_batch, dim=0)
v = torch.cat(v_batch, dim=0)
return q, k, v
def get_relation_type_index(self, type1, type2):
return type1 * self.num_types + type2
def get_hetero_edge_weights(self, x, types):
w_att_batch = []
w_msg_batch = []
for b in range(x.shape[0]):
w_att_list = []
w_msg_list = []
for i in range(x.shape[-2]):
w_att_i_list = []
w_msg_i_list = []
for j in range(x.shape[-2]):
e_type = self.get_relation_type_index(types[b, i],
types[b, j])
w_att_i_list.append(self.relation_att[e_type].unsqueeze(0))
w_msg_i_list.append(self.relation_msg[e_type].unsqueeze(0))
w_att_list.append(torch.cat(w_att_i_list, dim=0).unsqueeze(0))
w_msg_list.append(torch.cat(w_msg_i_list, dim=0).unsqueeze(0))
w_att_batch.append(torch.cat(w_att_list, dim=0).unsqueeze(0))
w_msg_batch.append(torch.cat(w_msg_list, dim=0).unsqueeze(0))
# (B,M,L,L,C_head,C_head)
w_att = torch.cat(w_att_batch, dim=0).permute(0, 3, 1, 2, 4, 5)
w_msg = torch.cat(w_msg_batch, dim=0).permute(0, 3, 1, 2, 4, 5)
return w_att, w_msg
def to_out(self, x, types):
out_batch = []
for b in range(x.shape[0]):
out_list = []
for i in range(x.shape[-2]):
out_list.append(
self.a_linears[types[b, i]](x[b, :, :, i, :].unsqueeze(2)))
out_batch.append(torch.cat(out_list, dim=2).unsqueeze(0))
out = torch.cat(out_batch, dim=0)
return out
def forward(self, x, mask, prior_encoding):
# x: (B, L, H, W, C) -> (B, H, W, L, C)
# mask: (B, H, W, L, 1)
# prior_encoding: (B,L,H,W,3)
x = x.permute(0, 2, 3, 1, 4)
# mask: (B, 1, H, W, L, 1)
mask = mask.unsqueeze(1)
# (B,L)
velocities, dts, types = [itm.squeeze(-1) for itm in
prior_encoding[:, :, 0, 0, :].split(
[1, 1, 1], dim=-1)]
types = types.to(torch.int)
dts = dts.to(torch.int)
qkv = self.to_qkv(x, types)
# (B,M,L,L,C_head,C_head)
w_att, w_msg = self.get_hetero_edge_weights(x, types)
# q: (B, M, H, W, L, C)
q, k, v = map(lambda t: rearrange(t, 'b h w l (m c) -> b m h w l c',
m=self.heads), (qkv))
# attention, (B, M, H, W, L, L)
att_map = torch.einsum(
'b m h w i p, b m i j p q, bm h w j q -> b m h w i j',
[q, w_att, k]) * self.scale
# add mask
att_map = att_map.masked_fill(mask == 0, -float('inf'))
# softmax
att_map = self.attend(att_map)
# out:(B, M, H, W, L, C_head)
v_msg = torch.einsum('b m i j p c, b m h w j p -> b m h w i j c',
w_msg, v)
out = torch.einsum('b m h w i j, b m h w i j c -> b m h w i c',
att_map, v_msg)
out = rearrange(out, 'b m h w l c -> b h w l (m c)',
m=self.heads)
out = self.to_out(out, types)
out = self.drop_out(out)
# (B L H W C)
out = out.permute(0, 3, 1, 2, 4)
return out
"""
Multi-scale window transformer
"""
import torch
import torch.nn as nn
import numpy as np
from einops import rearrange
from v2xvit.models.sub_modules.split_attn import SplitAttn
def get_relative_distances(window_size):
indices = torch.tensor(np.array(
[[x, y] for x in range(window_size) for y in range(window_size)]))
distances = indices[None, :, :] - indices[:, None, :]
return distances
class BaseWindowAttention(nn.Module):
def __init__(self, dim, heads, dim_head, drop_out, window_size,
relative_pos_embedding):
super().__init__()
inner_dim = dim_head * heads
self.heads = heads
self.scale = dim_head ** -0.5
self.window_size = window_size
self.relative_pos_embedding = relative_pos_embedding
self.to_qkv = nn.Linear(dim, inner_dim * 3, bias=False)
if self.relative_pos_embedding:
self.relative_indices = get_relative_distances(window_size) + \
window_size - 1
self.pos_embedding = nn.Parameter(torch.randn(2 * window_size - 1,
2 * window_size - 1))
else:
self.pos_embedding = nn.Parameter(torch.randn(window_size ** 2,
window_size ** 2))
self.to_out = nn.Sequential(
nn.Linear(inner_dim, dim),
nn.Dropout(drop_out)
)
def forward(self, x):
b, l, h, w, c, m = *x.shape, self.heads
qkv = self.to_qkv(x).chunk(3, dim=-1)
new_h = h // self.window_size
new_w = w // self.window_size
# q : (b, l, m, new_h*new_w, window_size^2, c_head)
q, k, v = map(
lambda t: rearrange(t,
'b l (new_h w_h) (new_w w_w) (m c) -> b l m (new_h new_w) (w_h w_w) c',
m=m, w_h=self.window_size,
w_w=self.window_size), qkv)
# b l m h window_size window_size
dots = torch.einsum('b l m h i c, b l m h j c -> b l m h i j',
q, k, ) * self.scale
# consider prior knowledge of the local window
if self.relative_pos_embedding:
dots += self.pos_embedding[self.relative_indices[:, :, 0],
self.relative_indices[:, :, 1]]
else:
dots += self.pos_embedding
attn = dots.softmax(dim=-1)
out = torch.einsum('b l m h i j, b l m h j c -> b l m h i c', attn, v)
# b l h w c
out = rearrange(out,
'b l m (new_h new_w) (w_h w_w) c -> b l (new_h w_h) (new_w w_w) (m c)',
m=self.heads, w_h=self.window_size,
w_w=self.window_size,
new_w=new_w, new_h=new_h)
out = self.to_out(out)
return out
class PyramidWindowAttention(nn.Module):
def __init__(self, dim, heads, dim_heads, drop_out, window_size,
relative_pos_embedding, fuse_method='naive'):
super().__init__()
assert isinstance(window_size, list)
assert isinstance(heads, list)
assert isinstance(dim_heads, list)
assert len(dim_heads) == len(heads)
self.pwmsa = nn.ModuleList([])
for (head, dim_head, ws) in zip(heads, dim_heads, window_size):
self.pwmsa.append(BaseWindowAttention(dim,
head,
dim_head,
drop_out,
ws,
relative_pos_embedding))
self.fuse_mehod = fuse_method
if fuse_method == 'split_attn':
self.split_attn = SplitAttn(256)
def forward(self, x):
output = None
# naive fusion will just sum up all window attention output and do a
# mean
if self.fuse_mehod == 'naive':
for wmsa in self.pwmsa:
output = wmsa(x) if output is None else output + wmsa(x)
return output / len(self.pwmsa)
elif self.fuse_mehod == 'split_attn':
window_list = []
for wmsa in self.pwmsa:
window_list.append(wmsa(x))
return self.split_attn(window_list)
import os
from collections import OrderedDict
import numpy as np
import torch
from v2xvit.utils.common_utils import torch_tensor_to_numpy
def inference_late_fusion(batch_data, model, dataset):
"""
Model inference for late fusion.
Parameters
----------
batch_data : dict
model : opencood.object
dataset : opencood.LateFusionDataset
Returns
-------
pred_box_tensor : torch.Tensor
The tensor of prediction bounding box after NMS.
gt_box_tensor : torch.Tensor
The tensor of gt bounding box.
"""
output_dict = OrderedDict()
for cav_id, cav_content in batch_data.items():
output_dict[cav_id] = model(cav_content)
pred_box_tensor, pred_score, gt_box_tensor = \
dataset.post_process(batch_data,
output_dict)
return pred_box_tensor, pred_score, gt_box_tensor
def inference_early_fusion(batch_data, model, dataset):
"""
Model inference for early fusion.
Parameters
----------
batch_data : dict
model : opencood.object
dataset : opencood.EarlyFusionDataset
Returns
-------
pred_box_tensor : torch.Tensor
The tensor of prediction bounding box after NMS.
gt_box_tensor : torch.Tensor
The tensor of gt bounding box.
"""
output_dict = OrderedDict()
cav_content = batch_data['ego']
output_dict['ego'] = model(cav_content)
pred_box_tensor, pred_score, gt_box_tensor = \
dataset.post_process(batch_data,
output_dict)
return pred_box_tensor, pred_score, gt_box_tensor
def inference_intermediate_fusion(batch_data, model, dataset):
"""
Model inference for early fusion.
Parameters
----------
batch_data : dict
model : opencood.object
dataset : opencood.EarlyFusionDataset
Returns
-------
pred_box_tensor : torch.Tensor
The tensor of prediction bounding box after NMS.
gt_box_tensor : torch.Tensor
The tensor of gt bounding box.
"""
return inference_early_fusion(batch_data, model, dataset)
def save_prediction_gt(pred_tensor, gt_tensor, pcd, timestamp, save_path):
"""
Save prediction and gt tensor to txt file.
"""
pred_np = torch_tensor_to_numpy(pred_tensor)
gt_np = torch_tensor_to_numpy(gt_tensor)
pcd_np = torch_tensor_to_numpy(pcd)
np.save(os.path.join(save_path, '%04d_pcd.npy' % timestamp), pcd_np)
np.save(os.path.join(save_path, '%04d_pred.npy' % timestamp), pred_np)
np.save(os.path.join(save_path, '%04d_gt.npy' % timestamp), gt_np)
import argparse
import os
import time
import torch
import open3d as o3d
from torch.utils.data import DataLoader
import v2xvit.hypes_yaml.yaml_utils as yaml_utils
from v2xvit.tools import train_utils, infrence_utils
from v2xvit.data_utils.datasets import build_dataset
from v2xvit.visualization import vis_utils
from v2xvit.utils import eval_utils
def test_parser():
parser = argparse.ArgumentParser(description="synthetic data generation")
parser.add_argument('--model_dir', type=str, required=True,
help='Continued training path')
parser.add_argument('--fusion_method', required=True, type=str,
default='late',
help='late, early or intermediate')
parser.add_argument('--show_vis', action='store_true',
help='whether to show image visualization result')
parser.add_argument('--show_sequence', action='store_true',
help='whether to show video visualization result.'
'it can note be set true with show_vis together ')
parser.add_argument('--save_vis', action='store_true',
help='whether to save visualization result')
parser.add_argument('--save_npy', action='store_true',
help='whether to save prediction and gt result'
'in npy file')
opt = parser.parse_args()
return opt
def main():
opt = test_parser()
assert opt.fusion_method in ['late', 'early', 'intermediate']
assert not (opt.show_vis and opt.show_sequence), \
'you can only visualize ' \
'the results in single ' \
'image mode or video mode'
hypes = yaml_utils.load_yaml(None, opt)
print('Dataset Building')
opencood_dataset = build_dataset(hypes, visualize=True, train=False)
data_loader = DataLoader(opencood_dataset,
batch_size=1,
num_workers=10,
collate_fn=opencood_dataset.collate_batch_test,
shuffle=False,
pin_memory=False,
drop_last=False)
print('Creating Model')
model = train_utils.create_model(hypes)
# we assume gpu is necessary
if torch.cuda.is_available():
model.cuda()
device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
print('Loading Model from checkpoint')
saved_path = opt.model_dir
_, model = train_utils.load_saved_model(saved_path, model)
model.eval()
# Create the dictionary for evaluation
result_stat = {0.3: {'tp': [], 'fp': [], 'gt': 0},
0.5: {'tp': [], 'fp': [], 'gt': 0},
0.7: {'tp': [], 'fp': [], 'gt': 0}}
if opt.show_sequence:
vis = o3d.visualization.Visualizer()
vis.create_window()
vis.get_render_option().background_color = [0.05, 0.05, 0.05]
vis.get_render_option().point_size = 1.0
vis.get_render_option().show_coordinate_frame = True
# used to visualize lidar points
vis_pcd = o3d.geometry.PointCloud()
# used to visualize object bounding box, maximum 50
vis_aabbs_gt = []
vis_aabbs_pred = []
for _ in range(50):
vis_aabbs_gt.append(o3d.geometry.LineSet())
vis_aabbs_pred.append(o3d.geometry.LineSet())
for i, batch_data in enumerate(data_loader):
print(i)
with torch.no_grad():
torch.cuda.synchronize()
batch_data = train_utils.to_device(batch_data, device)
if opt.fusion_method == 'late':
pred_box_tensor, pred_score, gt_box_tensor = \
infrence_utils.inference_late_fusion(batch_data,
model,
opencood_dataset)
elif opt.fusion_method == 'early':
pred_box_tensor, pred_score, gt_box_tensor = \
infrence_utils.inference_early_fusion(batch_data,
model,
opencood_dataset)
elif opt.fusion_method == 'intermediate':
pred_box_tensor, pred_score, gt_box_tensor = \
infrence_utils.inference_intermediate_fusion(batch_data,
model,
opencood_dataset)
else:
raise NotImplementedError('Only early, late and intermediate'
'fusion is supported.')
eval_utils.caluclate_tp_fp(pred_box_tensor,
pred_score,
gt_box_tensor,
result_stat,
0.3)
eval_utils.caluclate_tp_fp(pred_box_tensor,
pred_score,
gt_box_tensor,
result_stat,
0.5)
eval_utils.caluclate_tp_fp(pred_box_tensor,
pred_score,
gt_box_tensor,
result_stat,
0.7)
if opt.save_npy:
npy_save_path = os.path.join(opt.model_dir, 'npy')
if not os.path.exists(npy_save_path):
os.makedirs(npy_save_path)
infrence_utils.save_prediction_gt(pred_box_tensor,
gt_box_tensor,
batch_data['ego'][
'origin_lidar'][0],
i,
npy_save_path)
if opt.show_vis or opt.save_vis:
vis_save_path = ''
if opt.save_vis:
vis_save_path = os.path.join(opt.model_dir, 'vis')
if not os.path.exists(vis_save_path):
os.makedirs(vis_save_path)
vis_save_path = os.path.join(vis_save_path, '%05d.png' % i)
opencood_dataset.visualize_result(pred_box_tensor,
gt_box_tensor,
batch_data['ego'][
'origin_lidar'][0],
opt.show_vis,
vis_save_path,
dataset=opencood_dataset)
if opt.show_sequence:
pcd, pred_o3d_box, gt_o3d_box = \
vis_utils.visualize_inference_sample_dataloader(
pred_box_tensor,
gt_box_tensor,
batch_data['ego']['origin_lidar'][0],
vis_pcd,
mode='constant'
)
if i == 0:
vis.add_geometry(pcd)
vis_utils.linset_assign_list(vis,
vis_aabbs_pred,
pred_o3d_box,
update_mode='add')
vis_utils.linset_assign_list(vis,
vis_aabbs_gt,
gt_o3d_box,
update_mode='add')
vis_utils.linset_assign_list(vis,
vis_aabbs_pred,
pred_o3d_box)
vis_utils.linset_assign_list(vis,
vis_aabbs_gt,
gt_o3d_box)
vis.update_geometry(pcd)
vis.poll_events()
vis.update_renderer()
time.sleep(0.001)
eval_utils.eval_final_results(result_stat,
opt.model_dir)
if opt.show_sequence:
vis.destroy_window()
if __name__ == '__main__':
main()
import glob
import importlib
import yaml
import os
import re
from datetime import datetime
import torch
import torch.optim as optim
def load_saved_model(saved_path, model):
"""
Load saved model if exiseted
Parameters
__________
saved_path : str
model saved path
model : opencood object
The model instance.
Returns
-------
model : opencood object
The model instance loaded pretrained params.
"""
assert os.path.exists(saved_path), '{} not found'.format(saved_path)
def findLastCheckpoint(save_dir):
file_list = glob.glob(os.path.join(save_dir, '*epoch*.pth'))
if file_list:
epochs_exist = []
for file_ in file_list:
result = re.findall(".*epoch(.*).pth.*", file_)
epochs_exist.append(int(result[0]))
initial_epoch_ = max(epochs_exist)
else:
initial_epoch_ = 0
return initial_epoch_
initial_epoch = findLastCheckpoint(saved_path)
if initial_epoch > 0:
print('resuming by loading epoch %d' % initial_epoch)
model.load_state_dict(torch.load(
os.path.join(saved_path,
'net_epoch%d.pth' % initial_epoch)), strict=False)
return initial_epoch, model
def setup_train(hypes):
"""
Create folder for saved model based on current timestep and model name
Parameters
----------
hypes: dict
Config yaml dictionary for training:
"""
model_name = hypes['name']
current_time = datetime.now()
folder_name = current_time.strftime("_%Y_%m_%d_%H_%M_%S")
folder_name = model_name + folder_name
current_path = os.path.dirname(__file__)
current_path = os.path.join(current_path, '../logs')
full_path = os.path.join(current_path, folder_name)
if not os.path.exists(full_path):
os.makedirs(full_path)
# save the yaml file
save_name = os.path.join(full_path, 'config.yaml')
with open(save_name, 'w') as outfile:
yaml.dump(hypes, outfile)
return full_path
def create_model(hypes):
"""
Import the module "models/[model_name].py
Parameters
__________
hypes : dict
Dictionary containing parameters.
Returns
-------
model : opencood,object
Model object.
"""
backbone_name = hypes['model']['core_method']
backbone_config = hypes['model']['args']
model_filename = "v2xvit.models." + backbone_name
model_lib = importlib.import_module(model_filename)
model = None
target_model_name = backbone_name.replace('_', '')
for name, cls in model_lib.__dict__.items():
if name.lower() == target_model_name.lower():
model = cls
if model is None:
print('backbone not found in models folder. Please make sure you '
'have a python file named %s and has a class '
'called %s ignoring upper/lower case' % (model_filename,
target_model_name))
exit(0)
instance = model(backbone_config)
return instance
def create_loss(hypes):
"""
Create the loss function based on the given loss name.
Parameters
----------
hypes : dict
Configuration params for training.
Returns
-------
criterion : opencood.object
The loss function.
"""
loss_func_name = hypes['loss']['core_method']
loss_func_config = hypes['loss']['args']
loss_filename = "v2xvit.loss." + loss_func_name
loss_lib = importlib.import_module(loss_filename)
loss_func = None
target_loss_name = loss_func_name.replace('_', '')
for name, lfunc in loss_lib.__dict__.items():
if name.lower() == target_loss_name.lower():
loss_func = lfunc
if loss_func is None:
print('loss function not found in loss folder. Please make sure you '
'have a python file named %s and has a class '
'called %s ignoring upper/lower case' % (loss_filename,
target_loss_name))
exit(0)
criterion = loss_func(loss_func_config)
return criterion
def setup_optimizer(hypes, model):
"""
Create optimizer corresponding to the yaml file
Parameters
----------
hypes : dict
The training configurations.
model : opencood model
The pytorch model
"""
method_dict = hypes['optimizer']
optimizer_method = getattr(optim, method_dict['core_method'], None)
if not optimizer_method:
raise ValueError('{} is not supported'.format(method_dict['name']))
if 'args' in method_dict:
return optimizer_method(filter(lambda p: p.requires_grad,
model.parameters()),
lr=method_dict['lr'],
**method_dict['args'])
else:
return optimizer_method(filter(lambda p: p.requires_grad,
model.parameters()),
lr=method_dict['lr'])
def setup_lr_schedular(hypes, optimizer):
"""
Set up the learning rate schedular.
Parameters
----------
hypes : dict
The training configurations.
optimizer : torch.optimizer
"""
lr_schedule_config = hypes['lr_scheduler']
if lr_schedule_config['core_method'] == 'step':
from torch.optim.lr_scheduler import StepLR
step_size = lr_schedule_config['step_size']
gamma = lr_schedule_config['gamma']
scheduler = StepLR(optimizer, step_size=step_size, gamma=gamma)
elif lr_schedule_config['core_method'] == 'multistep':
from torch.optim.lr_scheduler import MultiStepLR
milestones = lr_schedule_config['step_size']
gamma = lr_schedule_config['gamma']
scheduler = MultiStepLR(optimizer,
milestones=milestones,
gamma=gamma)
else:
from torch.optim.lr_scheduler import ExponentialLR
gamma = lr_schedule_config['gamma']
scheduler = ExponentialLR(optimizer, gamma)
return scheduler
def to_device(inputs, device):
if isinstance(inputs, list):
return [to_device(x, device) for x in inputs]
elif isinstance(inputs, dict):
return {k: to_device(v, device) for k, v in inputs.items()}
else:
if isinstance(inputs, int) or isinstance(inputs, float) \
or isinstance(inputs, str):
return inputs
return inputs.to(device)
import argparse
import torch
from torch.utils.data import DataLoader
import v2xvit.hypes_yaml.yaml_utils as yaml_utils
from v2xvit.tools import train_utils
from v2xvit.data_utils.datasets import build_dataset
from v2xvit.visualization import vis_utils
def test_parser():
parser = argparse.ArgumentParser(description="synthetic data generation")
parser.add_argument('--model_dir', type=str, required=True,
help='Continued training path')
parser.add_argument('--fusion_method', type=str, default='late',
help='late, early or intermediate')
opt = parser.parse_args()
return opt
def test_bev_post_processing():
opt = test_parser()
assert opt.fusion_method in ['late', 'early', 'intermediate']
hypes = yaml_utils.load_yaml(None, opt)
print('Dataset Building')
opencood_dataset = build_dataset(hypes, visualize=True, train=False)
data_loader = DataLoader(opencood_dataset,
batch_size=1,
num_workers=0,
collate_fn=opencood_dataset.collate_batch_test,
shuffle=False,
pin_memory=False,
drop_last=False)
print('Creating Model')
model = train_utils.create_model(hypes)
# we assume gpu is necessary
if torch.cuda.is_available():
model.cuda()
device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
print('Loading Model from checkpoint')
saved_path = opt.model_dir
_, model = train_utils.load_saved_model(saved_path, model)
model.eval()
for i, batch_data in enumerate(data_loader):
batch_data = train_utils.to_device(batch_data, device)
label_map = batch_data["ego"]["label_dict"]["label_map"]
output_dict = {
"cls": label_map[:, 0, :, :],
"reg": label_map[:, 1:, :, :]
}
gt_box_tensor, _ = opencood_dataset.post_processor.post_process_debug(
batch_data["ego"], output_dict)
vis_utils.visualize_single_sample_output_bev(gt_box_tensor,
batch_data['ego'][
'origin_lidar'].squeeze(
0),
opencood_dataset)
if __name__ == '__main__':
test_bev_post_processing()
import argparse
import os
import statistics
import torch
import tqdm
from torch.utils.data import DataLoader
from tensorboardX import SummaryWriter
import v2xvit.hypes_yaml.yaml_utils as yaml_utils
from v2xvit.tools import train_utils
from v2xvit.data_utils.datasets import build_dataset
def train_parser():
parser = argparse.ArgumentParser(description="synthetic data generation")
parser.add_argument("--hypes_yaml", type=str, required=True,
help='data generation yaml file needed ')
parser.add_argument('--model_dir', default='',
help='Continued training path')
parser.add_argument("--half", action='store_true', help="whether train with half precision")
opt = parser.parse_args()
return opt
def main():
opt = train_parser()
hypes = yaml_utils.load_yaml(opt.hypes_yaml, opt)
print('Dataset Building')
opencood_train_dataset = build_dataset(hypes, visualize=False, train=True)
opencood_validate_dataset = build_dataset(hypes,
visualize=False,
train=False)
train_loader = DataLoader(opencood_train_dataset,
batch_size=hypes['train_params']['batch_size'],
num_workers=8,
collate_fn=opencood_train_dataset.collate_batch_train,
shuffle=True,
pin_memory=False,
drop_last=True)
val_loader = DataLoader(opencood_validate_dataset,
batch_size=hypes['train_params']['batch_size'],
num_workers=8,
collate_fn=opencood_train_dataset.collate_batch_train,
shuffle=False,
pin_memory=False,
drop_last=True)
print('Creating Model')
model = train_utils.create_model(hypes)
device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
# we assume gpu is necessary
if torch.cuda.is_available():
model.to(device)
# define the loss
criterion = train_utils.create_loss(hypes)
# optimizer setup
optimizer = train_utils.setup_optimizer(hypes, model)
# lr scheduler setup
scheduler = train_utils.setup_lr_schedular(hypes, optimizer)
# if we want to train from last checkpoint.
if opt.model_dir:
saved_path = opt.model_dir
init_epoch, model = train_utils.load_saved_model(saved_path, model)
else:
init_epoch = 0
# if we train the model from scratch, we need to create a folder
# to save the model,
saved_path = train_utils.setup_train(hypes)
# record training
writer = SummaryWriter(saved_path)
# half precision training
if opt.half:
scaler = torch.cuda.amp.GradScaler()
print('Training start')
epoches = hypes['train_params']['epoches']
# used to help schedule learning rate
for epoch in range(init_epoch, max(epoches, init_epoch)):
scheduler.step(epoch)
for param_group in optimizer.param_groups:
print('learning rate %f' % param_group["lr"])
pbar2 = tqdm.tqdm(total=len(train_loader), leave=True)
for i, batch_data in enumerate(train_loader):
# the model will be evaluation mode during validation
model.train()
model.zero_grad()
optimizer.zero_grad()
batch_data = train_utils.to_device(batch_data, device)
# case1 : late fusion train --> only ego needed
# case2 : early fusion train --> all data projected to ego
# case3 : intermediate fusion --> ['ego']['processed_lidar']
# becomes a list, which containing all data from other cavs
# as well
if not opt.half:
ouput_dict = model(batch_data['ego'])
# first argument is always your output dictionary,
# second argument is always your label dictionary.
final_loss = criterion(ouput_dict, batch_data['ego']['label_dict'])
else:
with torch.cuda.amp.autocast():
ouput_dict = model(batch_data['ego'])
final_loss = criterion(ouput_dict, batch_data['ego']['label_dict'])
criterion.logging(epoch, i, len(train_loader), writer, pbar=pbar2)
pbar2.update(1)
# back-propagation
if not opt.half:
final_loss.backward()
optimizer.step()
else:
scaler.scale(final_loss).backward()
scaler.step(optimizer)
scaler.update()
if epoch % hypes['train_params']['eval_freq'] == 0:
valid_ave_loss = []
with torch.no_grad():
for i, batch_data in enumerate(val_loader):
model.eval()
batch_data = train_utils.to_device(batch_data, device)
ouput_dict = model(batch_data['ego'])
final_loss = criterion(ouput_dict,
batch_data['ego']['label_dict'])
valid_ave_loss.append(final_loss.item())
valid_ave_loss = statistics.mean(valid_ave_loss)
print('At epoch %d, the validation loss is %f' % (epoch,
valid_ave_loss))
writer.add_scalar('Validate_Loss', valid_ave_loss, epoch)
if epoch % hypes['train_params']['save_freq'] == 0:
torch.save(model.state_dict(),
os.path.join(saved_path,
'net_epoch%d.pth' % (epoch + 1)))
print('Training Finished, checkpoints saved to %s' % saved_path)
if __name__ == '__main__':
main()
import torch
import torch.nn as nn
import torch.nn.functional as F
class VoxelNetLoss(nn.Module):
def __init__(self, args):
super(VoxelNetLoss, self).__init__()
self.smoothl1loss = nn.SmoothL1Loss(size_average=False)
self.alpha = args['alpha']
self.beta = args['beta']
self.reg_coe = args['reg']
self.loss_dict = {}
def forward(self, output_dict, target_dict):
"""
Parameters
----------
output_dict : dict
target_dict : dict
"""
rm = output_dict['rm']
psm = output_dict['psm']
pos_equal_one = target_dict['pos_equal_one']
neg_equal_one = target_dict['neg_equal_one']
targets = target_dict['targets']
p_pos = F.sigmoid(psm.permute(0, 2, 3, 1))
rm = rm.permute(0, 2, 3, 1).contiguous()
rm = rm.view(rm.size(0), rm.size(1), rm.size(2), -1, 7)
targets = targets.view(targets.size(0), targets.size(1),
targets.size(2), -1, 7)
pos_equal_one_for_reg = pos_equal_one.unsqueeze(
pos_equal_one.dim()).expand(-1, -1, -1, -1, 7)
rm_pos = rm * pos_equal_one_for_reg
targets_pos = targets * pos_equal_one_for_reg
cls_pos_loss = -pos_equal_one * torch.log(p_pos + 1e-6)
cls_pos_loss = cls_pos_loss.sum() / (pos_equal_one.sum() + 1e-6)
cls_neg_loss = -neg_equal_one * torch.log(1 - p_pos + 1e-6)
cls_neg_loss = cls_neg_loss.sum() / (neg_equal_one.sum() + 1e-6)
reg_loss = self.smoothl1loss(rm_pos, targets_pos)
reg_loss = reg_loss / (pos_equal_one.sum() + 1e-6)
conf_loss = self.alpha * cls_pos_loss + self.beta * cls_neg_loss
total_loss = self.reg_coe * reg_loss + conf_loss
self.loss_dict.update({'total_loss': total_loss,
'reg_loss': reg_loss,
'conf_loss': conf_loss})
return total_loss
def logging(self, epoch, batch_id, batch_len, writer):
"""
Print out the loss function for current iteration.
Parameters
----------
epoch : int
Current epoch for training.
batch_id : int
The current batch.
batch_len : int
Total batch length in one iteration of training,
writer : SummaryWriter
Used to visualize on tensorboard
"""
total_loss = self.loss_dict['total_loss']
reg_loss = self.loss_dict['reg_loss']
conf_loss = self.loss_dict['conf_loss']
print("[epoch %d][%d/%d], || Loss: %.4f || Conf Loss: %.4f"
" || Loc Loss: %.4f" % (
epoch, batch_id + 1, batch_len,
total_loss.item(), conf_loss.item(), reg_loss.item()))
writer.add_scalar('Regression_loss', reg_loss.item(),
epoch*batch_len + batch_id)
writer.add_scalar('Confidence_loss', conf_loss.item(),
epoch*batch_len + batch_id)
from functools import reduce
import torch
import torch.nn as nn
import torch.nn.functional as F
class PixorLoss(nn.Module):
def __init__(self, args):
super(PixorLoss, self).__init__()
self.alpha = args["alpha"]
self.beta = args["beta"]
self.loss_dict = {}
def forward(self, output_dict, target_dict):
"""
Compute loss for pixor network
Parameters
----------
output_dict : dict
The dictionary that contains the output.
target_dict : dict
The dictionary that contains the target.
Returns
-------
total_loss : torch.Tensor
Total loss.
"""
targets = target_dict["label_map"]
cls_preds, loc_preds = output_dict["cls"], output_dict["reg"]
cls_targets, loc_targets = targets.split([1, 6], dim=1)
pos_count = cls_targets.sum()
neg_count = (cls_targets == 0).sum()
w1, w2 = neg_count / (pos_count + neg_count), pos_count / (
pos_count + neg_count)
weights = torch.ones_like(cls_preds.reshape(-1))
weights[cls_targets.reshape(-1) == 1] = w1
weights[cls_targets.reshape(-1) == 0] = w2
# cls_targets = cls_targets.float()
# cls_loss = F.binary_cross_entropy_with_logits(input=cls_preds.reshape(-1), target=cls_targets.reshape(-1), weight=weights,
# reduction='mean')
cls_loss = F.binary_cross_entropy_with_logits(
input=cls_preds, target=cls_targets,
reduction='mean')
pos_pixels = cls_targets.sum()
loc_loss = F.smooth_l1_loss(cls_targets * loc_preds,
cls_targets * loc_targets,
reduction='sum')
loc_loss = loc_loss / pos_pixels if pos_pixels > 0 else loc_loss
total_loss = self.alpha * cls_loss + self.beta * loc_loss
self.loss_dict.update({'total_loss': total_loss,
'reg_loss': loc_loss,
'cls_loss': cls_loss})
return total_loss
def logging(self, epoch, batch_id, batch_len, writer):
"""
Print out the loss function for current iteration.
Parameters
----------
epoch : int
Current epoch for training.
batch_id : int
The current batch.
batch_len : int
Total batch length in one iteration of training,
writer : SummaryWriter
Used to visualize on tensorboard
"""
total_loss = self.loss_dict['total_loss']
reg_loss = self.loss_dict['reg_loss']
cls_loss = self.loss_dict['cls_loss']
print("[epoch %d][%d/%d], || Loss: %.4f || cls Loss: %.4f"
" || reg Loss: %.4f" % (
epoch, batch_id + 1, batch_len,
total_loss.item(), cls_loss.item(), reg_loss.item()))
writer.add_scalar('Regression_loss', reg_loss.item(),
epoch * batch_len + batch_id)
writer.add_scalar('Confidence_loss', cls_loss.item(),
epoch * batch_len + batch_id)
def test():
torch.manual_seed(0)
loss = PixorLoss(None)
pred = torch.sigmoid(torch.randn(1, 7, 2, 3))
label = torch.zeros(1, 7, 2, 3)
loss = loss(pred, label)
print(loss)
if __name__ == "__main__":
test()
import torch
import torch.nn as nn
import torch.nn.functional as F
import numpy as np
class WeightedSmoothL1Loss(nn.Module):
"""
Code-wise Weighted Smooth L1 Loss modified based on fvcore.nn.smooth_l1_loss
https://github.com/facebookresearch/fvcore/blob/master/fvcore/nn/smooth_l1_loss.py
| 0.5 * x ** 2 / beta if abs(x) < beta
smoothl1(x) = |
| abs(x) - 0.5 * beta otherwise,
where x = input - target.
"""
def __init__(self, beta: float = 1.0 / 9.0, code_weights: list = None):
"""
Args:
beta: Scalar float.
L1 to L2 change point.
For beta values < 1e-5, L1 loss is computed.
code_weights: (#codes) float list if not None.
Code-wise weights.
"""
super(WeightedSmoothL1Loss, self).__init__()
self.beta = beta
if code_weights is not None:
self.code_weights = np.array(code_weights, dtype=np.float32)
self.code_weights = torch.from_numpy(self.code_weights).cuda()
@staticmethod
def smooth_l1_loss(diff, beta):
if beta < 1e-5:
loss = torch.abs(diff)
else:
n = torch.abs(diff)
loss = torch.where(n < beta, 0.5 * n ** 2 / beta, n - 0.5 * beta)
return loss
def forward(self, input: torch.Tensor,
target: torch.Tensor, weights: torch.Tensor = None):
"""
Args:
input: (B, #anchors, #codes) float tensor.
Ecoded predicted locations of objects.
target: (B, #anchors, #codes) float tensor.
Regression targets.
weights: (B, #anchors) float tensor if not None.
Returns:
loss: (B, #anchors) float tensor.
Weighted smooth l1 loss without reduction.
"""
target = torch.where(torch.isnan(target), input, target) # ignore nan targets
diff = input - target
loss = self.smooth_l1_loss(diff, self.beta)
# anchor-wise weighting
if weights is not None:
assert weights.shape[0] == loss.shape[0] and weights.shape[1] == loss.shape[1]
loss = loss * weights.unsqueeze(-1)
return loss
class PointPillarLoss(nn.Module):
def __init__(self, args):
super(PointPillarLoss, self).__init__()
self.reg_loss_func = WeightedSmoothL1Loss()
self.alpha = 0.25
self.gamma = 2.0
self.cls_weight = args['cls_weight']
self.reg_coe = args['reg']
self.loss_dict = {}
def forward(self, output_dict, target_dict):
"""
Parameters
----------
output_dict : dict
target_dict : dict
"""
rm = output_dict['rm']
psm = output_dict['psm']
targets = target_dict['targets']
cls_preds = psm.permute(0, 2, 3, 1).contiguous()
box_cls_labels = target_dict['pos_equal_one']
box_cls_labels = box_cls_labels.view(psm.shape[0], -1).contiguous()
positives = box_cls_labels > 0
negatives = box_cls_labels == 0
negative_cls_weights = negatives * 1.0
cls_weights = (negative_cls_weights + 1.0 * positives).float()
reg_weights = positives.float()
pos_normalizer = positives.sum(1, keepdim=True).float()
reg_weights /= torch.clamp(pos_normalizer, min=1.0)
cls_weights /= torch.clamp(pos_normalizer, min=1.0)
cls_targets = box_cls_labels
cls_targets = cls_targets.unsqueeze(dim=-1)
cls_targets = cls_targets.squeeze(dim=-1)
one_hot_targets = torch.zeros(
*list(cls_targets.shape), 2,
dtype=cls_preds.dtype, device=cls_targets.device
)
one_hot_targets.scatter_(-1, cls_targets.unsqueeze(dim=-1).long(), 1.0)
cls_preds = cls_preds.view(psm.shape[0], -1, 1)
one_hot_targets = one_hot_targets[..., 1:]
cls_loss_src = self.cls_loss_func(cls_preds,
one_hot_targets,
weights=cls_weights) # [N, M]
cls_loss = cls_loss_src.sum() / psm.shape[0]
conf_loss = cls_loss * self.cls_weight
# regression
rm = rm.permute(0, 2, 3, 1).contiguous()
rm = rm.view(rm.size(0), -1, 7)
targets = targets.view(targets.size(0), -1, 7)
box_preds_sin, reg_targets_sin = self.add_sin_difference(rm,
targets)
loc_loss_src =\
self.reg_loss_func(box_preds_sin,
reg_targets_sin,
weights=reg_weights)
reg_loss = loc_loss_src.sum() / rm.shape[0]
reg_loss *= self.reg_coe
total_loss = reg_loss + conf_loss
self.loss_dict.update({'total_loss': total_loss,
'reg_loss': reg_loss,
'conf_loss': conf_loss})
return total_loss
def cls_loss_func(self, input: torch.Tensor,
target: torch.Tensor,
weights: torch.Tensor):
"""
Args:
input: (B, #anchors, #classes) float tensor.
Predicted logits for each class
target: (B, #anchors, #classes) float tensor.
One-hot encoded classification targets
weights: (B, #anchors) float tensor.
Anchor-wise weights.
Returns:
weighted_loss: (B, #anchors, #classes) float tensor after weighting.
"""
pred_sigmoid = torch.sigmoid(input)
alpha_weight = target * self.alpha + (1 - target) * (1 - self.alpha)
pt = target * (1.0 - pred_sigmoid) + (1.0 - target) * pred_sigmoid
focal_weight = alpha_weight * torch.pow(pt, self.gamma)
bce_loss = self.sigmoid_cross_entropy_with_logits(input, target)
loss = focal_weight * bce_loss
if weights.shape.__len__() == 2 or \
(weights.shape.__len__() == 1 and target.shape.__len__() == 2):
weights = weights.unsqueeze(-1)
assert weights.shape.__len__() == loss.shape.__len__()
return loss * weights
@staticmethod
def sigmoid_cross_entropy_with_logits(input: torch.Tensor, target: torch.Tensor):
""" PyTorch Implementation for tf.nn.sigmoid_cross_entropy_with_logits:
max(x, 0) - x * z + log(1 + exp(-abs(x))) in
https://www.tensorflow.org/api_docs/python/tf/nn/sigmoid_cross_entropy_with_logits
Args:
input: (B, #anchors, #classes) float tensor.
Predicted logits for each class
target: (B, #anchors, #classes) float tensor.
One-hot encoded classification targets
Returns:
loss: (B, #anchors, #classes) float tensor.
Sigmoid cross entropy loss without reduction
"""
loss = torch.clamp(input, min=0) - input * target + \
torch.log1p(torch.exp(-torch.abs(input)))
return loss
@staticmethod
def add_sin_difference(boxes1, boxes2, dim=6):
assert dim != -1
rad_pred_encoding = torch.sin(boxes1[..., dim:dim + 1]) * \
torch.cos(boxes2[..., dim:dim + 1])
rad_tg_encoding = torch.cos(boxes1[..., dim:dim + 1]) * \
torch.sin(boxes2[..., dim:dim + 1])
boxes1 = torch.cat([boxes1[..., :dim], rad_pred_encoding,
boxes1[..., dim + 1:]], dim=-1)
boxes2 = torch.cat([boxes2[..., :dim], rad_tg_encoding,
boxes2[..., dim + 1:]], dim=-1)
return boxes1, boxes2
def logging(self, epoch, batch_id, batch_len, writer, pbar=None):
"""
Print out the loss function for current iteration.
Parameters
----------
epoch : int
Current epoch for training.
batch_id : int
The current batch.
batch_len : int
Total batch length in one iteration of training,
writer : SummaryWriter
Used to visualize on tensorboard
"""
total_loss = self.loss_dict['total_loss']
reg_loss = self.loss_dict['reg_loss']
conf_loss = self.loss_dict['conf_loss']
if pbar is None:
print("[epoch %d][%d/%d], || Loss: %.4f || Conf Loss: %.4f"
" || Loc Loss: %.4f" % (
epoch, batch_id + 1, batch_len,
total_loss.item(), conf_loss.item(), reg_loss.item()))
else:
pbar.set_description("[epoch %d][%d/%d], || Loss: %.4f || Conf Loss: %.4f"
" || Loc Loss: %.4f" % (
epoch, batch_id + 1, batch_len,
total_loss.item(), conf_loss.item(), reg_loss.item()))
writer.add_scalar('Regression_loss', reg_loss.item(),
epoch*batch_len + batch_id)
writer.add_scalar('Confidence_loss', conf_loss.item(),
epoch*batch_len + batch_id)
"""
Common utilities
"""
import numpy as np
import torch
from shapely.geometry import Polygon
def check_numpy_to_torch(x):
if isinstance(x, np.ndarray):
return torch.from_numpy(x).float(), True
return x, False
def check_contain_nan(x):
if isinstance(x, dict):
return any(check_contain_nan(v) for k, v in x.items())
if isinstance(x, list):
return any(check_contain_nan(itm) for itm in x)
if isinstance(x, int) or isinstance(x, float):
return False
if isinstance(x, np.ndarray):
return np.any(np.isnan(x))
return torch.any(x.isnan()).detach().cpu().item()
def rotate_points_along_z(points, angle):
"""
Args:
points: (B, N, 3 + C)
angle: (B), radians, angle along z-axis, angle increases x ==> y
Returns:
"""
points, is_numpy = check_numpy_to_torch(points)
angle, _ = check_numpy_to_torch(angle)
cosa = torch.cos(angle)
sina = torch.sin(angle)
zeros = angle.new_zeros(points.shape[0])
ones = angle.new_ones(points.shape[0])
rot_matrix = torch.stack((
cosa, sina, zeros,
-sina, cosa, zeros,
zeros, zeros, ones
), dim=1).view(-1, 3, 3).float()
points_rot = torch.matmul(points[:, :, 0:3].float(), rot_matrix)
points_rot = torch.cat((points_rot, points[:, :, 3:]), dim=-1)
return points_rot.numpy() if is_numpy else points_rot
def rotate_points_along_z_2d(points, angle):
"""
Rorate the points along z-axis.
Parameters
----------
points : torch.Tensor / np.ndarray
(N, 2).
angle : torch.Tensor / np.ndarray
(N,)
Returns
-------
points_rot : torch.Tensor / np.ndarray
Rorated points with shape (N, 2)
"""
points, is_numpy = check_numpy_to_torch(points)
angle, _ = check_numpy_to_torch(angle)
cosa = torch.cos(angle)
sina = torch.sin(angle)
# (N, 2, 2)
rot_matrix = torch.stack((cosa, sina, -sina, cosa), dim=1).view(-1, 2,
2).float()
points_rot = torch.einsum("ik, ikj->ij", points.float(), rot_matrix)
return points_rot.numpy() if is_numpy else points_rot
def remove_ego_from_objects(objects, ego_id):
"""
Avoid adding ego vehicle to the object dictionary.
Parameters
----------
objects : dict
The dictionary contained all objects.
ego_id : int
Ego id.
"""
if ego_id in objects:
del objects[ego_id]
def retrieve_ego_id(base_data_dict):
"""
Retrieve the ego vehicle id from sample(origin format).
Parameters
----------
base_data_dict : dict
Data sample in origin format.
Returns
-------
ego_id : str
The id of ego vehicle.
"""
ego_id = None
for cav_id, cav_content in base_data_dict.items():
if cav_content['ego']:
ego_id = cav_id
break
return ego_id
def compute_iou(box, boxes):
"""
Compute iou between box and boxes list
Parameters
----------
box : shapely.geometry.Polygon
Bounding box Polygon.
boxes : list
List of shapely.geometry.Polygon.
Returns
-------
iou : np.ndarray
Array of iou between box and boxes.
"""
# Calculate intersection areas
iou = [box.intersection(b).area / box.union(b).area for b in boxes]
return np.array(iou, dtype=np.float32)
def convert_format(boxes_array):
"""
Convert boxes array to shapely.geometry.Polygon format.
Parameters
----------
boxes_array : np.ndarray
(N, 4, 2) or (N, 8, 3).
Returns
-------
list of converted shapely.geometry.Polygon object.
"""
polygons = [Polygon([(box[i, 0], box[i, 1]) for i in range(4)]) for box in
boxes_array]
return np.array(polygons)
def torch_tensor_to_numpy(torch_tensor):
"""
Convert a torch tensor to numpy.
Parameters
----------
torch_tensor : torch.Tensor
Returns
-------
A numpy array.
"""
return torch_tensor.numpy() if not torch_tensor.is_cuda else \
torch_tensor.cpu().detach().numpy()
"""
Transformation utils
"""
import numpy as np
def x_to_world(pose):
"""
The transformation matrix from x-coordinate system to carla world system
Parameters
----------
pose : list
[x, y, z, roll, yaw, pitch]
Returns
-------
matrix : np.ndarray
The transformation matrix.
"""
x, y, z, roll, yaw, pitch = pose[:]
# used for rotation matrix
c_y = np.cos(np.radians(yaw))
s_y = np.sin(np.radians(yaw))
c_r = np.cos(np.radians(roll))
s_r = np.sin(np.radians(roll))
c_p = np.cos(np.radians(pitch))
s_p = np.sin(np.radians(pitch))
matrix = np.identity(4)
# translation matrix
matrix[0, 3] = x
matrix[1, 3] = y
matrix[2, 3] = z
# rotation matrix
matrix[0, 0] = c_p * c_y
matrix[0, 1] = c_y * s_p * s_r - s_y * c_r
matrix[0, 2] = -c_y * s_p * c_r - s_y * s_r
matrix[1, 0] = s_y * c_p
matrix[1, 1] = s_y * s_p * s_r + c_y * c_r
matrix[1, 2] = -s_y * s_p * c_r + c_y * s_r
matrix[2, 0] = s_p
matrix[2, 1] = -c_p * s_r
matrix[2, 2] = c_p * c_r
return matrix
def x1_to_x2(x1, x2):
"""
Transformation matrix from x1 to x2.
Parameters
----------
x1 : list
The pose of x1 under world coordinates.
x2 : list
The pose of x2 under world coordinates.
Returns
-------
transformation_matrix : np.ndarray
The transformation matrix.
"""
x1_to_world = x_to_world(x1)
x2_to_world = x_to_world(x2)
world_to_x2 = np.linalg.inv(x2_to_world)
transformation_matrix = np.dot(world_to_x2, x1_to_world)
return transformation_matrix
def dist_to_continuous(p_dist, displacement_dist, res, downsample_rate):
"""
Convert points discretized format to continuous space for BEV representation.
Parameters
----------
p_dist : numpy.array
Points in discretized coorindates.
displacement_dist : numpy.array
Discretized coordinates of bottom left origin.
res : float
Discretization resolution.
downsample_rate : int
Dowmsamping rate.
Returns
-------
p_continuous : numpy.array
Points in continuous coorindates.
"""
p_dist = np.copy(p_dist)
p_dist = p_dist + displacement_dist
p_continuous = p_dist * res * downsample_rate
return p_continuous
"""
Bounding box related utility functions
"""
import sys
import numpy as np
import torch
import torch.nn.functional as F
import v2xvit.utils.common_utils as common_utils
from v2xvit.utils.transformation_utils import x1_to_x2
def corner_to_center(corner3d, order='lwh'):
"""
Convert 8 corners to x, y, z, dx, dy, dz, yaw.
Parameters
----------
corner3d : np.ndarray
(N, 8, 3)
order : str
'lwh' or 'hwl'
Returns
-------
box3d : np.ndarray
(N, 7)
"""
assert corner3d.ndim == 3
batch_size = corner3d.shape[0]
xyz = np.mean(corner3d[:, [0, 3, 5, 6], :], axis=1)
h = abs(np.mean(corner3d[:, 4:, 2] - corner3d[:, :4, 2], axis=1,
keepdims=True))
l = (np.sqrt(np.sum((corner3d[:, 0, [0, 1]] - corner3d[:, 3, [0, 1]]) ** 2,
axis=1, keepdims=True)) +
np.sqrt(np.sum((corner3d[:, 2, [0, 1]] - corner3d[:, 1, [0, 1]]) ** 2,
axis=1, keepdims=True)) +
np.sqrt(np.sum((corner3d[:, 4, [0, 1]] - corner3d[:, 7, [0, 1]]) ** 2,
axis=1, keepdims=True)) +
np.sqrt(np.sum((corner3d[:, 5, [0, 1]] - corner3d[:, 6, [0, 1]]) ** 2,
axis=1, keepdims=True))) / 4
w = (np.sqrt(
np.sum((corner3d[:, 0, [0, 1]] - corner3d[:, 1, [0, 1]]) ** 2, axis=1,
keepdims=True)) +
np.sqrt(np.sum((corner3d[:, 2, [0, 1]] - corner3d[:, 3, [0, 1]]) ** 2,
axis=1, keepdims=True)) +
np.sqrt(np.sum((corner3d[:, 4, [0, 1]] - corner3d[:, 5, [0, 1]]) ** 2,
axis=1, keepdims=True)) +
np.sqrt(np.sum((corner3d[:, 6, [0, 1]] - corner3d[:, 7, [0, 1]]) ** 2,
axis=1, keepdims=True))) / 4
theta = (np.arctan2(corner3d[:, 1, 1] - corner3d[:, 2, 1],
corner3d[:, 1, 0] - corner3d[:, 2, 0]) +
np.arctan2(corner3d[:, 0, 1] - corner3d[:, 3, 1],
corner3d[:, 0, 0] - corner3d[:, 3, 0]) +
np.arctan2(corner3d[:, 5, 1] - corner3d[:, 6, 1],
corner3d[:, 5, 0] - corner3d[:, 6, 0]) +
np.arctan2(corner3d[:, 4, 1] - corner3d[:, 7, 1],
corner3d[:, 4, 0] - corner3d[:, 7, 0]))[:,
np.newaxis] / 4
if order == 'lwh':
return np.concatenate([xyz, l, w, h, theta], axis=1).reshape(
batch_size, 7)
elif order == 'hwl':
return np.concatenate([xyz, h, w, l, theta], axis=1).reshape(
batch_size, 7)
else:
sys.exit('Unknown order')
def boxes_to_corners2d(boxes3d, order):
"""
0 -------- 1
| |
| |
| |
3 -------- 2
Parameters
__________
boxes3d: np.ndarray or torch.Tensor
(N, 7) [x, y, z, dx, dy, dz, heading], (x, y, z) is the box center.
order : str
'lwh' or 'hwl'
Returns:
corners2d: np.ndarray or torch.Tensor
(N, 4, 3), the 4 corners of the bounding box.
"""
corners3d = boxes_to_corners_3d(boxes3d, order)
corners2d = corners3d[:, :4, :]
return corners2d
def boxes2d_to_corners2d(boxes2d, order="lwh"):
"""
0 -------- 1
| |
| |
| |
3 -------- 2
Parameters
__________
boxes2d: np.ndarray or torch.Tensor
(..., 5) [x, y, dx, dy, heading], (x, y) is the box center.
order : str
'lwh' or 'hwl'
Returns:
corners2d: np.ndarray or torch.Tensor
(..., 4, 2), the 4 corners of the bounding box.
"""
assert order == "lwh", "boxes2d_to_corners_2d only supports lwh order for now."
boxes2d, is_numpy = common_utils.check_numpy_to_torch(boxes2d)
template = boxes2d.new_tensor((
[1, -1], [1, 1], [-1, 1], [-1, -1]
)) / 2
input_shape = boxes2d.shape
boxes2d = boxes2d.view(-1, 5)
corners2d = boxes2d[:, None, 2:4].repeat(1, 4, 1) * template[None, :, :]
corners2d = common_utils.rotate_points_along_z_2d(corners2d.view(-1, 2),
boxes2d[:,
4].repeat_interleave(
4)).view(-1, 4,
2)
corners2d += boxes2d[:, None, 0:2]
corners2d = corners2d.view(*(input_shape[:-1]), 4, 2)
return corners2d
def boxes_to_corners_3d(boxes3d, order):
"""
4 -------- 5
/| /|
7 -------- 6 .
| | | |
. 0 -------- 1
|/ |/
3 -------- 2
Parameters
__________
boxes3d: np.ndarray or torch.Tensor
(N, 7) [x, y, z, dx, dy, dz, heading], (x, y, z) is the box center.
order : str
'lwh' or 'hwl'
Returns:
corners3d: np.ndarray or torch.Tensor
(N, 8, 3), the 8 corners of the bounding box.
"""
# ^ z
# |
# |
# | . x
# |/
# +-------> y
boxes3d, is_numpy = common_utils.check_numpy_to_torch(boxes3d)
if order == 'hwl':
boxes3d[:, 3:6] = boxes3d[:, [5, 4, 3]]
template = boxes3d.new_tensor((
[1, -1, -1], [1, 1, -1], [-1, 1, -1], [-1, -1, -1],
[1, -1, 1], [1, 1, 1], [-1, 1, 1], [-1, -1, 1],
)) / 2
corners3d = boxes3d[:, None, 3:6].repeat(1, 8, 1) * template[None, :, :]
corners3d = common_utils.rotate_points_along_z(corners3d.view(-1, 8, 3),
boxes3d[:, 6]).view(-1, 8,
3)
corners3d += boxes3d[:, None, 0:3]
return corners3d.numpy() if is_numpy else corners3d
def box3d_to_2d(box3d):
"""
Convert 3D bounding box to 2D.
Parameters
----------
box3d : np.ndarray
(n, 8, 3)
Returns
-------
box2d : np.ndarray
(n, 4, 2), project 3d to 2d.
"""
box2d = box3d[:, :4, :2]
return box2d
def corner2d_to_standup_box(box2d):
"""
Find the minmaxx, minmaxy for each 2d box. (N, 4, 2) -> (N, 4)
x1, y1, x2, y2
Parameters
----------
box2d : np.ndarray
(n, 4, 2), four corners of the 2d bounding box.
Returns
-------
standup_box2d : np.ndarray
(n, 4)
"""
N = box2d.shape[0]
standup_boxes2d = np.zeros((N, 4))
standup_boxes2d[:, 0] = np.min(box2d[:, :, 0], axis=1)
standup_boxes2d[:, 1] = np.min(box2d[:, :, 1], axis=1)
standup_boxes2d[:, 2] = np.max(box2d[:, :, 0], axis=1)
standup_boxes2d[:, 3] = np.max(box2d[:, :, 1], axis=1)
return standup_boxes2d
def corner_to_standup_box_torch(box_corner):
"""
Find the minmax x and y for each bounding box.
Parameters
----------
box_corner : torch.Tensor
Shape: (N, 8, 3) or (N, 4)
Returns
-------
standup_box2d : torch.Tensor
(n, 4)
"""
N = box_corner.shape[0]
standup_boxes2d = torch.zeros((N, 4))
standup_boxes2d = standup_boxes2d.to(box_corner.device)
standup_boxes2d[:, 0] = torch.min(box_corner[:, :, 0], dim=1).values
standup_boxes2d[:, 1] = torch.min(box_corner[:, :, 1], dim=1).values
standup_boxes2d[:, 2] = torch.max(box_corner[:, :, 0], dim=1).values
standup_boxes2d[:, 3] = torch.max(box_corner[:, :, 1], dim=1).values
return standup_boxes2d
def project_box3d(box3d, transformation_matrix):
"""
Project the 3d bounding box to another coordinate system based on the
transfomration matrix.
Parameters
----------
box3d : torch.Tensor or np.ndarray
3D bounding box, (N, 8, 3)
transformation_matrix : torch.Tensor or np.ndarray
Transformation matrix, (4, 4)
Returns
-------
projected_box3d : torch.Tensor
The projected bounding box, (N, 8, 3)
"""
assert transformation_matrix.shape == (4, 4)
box3d, is_numpy = \
common_utils.check_numpy_to_torch(box3d)
transformation_matrix, _ = \
common_utils.check_numpy_to_torch(transformation_matrix)
# (N, 3, 8)
box3d_corner = box3d.transpose(1, 2)
# (N, 1, 8)
torch_ones = torch.ones((box3d_corner.shape[0], 1, 8))
torch_ones = torch_ones.to(box3d_corner.device)
# (N, 4, 8)
box3d_corner = torch.cat((box3d_corner, torch_ones),
dim=1)
# (N, 4, 8)
projected_box3d = torch.matmul(transformation_matrix,
box3d_corner)
# (N, 8, 3)
projected_box3d = projected_box3d[:, :3, :].transpose(1, 2)
return projected_box3d if not is_numpy else projected_box3d.numpy()
def project_points_by_matrix_torch(points, transformation_matrix):
"""
Project the points to another coordinate system based on the
transfomration matrix.
Parameters
----------
points : torch.Tensor
3D points, (N, 3)
transformation_matrix : torch.Tensor
Transformation matrix, (4, 4)
Returns
-------
projected_points : torch.Tensor
The projected points, (N, 3)
"""
# convert to homogeneous coordinates via padding 1 at the last dimension.
# (N, 4)
points_homogeneous = F.pad(points, (0, 1), mode="constant", value=1)
# (N, 4)
projected_points = torch.einsum("ik, jk->ij", points_homogeneous,
transformation_matrix)
return projected_points[:, :3]
def get_mask_for_boxes_within_range_torch(boxes):
"""
Generate mask to remove the bounding boxes
outside the range.
Parameters
----------
boxes : torch.Tensor
Groundtruth bbx, shape: N,8,3 or N,4,2
Returns
-------
mask: torch.Tensor
The mask for bounding box -- True means the
bbx is within the range and False means the
bbx is outside the range.
"""
from v2xvit.data_utils.datasets import GT_RANGE
# mask out the gt bounding box out fixed range (-140, -40, -3, 140, 40 1)
device = boxes.device
boundary_lower_range = \
torch.Tensor(GT_RANGE[:2]).reshape(1, 1, -1).to(device)
boundary_higher_range = \
torch.Tensor(GT_RANGE[3:5]).reshape(1, 1, -1).to(device)
mask = torch.all(
torch.all(boxes[:, :, :2] >= boundary_lower_range,
dim=-1) & \
torch.all(boxes[:, :, :2] <= boundary_higher_range,
dim=-1), dim=-1)
return mask
def mask_boxes_outside_range_numpy(boxes, limit_range, order,
min_num_corners=8):
"""
Parameters
----------
boxes: np.ndarray
(N, 7) [x, y, z, dx, dy, dz, heading], (x, y, z) is the box center
limit_range: list
[minx, miny, minz, maxx, maxy, maxz]
min_num_corners: int
The required minimum number of corners to be considered as in range.
order : str
'lwh' or 'hwl'
Returns
-------
boxes: np.ndarray
The filtered boxes.
"""
assert boxes.shape[1] == 8 or boxes.shape[1] == 7
new_boxes = boxes.copy()
if boxes.shape[1] == 7:
new_boxes = boxes_to_corners_3d(new_boxes, order)
mask = ((new_boxes >= limit_range[0:3]) &
(new_boxes <= limit_range[3:6])).all(axis=2)
mask = mask.sum(axis=1) >= min_num_corners # (N)
return boxes[mask]
def create_bbx(extent):
"""
Create bounding box with 8 corners under obstacle vehicle reference.
Parameters
----------
extent : list
Width, height, length of the bbx.
Returns
-------
bbx : np.array
The bounding box with 8 corners, shape: (8, 3)
"""
bbx = np.array([[extent[0], -extent[1], -extent[2]],
[extent[0], extent[1], -extent[2]],
[-extent[0], extent[1], -extent[2]],
[-extent[0], -extent[1], -extent[2]],
[extent[0], -extent[1], extent[2]],
[extent[0], extent[1], extent[2]],
[-extent[0], extent[1], extent[2]],
[-extent[0], -extent[1], extent[2]]])
return bbx
def project_world_objects(object_dict,
output_dict,
lidar_pose,
lidar_range,
order):
"""
Project the objects under world coordinates into another coordinate
based on the provided extrinsic.
Parameters
----------
object_dict : dict
The dictionary contains all objects surrounding a certain cav.
output_dict : dict
key: object id, value: object bbx (xyzlwhyaw).
lidar_pose : list
(6, ), lidar pose under world coordinate, [x, y, z, roll, yaw, pitch].
lidar_range : list
[minx, miny, minz, maxx, maxy, maxz]
order : str
'lwh' or 'hwl'
"""
for object_id, object_content in object_dict.items():
location = object_content['location']
rotation = object_content['angle']
center = object_content['center']
extent = object_content['extent']
object_pose = [location[0] + center[0],
location[1] + center[1],
location[2] + center[2],
rotation[0], rotation[1], rotation[2]]
object2lidar = x1_to_x2(object_pose, lidar_pose)
# shape (3, 8)
bbx = create_bbx(extent).T
# bounding box under ego coordinate shape (4, 8)
bbx = np.r_[bbx, [np.ones(bbx.shape[1])]]
# project the 8 corners to world coordinate
bbx_lidar = np.dot(object2lidar, bbx).T
bbx_lidar = np.expand_dims(bbx_lidar[:, :3], 0)
bbx_lidar = corner_to_center(bbx_lidar, order=order)
bbx_lidar = mask_boxes_outside_range_numpy(bbx_lidar,
lidar_range,
order)
if bbx_lidar.shape[0] > 0:
output_dict.update({object_id: bbx_lidar})
def get_points_in_rotated_box(p, box_corner):
"""
Get points within a rotated bounding box (2D version).
Parameters
----------
p : numpy.array
Points to be tested with shape (N, 2).
box_corner : numpy.array
Corners of bounding box with shape (4, 2).
Returns
-------
p_in_box : numpy.array
Points within the box.
"""
edge1 = box_corner[1, :] - box_corner[0, :]
edge2 = box_corner[3, :] - box_corner[0, :]
p_rel = p - box_corner[0, :].reshape(1, -1)
l1 = get_projection_length_for_vector_projection(p_rel, edge1)
l2 = get_projection_length_for_vector_projection(p_rel, edge2)
# A point is within the box, if and only after projecting the
# point onto the two edges s.t. p_rel = [edge1, edge2] @ [l1, l2]^T,
# we have 0<=l1<=1 and 0<=l2<=1.
mask = np.logical_and(l1 >= 0, l1 <= 1)
mask = np.logical_and(mask, l2 >= 0)
mask = np.logical_and(mask, l2 <= 1)
p_in_box = p[mask, :]
return p_in_box
def get_points_in_rotated_box_3d(p, box_corner):
"""
Get points within a rotated bounding box (3D version).
Parameters
----------
p : numpy.array
Points to be tested with shape (N, 3).
box_corner : numpy.array
Corners of bounding box with shape (8, 3).
Returns
-------
p_in_box : numpy.array
Points within the box.
"""
edge1 = box_corner[1, :] - box_corner[0, :]
edge2 = box_corner[3, :] - box_corner[0, :]
edge3 = box_corner[4, :] - box_corner[0, :]
p_rel = p - box_corner[0, :].reshape(1, -1)
l1 = get_projection_length_for_vector_projection(p_rel, edge1)
l2 = get_projection_length_for_vector_projection(p_rel, edge2)
l3 = get_projection_length_for_vector_projection(p_rel, edge3)
# A point is within the box, if and only after projecting the
# point onto the two edges s.t. p_rel = [edge1, edge2] @ [l1, l2]^T,
# we have 0<=l1<=1 and 0<=l2<=1.
mask1 = np.logical_and(l1 >= 0, l1 <= 1)
mask2 = np.logical_and(l2 >= 0, l2 <= 1)
mask3 = np.logical_and(l3 >= 0, l3 <= 1)
mask = np.logical_and(mask1, mask2)
mask = np.logical_and(mask, mask3)
p_in_box = p[mask, :]
return p_in_box
def get_projection_length_for_vector_projection(a, b):
"""
Get projection length for the Vector projection of a onto b s.t.
a_projected = length * b. (2D version) See
https://en.wikipedia.org/wiki/Vector_projection#Vector_projection_2
for more details.
Parameters
----------
a : numpy.array
The vectors to be projected with shape (N, 2).
b : numpy.array
The vector that is projected onto with shape (2).
Returns
-------
length : numpy.array
The length of projected a with respect to b.
"""
assert np.sum(b ** 2, axis=-1) > 1e-6
length = a.dot(b) / np.sum(b ** 2, axis=-1)
return length
def nms_rotated(boxes, scores, threshold):
"""Performs rorated non-maximum suppression and returns indices of kept
boxes.
Parameters
----------
boxes : torch.tensor
The location preds with shape (N, 4, 2).
scores : torch.tensor
The predicted confidence score with shape (N,)
threshold: float
IoU threshold to use for filtering.
Returns
-------
An array of index
"""
if boxes.shape[0] == 0:
return np.array([], dtype=np.int32)
boxes = boxes.cpu().detach().numpy()
scores = scores.cpu().detach().numpy()
polygons = common_utils.convert_format(boxes)
top = 1000
# Get indicies of boxes sorted by scores (highest first)
ixs = scores.argsort()[::-1][:top]
pick = []
while len(ixs) > 0:
# Pick top box and add its index to the list
i = ixs[0]
pick.append(i)
# Compute IoU of the picked box with the rest
iou = common_utils.compute_iou(polygons[i], polygons[ixs[1:]])
# Identify boxes with IoU over the threshold. This
# returns indices into ixs[1:], so add 1 to get
# indices into ixs.
remove_ixs = np.where(iou > threshold)[0] + 1
# Remove indices of the picked and overlapped boxes.
ixs = np.delete(ixs, remove_ixs)
ixs = np.delete(ixs, 0)
return np.array(pick, dtype=np.int32)
def nms_pytorch(boxes: torch.tensor, thresh_iou: float):
"""
Apply non-maximum suppression to avoid detecting too many
overlapping bounding boxes for a given object.
Parameters
----------
boxes : torch.tensor
The location preds along with the class predscores,
Shape: [num_boxes,5].
thresh_iou : float
(float) The overlap thresh for suppressing unnecessary boxes.
Returns
-------
A list of index
"""
# we extract coordinates for every
# prediction box present in P
x1 = boxes[:, 0]
y1 = boxes[:, 1]
x2 = boxes[:, 2]
y2 = boxes[:, 3]
# we extract the confidence scores as well
scores = boxes[:, 4]
# calculate area of every block in P
areas = (x2 - x1) * (y2 - y1)
# sort the prediction boxes in P
# according to their confidence scores
order = scores.argsort()
# initialise an empty list for
# filtered prediction boxes
keep = []
while len(order) > 0:
# extract the index of the
# prediction with highest score
# we call this prediction S
idx = order[-1]
# push S in filtered predictions list
keep.append(idx.numpy().item()
if not idx.is_cuda else idx.cpu().detach().numpy().item())
# remove S from P
order = order[:-1]
# sanity check
if len(order) == 0:
break
# select coordinates of BBoxes according to
# the indices in order
xx1 = torch.index_select(x1, dim=0, index=order)
xx2 = torch.index_select(x2, dim=0, index=order)
yy1 = torch.index_select(y1, dim=0, index=order)
yy2 = torch.index_select(y2, dim=0, index=order)
# find the coordinates of the intersection boxes
xx1 = torch.max(xx1, x1[idx])
yy1 = torch.max(yy1, y1[idx])
xx2 = torch.min(xx2, x2[idx])
yy2 = torch.min(yy2, y2[idx])
# find height and width of the intersection boxes
w = xx2 - xx1
h = yy2 - yy1
# take max with 0.0 to avoid negative w and h
# due to non-overlapping boxes
w = torch.clamp(w, min=0.0)
h = torch.clamp(h, min=0.0)
# find the intersection area
inter = w * h
# find the areas of BBoxes according the indices in order
rem_areas = torch.index_select(areas, dim=0, index=order)
# find the union of every prediction T in P
# with the prediction S
# Note that areas[idx] represents area of S
union = (rem_areas - inter) + areas[idx]
# find the IoU of every prediction in P with S
IoU = inter / union
# keep the boxes with IoU less than thresh_iou
mask = IoU < thresh_iou
order = order[mask]
return keep
def remove_large_pred_bbx(bbx_3d):
"""
Remove large bounding box.
Parameters
----------
bbx_3d : torch.Tensor
Predcited 3d bounding box, shape:(N,8,3)
Returns
-------
index : torch.Tensor
The keep index.
"""
bbx_x_max = torch.max(bbx_3d[:, :, 0], dim=1)[0]
bbx_x_min = torch.min(bbx_3d[:, :, 0], dim=1)[0]
x_len = bbx_x_max - bbx_x_min
bbx_y_max = torch.max(bbx_3d[:, :, 1], dim=1)[0]
bbx_y_min = torch.min(bbx_3d[:, :, 1], dim=1)[0]
y_len = bbx_y_max - bbx_y_min
bbx_z_max = torch.max(bbx_3d[:, :, 1], dim=1)[0]
bbx_z_min = torch.min(bbx_3d[:, :, 1], dim=1)[0]
z_len = bbx_z_max - bbx_z_min
index = torch.logical_and(x_len <= 6, y_len <= 6)
index = torch.logical_and(index, z_len)
return index
def remove_bbx_abnormal_z(bbx_3d):
"""
Remove bounding box that has negative z axis.
Parameters
----------
bbx_3d : torch.Tensor
Predcited 3d bounding box, shape:(N,8,3)
Returns
-------
index : torch.Tensor
The keep index.
"""
bbx_z_min = torch.min(bbx_3d[:, :, 2], dim=1)[0]
bbx_z_max = torch.max(bbx_3d[:, :, 2], dim=1)[0]
index = torch.logical_and(bbx_z_min >= -3, bbx_z_max <= 1)
return index
def project_points_by_matrix_torch(points, transformation_matrix):
"""
Project the points to another coordinate system based on the
transformation matrix.
Parameters
----------
points : torch.Tensor
3D points, (N, 3)
transformation_matrix : torch.Tensor
Transformation matrix, (4, 4)
Returns
-------
projected_points : torch.Tensor
The projected points, (N, 3)
"""
points, is_numpy = \
common_utils.check_numpy_to_torch(points)
transformation_matrix, _ = \
common_utils.check_numpy_to_torch(transformation_matrix)
# convert to homogeneous coordinates via padding 1 at the last dimension.
# (N, 4)
points_homogeneous = F.pad(points, (0, 1), mode="constant", value=1)
# (N, 4)
projected_points = torch.einsum("ik, jk->ij", points_homogeneous,
transformation_matrix)
return projected_points[:, :3] if not is_numpy \
else projected_points[:, :3].numpy()
if __name__ == "__main__":
x = np.arange(-5, 5, 0.1)
y = np.arange(-5, 5, 0.1)
xx, yy = np.meshgrid(x, y)
points = np.concatenate([xx.reshape(-1, 1), yy.reshape(-1, 1)], axis=-1)
box_corners = np.array([
[2, -2], [2, 2], [-2, 2], [-2, -2]
])
temp = get_points_in_rotated_box(points, box_corners)
assert np.all(np.logical_and(temp[:, 0] >= -2, temp[:, 0] <= 2))
assert np.all(np.logical_and(temp[:, 1] >= -2, temp[:, 1] <= 2))
from distutils.core import setup
from Cython.Build import cythonize
import numpy
setup(
name='box overlaps',
ext_modules=cythonize('v2xvit/utils/box_overlaps.pyx'),
include_dirs=[numpy.get_include()]
)
"""
Utility functions related to point cloud
"""
import open3d as o3d
import numpy as np
def pcd_to_np(pcd_file):
"""
Read pcd and return numpy array.
Parameters
----------
pcd_file : str
The pcd file that contains the point cloud.
Returns
-------
pcd : o3d.PointCloud
PointCloud object, used for visualization
pcd_np : np.ndarray
The lidar data in numpy format, shape:(n, 4)
"""
pcd = o3d.io.read_point_cloud(pcd_file)
xyz = np.asarray(pcd.points)
# we save the intensity in the first channel
intensity = np.expand_dims(np.asarray(pcd.colors)[:, 0], -1)
pcd_np = np.hstack((xyz, intensity))
return np.asarray(pcd_np, dtype=np.float32)
def mask_points_by_range(points, limit_range):
"""
Remove the lidar points out of the boundary.
Parameters
----------
points : np.ndarray
Lidar points under lidar sensor coordinate system.
limit_range : list
[x_min, y_min, z_min, x_max, y_max, z_max]
Returns
-------
points : np.ndarray
Filtered lidar points.
"""
mask = (points[:, 0] > limit_range[0]) & (points[:, 0] < limit_range[3])\
& (points[:, 1] > limit_range[1]) & (
points[:, 1] < limit_range[4]) \
& (points[:, 2] > limit_range[2]) & (
points[:, 2] < limit_range[5])
points = points[mask]
return points
def mask_ego_points(points):
"""
Remove the lidar points of the ego vehicle itself.
Parameters
----------
points : np.ndarray
Lidar points under lidar sensor coordinate system.
Returns
-------
points : np.ndarray
Filtered lidar points.
"""
mask = (points[:, 0] >= -1.95) & (points[:, 0] <= 2.95) \
& (points[:, 1] >= -1.1) & (points[:, 1] <= 1.1)
points = points[np.logical_not(mask)]
return points
def shuffle_points(points):
shuffle_idx = np.random.permutation(points.shape[0])
points = points[shuffle_idx]
return points
def lidar_project(lidar_data, extrinsic):
"""
Given the extrinsic matrix, project lidar data to another space.
Parameters
----------
lidar_data : np.ndarray
Lidar data, shape: (n, 4)
extrinsic : np.ndarray
Extrinsic matrix, shape: (4, 4)
Returns
-------
projected_lidar : np.ndarray
Projected lida data, shape: (n, 4)
"""
lidar_xyz = lidar_data[:, :3].T
# (3, n) -> (4, n), homogeneous transformation
lidar_xyz = np.r_[lidar_xyz, [np.ones(lidar_xyz.shape[1])]]
lidar_int = lidar_data[:, 3]
# transform to ego vehicle space, (3, n)
project_lidar_xyz = np.dot(extrinsic, lidar_xyz)[:3, :]
# (n, 3)
project_lidar_xyz = project_lidar_xyz.T
# concatenate the intensity with xyz, (n, 4)
projected_lidar = np.hstack((project_lidar_xyz,
np.expand_dims(lidar_int, -1)))
return projected_lidar
def projected_lidar_stack(projected_lidar_list):
"""
Stack all projected lidar together.
Parameters
----------
projected_lidar_list : list
The list containing all projected lidar.
Returns
-------
stack_lidar : np.ndarray
Stack all projected lidar data together.
"""
stack_lidar = []
for lidar_data in projected_lidar_list:
stack_lidar.append(lidar_data)
return np.vstack(stack_lidar)
def downsample_lidar(pcd_np, num):
"""
Downsample the lidar points to a certain number.
Parameters
----------
pcd_np : np.ndarray
The lidar points, (n, 4).
num : int
The downsample target number.
Returns
-------
pcd_np : np.ndarray
The downsampled lidar points.
"""
assert pcd_np.shape[0] >= num
selected_index = np.random.choice((pcd_np.shape[0]),
num,
replace=False)
pcd_np = pcd_np[selected_index]
return pcd_np
def downsample_lidar_minimum(pcd_np_list):
"""
Given a list of pcd, find the minimum number and downsample all
point clouds to the minimum number.
Parameters
----------
pcd_np_list : list
A list of pcd numpy array(n, 4).
Returns
-------
pcd_np_list : list
Downsampled point clouds.
"""
minimum = np.Inf
for i in range(len(pcd_np_list)):
num = pcd_np_list[i].shape[0]
minimum = num if minimum > num else minimum
for (i, pcd_np) in enumerate(pcd_np_list):
pcd_np_list[i] = downsample_lidar(pcd_np, minimum)
return pcd_np_list
import os
import numpy as np
import torch
from v2xvit.utils import common_utils
from v2xvit.hypes_yaml import yaml_utils
def voc_ap(rec, prec):
"""
VOC 2010 Average Precision.
"""
rec.insert(0, 0.0)
rec.append(1.0)
mrec = rec[:]
prec.insert(0, 0.0)
prec.append(0.0)
mpre = prec[:]
for i in range(len(mpre) - 2, -1, -1):
mpre[i] = max(mpre[i], mpre[i + 1])
i_list = []
for i in range(1, len(mrec)):
if mrec[i] != mrec[i - 1]:
i_list.append(i)
ap = 0.0
for i in i_list:
ap += ((mrec[i] - mrec[i - 1]) * mpre[i])
return ap, mrec, mpre
def caluclate_tp_fp(det_boxes, det_score, gt_boxes, result_stat, iou_thresh):
"""
Calculate the true positive and false positive numbers of the current
frames.
Parameters
----------
det_boxes : torch.Tensor
The detection bounding box, shape (N, 8, 3) or (N, 4, 2).
det_score :torch.Tensor
The confidence score for each preditect bounding box.
gt_boxes : torch.Tensor
The groundtruth bounding box.
result_stat: dict
A dictionary contains fp, tp and gt number.
iou_thresh : float
The iou thresh.
"""
# fp, tp and gt in the current frame
fp = []
tp = []
gt = gt_boxes.shape[0]
if det_boxes is not None:
# convert bounding boxes to numpy array
det_boxes = common_utils.torch_tensor_to_numpy(det_boxes)
det_score = common_utils.torch_tensor_to_numpy(det_score)
gt_boxes = common_utils.torch_tensor_to_numpy(gt_boxes)
# sort the prediction bounding box by score
score_order_descend = np.argsort(-det_score)
det_polygon_list = list(common_utils.convert_format(det_boxes))
gt_polygon_list = list(common_utils.convert_format(gt_boxes))
# match prediction and gt bounding box
for i in range(score_order_descend.shape[0]):
det_polygon = det_polygon_list[score_order_descend[i]]
ious = common_utils.compute_iou(det_polygon, gt_polygon_list)
if len(gt_polygon_list) == 0 or np.max(ious) < iou_thresh:
fp.append(1)
tp.append(0)
continue
fp.append(0)
tp.append(1)
gt_index = np.argmax(ious)
gt_polygon_list.pop(gt_index)
result_stat[iou_thresh]['fp'] += fp
result_stat[iou_thresh]['tp'] += tp
result_stat[iou_thresh]['gt'] += gt
def calculate_ap(result_stat, iou):
"""
Calculate the average precision and recall, and save them into a txt.
Parameters
----------
result_stat : dict
A dictionary contains fp, tp and gt number.
iou : float
"""
iou_5 = result_stat[iou]
fp = iou_5['fp']
tp = iou_5['tp']
assert len(fp) == len(tp)
gt_total = iou_5['gt']
cumsum = 0
for idx, val in enumerate(fp):
fp[idx] += cumsum
cumsum += val
cumsum = 0
for idx, val in enumerate(tp):
tp[idx] += cumsum
cumsum += val
rec = tp[:]
for idx, val in enumerate(tp):
rec[idx] = float(tp[idx]) / gt_total
prec = tp[:]
for idx, val in enumerate(tp):
prec[idx] = float(tp[idx]) / (fp[idx] + tp[idx])
ap, mrec, mprec = voc_ap(rec[:], prec[:])
return ap, mrec, mprec
def eval_final_results(result_stat, save_path):
dump_dict = {}
ap_30, mrec_30, mpre_30 = calculate_ap(result_stat, 0.30)
ap_50, mrec_50, mpre_50 = calculate_ap(result_stat, 0.50)
ap_70, mrec_70, mpre_70 = calculate_ap(result_stat, 0.70)
dump_dict.update({'ap30': ap_30,
'ap_50': ap_50,
'ap_70': ap_70,
'mpre_50': mpre_50,
'mrec_50': mrec_50,
'mpre_70': mpre_70,
'mrec_70': mrec_70,
})
yaml_utils.save_yaml(dump_dict, os.path.join(save_path, 'eval.yaml'))
print('The Average Precision at IOU 0.3 is %.2f, '
'The Average Precision at IOU 0.5 is %.2f, '
'The Average Precision at IOU 0.7 is %.2f' % (ap_30, ap_50, ap_70))
{
"class_name" : "PinholeCameraParameters",
"extrinsic" :
[
1.0,
-0.0,
-0.0,
0.0,
0.0,
-1.0,
-0.0,
0.0,
0.0,
-0.0,
-1.0,
0.0,
14.870189666748047,
0.0001621246337890625,
141.0903074604017,
1.0
],
"intrinsic" :
{
"height" : 1025,
"intrinsic_matrix" :
[
887.67603887904966,
0.0,
0.0,
0.0,
887.67603887904966,
0.0,
926.0,
512.0,
1.0
],
"width" : 1853
},
"version_major" : 1,
"version_minor" : 0
}
import time
import cv2
import numpy as np
import open3d as o3d
import matplotlib
import matplotlib.pyplot as plt
from matplotlib import cm
from v2xvit.utils import box_utils
from v2xvit.utils import common_utils
VIRIDIS = np.array(cm.get_cmap('plasma').colors)
VID_RANGE = np.linspace(0.0, 1.0, VIRIDIS.shape[0])
def bbx2linset(bbx_corner, order='hwl', color=(0, 1, 0)):
"""
Convert the torch tensor bounding box to o3d lineset for visualization.
Parameters
----------
bbx_corner : torch.Tensor
shape: (n, 8, 3).
order : str
The order of the bounding box if shape is (n, 7)
color : tuple
The bounding box color.
Returns
-------
line_set : list
The list containing linsets.
"""
if not isinstance(bbx_corner, np.ndarray):
bbx_corner = common_utils.torch_tensor_to_numpy(bbx_corner)
if len(bbx_corner.shape) == 2:
bbx_corner = box_utils.boxes_to_corners_3d(bbx_corner,
order)
# Our lines span from points 0 to 1, 1 to 2, 2 to 3, etc...
lines = [[0, 1], [1, 2], [2, 3], [0, 3],
[4, 5], [5, 6], [6, 7], [4, 7],
[0, 4], [1, 5], [2, 6], [3, 7]]
# Use the same color for all lines
colors = [list(color) for _ in range(len(lines))]
bbx_linset = []
for i in range(bbx_corner.shape[0]):
bbx = bbx_corner[i]
# o3d use right-hand coordinate
bbx[:, :1] = - bbx[:, :1]
line_set = o3d.geometry.LineSet()
line_set.points = o3d.utility.Vector3dVector(bbx)
line_set.lines = o3d.utility.Vector2iVector(lines)
line_set.colors = o3d.utility.Vector3dVector(colors)
bbx_linset.append(line_set)
return bbx_linset
def bbx2oabb(bbx_corner, order='hwl', color=(0, 0, 1)):
"""
Convert the torch tensor bounding box to o3d oabb for visualization.
Parameters
----------
bbx_corner : torch.Tensor
shape: (n, 8, 3).
order : str
The order of the bounding box if shape is (n, 7)
color : tuple
The bounding box color.
Returns
-------
oabbs : list
The list containing all oriented bounding boxes.
"""
if not isinstance(bbx_corner, np.ndarray):
bbx_corner = common_utils.torch_tensor_to_numpy(bbx_corner)
if len(bbx_corner.shape) == 2:
bbx_corner = box_utils.boxes_to_corners_3d(bbx_corner,
order)
oabbs = []
for i in range(bbx_corner.shape[0]):
bbx = bbx_corner[i]
# o3d use right-hand coordinate
bbx[:, :1] = - bbx[:, :1]
tmp_pcd = o3d.geometry.PointCloud()
tmp_pcd.points = o3d.utility.Vector3dVector(bbx)
oabb = tmp_pcd.get_oriented_bounding_box()
oabb.color = color
oabbs.append(oabb)
return oabbs
def bbx2aabb(bbx_center, order):
"""
Convert the torch tensor bounding box to o3d aabb for visualization.
Parameters
----------
bbx_center : torch.Tensor
shape: (n, 7).
order: str
hwl or lwh.
Returns
-------
aabbs : list
The list containing all o3d.aabb
"""
if not isinstance(bbx_center, np.ndarray):
bbx_center = common_utils.torch_tensor_to_numpy(bbx_center)
bbx_corner = box_utils.boxes_to_corners_3d(bbx_center, order)
aabbs = []
for i in range(bbx_corner.shape[0]):
bbx = bbx_corner[i]
# o3d use right-hand coordinate
bbx[:, :1] = - bbx[:, :1]
tmp_pcd = o3d.geometry.PointCloud()
tmp_pcd.points = o3d.utility.Vector3dVector(bbx)
aabb = tmp_pcd.get_axis_aligned_bounding_box()
aabb.color = (0, 0, 1)
aabbs.append(aabb)
return aabbs
def linset_assign_list(vis,
lineset_list1,
lineset_list2,
update_mode='update'):
"""
Associate two lists of lineset.
Parameters
----------
vis : open3d.Visualizer
lineset_list1 : list
lineset_list2 : list
update_mode : str
Add or update the geometry.
"""
for j in range(len(lineset_list1)):
index = j if j < len(lineset_list2) else -1
lineset_list1[j] = \
lineset_assign(lineset_list1[j],
lineset_list2[index])
if update_mode == 'add':
vis.add_geometry(lineset_list1[j])
else:
vis.update_geometry(lineset_list1[j])
def lineset_assign(lineset1, lineset2):
"""
Assign the attributes of lineset2 to lineset1.
Parameters
----------
lineset1 : open3d.LineSet
lineset2 : open3d.LineSet
Returns
-------
The lineset1 object with 2's attributes.
"""
lineset1.points = lineset2.points
lineset1.lines = lineset2.lines
lineset1.colors = lineset2.colors
return lineset1
def color_encoding(intensity, mode='intensity'):
"""
Encode the single-channel intensity to 3 channels rgb color.
Parameters
----------
intensity : np.ndarray
Lidar intensity, shape (n,)
mode : str
The color rendering mode. intensity, z-value and constant are
supported.
Returns
-------
color : np.ndarray
Encoded Lidar color, shape (n, 3)
"""
assert mode in ['intensity', 'z-value', 'constant']
if mode == 'intensity':
intensity_col = 1.0 - np.log(intensity) / np.log(np.exp(-0.004 * 100))
int_color = np.c_[
np.interp(intensity_col, VID_RANGE, VIRIDIS[:, 0]),
np.interp(intensity_col, VID_RANGE, VIRIDIS[:, 1]),
np.interp(intensity_col, VID_RANGE, VIRIDIS[:, 2])]
elif mode == 'z-value':
min_value = -1.5
max_value = 0.5
norm = matplotlib.colors.Normalize(vmin=min_value, vmax=max_value)
cmap = cm.jet
m = cm.ScalarMappable(norm=norm, cmap=cmap)
colors = m.to_rgba(intensity)
colors[:, [2, 1, 0, 3]] = colors[:, [0, 1, 2, 3]]
colors[:, 3] = 0.5
int_color = colors[:, :3]
elif mode == 'constant':
# regard all point cloud the same color
int_color = np.ones((intensity.shape[0], 3))
int_color[:, 0] *= 247 / 255
int_color[:, 1] *= 244 / 255
int_color[:, 2] *= 237 / 255
return int_color
def visualize_single_sample_output_gt(pred_tensor,
gt_tensor,
pcd,
show_vis=True,
save_path='',
mode='constant'):
"""
Visualize the prediction, groundtruth with point cloud together.
Parameters
----------
pred_tensor : torch.Tensor
(N, 8, 3) prediction.
gt_tensor : torch.Tensor
(N, 8, 3) groundtruth bbx
pcd : torch.Tensor
PointCloud, (N, 4).
show_vis : bool
Whether to show visualization.
save_path : str
Save the visualization results to given path.
mode : str
Color rendering mode.
"""
def custom_draw_geometry(pcd, pred, gt):
vis = o3d.visualization.Visualizer()
vis.create_window()
opt = vis.get_render_option()
opt.background_color = np.asarray([0, 0, 0])
opt.point_size = 1.0
vis.add_geometry(pcd)
for ele in pred:
vis.add_geometry(ele)
for ele in gt:
vis.add_geometry(ele)
vis.run()
vis.destroy_window()
origin_lidar = pcd
if not isinstance(pcd, np.ndarray):
origin_lidar = common_utils.torch_tensor_to_numpy(pcd)
origin_lidar_intcolor = \
color_encoding(origin_lidar[:, -1] if mode == 'intensity'
else origin_lidar[:, 2], mode=mode)
# left -> right hand
origin_lidar[:, :1] = -origin_lidar[:, :1]
o3d_pcd = o3d.geometry.PointCloud()
o3d_pcd.points = o3d.utility.Vector3dVector(origin_lidar[:, :3])
o3d_pcd.colors = o3d.utility.Vector3dVector(origin_lidar_intcolor)
oabbs_pred = bbx2oabb(pred_tensor, color=(1, 0, 0))
oabbs_gt = bbx2oabb(gt_tensor, color=(0, 1, 0))
visualize_elements = [o3d_pcd] + oabbs_pred + oabbs_gt
if show_vis:
custom_draw_geometry(o3d_pcd, oabbs_pred, oabbs_gt)
if save_path:
save_o3d_visualization(visualize_elements, save_path)
def visualize_sequence_sample_output(pred_tensor_list,
gt_tensor_list,
pcd_list):
vis = o3d.visualization.Visualizer()
vis.create_window()
vis.get_render_option().background_color = [0.05, 0.05, 0.05]
vis.get_render_option().point_size = 1.0
vis.get_render_option().show_coordinate_frame = True
# used to visualize lidar points
vis_pcd = o3d.geometry.PointCloud()
while True:
for i, (pred_tensor, gt_tensor, pcd) in \
enumerate(zip(pred_tensor_list, gt_tensor_list, pcd_list)):
pred_tensor = pred_tensor.copy()
gt_tensor = gt_tensor.copy()
pcd = pcd.copy()
pcd_intcolor = color_encoding(pcd[:, -1])
pcd[:, :1] = -pcd[:, :1]
vis_pcd.points = o3d.utility.Vector3dVector(pcd[:, :3])
vis_pcd.colors = o3d.utility.Vector3dVector(pcd_intcolor)
oabbs_pred = bbx2oabb(pred_tensor, 'hwl')
oabbs_gt = bbx2oabb(gt_tensor, 'hwl', color=(0, 1, 0))
oabbs = oabbs_pred + oabbs_gt
if i == 0:
vis.add_geometry(vis_pcd)
for oabb in oabbs:
vis.add_geometry(oabb)
vis.update_geometry(vis_pcd)
ctr = vis.get_view_control()
param = o3d.io.read_pinhole_camera_parameters('pinhole_param.json')
ctr.convert_from_pinhole_camera_parameters(param)
vis.poll_events()
vis.update_renderer()
for oabb in oabbs:
vis.remove_geometry(oabb)
time.sleep(0.01)
vis.destroy_window()
def visualize_single_sample_output_bev(pred_box, gt_box, pcd, dataset,
show_vis=True,
save_path=''):
"""
Visualize the prediction, groundtruth with point cloud together in
a bev format.
Parameters
----------
pred_box : torch.Tensor
(N, 4, 2) prediction.
gt_box : torch.Tensor
(N, 4, 2) groundtruth bbx
pcd : torch.Tensor
PointCloud, (N, 4).
show_vis : bool
Whether to show visualization.
save_path : str
Save the visualization results to given path.
"""
if not isinstance(pcd, np.ndarray):
pcd = common_utils.torch_tensor_to_numpy(pcd)
if pred_box is not None and not isinstance(pred_box, np.ndarray):
pred_box = common_utils.torch_tensor_to_numpy(pred_box)
if gt_box is not None and not isinstance(gt_box, np.ndarray):
gt_box = common_utils.torch_tensor_to_numpy(gt_box)
ratio = dataset.params["preprocess"]["args"]["res"]
L1, W1, H1, L2, W2, H2 = dataset.params["preprocess"]["cav_lidar_range"]
bev_origin = np.array([L1, W1]).reshape(1, -1)
# (img_row, img_col)
bev_map = dataset.project_points_to_bev_map(pcd, ratio)
# (img_row, img_col, 3)
bev_map = \
np.repeat(bev_map[:, :, np.newaxis], 3, axis=-1).astype(np.float32)
bev_map = bev_map * 255
if pred_box is not None:
num_bbx = pred_box.shape[0]
for i in range(num_bbx):
bbx = pred_box[i]
bbx = ((bbx - bev_origin) / ratio).astype(int)
bbx = bbx[:, ::-1]
cv2.polylines(bev_map, [bbx], True, (0, 0, 255), 1)
if gt_box is not None and len(gt_box):
for i in range(gt_box.shape[0]):
bbx = gt_box[i][:4, :2]
bbx = (((bbx - bev_origin)) / ratio).astype(int)
bbx = bbx[:, ::-1]
cv2.polylines(bev_map, [bbx], True, (255, 0, 0), 1)
if show_vis:
plt.axis("off")
plt.imshow(bev_map)
plt.show()
if save_path:
plt.axis("off")
plt.imshow(bev_map)
plt.savefig(save_path)
def visualize_single_sample_dataloader(batch_data,
o3d_pcd,
order,
key='origin_lidar',
visualize=False,
save_path='',
oabb=False,
mode='constant'):
"""
Visualize a single frame of a single CAV for validation of data pipeline.
Parameters
----------
o3d_pcd : o3d.PointCloud
Open3d PointCloud.
order : str
The bounding box order.
key : str
origin_lidar for late fusion and stacked_lidar for early fusion.
todo: consider intermediate fusion in the future.
visualize : bool
Whether to visualize the sample.
batch_data : dict
The dictionary that contains current timestamp's data.
save_path : str
If set, save the visualization image to the path.
oabb : bool
If oriented bounding box is used.
"""
origin_lidar = batch_data[key]
if not isinstance(origin_lidar, np.ndarray):
origin_lidar = common_utils.torch_tensor_to_numpy(origin_lidar)
# we only visualize the first cav for single sample
if len(origin_lidar.shape) > 2:
origin_lidar = origin_lidar[0]
origin_lidar_intcolor = \
color_encoding(origin_lidar[:, -1] if mode == 'intensity'
else origin_lidar[:, 2], mode=mode)
# left -> right hand
origin_lidar[:, :1] = -origin_lidar[:, :1]
o3d_pcd.points = o3d.utility.Vector3dVector(origin_lidar[:, :3])
o3d_pcd.colors = o3d.utility.Vector3dVector(origin_lidar_intcolor)
object_bbx_center = batch_data['object_bbx_center']
object_bbx_mask = batch_data['object_bbx_mask']
object_bbx_center = object_bbx_center[object_bbx_mask == 1]
aabbs = bbx2linset(object_bbx_center, order) if not oabb else \
bbx2oabb(object_bbx_center, order)
visualize_elements = [o3d_pcd] + aabbs
if visualize:
o3d.visualization.draw_geometries(visualize_elements)
if save_path:
save_o3d_visualization(visualize_elements, save_path)
return o3d_pcd, aabbs
def visualize_inference_sample_dataloader(pred_box_tensor,
gt_box_tensor,
origin_lidar,
o3d_pcd,
mode='constant'):
"""
Visualize a frame during inference for video stream.
Parameters
----------
pred_box_tensor : torch.Tensor
(N, 8, 3) prediction.
gt_box_tensor : torch.Tensor
(N, 8, 3) groundtruth bbx
origin_lidar : torch.Tensor
PointCloud, (N, 4).
o3d_pcd : open3d.PointCloud
Used to visualize the pcd.
mode : str
lidar point rendering mode.
"""
if not isinstance(origin_lidar, np.ndarray):
origin_lidar = common_utils.torch_tensor_to_numpy(origin_lidar)
# we only visualize the first cav for single sample
if len(origin_lidar.shape) > 2:
origin_lidar = origin_lidar[0]
origin_lidar_intcolor = \
color_encoding(origin_lidar[:, -1] if mode == 'intensity'
else origin_lidar[:, 2], mode=mode)
if not isinstance(pred_box_tensor, np.ndarray):
pred_box_tensor = common_utils.torch_tensor_to_numpy(pred_box_tensor)
if not isinstance(gt_box_tensor, np.ndarray):
gt_box_tensor = common_utils.torch_tensor_to_numpy(gt_box_tensor)
# left -> right hand
origin_lidar[:, :1] = -origin_lidar[:, :1]
o3d_pcd.points = o3d.utility.Vector3dVector(origin_lidar[:, :3])
o3d_pcd.colors = o3d.utility.Vector3dVector(origin_lidar_intcolor)
gt_o3d_box = bbx2linset(gt_box_tensor, order='hwl', color=(0, 1, 0))
pred_o3d_box = bbx2linset(pred_box_tensor, color=(1, 0, 0))
return o3d_pcd, pred_o3d_box, gt_o3d_box
def visualize_sequence_dataloader(dataloader, order, color_mode='constant'):
"""
Visualize the batch data in animation.
Parameters
----------
dataloader : torch.Dataloader
Pytorch dataloader
order : str
Bounding box order(N, 7).
color_mode : str
Color rendering mode.
"""
vis = o3d.visualization.Visualizer()
vis.create_window()
vis.get_render_option().background_color = [0.05, 0.05, 0.05]
vis.get_render_option().point_size = 1.0
vis.get_render_option().show_coordinate_frame = True
# used to visualize lidar points
vis_pcd = o3d.geometry.PointCloud()
# used to visualize object bounding box, maximum 50
vis_aabbs = []
for _ in range(50):
vis_aabbs.append(o3d.geometry.LineSet())
while True:
for i_batch, sample_batched in enumerate(dataloader):
print(i_batch)
pcd, aabbs = \
visualize_single_sample_dataloader(sample_batched['ego'],
vis_pcd,
order,
mode=color_mode)
if i_batch == 0:
vis.add_geometry(pcd)
for i in range(len(vis_aabbs)):
index = i if i < len(aabbs) else -1
vis_aabbs[i] = lineset_assign(vis_aabbs[i], aabbs[index])
vis.add_geometry(vis_aabbs[i])
for i in range(len(vis_aabbs)):
index = i if i < len(aabbs) else -1
vis_aabbs[i] = lineset_assign(vis_aabbs[i], aabbs[index])
vis.update_geometry(vis_aabbs[i])
vis.update_geometry(pcd)
vis.poll_events()
vis.update_renderer()
time.sleep(0.001)
vis.destroy_window()
def save_o3d_visualization(element, save_path):
"""
Save the open3d drawing to folder.
Parameters
----------
element : list
List of o3d.geometry objects.
save_path : str
The save path.
"""
vis = o3d.visualization.Visualizer()
vis.create_window()
for i in range(len(element)):
vis.add_geometry(element[i])
vis.update_geometry(element[i])
vis.poll_events()
vis.update_renderer()
vis.capture_screen_image(save_path)
vis.destroy_window()
def visualize_bev(batch_data):
bev_input = batch_data["processed_lidar"]["bev_input"]
label_map = batch_data["label_dict"]["label_map"]
if not isinstance(bev_input, np.ndarray):
bev_input = common_utils.torch_tensor_to_numpy(bev_input)
if not isinstance(label_map, np.ndarray):
label_map = label_map[0].numpy() if not label_map[0].is_cuda else \
label_map[0].cpu().detach().numpy()
if len(bev_input.shape) > 3:
bev_input = bev_input[0, ...]
plt.matshow(np.sum(bev_input, axis=0))
plt.axis("off")
plt.matshow(label_map[0, :, :])
plt.axis("off")
plt.show()
import os
import argparse
from torch.utils.data import DataLoader
from v2xvit.hypes_yaml.yaml_utils import load_yaml
from v2xvit.visualization import vis_utils
from v2xvit.data_utils.datasets.early_fusion_vis_dataset import \
EarlyFusionVisDataset
def vis_parser():
parser = argparse.ArgumentParser(description="data visualization")
parser.add_argument('--color_mode', type=str, default="intensity",
help='lidar color rendering mode, e.g. intensity,'
'z-value or constant.')
opt = parser.parse_args()
return opt
if __name__ == '__main__':
current_path = os.path.dirname(os.path.realpath(__file__))
params = load_yaml(os.path.join(current_path,
'../hypes_yaml/visualization.yaml'))
opencda_dataset = EarlyFusionVisDataset(params, visualize=True,
train=False)
data_loader = DataLoader(opencda_dataset, batch_size=1, num_workers=8,
collate_fn=opencda_dataset.collate_batch_train,
shuffle=False,
pin_memory=False)
opt = vis_parser()
vis_utils.visualize_sequence_dataloader(data_loader,
params['postprocess']['order'],
color_mode=opt.color_mode)
"""
Convert lidar to bev
"""
import numpy as np
import torch
from v2xvit.data_utils.pre_processor.base_preprocessor import \
BasePreprocessor
class BevPreprocessor(BasePreprocessor):
def __init__(self, preprocess_params, train):
super(BevPreprocessor, self).__init__(preprocess_params, train)
self.lidar_range = self.params['cav_lidar_range']
self.geometry_param = preprocess_params["geometry_param"]
def preprocess(self, pcd_raw):
"""
Preprocess the lidar points to BEV representations.
Parameters
----------
pcd_raw : np.ndarray
The raw lidar.
Returns
-------
data_dict : the structured output dictionary.
"""
bev = np.zeros(self.geometry_param['input_shape'], dtype=np.float32)
intensity_map_count = np.zeros((bev.shape[0], bev.shape[1]), dtype=np.int)
bev_origin = np.array(
[self.geometry_param["L1"], self.geometry_param["W1"],
self.geometry_param["H1"]]).reshape(1, -1)
indices = ((pcd_raw[:, :3] - bev_origin) / self.geometry_param[
"res"]).astype(int)
## bev[indices[:, 0], indices[:, 1], indices[:, 2]] = 1
# np.add.at(bev, (indices[:, 0], indices[:, 1], indices[:, 2]), 1)
# bev[indices[:, 0], indices[:, 1], -1] += pcd_raw[:, 3]
# intensity_map_count[indices[:, 0], indices[:, 1]] += 1
for i in range(indices.shape[0]):
bev[indices[i, 0], indices[i, 1], indices[i, 2]] = 1
bev[indices[i, 0], indices[i, 1], -1] += pcd_raw[i, 3]
intensity_map_count[indices[i, 0], indices[i, 1]] += 1
divide_mask = intensity_map_count!=0
bev[divide_mask, -1] = np.divide(bev[divide_mask, -1], intensity_map_count[divide_mask])
data_dict = {
"bev_input": np.transpose(bev, (2, 0, 1))
}
return data_dict
@staticmethod
def collate_batch_list(batch):
"""
Customized pytorch data loader collate function.
Parameters
----------
batch : list
List of dictionary. Each dictionary represent a single frame.
Returns
-------
processed_batch : dict
Updated lidar batch.
"""
bev_input_list = [
x["bev_input"][np.newaxis, ...] for x in batch
]
processed_batch = {
"bev_input": torch.from_numpy(
np.concatenate(bev_input_list, axis=0))
}
return processed_batch
@staticmethod
def collate_batch_dict(batch):
"""
Customized pytorch data loader collate function.
Parameters
----------
batch : dict
Dict of list. Each element represents a CAV.
Returns
-------
processed_batch : dict
Updated lidar batch.
"""
bev_input_list = [
x[np.newaxis, ...] for x in batch["bev_input"]
]
processed_batch = {
"bev_input": torch.from_numpy(
np.concatenate(bev_input_list, axis=0))
}
return processed_batch
def collate_batch(self, batch):
"""
Customized pytorch data loader collate function.
Parameters
----------
batch : list / dict
Batched data.
Returns
-------
processed_batch : dict
Updated lidar batch.
"""
if isinstance(batch, list):
return self.collate_batch_list(batch)
elif isinstance(batch, dict):
return self.collate_batch_dict(batch)
else:
raise NotImplemented
import numpy as np
from v2xvit.utils import pcd_utils
class BasePreprocessor(object):
"""
Basic Lidar pre-processor.
Parameters
----------
preprocess_params : dict
The dictionary containing all parameters of the preprocessing.
train : bool
Train or test mode.
"""
def __init__(self, preprocess_params, train):
self.params = preprocess_params
self.train = train
def preprocess(self, pcd_np):
"""
Preprocess the lidar points by simple sampling.
Parameters
----------
pcd_np : np.ndarray
The raw lidar.
Returns
-------
data_dict : the output dictionary.
"""
data_dict = {}
sample_num = self.params['args']['sample_num']
pcd_np = pcd_utils.downsample_lidar(pcd_np, sample_num)
data_dict['downsample_lidar'] = pcd_np
return data_dict
def project_points_to_bev_map(self, points, ratio=0.1):
"""
Project points to BEV occupancy map with default ratio=0.1.
Parameters
----------
points : np.ndarray
(N, 3) / (N, 4)
ratio : float
Discretization parameters. Default is 0.1.
Returns
-------
bev_map : np.ndarray
BEV occupancy map including projected points with shape
(img_row, img_col).
"""
L1, W1, H1, L2, W2, H2 = self.params["cav_lidar_range"]
img_row = int((L2 - L1) / ratio)
img_col = int((W2 - W1) / ratio)
bev_map = np.zeros((img_row, img_col))
bev_origin = np.array([L1, W1, H1]).reshape(1, -1)
# (N, 3)
indices = ((points[:, :3] - bev_origin) / ratio).astype(int)
mask = np.logical_and(indices[:, 0] > 0, indices[:, 0] < img_row)
mask = np.logical_and(mask, np.logical_and(indices[:, 1] > 0,
indices[:, 1] < img_col))
indices = indices[mask, :]
bev_map[indices[:, 0], indices[:, 1]] = 1
return bev_map
"""
Transform points to voxels using sparse conv library
"""
import sys
import numpy as np
import torch
from cumm import tensorview as tv
from spconv.utils import Point2VoxelCPU3d
from v2xvit.data_utils.pre_processor.base_preprocessor import \
BasePreprocessor
class SpVoxelPreprocessor(BasePreprocessor):
def __init__(self, preprocess_params, train):
super(SpVoxelPreprocessor, self).__init__(preprocess_params,
train)
self.lidar_range = self.params['cav_lidar_range']
self.voxel_size = self.params['args']['voxel_size']
self.max_points_per_voxel = self.params['args']['max_points_per_voxel']
if train:
self.max_voxels = self.params['args']['max_voxel_train']
else:
self.max_voxels = self.params['args']['max_voxel_test']
grid_size = (np.array(self.lidar_range[3:6]) -
np.array(self.lidar_range[0:3])) / np.array(self.voxel_size)
self.grid_size = np.round(grid_size).astype(np.int64)
# use sparse conv library to generate voxel
self.voxel_generator = Point2VoxelCPU3d(
vsize_xyz=self.voxel_size,
coors_range_xyz=self.lidar_range,
max_num_points_per_voxel=self.max_points_per_voxel,
num_point_features=4,
max_num_voxels=self.max_voxels
)
def preprocess(self, pcd_np):
data_dict = {}
pcd_tv = tv.from_numpy(pcd_np)
voxel_output = self.voxel_generator.point_to_voxel(pcd_tv)
if isinstance(voxel_output, dict):
voxels, coordinates, num_points = \
voxel_output['voxels'], voxel_output['coordinates'], \
voxel_output['num_points_per_voxel']
else:
voxels, coordinates, num_points = voxel_output
data_dict['voxel_features'] = voxels.numpy()
data_dict['voxel_coords'] = coordinates.numpy()
data_dict['voxel_num_points'] = num_points.numpy()
return data_dict
def collate_batch(self, batch):
"""
Customized pytorch data loader collate function.
Parameters
----------
batch : list or dict
List or dictionary.
Returns
-------
processed_batch : dict
Updated lidar batch.
"""
if isinstance(batch, list):
return self.collate_batch_list(batch)
elif isinstance(batch, dict):
return self.collate_batch_dict(batch)
else:
sys.exit('Batch has too be a list or a dictionarn')
@staticmethod
def collate_batch_list(batch):
"""
Customized pytorch data loader collate function.
Parameters
----------
batch : list
List of dictionary. Each dictionary represent a single frame.
Returns
-------
processed_batch : dict
Updated lidar batch.
"""
voxel_features = []
voxel_num_points = []
voxel_coords = []
for i in range(len(batch)):
voxel_features.append(batch[i]['voxel_features'])
voxel_num_points.append(batch[i]['voxel_num_points'])
coords = batch[i]['voxel_coords']
voxel_coords.append(
np.pad(coords, ((0, 0), (1, 0)),
mode='constant', constant_values=i))
voxel_num_points = torch.from_numpy(np.concatenate(voxel_num_points))
voxel_features = torch.from_numpy(np.concatenate(voxel_features))
voxel_coords = torch.from_numpy(np.concatenate(voxel_coords))
return {'voxel_features': voxel_features,
'voxel_coords': voxel_coords,
'voxel_num_points': voxel_num_points}
@staticmethod
def collate_batch_dict(batch: dict):
"""
Collate batch if the batch is a dictionary,
eg: {'voxel_features': [feature1, feature2...., feature n]}
Parameters
----------
batch : dict
Returns
-------
processed_batch : dict
Updated lidar batch.
"""
voxel_features = \
torch.from_numpy(np.concatenate(batch['voxel_features']))
voxel_num_points = \
torch.from_numpy(np.concatenate(batch['voxel_num_points']))
coords = batch['voxel_coords']
voxel_coords = []
for i in range(len(coords)):
voxel_coords.append(
np.pad(coords[i], ((0, 0), (1, 0)),
mode='constant', constant_values=i))
voxel_coords = torch.from_numpy(np.concatenate(voxel_coords))
return {'voxel_features': voxel_features,
'voxel_coords': voxel_coords,
'voxel_num_points': voxel_num_points}
"""
Convert lidar to voxel
"""
import sys
import numpy as np
import torch
from v2xvit.data_utils.pre_processor.base_preprocessor import \
BasePreprocessor
class VoxelPreprocessor(BasePreprocessor):
def __init__(self, preprocess_params, train):
super(VoxelPreprocessor, self).__init__(preprocess_params, train)
self.lidar_range = self.params['cav_lidar_range']
self.vw = self.params['args']['vw']
self.vh = self.params['args']['vh']
self.vd = self.params['args']['vd']
self.T = self.params['args']['T']
def preprocess(self, pcd_np):
"""
Preprocess the lidar points by voxelization.
Parameters
----------
pcd_np : np.ndarray
The raw lidar.
Returns
-------
data_dict : the structured output dictionary.
"""
data_dict = {}
# calculate the voxel coordinates
voxel_coords = ((pcd_np[:, :3] -
np.floor(np.array([self.lidar_range[0],
self.lidar_range[1],
self.lidar_range[2]])) / (
self.vw, self.vh, self.vd))).astype(np.int32)
# convert to (D, H, W) as the paper
voxel_coords = voxel_coords[:, [2, 1, 0]]
voxel_coords, inv_ind, voxel_counts = np.unique(voxel_coords, axis=0,
return_inverse=True,
return_counts=True)
voxel_features = []
for i in range(len(voxel_coords)):
voxel = np.zeros((self.T, 7), dtype=np.float32)
pts = pcd_np[inv_ind == i]
if voxel_counts[i] > self.T:
pts = pts[:self.T, :]
voxel_counts[i] = self.T
# augment the points
voxel[:pts.shape[0], :] = np.concatenate((pts, pts[:, :3] -
np.mean(pts[:, :3], 0)),
axis=1)
voxel_features.append(voxel)
data_dict['voxel_features'] = np.array(voxel_features)
data_dict['voxel_coords'] = voxel_coords
return data_dict
def collate_batch(self, batch):
"""
Customized pytorch data loader collate function.
Parameters
----------
batch : list or dict
List or dictionary.
Returns
-------
processed_batch : dict
Updated lidar batch.
"""
if isinstance(batch, list):
return self.collate_batch_list(batch)
elif isinstance(batch, dict):
return self.collate_batch_dict(batch)
else:
sys.exit('Batch has too be a list or a dictionarn')
@staticmethod
def collate_batch_list(batch):
"""
Customized pytorch data loader collate function.
Parameters
----------
batch : list
List of dictionary. Each dictionary represent a single frame.
Returns
-------
processed_batch : dict
Updated lidar batch.
"""
voxel_features = []
voxel_coords = []
for i in range(len(batch)):
voxel_features.append(batch[i]['voxel_features'])
coords = batch[i]['voxel_coords']
voxel_coords.append(
np.pad(coords, ((0, 0), (1, 0)),
mode='constant', constant_values=i))
voxel_features = torch.from_numpy(np.concatenate(voxel_features))
voxel_coords = torch.from_numpy(np.concatenate(voxel_coords))
return {'voxel_features': voxel_features,
'voxel_coords': voxel_coords}
@staticmethod
def collate_batch_dict(batch: dict):
"""
Collate batch if the batch is a dictionary,
eg: {'voxel_features': [feature1, feature2...., feature n]}
Parameters
----------
batch : dict
Returns
-------
processed_batch : dict
Updated lidar batch.
"""
voxel_features = \
torch.from_numpy(np.concatenate(batch['voxel_features']))
coords = batch['voxel_coords']
voxel_coords = []
for i in range(len(coords)):
voxel_coords.append(
np.pad(coords[i], ((0, 0), (1, 0)),
mode='constant', constant_values=i))
voxel_coords = torch.from_numpy(np.concatenate(voxel_coords))
return {'voxel_features': voxel_features,
'voxel_coords': voxel_coords}
from v2xvit.data_utils.pre_processor.base_preprocessor import BasePreprocessor
from v2xvit.data_utils.pre_processor.voxel_preprocessor import VoxelPreprocessor
from v2xvit.data_utils.pre_processor.bev_preprocessor import BevPreprocessor
from v2xvit.data_utils.pre_processor.sp_voxel_preprocessor import SpVoxelPreprocessor
__all__ = {
'BasePreprocessor': BasePreprocessor,
'VoxelPreprocessor': VoxelPreprocessor,
'BevPreprocessor': BevPreprocessor,
'SpVoxelPreprocessor': SpVoxelPreprocessor
}
def build_preprocessor(preprocess_cfg, train):
process_method_name = preprocess_cfg['core_method']
error_message = f"{process_method_name} is not found. " \
f"Please add your processor file's name in opencood/" \
f"data_utils/processor/init.py"
assert process_method_name in ['BasePreprocessor', 'VoxelPreprocessor',
'BevPreprocessor', 'SpVoxelPreprocessor'], \
error_message
processor = __all__[process_method_name](
preprocess_params=preprocess_cfg,
train=train
)
return processor
"""
Template for AnchorGenerator
"""
import numpy as np
import torch
from v2xvit.utils import box_utils
class BasePostprocessor(object):
"""
Template for Anchor generator.
Parameters
----------
anchor_params : dict
The dictionary containing all anchor-related parameters.
train : bool
Indicate train or test mode.
Attributes
----------
bbx_dict : dictionary
Contain all objects information across the cav, key: id, value: bbx
coordinates (1, 7)
"""
def __init__(self, anchor_params, train=True):
self.params = anchor_params
self.bbx_dict = {}
self.train = train
def generate_anchor_box(self):
# needs to be overloaded
return None
def generate_label(self, *argv):
return None
def generate_gt_bbx(self, data_dict):
"""
The base postprocessor will generate 3d groundtruth bounding box.
Parameters
----------
data_dict : dict
The dictionary containing the origin input data of model.
Returns
-------
gt_box3d_tensor : torch.Tensor
The groundtruth bounding box tensor, shape (N, 8, 3).
"""
gt_box3d_list = []
# used to avoid repetitive bounding box
object_id_list = []
for cav_id, cav_content in data_dict.items():
# used to project gt bounding box to ego space.
# the transformation matrix for gt should always be based on
# current timestamp (object transformation matrix is for
# late fusion only since other fusion method already did
# the transformation in the preprocess)
transformation_matrix = cav_content['transformation_matrix'] \
if 'gt_transformation_matrix' not in cav_content \
else cav_content['gt_transformation_matrix']
object_bbx_center = cav_content['object_bbx_center']
object_bbx_mask = cav_content['object_bbx_mask']
object_ids = cav_content['object_ids']
object_bbx_center = object_bbx_center[object_bbx_mask == 1]
# convert center to corner
object_bbx_corner = \
box_utils.boxes_to_corners_3d(object_bbx_center,
self.params['order'])
projected_object_bbx_corner = \
box_utils.project_box3d(object_bbx_corner.float(),
transformation_matrix)
gt_box3d_list.append(projected_object_bbx_corner)
# append the corresponding ids
object_id_list += object_ids
# gt bbx 3d
gt_box3d_list = torch.vstack(gt_box3d_list)
# some of the bbx may be repetitive, use the id list to filter
gt_box3d_selected_indices = \
[object_id_list.index(x) for x in set(object_id_list)]
gt_box3d_tensor = gt_box3d_list[gt_box3d_selected_indices]
# filter the gt_box to make sure all bbx are in the range
mask = \
box_utils.get_mask_for_boxes_within_range_torch(gt_box3d_tensor)
gt_box3d_tensor = gt_box3d_tensor[mask, :, :]
return gt_box3d_tensor
def generate_object_center(self,
cav_contents,
reference_lidar_pose):
"""
Retrieve all objects in a format of (n, 7), where 7 represents
x, y, z, l, w, h, yaw or x, y, z, h, w, l, yaw.
Parameters
----------
cav_contents : list
List of dictionary, save all cavs' information.
reference_lidar_pose : list
The final target lidar pose with length 6.
Returns
-------
object_np : np.ndarray
Shape is (max_num, 7).
mask : np.ndarray
Shape is (max_num,).
object_ids : list
Length is number of bbx in current sample.
"""
from v2xvit.data_utils.datasets import GT_RANGE
tmp_object_dict = {}
for cav_content in cav_contents:
tmp_object_dict.update(cav_content['params']['vehicles'])
output_dict = {}
filter_range = self.params['anchor_args']['cav_lidar_range'] \
if self.train else GT_RANGE
box_utils.project_world_objects(tmp_object_dict,
output_dict,
reference_lidar_pose,
filter_range,
self.params['order'])
object_np = np.zeros((self.params['max_num'], 7))
mask = np.zeros(self.params['max_num'])
object_ids = []
for i, (object_id, object_bbx) in enumerate(output_dict.items()):
object_np[i] = object_bbx[0, :]
mask[i] = 1
object_ids.append(object_id)
return object_np, mask, object_ids
"""
3D Anchor Generator for Voxel
"""
import math
import sys
import numpy as np
import torch
import torch.nn.functional as F
from v2xvit.data_utils.post_processor.base_postprocessor \
import BasePostprocessor
from v2xvit.utils import box_utils
from v2xvit.utils.box_overlaps import bbox_overlaps
from v2xvit.visualization import vis_utils
class VoxelPostprocessor(BasePostprocessor):
def __init__(self, anchor_params, train):
super(VoxelPostprocessor, self).__init__(anchor_params, train)
self.anchor_num = self.params['anchor_args']['num']
def generate_anchor_box(self):
W = self.params['anchor_args']['W']
H = self.params['anchor_args']['H']
l = self.params['anchor_args']['l']
w = self.params['anchor_args']['w']
h = self.params['anchor_args']['h']
r = self.params['anchor_args']['r']
assert self.anchor_num == len(r)
r = [math.radians(ele) for ele in r]
vh = self.params['anchor_args']['vh']
vw = self.params['anchor_args']['vw']
xrange = [self.params['anchor_args']['cav_lidar_range'][0],
self.params['anchor_args']['cav_lidar_range'][3]]
yrange = [self.params['anchor_args']['cav_lidar_range'][1],
self.params['anchor_args']['cav_lidar_range'][4]]
if 'feature_stride' in self.params['anchor_args']:
feature_stride = self.params['anchor_args']['feature_stride']
else:
feature_stride = 2
x = np.linspace(xrange[0] + vw, xrange[1] - vw, W // feature_stride)
y = np.linspace(yrange[0] + vh, yrange[1] - vh, H // feature_stride)
cx, cy = np.meshgrid(x, y)
cx = np.tile(cx[..., np.newaxis], self.anchor_num)
cy = np.tile(cy[..., np.newaxis], self.anchor_num)
cz = np.ones_like(cx) * -1.0
w = np.ones_like(cx) * w
l = np.ones_like(cx) * l
h = np.ones_like(cx) * h
r_ = np.ones_like(cx)
for i in range(self.anchor_num):
r_[..., i] = r[i]
if self.params['order'] == 'hwl':
anchors = np.stack([cx, cy, cz, h, w, l, r_], axis=-1)
elif self.params['order'] == 'lhw':
anchors = np.stack([cx, cy, cz, l, h, w, r_], axis=-1)
else:
sys.exit('Unknown bbx order.')
return anchors
def generate_label(self, **kwargs):
"""
Generate targets for training.
Parameters
----------
argv : list
gt_box_center:(max_num, 7), anchor:(H, W, anchor_num, 7)
Returns
-------
label_dict : dict
Dictionary that contains all target related info.
"""
assert self.params['order'] == 'hwl', 'Currently Voxel only support' \
'hwl bbx order.'
# (max_num, 7)
gt_box_center = kwargs['gt_box_center']
# (H, W, anchor_num, 7)
anchors = kwargs['anchors']
# (max_num)
masks = kwargs['mask']
# (H, W)
feature_map_shape = anchors.shape[:2]
# (H*W*anchor_num, 7)
anchors = anchors.reshape(-1, 7)
# normalization factor, (H * W * anchor_num)
anchors_d = np.sqrt(anchors[:, 4] ** 2 + anchors[:, 5] ** 2)
# (H, W, 2)
pos_equal_one = np.zeros((*feature_map_shape, self.anchor_num))
neg_equal_one = np.zeros((*feature_map_shape, self.anchor_num))
# (H, W, self.anchor_num * 7)
targets = np.zeros((*feature_map_shape, self.anchor_num * 7))
# (n, 7)
gt_box_center_valid = gt_box_center[masks == 1]
# (n, 8, 3)
gt_box_corner_valid = \
box_utils.boxes_to_corners_3d(gt_box_center_valid,
self.params['order'])
# (H*W*anchor_num, 8, 3)
anchors_corner = \
box_utils.boxes_to_corners_3d(anchors,
order=self.params['order'])
# (H*W*anchor_num, 4)
anchors_standup_2d = \
box_utils.corner2d_to_standup_box(anchors_corner)
# (n, 4)
gt_standup_2d = \
box_utils.corner2d_to_standup_box(gt_box_corner_valid)
# (H*W*anchor_n)
iou = bbox_overlaps(
np.ascontiguousarray(anchors_standup_2d).astype(np.float32),
np.ascontiguousarray(gt_standup_2d).astype(np.float32),
)
# the anchor boxes has the largest iou across
# shape: (n)
id_highest = np.argmax(iou.T, axis=1)
# [0, 1, 2, ..., n-1]
id_highest_gt = np.arange(iou.T.shape[0])
# make sure all highest iou is larger than 0
mask = iou.T[id_highest_gt, id_highest] > 0
id_highest, id_highest_gt = id_highest[mask], id_highest_gt[mask]
# find anchors iou > params['pos_iou']
id_pos, id_pos_gt = \
np.where(iou >
self.params['target_args']['pos_threshold'])
# find anchors iou params['neg_iou']
id_neg = np.where(np.sum(iou <
self.params['target_args']['neg_threshold'],
axis=1) == iou.shape[1])[0]
id_pos = np.concatenate([id_pos, id_highest])
id_pos_gt = np.concatenate([id_pos_gt, id_highest_gt])
id_pos, index = np.unique(id_pos, return_index=True)
id_pos_gt = id_pos_gt[index]
id_neg.sort()
# cal the target and set the equal one
index_x, index_y, index_z = np.unravel_index(
id_pos, (*feature_map_shape, self.anchor_num))
pos_equal_one[index_x, index_y, index_z] = 1
# calculate the targets
targets[index_x, index_y, np.array(index_z) * 7] = \
(gt_box_center[id_pos_gt, 0] - anchors[id_pos, 0]) / anchors_d[
id_pos]
targets[index_x, index_y, np.array(index_z) * 7 + 1] = \
(gt_box_center[id_pos_gt, 1] - anchors[id_pos, 1]) / anchors_d[
id_pos]
targets[index_x, index_y, np.array(index_z) * 7 + 2] = \
(gt_box_center[id_pos_gt, 2] - anchors[id_pos, 2]) / anchors[
id_pos, 3]
targets[index_x, index_y, np.array(index_z) * 7 + 3] = np.log(
gt_box_center[id_pos_gt, 3] / anchors[id_pos, 3])
targets[index_x, index_y, np.array(index_z) * 7 + 4] = np.log(
gt_box_center[id_pos_gt, 4] / anchors[id_pos, 4])
targets[index_x, index_y, np.array(index_z) * 7 + 5] = np.log(
gt_box_center[id_pos_gt, 5] / anchors[id_pos, 5])
targets[index_x, index_y, np.array(index_z) * 7 + 6] = (
gt_box_center[id_pos_gt, 6] - anchors[id_pos, 6])
index_x, index_y, index_z = np.unravel_index(
id_neg, (*feature_map_shape, self.anchor_num))
neg_equal_one[index_x, index_y, index_z] = 1
# to avoid a box be pos/neg in the same time
index_x, index_y, index_z = np.unravel_index(
id_highest, (*feature_map_shape, self.anchor_num))
neg_equal_one[index_x, index_y, index_z] = 0
label_dict = {'pos_equal_one': pos_equal_one,
'neg_equal_one': neg_equal_one,
'targets': targets}
return label_dict
@staticmethod
def collate_batch(label_batch_list):
"""
Customized collate function for target label generation.
Parameters
----------
label_batch_list : list
The list of dictionary that contains all labels for several
frames.
Returns
-------
target_batch : dict
Reformatted labels in torch tensor.
"""
pos_equal_one = []
neg_equal_one = []
targets = []
for i in range(len(label_batch_list)):
pos_equal_one.append(label_batch_list[i]['pos_equal_one'])
neg_equal_one.append(label_batch_list[i]['neg_equal_one'])
targets.append(label_batch_list[i]['targets'])
pos_equal_one = \
torch.from_numpy(np.array(pos_equal_one))
neg_equal_one = \
torch.from_numpy(np.array(neg_equal_one))
targets = \
torch.from_numpy(np.array(targets))
return {'targets': targets,
'pos_equal_one': pos_equal_one,
'neg_equal_one': neg_equal_one}
def post_process(self, data_dict, output_dict):
"""
Process the outputs of the model to 2D/3D bounding box.
Step1: convert each cav's output to bounding box format
Step2: project the bounding boxes to ego space.
Step:3 NMS
Parameters
----------
data_dict : dict
The dictionary containing the origin input data of model.
output_dict :dict
The dictionary containing the output of the model.
Returns
-------
pred_box3d_tensor : torch.Tensor
The prediction bounding box tensor after NMS.
gt_box3d_tensor : torch.Tensor
The groundtruth bounding box tensor.
"""
# the final bounding box list
pred_box3d_list = []
pred_box2d_list = []
for cav_id, cav_content in data_dict.items():
assert cav_id in output_dict
# the transformation matrix to ego space
transformation_matrix = cav_content['transformation_matrix']
# (H, W, anchor_num, 7)
anchor_box = cav_content['anchor_box']
# classification probability
prob = output_dict[cav_id]['psm']
prob = F.sigmoid(prob.permute(0, 2, 3, 1))
prob = prob.reshape(1, -1)
# regression map
reg = output_dict[cav_id]['rm']
# convert regression map back to bounding box
# (N, W*L*anchor_num, 7)
batch_box3d = self.delta_to_boxes3d(reg, anchor_box)
mask = \
torch.gt(prob, self.params['target_args']['score_threshold'])
mask = mask.view(1, -1)
mask_reg = mask.unsqueeze(2).repeat(1, 1, 7)
# during validation/testing, the batch size should be 1
assert batch_box3d.shape[0] == 1
boxes3d = torch.masked_select(batch_box3d[0],
mask_reg[0]).view(-1, 7)
scores = torch.masked_select(prob[0], mask[0])
# convert output to bounding box
if len(boxes3d) != 0:
# (N, 8, 3)
boxes3d_corner = \
box_utils.boxes_to_corners_3d(boxes3d,
order=self.params['order'])
# (N, 8, 3)
projected_boxes3d = \
box_utils.project_box3d(boxes3d_corner,
transformation_matrix)
# convert 3d bbx to 2d, (N,4)
projected_boxes2d = \
box_utils.corner_to_standup_box_torch(projected_boxes3d)
# (N, 5)
boxes2d_score = \
torch.cat((projected_boxes2d, scores.unsqueeze(1)), dim=1)
pred_box2d_list.append(boxes2d_score)
pred_box3d_list.append(projected_boxes3d)
if len(pred_box2d_list) ==0 or len(pred_box3d_list) == 0:
return None, None
# shape: (N, 5)
pred_box2d_list = torch.vstack(pred_box2d_list)
# scores
scores = pred_box2d_list[:, -1]
# predicted 3d bbx
pred_box3d_tensor = torch.vstack(pred_box3d_list)
# remove large bbx
keep_index_1 = box_utils.remove_large_pred_bbx(pred_box3d_tensor)
keep_index_2 = box_utils.remove_bbx_abnormal_z(pred_box3d_tensor)
keep_index = torch.logical_and(keep_index_1, keep_index_2)
pred_box3d_tensor = pred_box3d_tensor[keep_index]
scores = scores[keep_index]
# nms
keep_index = box_utils.nms_rotated(pred_box3d_tensor,
scores,
self.params['nms_thresh']
)
pred_box3d_tensor = pred_box3d_tensor[keep_index]
# select cooresponding score
scores = scores[keep_index]
# filter out the prediction out of the range.
mask = \
box_utils.get_mask_for_boxes_within_range_torch(pred_box3d_tensor)
pred_box3d_tensor = pred_box3d_tensor[mask, :, :]
scores = scores[mask]
assert scores.shape[0] == pred_box3d_tensor.shape[0]
return pred_box3d_tensor, scores
@staticmethod
def delta_to_boxes3d(deltas, anchors):
"""
Convert the output delta to 3d bbx.
Parameters
----------
deltas : torch.Tensor
(N, W, L, 14)
anchors : torch.Tensor
(W, L, 2, 7) -> xyzhwlr
Returns
-------
box3d : torch.Tensor
(N, W*L*2, 7)
"""
# batch size
N = deltas.shape[0]
deltas = deltas.permute(0, 2, 3, 1).contiguous().view(N, -1, 7)
boxes3d = torch.zeros_like(deltas)
if deltas.is_cuda:
anchors = anchors.cuda()
boxes3d = boxes3d.cuda()
# (W*L*2, 7)
anchors_reshaped = anchors.view(-1, 7).float()
# the diagonal of the anchor 2d box, (W*L*2)
anchors_d = torch.sqrt(
anchors_reshaped[:, 4] ** 2 + anchors_reshaped[:, 5] ** 2)
anchors_d = anchors_d.repeat(N, 2, 1).transpose(1, 2)
anchors_reshaped = anchors_reshaped.repeat(N, 1, 1)
# Inv-normalize to get xyz
boxes3d[..., [0, 1]] = torch.mul(deltas[..., [0, 1]], anchors_d) + \
anchors_reshaped[..., [0, 1]]
boxes3d[..., [2]] = torch.mul(deltas[..., [2]],
anchors_reshaped[..., [3]]) + \
anchors_reshaped[..., [2]]
# hwl
boxes3d[..., [3, 4, 5]] = torch.exp(
deltas[..., [3, 4, 5]]) * anchors_reshaped[..., [3, 4, 5]]
# yaw angle
boxes3d[..., 6] = deltas[..., 6] + anchors_reshaped[..., 6]
return boxes3d
@staticmethod
def visualize(pred_box_tensor, gt_tensor, pcd, show_vis, save_path, dataset=None):
"""
Visualize the prediction, ground truth with point cloud together.
Parameters
----------
pred_box_tensor : torch.Tensor
(N, 8, 3) prediction.
gt_tensor : torch.Tensor
(N, 8, 3) groundtruth bbx
pcd : torch.Tensor
PointCloud, (N, 4).
show_vis : bool
Whether to show visualization.
save_path : str
Save the visualization results to given path.
dataset : BaseDataset
opencood dataset object.
"""
vis_utils.visualize_single_sample_output_gt(pred_box_tensor,
gt_tensor,
pcd,
show_vis,
save_path)
"""
Anchor-free 2d Generator
"""
import numpy as np
import torch
import torch.nn.functional as F
from v2xvit.utils.transformation_utils import dist_to_continuous
from v2xvit.data_utils.post_processor.base_postprocessor \
import BasePostprocessor
from v2xvit.utils import box_utils
from v2xvit.visualization import vis_utils
class BevPostprocessor(BasePostprocessor):
def __init__(self, anchor_params, train):
super(BevPostprocessor, self).__init__(anchor_params, train)
# self.geometry_param = anchor_params["geometry"]
self.geometry_param = anchor_params["geometry_param"]
# TODO
# Hard coded for now. Need to calculate for our own training dataset
self.target_mean = np.array([0.008, 0.001, 0.202, 0.2, 0.43, 1.368])
self.target_std_dev = np.array([0.866, 0.5, 0.954, 0.668, 0.09, 0.111])
def generate_anchor_box(self):
return None
def generate_label(self, **kwargs):
"""
Generate targets for training.
Parameters
----------
kwargs : list
gt_box_center:(max_num, 7)
Returns
-------
label_dict : dict
Dictionary that contains all target related info.
"""
assert self.params['order'] == 'lwh', \
'Currently BEV only support lwh bbx order.'
# (max_num, 7)
gt_box_center = kwargs['gt_box_center']
# (max_num)
masks = kwargs['mask']
# (n, 7)
gt_box_center_valid = gt_box_center[masks == 1]
# (n, 4, 3)
bev_corners = box_utils.boxes_to_corners2d(gt_box_center_valid,
self.params['order'])
n = gt_box_center_valid.shape[0]
# (n, 4, 2)
bev_corners = bev_corners[:, :, :2]
yaw = gt_box_center_valid[:, -1]
x, y = gt_box_center_valid[:, 0], gt_box_center_valid[:, 1]
dx, dy = gt_box_center_valid[:, 3], gt_box_center_valid[:, 4]
# (n, 6)
reg_targets = np.column_stack([np.cos(yaw), np.sin(yaw), x, y, dx, dy])
# target label map including classification and regression targets
# shape -- (label_shape[0], label_shape[1], 7)
# (binary, cos(yaw), sin(yaw), displacement_x, displacement_y, log(dx), log(dy))
label_map = np.zeros(self.geometry_param["label_shape"])
self.update_label_map(label_map, bev_corners, reg_targets)
label_map = self.normalize_targets(label_map)
label_dict = {
# (7, label_shape[0], label_shape[1])
"label_map": np.transpose(label_map, (2, 0, 1)).astype(np.float32),
"bev_corners": bev_corners
}
return label_dict
def update_label_map(self, label_map, bev_corners, reg_targets):
"""
Update label_map based on bbx and regression targets.
Parameters
----------
label_map : numpy.array
Targets array for classification and regression tasks with
the shape of label_shape.
bev_corners : numpy.array
The bbx corners in lidar frame with shape (n, 4, 2)
reg_targets : numpy.array
Array containing the regression targets information. It need to be
further processed.
"""
res = self.geometry_param["res"]
downsample_rate = self.geometry_param["downsample_rate"]
bev_origin = np.array([self.geometry_param["L1"],
self.geometry_param["W1"]]).reshape(1, -1)
# discretized bbx corner representations -- (n, 4, 2)
bev_corners_dist = (bev_corners - bev_origin) / res / downsample_rate
# generate the coordinates of m
x = np.arange(self.geometry_param["label_shape"][0])
y = np.arange(self.geometry_param["label_shape"][1])
xx, yy = np.meshgrid(x, y)
# (label_shape[0]*label_shape[1], 2)
points = np.concatenate([xx.reshape(-1, 1), yy.reshape(-1, 1)], axis=-1)
bev_origin_dist = bev_origin / res / downsample_rate
# loop over each bbx, find the points within the bbx.
for i in range(bev_corners.shape[0]):
reg_target = reg_targets[i, :]
# find discredited points in bbx
points_in_box = \
box_utils.get_points_in_rotated_box(points,
bev_corners_dist[i, ...])
# convert points to continuous space
points_continuous = dist_to_continuous(points_in_box,
bev_origin_dist,
res,
downsample_rate)
actual_reg_target = np.repeat(reg_target.reshape(1, -1),
points_continuous.shape[0],
axis=0)
# build learning targets
actual_reg_target[:, 2:4] = \
actual_reg_target[:, 2:4] - points_continuous
actual_reg_target[:, 4:] = np.log(actual_reg_target[:, 4:])
# update label map
label_map[points_in_box[:, 0], points_in_box[:, 1], 0] = 1.0
label_map[points_in_box[:, 0], points_in_box[:, 1], 1:] = \
actual_reg_target
def normalize_targets(self, label_map):
"""
Normalize label_map
Parameters
----------
label_map : numpy.array
Targets array for classification and regression tasks with the
shape of label_shape.
Returns
-------
label_map: numpy.array
Nromalized label_map.
"""
label_map[..., 1:] = \
(label_map[..., 1:] - self.target_mean) / self.target_std_dev
return label_map
def denormalize_reg_map(self, reg_map):
"""
Denormalize the regression map
Parameters
----------
reg_map : np.ndarray / torch.Tensor
Regression output mapwith the shape of (label_shape[0],
label_shape[1], 6).
Returns
-------
reg_map : np.ndarray / torch.Tensor
Denormalized regression map.
"""
if isinstance(reg_map, np.ndarray):
target_mean = self.target_mean
target_std_dev = self.target_std_dev
else:
target_mean = \
torch.from_numpy(self.target_mean).to(reg_map.device)
target_std_dev =\
torch.from_numpy(self.target_std_dev).to(reg_map.device)
reg_map = reg_map * target_std_dev + target_mean
return reg_map
@staticmethod
def collate_batch(label_batch_list):
"""
Customized collate function for target label generation.
Parameters
----------
label_batch_list : list
The list of dictionary that contains all labels for several
frames.
Returns
-------
processed_batch : dict
Reformatted labels in torch tensor.
"""
label_map_list = [x["label_map"][np.newaxis, ...] for x in
label_batch_list]
processed_batch = {
# (batch_size, 7, label_shape[0], label_shape[1])
"label_map": torch.from_numpy(np.concatenate(label_map_list,
axis=0)),
"bev_corners": [torch.from_numpy(x["bev_corners"]) for x in
label_batch_list]
}
return processed_batch
def post_process(self, data_dict, output_dict):
"""
Process the outputs of the model to 2D bounding box.
Step1: convert each cav's output to bounding box format
Step2: project the bounding boxes to ego space.
Step:3 NMS
Parameters
----------
data_dict : dict
The dictionary containing the origin input data of model.
output_dict :dict
The dictionary containing the output of the model.
Returns
-------
pred_box2d_tensor : torch.Tensor
The prediction bounding box tensor after NMS.
gt_box2d_tensor : torch.Tensor
The groundtruth bounding box tensor.
"""
# the final bounding box list
pred_box2d_list = []
pred_score_list = []
for cav_id, cav_content in data_dict.items():
assert cav_id in output_dict
# the transformation matrix to ego space
transformation_matrix = cav_content['transformation_matrix']
# classification probability -- (label_shape[0], label_shape[1])
prob = output_dict[cav_id]['cls'].squeeze(0).squeeze(0)
prob = torch.sigmoid(prob)
# regression map -- (label_shape[0], label_shape[1], 6)
reg_map = output_dict[cav_id]['reg'].squeeze(0).permute(1, 2, 0)
reg_map = self.denormalize_reg_map(reg_map)
threshold = self.params['target_args']['score_threshold']
mask = torch.gt(prob, threshold)
if mask.sum() > 0:
# (number of high confidence bbx, 4, 2)
corners2d = self.reg_map_to_bbx_corners(reg_map, mask)
# assume the z-diviation in transformation_matrix is small,
# thus we can pad zeros to simulate the 3d transformation.
# (number of high confidence bbx, 4, 3)
box3d = F.pad(corners2d, (0, 1))
# (number of high confidence bbx, 4, 2)
projected_boxes2d = \
box_utils.project_points_by_matrix_torch(box3d.view(-1, 3),
transformation_matrix)[:, :2]
projected_boxes2d = projected_boxes2d.view(-1, 4, 2)
scores = prob[mask]
pred_box2d_list.append(projected_boxes2d)
pred_score_list.append(scores)
if len(pred_box2d_list):
pred_box2ds = torch.cat(pred_box2d_list, dim=0)
pred_scores = torch.cat(pred_score_list, dim=0)
else:
return None, None
keep_index = box_utils.nms_rotated(pred_box2ds, pred_scores,
self.params['nms_thresh'])
if len(keep_index):
pred_box2ds = pred_box2ds[keep_index]
pred_scores = pred_scores[keep_index]
# filter out the prediction out of the range.
mask = box_utils.get_mask_for_boxes_within_range_torch(pred_box2ds)
pred_box2ds = pred_box2ds[mask, :, :]
pred_scores = pred_scores[mask]
assert pred_scores.shape[0] == pred_box2ds.shape[0]
return pred_box2ds, pred_scores
def reg_map_to_bbx_corners(self, reg_map, mask):
"""
Construct bbx from the regression output of the model.
Parameters
----------
reg_map : torch.Tensor
Regression output of neural networks.
mask : torch.Tensor
Masks used to filter bbx.
Returns
-------
corners : torch.Tensor
Bbx output with shape (N, 4, 2).
"""
assert len(reg_map.shape) == 3,\
"only support shape of label_shape i.e. (*, *, 6)"
device = reg_map.device
cos_t, sin_t, x, y, log_dx, log_dy = \
[tt.squeeze(-1) for tt in torch.chunk(reg_map, 6, dim=-1)]
yaw = torch.atan2(sin_t, cos_t)
dx, dy = log_dx.exp(), log_dy.exp()
grid_size = self.geometry_param["res"] * \
self.geometry_param["downsample_rate"]
grid_x = torch.arange(self.geometry_param["L1"],
self.geometry_param["L2"],
grid_size, dtype=torch.float32, device=device)
grid_y = torch.arange(self.geometry_param["W1"],
self.geometry_param["W2"],
grid_size,
dtype=torch.float32,
device=device)
xx, yy = torch.meshgrid([grid_x, grid_y])
center_x = xx + x
center_y = yy + y
bbx2d = torch.stack([center_x, center_y, dx, dy, yaw], dim=-1)
bbx2d = bbx2d[mask, :]
corners = box_utils.boxes2d_to_corners2d(bbx2d)
return corners
def post_process_debug(self, data_dict, output_dict):
"""
Process the outputs of the model to 2D bounding box for debug purpose.
Step1: convert each cav's output to bounding box format
Step2: project the bounding boxes to ego space.
Step:3 NMS
Parameters
----------
data_dict : dict
The dictionary containing the origin input data of model.
output_dict :dict
The dictionary containing the output of the model.
Returns
-------
pred_box2d_tensor : torch.Tensor
The prediction bounding box tensor after NMS.
gt_box2d_tensor : torch.Tensor
The groundtruth bounding box tensor.
"""
# the final bounding box list
pred_box2d_list = []
pred_score_list = []
# the transformation matrix to ego space
transformation_matrix = data_dict['transformation_matrix']
# classification probability -- (label_shape[0], label_shape[1])
prob = output_dict['cls'].squeeze(0).squeeze(0)
prob = torch.sigmoid(prob)
# regression map -- (label_shape[0], label_shape[1], 6)
reg_map = output_dict['reg'].squeeze(0).permute(1, 2, 0)
reg_map = self.denormalize_reg_map(reg_map)
# threshold = self.params['target_args']['score_threshold']
threshold = 0.5
mask = torch.gt(prob, threshold)
if mask.sum() > 0:
# (number of high confidence bbx, 4, 2)
corners2d = self.reg_map_to_bbx_corners(reg_map, mask)
# assume the z-diviation in transformation_matrix is small,
# thus we can pad zeros to simulate the 3d transformation.
# (number of high confidence bbx, 4, 3)
box3d = F.pad(corners2d, (0, 1))
# (number of high confidence bbx, 4, 2)
projected_boxes2d = \
box_utils.project_points_by_matrix_torch(box3d.view(-1, 3),
transformation_matrix)[:, :2]
projected_boxes2d = projected_boxes2d.view(-1, 4, 2)
scores = prob[mask]
pred_box2d_list.append(projected_boxes2d)
pred_score_list.append(scores)
pred_box2ds = torch.cat(pred_box2d_list, dim=0)
pred_scores = torch.cat(pred_score_list, dim=0)
keep_index = box_utils.nms_rotated(pred_box2ds,
pred_scores,
self.params['nms_thresh'])
pred_box2ds = pred_box2ds[keep_index]
# filter out the prediction out of the range.
mask = box_utils.get_mask_for_boxes_within_range_torch(pred_box2ds)
pred_box2ds = pred_box2ds[mask, :, :]
return pred_box2ds
@staticmethod
def visualize(pred_box_tensor, gt_tensor, pcd, show_vis, save_path, dataset = None):
"""
Visualize the BEV 2D prediction, ground truth with point cloud together.
Parameters
----------
pred_box_tensor : torch.Tensor
(N, 8, 3) prediction.
gt_tensor : torch.Tensor
(N, 8, 3) groundtruth bbx
pcd : torch.Tensor
PointCloud, (N, 4).
show_vis : bool
Whether to show visualization.
save_path : str
Save the visualization results to given path.
dataset : BaseDataset
opencood dataset object.
"""
assert dataset is not None, "dataset argument can't be None"
vis_utils.visualize_single_sample_output_bev(pred_box_tensor,
gt_tensor,
pcd,
dataset,
show_vis,
save_path)
from v2xvit.data_utils.post_processor.voxel_postprocessor import VoxelPostprocessor
from v2xvit.data_utils.post_processor.bev_postprocessor import BevPostprocessor
__all__ = {
'VoxelPostprocessor': VoxelPostprocessor,
'BevPostprocessor': BevPostprocessor,
}
def build_postprocessor(anchor_cfg, train):
process_method_name = anchor_cfg['core_method']
assert process_method_name in ['VoxelPostprocessor', 'BevPostprocessor']
anchor_generator = __all__[process_method_name](
anchor_params=anchor_cfg,
train=train
)
return anchor_generator
"""
Dataset class for late fusion
"""
import random
import math
from collections import OrderedDict
import numpy as np
import torch
from torch.utils.data import DataLoader
import v2xvit
from v2xvit.data_utils.post_processor import build_postprocessor
from v2xvit.data_utils.datasets import basedataset
from v2xvit.data_utils.pre_processor import build_preprocessor
from v2xvit.hypes_yaml.yaml_utils import load_yaml
from v2xvit.utils import box_utils
from v2xvit.utils.pcd_utils import \
mask_points_by_range, mask_ego_points, shuffle_points, \
downsample_lidar_minimum
class LateFusionDataset(basedataset.BaseDataset):
def __init__(self, params, visualize, train=True):
super(LateFusionDataset, self).__init__(params, visualize, train)
self.pre_processor = build_preprocessor(params['preprocess'],
train)
self.post_processor = build_postprocessor(params['postprocess'], train)
def __getitem__(self, idx):
base_data_dict = self.retrieve_base_data(idx, cur_ego_pose_flag=True)
if self.train:
reformat_data_dict = self.get_item_train(base_data_dict)
else:
reformat_data_dict = self.get_item_test(base_data_dict)
return reformat_data_dict
def get_item_single_car(self, selected_cav_base):
"""
Process a single CAV's information for the train/test pipeline.
Parameters
----------
selected_cav_base : dict
The dictionary contains a single CAV's raw information.
Returns
-------
selected_cav_processed : dict
The dictionary contains the cav's processed information.
"""
selected_cav_processed = {}
# filter lidar
lidar_np = selected_cav_base['lidar_np']
lidar_np = shuffle_points(lidar_np)
lidar_np = mask_points_by_range(lidar_np,
self.params['preprocess'][
'cav_lidar_range'])
# remove points that hit ego vehicle
lidar_np = mask_ego_points(lidar_np)
# generate the bounding box(n, 7) under the cav's space
object_bbx_center, object_bbx_mask, object_ids = \
self.post_processor.generate_object_center([selected_cav_base],
selected_cav_base[
'params'][
'lidar_pose'])
# data augmentation
lidar_np, object_bbx_center, object_bbx_mask = \
self.augment(lidar_np, object_bbx_center, object_bbx_mask)
if self.visualize:
selected_cav_processed.update({'origin_lidar': lidar_np})
# pre-process the lidar to voxel/bev/downsampled lidar
lidar_dict = self.pre_processor.preprocess(lidar_np)
selected_cav_processed.update({'processed_lidar': lidar_dict})
# generate the anchor boxes
anchor_box = self.post_processor.generate_anchor_box()
selected_cav_processed.update({'anchor_box': anchor_box})
selected_cav_processed.update({'object_bbx_center': object_bbx_center,
'object_bbx_mask': object_bbx_mask,
'object_ids': object_ids})
# generate targets label
label_dict = \
self.post_processor.generate_label(
gt_box_center=object_bbx_center,
anchors=anchor_box,
mask=object_bbx_mask)
selected_cav_processed.update({'label_dict': label_dict})
return selected_cav_processed
def get_item_train(self, base_data_dict):
processed_data_dict = OrderedDict()
# during training, we return a random cav's data
if not self.visualize:
selected_cav_id, selected_cav_base = \
random.choice(list(base_data_dict.items()))
else:
selected_cav_id, selected_cav_base = \
list(base_data_dict.items())[0]
selected_cav_processed = self.get_item_single_car(selected_cav_base)
processed_data_dict.update({'ego': selected_cav_processed})
return processed_data_dict
def get_item_test(self, base_data_dict):
processed_data_dict = OrderedDict()
ego_id = -1
ego_lidar_pose = []
# first find the ego vehicle's lidar pose
for cav_id, cav_content in base_data_dict.items():
if cav_content['ego']:
ego_id = cav_id
ego_lidar_pose = cav_content['params']['lidar_pose']
break
assert ego_id != -1
assert len(ego_lidar_pose) > 0
# loop over all CAVs to process information
for cav_id, selected_cav_base in base_data_dict.items():
distance = \
math.sqrt((selected_cav_base['params']['lidar_pose'][0] -
ego_lidar_pose[0]) ** 2 + (
selected_cav_base['params'][
'lidar_pose'][1] - ego_lidar_pose[
1]) ** 2)
if distance > v2xvit.data_utils.datasets.COM_RANGE:
continue
# find the transformation matrix from current cav to ego.
# this is used to project prediction to the right space
transformation_matrix = \
selected_cav_base['params']['transformation_matrix']
# this is used to project gt objects to ego space
gt_transformation_matrix = \
selected_cav_base['params']['gt_transformation_matrix']
selected_cav_processed = \
self.get_item_single_car(selected_cav_base)
selected_cav_processed.update({'transformation_matrix':
transformation_matrix})
selected_cav_processed.update({'gt_transformation_matrix':
gt_transformation_matrix})
update_cav = "ego" if cav_id == ego_id else cav_id
processed_data_dict.update({update_cav: selected_cav_processed})
return processed_data_dict
def collate_batch_test(self, batch):
"""
Customized collate function for pytorch dataloader during testing
for late fusion dataset.
Parameters
----------
batch : dict
Returns
-------
batch : dict
Reformatted batch.
"""
# currently, we only support batch size of 1 during testing
assert len(batch) <= 1, "Batch size 1 is required during testing!"
batch = batch[0]
output_dict = {}
# for late fusion, we also need to stack the lidar for better
# visualization
if self.visualize:
projected_lidar_list = []
origin_lidar = []
for cav_id, cav_content in batch.items():
output_dict.update({cav_id: {}})
# shape: (1, max_num, 7)
object_bbx_center = \
torch.from_numpy(np.array([cav_content['object_bbx_center']]))
object_bbx_mask = \
torch.from_numpy(np.array([cav_content['object_bbx_mask']]))
object_ids = cav_content['object_ids']
# the anchor box is the same for all bounding boxes usually, thus
# we don't need the batch dimension.
if cav_content['anchor_box'] is not None:
output_dict[cav_id].update({'anchor_box':
torch.from_numpy(np.array(
cav_content[
'anchor_box']))})
if self.visualize:
transformation_matrix = cav_content['transformation_matrix']
origin_lidar = [cav_content['origin_lidar']]
projected_lidar = cav_content['origin_lidar']
projected_lidar[:, :3] = \
box_utils.project_points_by_matrix_torch(
projected_lidar[:, :3],
transformation_matrix)
projected_lidar_list.append(projected_lidar)
# processed lidar dictionary
processed_lidar_torch_dict = \
self.pre_processor.collate_batch(
[cav_content['processed_lidar']])
# label dictionary
label_torch_dict = \
self.post_processor.collate_batch([cav_content['label_dict']])
# save the transformation matrix (4, 4) to ego vehicle
transformation_matrix_torch = \
torch.from_numpy(
np.array(cav_content['transformation_matrix'])).float()
gt_transformation_matrix_torch = \
torch.from_numpy(
np.array(cav_content['gt_transformation_matrix'])).float()
output_dict[cav_id].update({'object_bbx_center': object_bbx_center,
'object_bbx_mask': object_bbx_mask,
'processed_lidar': processed_lidar_torch_dict,
'label_dict': label_torch_dict,
'object_ids': object_ids,
'transformation_matrix': transformation_matrix_torch,
'gt_transformation_matrix': gt_transformation_matrix_torch})
if self.visualize:
origin_lidar = \
np.array(
downsample_lidar_minimum(pcd_np_list=origin_lidar))
origin_lidar = torch.from_numpy(origin_lidar)
output_dict[cav_id].update({'origin_lidar': origin_lidar})
if self.visualize:
projected_lidar_stack = [torch.from_numpy(
np.vstack(projected_lidar_list))]
output_dict['ego'].update({'origin_lidar': projected_lidar_stack})
return output_dict
def post_process(self, data_dict, output_dict):
"""
Process the outputs of the model to 2D/3D bounding box.
Parameters
----------
data_dict : dict
The dictionary containing the origin input data of model.
output_dict :dict
The dictionary containing the output of the model.
Returns
-------
pred_box_tensor : torch.Tensor
The tensor of prediction bounding box after NMS.
gt_box_tensor : torch.Tensor
The tensor of gt bounding box.
"""
pred_box_tensor, pred_score = \
self.post_processor.post_process(data_dict, output_dict)
gt_box_tensor = self.post_processor.generate_gt_bbx(data_dict)
return pred_box_tensor, pred_score, gt_box_tensor
"""
Dataset class for early fusion
"""
import math
from collections import OrderedDict
import numpy as np
import torch
import v2xvit
import v2xvit.data_utils.post_processor as post_processor
from v2xvit.utils import box_utils
from v2xvit.data_utils.datasets import basedataset
from v2xvit.data_utils.pre_processor import build_preprocessor
from v2xvit.utils.pcd_utils import \
mask_points_by_range, mask_ego_points, shuffle_points, \
downsample_lidar_minimum
class IntermediateFusionDataset(basedataset.BaseDataset):
def __init__(self, params, visualize, train=True):
super(IntermediateFusionDataset, self). \
__init__(params, visualize, train)
self.cur_ego_pose_flag = params['fusion']['args']['cur_ego_pose_flag']
self.pre_processor = build_preprocessor(params['preprocess'],
train)
self.post_processor = post_processor.build_postprocessor(
params['postprocess'],
train)
def __getitem__(self, idx):
# when the cur_ego_pose_flag is set to True, there is no time gap
# between the time when the LiDAR data is captured by connected
# agents and when the extracted features are received by
# the ego vehicle. This is equal to implement STCM.
base_data_dict = \
self.retrieve_base_data(idx,
cur_ego_pose_flag=self.cur_ego_pose_flag)
processed_data_dict = OrderedDict()
processed_data_dict['ego'] = {}
ego_id = -1
ego_lidar_pose = []
# first find the ego vehicle's lidar pose
for cav_id, cav_content in base_data_dict.items():
if cav_content['ego']:
ego_id = cav_id
ego_lidar_pose = cav_content['params']['lidar_pose']
break
assert cav_id == list(base_data_dict.keys())[
0], "The first element in the OrderedDict must be ego"
assert ego_id != -1
assert len(ego_lidar_pose) > 0
# this is used for v2vnet and disconet
pairwise_t_matrix = \
self.get_pairwise_transformation(base_data_dict,
self.params['train_params'][
'max_cav'])
processed_features = []
object_stack = []
object_id_stack = []
# prior knowledge for time delay correction and indicating data type
# (V2V vs V2i)
velocity = []
time_delay = []
infra = []
spatial_correction_matrix = []
if self.visualize:
projected_lidar_stack = []
# loop over all CAVs to process information
for cav_id, selected_cav_base in base_data_dict.items():
# check if the cav is within the communication range with ego
distance = \
math.sqrt((selected_cav_base['params']['lidar_pose'][0] -
ego_lidar_pose[0]) ** 2 + (
selected_cav_base['params'][
'lidar_pose'][1] - ego_lidar_pose[
1]) ** 2)
if distance > v2xvit.data_utils.datasets.COM_RANGE:
continue
selected_cav_processed, void_lidar = self.get_item_single_car(
selected_cav_base,
ego_lidar_pose)
if void_lidar:
continue
object_stack.append(selected_cav_processed['object_bbx_center'])
object_id_stack += selected_cav_processed['object_ids']
processed_features.append(
selected_cav_processed['processed_features'])
velocity.append(selected_cav_processed['velocity'])
time_delay.append(float(selected_cav_base['time_delay']))
spatial_correction_matrix.append(
selected_cav_base['params']['spatial_correction_matrix'])
infra.append(1 if int(cav_id) < 0 else 0)
if self.visualize:
projected_lidar_stack.append(
selected_cav_processed['projected_lidar'])
# exclude all repetitive objects
unique_indices = \
[object_id_stack.index(x) for x in set(object_id_stack)]
object_stack = np.vstack(object_stack)
object_stack = object_stack[unique_indices]
# make sure bounding boxes across all frames have the same number
object_bbx_center = \
np.zeros((self.params['postprocess']['max_num'], 7))
mask = np.zeros(self.params['postprocess']['max_num'])
object_bbx_center[:object_stack.shape[0], :] = object_stack
mask[:object_stack.shape[0]] = 1
# merge preprocessed features from different cavs into the same dict
cav_num = len(processed_features)
merged_feature_dict = self.merge_features_to_dict(processed_features)
# generate the anchor boxes
anchor_box = self.post_processor.generate_anchor_box()
# generate targets label
label_dict = \
self.post_processor.generate_label(
gt_box_center=object_bbx_center,
anchors=anchor_box,
mask=mask)
# pad dv, dt, infra to max_cav
velocity = velocity + (self.max_cav - len(velocity)) * [0.]
time_delay = time_delay + (self.max_cav - len(time_delay)) * [0.]
infra = infra + (self.max_cav - len(infra)) * [0.]
spatial_correction_matrix = np.stack(spatial_correction_matrix)
padding_eye = np.tile(np.eye(4)[None],(self.max_cav - len(
spatial_correction_matrix),1,1))
spatial_correction_matrix = np.concatenate([spatial_correction_matrix, padding_eye], axis=0)
processed_data_dict['ego'].update(
{'object_bbx_center': object_bbx_center,
'object_bbx_mask': mask,
'object_ids': [object_id_stack[i] for i in unique_indices],
'anchor_box': anchor_box,
'processed_lidar': merged_feature_dict,
'label_dict': label_dict,
'cav_num': cav_num,
'velocity': velocity,
'time_delay': time_delay,
'infra': infra,
'spatial_correction_matrix': spatial_correction_matrix,
'pairwise_t_matrix': pairwise_t_matrix})
if self.visualize:
processed_data_dict['ego'].update({'origin_lidar':
np.vstack(
projected_lidar_stack)})
return processed_data_dict
@staticmethod
def get_pairwise_transformation(base_data_dict, max_cav):
"""
Get pair-wise transformation matrix across different agents.
This is only used for v2vnet and disconet. Currently we set
this as identity matrix as the pointcloud is projected to
ego vehicle first.
Parameters
----------
base_data_dict : dict
Key : cav id, item: transformation matrix to ego, lidar points.
max_cav : int
The maximum number of cav, default 5
Return
------
pairwise_t_matrix : np.array
The pairwise transformation matrix across each cav.
shape: (L, L, 4, 4)
"""
pairwise_t_matrix = np.zeros((max_cav, max_cav, 4, 4))
# default are identity matrix
pairwise_t_matrix[:, :] = np.identity(4)
return pairwise_t_matrix
def get_item_single_car(self, selected_cav_base, ego_pose):
"""
Project the lidar and bbx to ego space first, and then do clipping.
Parameters
----------
selected_cav_base : dict
The dictionary contains a single CAV's raw information.
ego_pose : list
The ego vehicle lidar pose under world coordinate.
Returns
-------
selected_cav_processed : dict
The dictionary contains the cav's processed information.
"""
selected_cav_processed = {}
# calculate the transformation matrix
transformation_matrix = \
selected_cav_base['params']['transformation_matrix']
# retrieve objects under ego coordinates
object_bbx_center, object_bbx_mask, object_ids = \
self.post_processor.generate_object_center([selected_cav_base],
ego_pose)
# filter lidar
lidar_np = selected_cav_base['lidar_np']
lidar_np = shuffle_points(lidar_np)
# remove points that hit itself
lidar_np = mask_ego_points(lidar_np)
# project the lidar to ego space
lidar_np[:, :3] = \
box_utils.project_points_by_matrix_torch(lidar_np[:, :3],
transformation_matrix)
lidar_np = mask_points_by_range(lidar_np,
self.params['preprocess'][
'cav_lidar_range'])
# Check if filtered LiDAR points are not void
void_lidar = True if lidar_np.shape[0] < 1 else False
processed_lidar = self.pre_processor.preprocess(lidar_np)
# velocity
velocity = selected_cav_base['params']['ego_speed']
# normalize veloccity by average speed 30 km/h
velocity = velocity / 30
selected_cav_processed.update(
{'object_bbx_center': object_bbx_center[object_bbx_mask == 1],
'object_ids': object_ids,
'projected_lidar': lidar_np,
'processed_features': processed_lidar,
'velocity': velocity})
return selected_cav_processed, void_lidar
@staticmethod
def merge_features_to_dict(processed_feature_list):
"""
Merge the preprocessed features from different cavs to the same
dictionary.
Parameters
----------
processed_feature_list : list
A list of dictionary containing all processed features from
different cavs.
Returns
-------
merged_feature_dict: dict
key: feature names, value: list of features.
"""
merged_feature_dict = OrderedDict()
for i in range(len(processed_feature_list)):
for feature_name, feature in processed_feature_list[i].items():
if feature_name not in merged_feature_dict:
merged_feature_dict[feature_name] = []
if isinstance(feature, list):
merged_feature_dict[feature_name] += feature
else:
merged_feature_dict[feature_name].append(feature)
return merged_feature_dict
def collate_batch_train(self, batch):
# Intermediate fusion is different the other two
output_dict = {'ego': {}}
object_bbx_center = []
object_bbx_mask = []
object_ids = []
processed_lidar_list = []
# used to record different scenario
record_len = []
label_dict_list = []
# used for PriorEncoding
velocity = []
time_delay = []
infra = []
# pairwise transformation matrix
pairwise_t_matrix_list = []
# used for correcting the spatial transformation between delayed timestamp
# and current timestamp
spatial_correction_matrix_list = []
if self.visualize:
origin_lidar = []
for i in range(len(batch)):
ego_dict = batch[i]['ego']
object_bbx_center.append(ego_dict['object_bbx_center'])
object_bbx_mask.append(ego_dict['object_bbx_mask'])
object_ids.append(ego_dict['object_ids'])
processed_lidar_list.append(ego_dict['processed_lidar'])
record_len.append(ego_dict['cav_num'])
label_dict_list.append(ego_dict['label_dict'])
velocity.append(ego_dict['velocity'])
time_delay.append(ego_dict['time_delay'])
infra.append(ego_dict['infra'])
spatial_correction_matrix_list.append(
ego_dict['spatial_correction_matrix'])
pairwise_t_matrix_list.append(ego_dict['pairwise_t_matrix'])
if self.visualize:
origin_lidar.append(ego_dict['origin_lidar'])
# convert to numpy, (B, max_num, 7)
object_bbx_center = torch.from_numpy(np.array(object_bbx_center))
object_bbx_mask = torch.from_numpy(np.array(object_bbx_mask))
# example: {'voxel_features':[np.array([1,2,3]]),
# np.array([3,5,6]), ...]}
merged_feature_dict = self.merge_features_to_dict(processed_lidar_list)
processed_lidar_torch_dict = \
self.pre_processor.collate_batch(merged_feature_dict)
# [2, 3, 4, ..., M]
record_len = torch.from_numpy(np.array(record_len, dtype=int))
label_torch_dict = \
self.post_processor.collate_batch(label_dict_list)
# (B, max_cav)
velocity = torch.from_numpy(np.array(velocity))
time_delay = torch.from_numpy(np.array(time_delay))
infra = torch.from_numpy(np.array(infra))
spatial_correction_matrix_list = \
torch.from_numpy(np.array(spatial_correction_matrix_list))
# (B, max_cav, 3)
prior_encoding = \
torch.stack([velocity, time_delay, infra], dim=-1).float()
# (B, max_cav)
pairwise_t_matrix = torch.from_numpy(np.array(pairwise_t_matrix_list))
# object id is only used during inference, where batch size is 1.
# so here we only get the first element.
output_dict['ego'].update({'object_bbx_center': object_bbx_center,
'object_bbx_mask': object_bbx_mask,
'processed_lidar': processed_lidar_torch_dict,
'record_len': record_len,
'label_dict': label_torch_dict,
'object_ids': object_ids[0],
'prior_encoding': prior_encoding,
'spatial_correction_matrix': spatial_correction_matrix_list,
'pairwise_t_matrix': pairwise_t_matrix})
if self.visualize:
origin_lidar = \
np.array(downsample_lidar_minimum(pcd_np_list=origin_lidar))
origin_lidar = torch.from_numpy(origin_lidar)
output_dict['ego'].update({'origin_lidar': origin_lidar})
return output_dict
def collate_batch_test(self, batch):
assert len(batch) <= 1, "Batch size 1 is required during testing!"
output_dict = self.collate_batch_train(batch)
# check if anchor box in the batch
if batch[0]['ego']['anchor_box'] is not None:
output_dict['ego'].update({'anchor_box':
torch.from_numpy(np.array(
batch[0]['ego'][
'anchor_box']))})
# save the transformation matrix (4, 4) to ego vehicle
transformation_matrix_torch = \
torch.from_numpy(np.identity(4)).float()
output_dict['ego'].update({'transformation_matrix':
transformation_matrix_torch})
return output_dict
def post_process(self, data_dict, output_dict):
"""
Process the outputs of the model to 2D/3D bounding box.
Parameters
----------
data_dict : dict
The dictionary containing the origin input data of model.
output_dict :dict
The dictionary containing the output of the model.
Returns
-------
pred_box_tensor : torch.Tensor
The tensor of prediction bounding box after NMS.
gt_box_tensor : torch.Tensor
The tensor of gt bounding box.
"""
pred_box_tensor, pred_score = \
self.post_processor.post_process(data_dict, output_dict)
gt_box_tensor = self.post_processor.generate_gt_bbx(data_dict)
return pred_box_tensor, pred_score, gt_box_tensor
"""
This is a dataset for early fusion visualization only.
"""
from collections import OrderedDict
import numpy as np
import torch
from v2xvit.utils import box_utils
from v2xvit.data_utils.post_processor import build_postprocessor
from v2xvit.data_utils.datasets import basedataset
from v2xvit.data_utils.pre_processor import build_preprocessor
from v2xvit.utils.pcd_utils import \
mask_points_by_range, mask_ego_points, shuffle_points, \
downsample_lidar_minimum
class EarlyFusionVisDataset(basedataset.BaseDataset):
def __init__(self, params, visualize, train=True):
super(EarlyFusionVisDataset, self).__init__(params, visualize, train)
self.pre_processor = build_preprocessor(params['preprocess'],
train)
self.post_processor = build_postprocessor(params['postprocess'], train)
def __getitem__(self, idx):
base_data_dict = self.retrieve_base_data(idx)
processed_data_dict = OrderedDict()
processed_data_dict['ego'] = {}
ego_id = -1
ego_lidar_pose = []
# first find the ego vehicle's lidar pose
for cav_id, cav_content in base_data_dict.items():
if cav_content['ego']:
ego_id = cav_id
ego_lidar_pose = cav_content['params']['lidar_pose']
break
assert ego_id != -1
assert len(ego_lidar_pose) > 0
projected_lidar_stack = []
object_stack = []
object_id_stack = []
# loop over all CAVs to process information
for cav_id, selected_cav_base in base_data_dict.items():
selected_cav_processed = self.get_item_single_car(
selected_cav_base,
ego_lidar_pose)
# all these lidar and object coordinates are projected to ego
# already.
projected_lidar_stack.append(
selected_cav_processed['projected_lidar'])
object_stack.append(selected_cav_processed['object_bbx_center'])
object_id_stack += selected_cav_processed['object_ids']
# exclude all repetitive objects
unique_indices = \
[object_id_stack.index(x) for x in set(object_id_stack)]
object_stack = np.vstack(object_stack)
object_stack = object_stack[unique_indices]
# make sure bounding boxes across all frames have the same number
object_bbx_center = \
np.zeros((self.params['postprocess']['max_num'], 7))
mask = np.zeros(self.params['postprocess']['max_num'])
object_bbx_center[:object_stack.shape[0], :] = object_stack
mask[:object_stack.shape[0]] = 1
# convert list to numpy array, (N, 4)
projected_lidar_stack = np.vstack(projected_lidar_stack)
# data augmentation
projected_lidar_stack, object_bbx_center, mask = \
self.augment(projected_lidar_stack, object_bbx_center, mask)
# we do lidar filtering in the stacked lidar
projected_lidar_stack = mask_points_by_range(projected_lidar_stack,
self.params['preprocess'][
'cav_lidar_range'])
# augmentation may remove some of the bbx out of range
object_bbx_center_valid = object_bbx_center[mask == 1]
object_bbx_center_valid = \
box_utils.mask_boxes_outside_range_numpy(object_bbx_center_valid,
self.params['preprocess'][
'cav_lidar_range'],
self.params['postprocess'][
'order']
)
mask[object_bbx_center_valid.shape[0]:] = 0
object_bbx_center[:object_bbx_center_valid.shape[0]] = \
object_bbx_center_valid
object_bbx_center[object_bbx_center_valid.shape[0]:] = 0
processed_data_dict['ego'].update(
{'object_bbx_center': object_bbx_center,
'object_bbx_mask': mask,
'object_ids': [object_id_stack[i] for i in unique_indices],
'origin_lidar': projected_lidar_stack
})
return processed_data_dict
def get_item_single_car(self, selected_cav_base, ego_pose):
"""
Project the lidar and bbx to ego space first, and then do clipping.
Parameters
----------
selected_cav_base : dict
The dictionary contains a single CAV's raw information.
ego_pose : list
The ego vehicle lidar pose under world coordinate.
Returns
-------
selected_cav_processed : dict
The dictionary contains the cav's processed information.
"""
selected_cav_processed = {}
# calculate the transformation matrix
transformation_matrix = \
selected_cav_base['params']['transformation_matrix']
# retrieve objects under ego coordinates
object_bbx_center, object_bbx_mask, object_ids = \
self.post_processor.generate_object_center([selected_cav_base],
ego_pose)
# filter lidar
lidar_np = selected_cav_base['lidar_np']
lidar_np = shuffle_points(lidar_np)
# remove points that hit itself
lidar_np = mask_ego_points(lidar_np)
# project the lidar to ego space
lidar_np[:, :3] = \
box_utils.project_points_by_matrix_torch(lidar_np[:, :3],
transformation_matrix)
selected_cav_processed.update(
{'object_bbx_center': object_bbx_center[object_bbx_mask == 1],
'object_ids': object_ids,
'projected_lidar': lidar_np})
return selected_cav_processed
def collate_batch_train(self, batch):
"""
Customized collate function for pytorch dataloader during training
for late fusion dataset.
Parameters
----------
batch : dict
Returns
-------
batch : dict
Reformatted batch.
"""
# during training, we only care about ego.
output_dict = {'ego': {}}
object_bbx_center = []
object_bbx_mask = []
origin_lidar = []
for i in range(len(batch)):
ego_dict = batch[i]['ego']
object_bbx_center.append(ego_dict['object_bbx_center'])
object_bbx_mask.append(ego_dict['object_bbx_mask'])
origin_lidar.append(ego_dict['origin_lidar'])
# convert to numpy, (B, max_num, 7)
object_bbx_center = torch.from_numpy(np.array(object_bbx_center))
object_bbx_mask = torch.from_numpy(np.array(object_bbx_mask))
output_dict['ego'].update({'object_bbx_center': object_bbx_center,
'object_bbx_mask': object_bbx_mask})
origin_lidar = \
np.array(downsample_lidar_minimum(pcd_np_list=origin_lidar))
origin_lidar = torch.from_numpy(origin_lidar)
output_dict['ego'].update({'origin_lidar': origin_lidar})
return output_dict
"""
Dataset class for early fusion
"""
import math
from collections import OrderedDict
import numpy as np
import torch
import v2xvit
from v2xvit.utils import box_utils
from v2xvit.data_utils.post_processor import build_postprocessor
from v2xvit.data_utils.datasets import basedataset
from v2xvit.data_utils.pre_processor import build_preprocessor
from v2xvit.hypes_yaml.yaml_utils import load_yaml
from v2xvit.utils.pcd_utils import \
mask_points_by_range, mask_ego_points, shuffle_points, \
downsample_lidar_minimum
class EarlyFusionDataset(basedataset.BaseDataset):
def __init__(self, params, visualize, train=True):
super(EarlyFusionDataset, self).__init__(params, visualize, train)
self.pre_processor = build_preprocessor(params['preprocess'],
train)
self.post_processor = build_postprocessor(params['postprocess'], train)
def __getitem__(self, idx):
base_data_dict = self.retrieve_base_data(idx, cur_ego_pose_flag=True)
processed_data_dict = OrderedDict()
processed_data_dict['ego'] = {}
ego_id = -1
ego_lidar_pose = []
# first find the ego vehicle's lidar pose
for cav_id, cav_content in base_data_dict.items():
if cav_content['ego']:
ego_id = cav_id
ego_lidar_pose = cav_content['params']['lidar_pose']
break
assert ego_id != -1
assert len(ego_lidar_pose) > 0
projected_lidar_stack = []
object_stack = []
object_id_stack = []
# loop over all CAVs to process information
for cav_id, selected_cav_base in base_data_dict.items():
# check if the cav is within the communication range with ego
distance = \
math.sqrt((selected_cav_base['params']['lidar_pose'][0] -
ego_lidar_pose[0]) ** 2 + (
selected_cav_base['params'][
'lidar_pose'][1] - ego_lidar_pose[
1]) ** 2)
if distance > v2xvit.data_utils.datasets.COM_RANGE:
continue
selected_cav_processed = self.get_item_single_car(
selected_cav_base,
ego_lidar_pose)
# all these lidar and object coordinates are projected to ego
# already.
projected_lidar_stack.append(
selected_cav_processed['projected_lidar'])
object_stack.append(selected_cav_processed['object_bbx_center'])
object_id_stack += selected_cav_processed['object_ids']
# exclude all repetitive objects
unique_indices = \
[object_id_stack.index(x) for x in set(object_id_stack)]
object_stack = np.vstack(object_stack)
object_stack = object_stack[unique_indices]
# make sure bounding boxes across all frames have the same number
object_bbx_center = \
np.zeros((self.params['postprocess']['max_num'], 7))
mask = np.zeros(self.params['postprocess']['max_num'])
object_bbx_center[:object_stack.shape[0], :] = object_stack
mask[:object_stack.shape[0]] = 1
# convert list to numpy array, (N, 4)
projected_lidar_stack = np.vstack(projected_lidar_stack)
# data augmentation
projected_lidar_stack, object_bbx_center, mask = \
self.augment(projected_lidar_stack, object_bbx_center, mask)
# we do lidar filtering in the stacked lidar
projected_lidar_stack = mask_points_by_range(projected_lidar_stack,
self.params['preprocess'][
'cav_lidar_range'])
# augmentation may remove some of the bbx out of range
object_bbx_center_valid = object_bbx_center[mask == 1]
object_bbx_center_valid = \
box_utils.mask_boxes_outside_range_numpy(object_bbx_center_valid,
self.params['preprocess'][
'cav_lidar_range'],
self.params[
'postprocess'][
'order']
)
mask[object_bbx_center_valid.shape[0]:] = 0
object_bbx_center[:object_bbx_center_valid.shape[0]] = \
object_bbx_center_valid
object_bbx_center[object_bbx_center_valid.shape[0]:] = 0
# pre-process the lidar to voxel/bev/downsampled lidar
lidar_dict = self.pre_processor.preprocess(projected_lidar_stack)
# generate the anchor boxes
anchor_box = self.post_processor.generate_anchor_box()
# generate targets label
label_dict = \
self.post_processor.generate_label(
gt_box_center=object_bbx_center,
anchors=anchor_box,
mask=mask)
processed_data_dict['ego'].update(
{'object_bbx_center': object_bbx_center,
'object_bbx_mask': mask,
'object_ids': [object_id_stack[i] for i in unique_indices],
'anchor_box': anchor_box,
'processed_lidar': lidar_dict,
'label_dict': label_dict})
if self.visualize:
processed_data_dict['ego'].update({'origin_lidar':
projected_lidar_stack})
return processed_data_dict
def get_item_single_car(self, selected_cav_base, ego_pose):
"""
Project the lidar and bbx to ego space first, and then do clipping.
Parameters
----------
selected_cav_base : dict
The dictionary contains a single CAV's raw information.
ego_pose : list
The ego vehicle lidar pose under world coordinate.
Returns
-------
selected_cav_processed : dict
The dictionary contains the cav's processed information.
"""
selected_cav_processed = {}
# calculate the transformation matrix
transformation_matrix = selected_cav_base['params'][
'transformation_matrix']
# retrieve objects under ego coordinates
object_bbx_center, object_bbx_mask, object_ids = \
self.post_processor.generate_object_center([selected_cav_base],
ego_pose)
# filter lidar
lidar_np = selected_cav_base['lidar_np']
lidar_np = shuffle_points(lidar_np)
# remove points that hit itself
lidar_np = mask_ego_points(lidar_np)
# project the lidar to ego space
lidar_np[:, :3] = \
box_utils.project_points_by_matrix_torch(lidar_np[:, :3],
transformation_matrix)
selected_cav_processed.update(
{'object_bbx_center': object_bbx_center[object_bbx_mask == 1],
'object_ids': object_ids,
'projected_lidar': lidar_np})
return selected_cav_processed
def collate_batch_test(self, batch):
"""
Customized collate function for pytorch dataloader during testing
for late fusion dataset.
Parameters
----------
batch : dict
Returns
-------
batch : dict
Reformatted batch.
"""
# currently, we only support batch size of 1 during testing
assert len(batch) <= 1, "Batch size 1 is required during testing!"
batch = batch[0]
output_dict = {}
for cav_id, cav_content in batch.items():
output_dict.update({cav_id: {}})
# shape: (1, max_num, 7)
object_bbx_center = \
torch.from_numpy(np.array([cav_content['object_bbx_center']]))
object_bbx_mask = \
torch.from_numpy(np.array([cav_content['object_bbx_mask']]))
object_ids = cav_content['object_ids']
# the anchor box is the same for all bounding boxes usually, thus
# we don't need the batch dimension.
if cav_content['anchor_box'] is not None:
output_dict[cav_id].update({'anchor_box':
torch.from_numpy(np.array(
cav_content[
'anchor_box']))})
if self.visualize:
origin_lidar = [cav_content['origin_lidar']]
# processed lidar dictionary
processed_lidar_torch_dict = \
self.pre_processor.collate_batch(
[cav_content['processed_lidar']])
# label dictionary
label_torch_dict = \
self.post_processor.collate_batch([cav_content['label_dict']])
# save the transformation matrix (4, 4) to ego vehicle
transformation_matrix_torch = \
torch.from_numpy(np.identity(4)).float()
output_dict[cav_id].update({'object_bbx_center': object_bbx_center,
'object_bbx_mask': object_bbx_mask,
'processed_lidar': processed_lidar_torch_dict,
'label_dict': label_torch_dict,
'object_ids': object_ids,
'transformation_matrix': transformation_matrix_torch})
if self.visualize:
origin_lidar = \
np.array(
downsample_lidar_minimum(pcd_np_list=origin_lidar))
origin_lidar = torch.from_numpy(origin_lidar)
output_dict[cav_id].update({'origin_lidar': origin_lidar})
return output_dict
def post_process(self, data_dict, output_dict):
"""
Process the outputs of the model to 2D/3D bounding box.
Parameters
----------
data_dict : dict
The dictionary containing the origin input data of model.
output_dict :dict
The dictionary containing the output of the model.
Returns
-------
pred_box_tensor : torch.Tensor
The tensor of prediction bounding box after NMS.
gt_box_tensor : torch.Tensor
The tensor of gt bounding box.
"""
pred_box_tensor, pred_score = \
self.post_processor.post_process(data_dict, output_dict)
gt_box_tensor = self.post_processor.generate_gt_bbx(data_dict)
return pred_box_tensor, pred_score, gt_box_tensor
"""
Basedataset class for lidar data pre-processing
"""
import os
import math
from collections import OrderedDict
import torch
import numpy as np
from torch.utils.data import Dataset
import v2xvit.utils.pcd_utils as pcd_utils
from v2xvit.data_utils.augmentor.data_augmentor import DataAugmentor
from v2xvit.hypes_yaml.yaml_utils import load_yaml
from v2xvit.utils.pcd_utils import downsample_lidar_minimum
from v2xvit.utils.transformation_utils import x1_to_x2
class BaseDataset(Dataset):
"""
Base dataset for all kinds of fusion. Mainly used to assign correct
index and add noise.
Parameters
__________
params : dict
The dictionary contains all parameters for training/testing.
visualize : false
If set to true, the dataset is used for visualization.
Attributes
----------
scenario_database : OrderedDict
A structured dictionary contains all file information.
len_record : list
The list to record each scenario's data length. This is used to
retrieve the correct index during training.
"""
def __init__(self, params, visualize, train=True):
self.params = params
self.visualize = visualize
self.train = train
self.pre_processor = None
self.post_processor = None
self.data_augmentor = DataAugmentor(params['data_augment'],
train)
# if the training/testing include noisy setting
if 'wild_setting' in params:
self.seed = params['wild_setting']['seed']
# whether to add time delay
self.async_flag = params['wild_setting']['async']
self.async_mode = \
'sim' if 'async_mode' not in params['wild_setting'] \
else params['wild_setting']['async_mode']
self.async_overhead = params['wild_setting']['async_overhead']
# localization error
self.loc_err_flag = params['wild_setting']['loc_err']
self.xyz_noise_std = params['wild_setting']['xyz_std']
self.ryp_noise_std = params['wild_setting']['ryp_std']
# transmission data size
self.data_size = \
params['wild_setting']['data_size'] \
if 'data_size' in params['wild_setting'] else 0
self.transmission_speed = \
params['wild_setting']['transmission_speed'] \
if 'transmission_speed' in params['wild_setting'] else 27
self.backbone_delay = \
params['wild_setting']['backbone_delay'] \
if 'backbone_delay' in params['wild_setting'] else 0
else:
self.async_flag = False
self.async_overhead = 0 # ms
self.async_mode = 'sim'
self.loc_err_flag = False
self.xyz_noise_std = 0
self.ryp_noise_std = 0
self.data_size = 0 # Mb
self.transmission_speed = 27 # Mbps
self.backbone_delay = 0 # ms
if self.train:
root_dir = params['root_dir']
else:
root_dir = params['validate_dir']
if 'max_cav' not in params['train_params']:
self.max_cav = 7
else:
self.max_cav = params['train_params']['max_cav']
# first load all paths of different scenarios
scenario_folders = sorted([os.path.join(root_dir, x)
for x in os.listdir(root_dir) if
os.path.isdir(os.path.join(root_dir, x))])
# Structure: {scenario_id : {cav_1 : {timestamp1 : {yaml: path,
# lidar: path, cameras:list of path}}}}
self.scenario_database = OrderedDict()
self.len_record = []
# loop over all scenarios
for (i, scenario_folder) in enumerate(scenario_folders):
self.scenario_database.update({i: OrderedDict()})
# at least 1 cav should show up
cav_list = sorted([x for x in os.listdir(scenario_folder)
if os.path.isdir(
os.path.join(scenario_folder, x))])
assert len(cav_list) > 0
# roadside unit data's id is always negative, so here we want to
# make sure they will be in the end of the list as they shouldn't
# be ego vehicle.
if int(cav_list[0]) < 0:
cav_list = cav_list[1:] + [cav_list[0]]
# loop over all CAV data
for (j, cav_id) in enumerate(cav_list):
if j > self.max_cav - 1:
print('too many cavs')
break
self.scenario_database[i][cav_id] = OrderedDict()
# save all yaml files to the dictionary
cav_path = os.path.join(scenario_folder, cav_id)
# use the frame number as key, the full path as the values
yaml_files = \
sorted([os.path.join(cav_path, x)
for x in os.listdir(cav_path) if
x.endswith('.yaml')])
timestamps = self.extract_timestamps(yaml_files)
for timestamp in timestamps:
self.scenario_database[i][cav_id][timestamp] = \
OrderedDict()
yaml_file = os.path.join(cav_path,
timestamp + '.yaml')
lidar_file = os.path.join(cav_path,
timestamp + '.pcd')
camera_files = self.load_camera_files(cav_path, timestamp)
self.scenario_database[i][cav_id][timestamp]['yaml'] = \
yaml_file
self.scenario_database[i][cav_id][timestamp]['lidar'] = \
lidar_file
self.scenario_database[i][cav_id][timestamp]['camera0'] = \
camera_files
# Assume all cavs will have the same timestamps length. Thus
# we only need to calculate for the first vehicle in the
# scene.
if j == 0:
self.scenario_database[i][cav_id]['ego'] = True
if not self.len_record:
self.len_record.append(len(timestamps))
else:
prev_last = self.len_record[-1]
self.len_record.append(prev_last + len(timestamps))
else:
self.scenario_database[i][cav_id]['ego'] = False
def __len__(self):
return self.len_record[-1]
def __getitem__(self, idx):
"""
Abstract method, needs to be define by the children class.
"""
pass
def retrieve_base_data(self, idx, cur_ego_pose_flag=True):
"""
Given the index, return the corresponding data.
Parameters
----------
idx : int
Index given by dataloader.
cur_ego_pose_flag : bool
Indicate whether to use current timestamp ego pose to calculate
transformation matrix.
Returns
-------
data : dict
The dictionary contains loaded yaml params and lidar data for
each cav.
"""
# we loop the accumulated length list to see get the scenario index
scenario_index = 0
for i, ele in enumerate(self.len_record):
if idx < ele:
scenario_index = i
break
scenario_database = self.scenario_database[scenario_index]
# check the timestamp index
timestamp_index = idx if scenario_index == 0 else \
idx - self.len_record[scenario_index - 1]
# retrieve the corresponding timestamp key
timestamp_key = self.return_timestamp_key(scenario_database,
timestamp_index)
# calculate distance to ego for each cav for time delay estimation
ego_cav_content = \
self.calc_dist_to_ego(scenario_database, timestamp_key)
data = OrderedDict()
# load files for all CAVs
for cav_id, cav_content in scenario_database.items():
data[cav_id] = OrderedDict()
data[cav_id]['ego'] = cav_content['ego']
# calculate delay for this vehicle
timestamp_delay = \
self.time_delay_calculation(cav_content['ego'])
if timestamp_index - timestamp_delay <= 0:
timestamp_delay = timestamp_index
timestamp_index_delay = max(0, timestamp_index - timestamp_delay)
timestamp_key_delay = self.return_timestamp_key(scenario_database,
timestamp_index_delay)
# add time delay to vehicle parameters
data[cav_id]['time_delay'] = timestamp_delay
# load the corresponding data into the dictionary
data[cav_id]['params'] = self.reform_param(cav_content,
ego_cav_content,
timestamp_key,
timestamp_key_delay,
cur_ego_pose_flag)
data[cav_id]['lidar_np'] = \
pcd_utils.pcd_to_np(cav_content[timestamp_key_delay]['lidar'])
return data
@staticmethod
def extract_timestamps(yaml_files):
"""
Given the list of the yaml files, extract the mocked timestamps.
Parameters
----------
yaml_files : list
The full path of all yaml files of ego vehicle
Returns
-------
timestamps : list
The list containing timestamps only.
"""
timestamps = []
for file in yaml_files:
res = file.split('/')[-1]
timestamp = res.replace('.yaml', '')
timestamps.append(timestamp)
return timestamps
@staticmethod
def return_timestamp_key(scenario_database, timestamp_index):
"""
Given the timestamp index, return the correct timestamp key, e.g.
2 --> '000078'.
Parameters
----------
scenario_database : OrderedDict
The dictionary contains all contents in the current scenario.
timestamp_index : int
The index for timestamp.
Returns
-------
timestamp_key : str
The timestamp key saved in the cav dictionary.
"""
# get all timestamp keys
timestamp_keys = list(scenario_database.items())[0][1]
# retrieve the correct index
timestamp_key = list(timestamp_keys.items())[timestamp_index][0]
return timestamp_key
def calc_dist_to_ego(self, scenario_database, timestamp_key):
"""
Calculate the distance to ego for each cav.
"""
ego_lidar_pose = None
ego_cav_content = None
# Find ego pose first
for cav_id, cav_content in scenario_database.items():
if cav_content['ego']:
ego_cav_content = cav_content
ego_lidar_pose = \
load_yaml(cav_content[timestamp_key]['yaml'])['lidar_pose']
break
assert ego_lidar_pose is not None
# calculate the distance
for cav_id, cav_content in scenario_database.items():
cur_lidar_pose = \
load_yaml(cav_content[timestamp_key]['yaml'])['lidar_pose']
distance = \
math.sqrt((cur_lidar_pose[0] -
ego_lidar_pose[0]) ** 2 +
(cur_lidar_pose[1] - ego_lidar_pose[1]) ** 2)
cav_content['distance_to_ego'] = distance
scenario_database.update({cav_id: cav_content})
return ego_cav_content
def time_delay_calculation(self, ego_flag):
"""
Calculate the time delay for a certain vehicle.
Parameters
----------
ego_flag : boolean
Whether the current cav is ego.
Return
------
time_delay : int
The time delay quantization.
"""
# there is not time delay for ego vehicle
if ego_flag:
return 0
# time delay real mode
if self.async_mode == 'real':
# noise/time is in ms unit
overhead_noise = np.random.uniform(0, self.async_overhead)
tc = self.data_size / self.transmission_speed * 1000
time_delay = int(overhead_noise + tc + self.backbone_delay)
elif self.async_mode == 'sim':
time_delay = np.abs(self.async_overhead)
time_delay = time_delay // 100
return time_delay if self.async_flag else 0
def add_loc_noise(self, pose, xyz_std, ryp_std):
"""
Add localization noise to the pose.
Parameters
----------
pose : list
x,y,z,roll,yaw,pitch
xyz_std : float
std of the gaussian noise on xyz
ryp_std : float
std of the gaussian noise
"""
np.random.seed(self.seed)
xyz_noise = np.random.normal(0, xyz_std, 3)
ryp_std = np.random.normal(0, ryp_std, 3)
noise_pose = [pose[0] + xyz_noise[0],
pose[1] + xyz_noise[1],
pose[2] + xyz_noise[2],
pose[3],
pose[4] + ryp_std[1],
pose[5]]
return noise_pose
def reform_param(self, cav_content, ego_content, timestamp_cur,
timestamp_delay, cur_ego_pose_flag):
"""
Reform the data params with current timestamp object groundtruth and
delay timestamp LiDAR pose.
Parameters
----------
cav_content : dict
Dictionary that contains all file paths in the current cav/rsu.
ego_content : dict
Ego vehicle content.
timestamp_cur : str
The current timestamp.
timestamp_delay : str
The delayed timestamp.
cur_ego_pose_flag : bool
Whether use current ego pose to calculate transformation matrix.
Return
------
The merged parameters.
"""
cur_params = load_yaml(cav_content[timestamp_cur]['yaml'])
delay_params = load_yaml(cav_content[timestamp_delay]['yaml'])
cur_ego_params = load_yaml(ego_content[timestamp_cur]['yaml'])
delay_ego_params = load_yaml(ego_content[timestamp_delay]['yaml'])
# we need to calculate the transformation matrix from cav to ego
# at the delayed timestamp
delay_cav_lidar_pose = delay_params['lidar_pose']
delay_ego_lidar_pose = delay_ego_params["lidar_pose"]
cur_ego_lidar_pose = cur_ego_params['lidar_pose']
cur_cav_lidar_pose = cur_params['lidar_pose']
if not cav_content['ego'] and self.loc_err_flag:
delay_cav_lidar_pose = self.add_loc_noise(delay_cav_lidar_pose,
self.xyz_noise_std,
self.ryp_noise_std)
cur_cav_lidar_pose = self.add_loc_noise(cur_cav_lidar_pose,
self.xyz_noise_std,
self.ryp_noise_std)
if cur_ego_pose_flag:
transformation_matrix = x1_to_x2(delay_cav_lidar_pose,
cur_ego_lidar_pose)
spatial_correction_matrix = np.eye(4)
else:
transformation_matrix = x1_to_x2(delay_cav_lidar_pose,
delay_ego_lidar_pose)
spatial_correction_matrix = x1_to_x2(delay_ego_lidar_pose,
cur_ego_lidar_pose)
# This is only used for late fusion, as it did the transformation
# in the postprocess, so we want the gt object transformation use
# the correct one
gt_transformation_matrix = x1_to_x2(cur_cav_lidar_pose,
cur_ego_lidar_pose)
# we always use current timestamp's gt bbx to gain a fair evaluation
delay_params['vehicles'] = cur_params['vehicles']
delay_params['transformation_matrix'] = transformation_matrix
delay_params['gt_transformation_matrix'] = \
gt_transformation_matrix
delay_params['spatial_correction_matrix'] = spatial_correction_matrix
return delay_params
@staticmethod
def load_camera_files(cav_path, timestamp):
"""
Retrieve the paths to all camera files.
Parameters
----------
cav_path : str
The full file path of current cav.
timestamp : str
Current timestamp
Returns
-------
camera_files : list
The list containing all camera png file paths.
"""
camera0_file = os.path.join(cav_path,
timestamp + '_camera0.png')
camera1_file = os.path.join(cav_path,
timestamp + '_camera1.png')
camera2_file = os.path.join(cav_path,
timestamp + '_camera2.png')
camera3_file = os.path.join(cav_path,
timestamp + '_camera3.png')
return [camera0_file, camera1_file, camera2_file, camera3_file]
def project_points_to_bev_map(self, points, ratio=0.1):
"""
Project points to BEV occupancy map with default ratio=0.1.
Parameters
----------
points : np.ndarray
(N, 3) / (N, 4)
ratio : float
Discretization parameters. Default is 0.1.
Returns
-------
bev_map : np.ndarray
BEV occupancy map including projected points
with shape (img_row, img_col).
"""
return self.pre_processor.project_points_to_bev_map(points, ratio)
def augment(self, lidar_np, object_bbx_center, object_bbx_mask):
"""
Data augmentation operation.
"""
tmp_dict = {'lidar_np': lidar_np,
'object_bbx_center': object_bbx_center,
'object_bbx_mask': object_bbx_mask}
tmp_dict = self.data_augmentor.forward(tmp_dict)
lidar_np = tmp_dict['lidar_np']
object_bbx_center = tmp_dict['object_bbx_center']
object_bbx_mask = tmp_dict['object_bbx_mask']
return lidar_np, object_bbx_center, object_bbx_mask
def collate_batch_train(self, batch):
"""
Customized collate function for pytorch dataloader during training
for late fusion dataset.
Parameters
----------
batch : dict
Returns
-------
batch : dict
Reformatted batch.
"""
# during training, we only care about ego.
output_dict = {'ego': {}}
object_bbx_center = []
object_bbx_mask = []
processed_lidar_list = []
label_dict_list = []
if self.visualize:
origin_lidar = []
for i in range(len(batch)):
ego_dict = batch[i]['ego']
object_bbx_center.append(ego_dict['object_bbx_center'])
object_bbx_mask.append(ego_dict['object_bbx_mask'])
processed_lidar_list.append(ego_dict['processed_lidar'])
label_dict_list.append(ego_dict['label_dict'])
if self.visualize:
origin_lidar.append(ego_dict['origin_lidar'])
# convert to numpy, (B, max_num, 7)
object_bbx_center = torch.from_numpy(np.array(object_bbx_center))
object_bbx_mask = torch.from_numpy(np.array(object_bbx_mask))
processed_lidar_torch_dict = \
self.pre_processor.collate_batch(processed_lidar_list)
label_torch_dict = \
self.post_processor.collate_batch(label_dict_list)
output_dict['ego'].update({'object_bbx_center': object_bbx_center,
'object_bbx_mask': object_bbx_mask,
'processed_lidar': processed_lidar_torch_dict,
'label_dict': label_torch_dict})
if self.visualize:
origin_lidar = \
np.array(downsample_lidar_minimum(pcd_np_list=origin_lidar))
origin_lidar = torch.from_numpy(origin_lidar)
output_dict['ego'].update({'origin_lidar': origin_lidar})
return output_dict
def visualize_result(self, pred_box_tensor,
gt_tensor,
pcd,
show_vis,
save_path,
dataset=None):
self.post_processor.visualize(pred_box_tensor,
gt_tensor,
pcd,
show_vis,
save_path,
dataset=dataset)
from v2xvit.data_utils.datasets.late_fusion_dataset import LateFusionDataset
from v2xvit.data_utils.datasets.early_fusion_dataset import EarlyFusionDataset
from v2xvit.data_utils.datasets.intermediate_fusion_dataset import IntermediateFusionDataset
__all__ = {
'LateFusionDataset': LateFusionDataset,
'EarlyFusionDataset': EarlyFusionDataset,
'IntermediateFusionDataset': IntermediateFusionDataset
}
# the final range for evaluation
GT_RANGE = [-140, -40, -3, 140, 40, 1]
# The communication range for cavs
COM_RANGE = 70
def build_dataset(dataset_cfg, visualize=False, train=True):
dataset_name = dataset_cfg['fusion']['core_method']
error_message = f"{dataset_name} is not found. " \
f"Please add your processor file's name in opencood/" \
f"data_utils/datasets/init.py"
assert dataset_name in ['LateFusionDataset', 'EarlyFusionDataset',
'IntermediateFusionDataset'], error_message
dataset = __all__[dataset_name](
params=dataset_cfg,
visualize=visualize,
train=train
)
return dataset
"""
Class for data augmentation
"""
from functools import partial
from v2xvit.data_utils.augmentor import augment_utils
class DataAugmentor(object):
"""
Data Augmentor.
Parameters
----------
augment_config : list
A list of augmentation configuration.
Attributes
----------
data_augmentor_queue : list
The list of data augmented functions.
"""
def __init__(self, augment_config, train=True):
self.data_augmentor_queue = []
self.train = train
for cur_cfg in augment_config:
cur_augmentor = getattr(self, cur_cfg['NAME'])(config=cur_cfg)
self.data_augmentor_queue.append(cur_augmentor)
def random_world_flip(self, data_dict=None, config=None):
if data_dict is None:
return partial(self.random_world_flip, config=config)
gt_boxes, gt_mask, points = data_dict['object_bbx_center'], \
data_dict['object_bbx_mask'], \
data_dict['lidar_np']
gt_boxes_valid = gt_boxes[gt_mask == 1]
for cur_axis in config['ALONG_AXIS_LIST']:
assert cur_axis in ['x', 'y']
gt_boxes_valid, points = getattr(augment_utils,
'random_flip_along_%s' % cur_axis)(
gt_boxes_valid, points,
)
gt_boxes[:gt_boxes_valid.shape[0], :] = gt_boxes_valid
data_dict['object_bbx_center'] = gt_boxes
data_dict['object_bbx_mask'] = gt_mask
data_dict['lidar_np'] = points
return data_dict
def random_world_rotation(self, data_dict=None, config=None):
if data_dict is None:
return partial(self.random_world_rotation, config=config)
rot_range = config['WORLD_ROT_ANGLE']
if not isinstance(rot_range, list):
rot_range = [-rot_range, rot_range]
gt_boxes, gt_mask, points = data_dict['object_bbx_center'], \
data_dict['object_bbx_mask'], \
data_dict['lidar_np']
gt_boxes_valid = gt_boxes[gt_mask == 1]
gt_boxes_valid, points = augment_utils.global_rotation(
gt_boxes_valid, points, rot_range=rot_range
)
gt_boxes[:gt_boxes_valid.shape[0], :] = gt_boxes_valid
data_dict['object_bbx_center'] = gt_boxes
data_dict['object_bbx_mask'] = gt_mask
data_dict['lidar_np'] = points
return data_dict
def random_world_scaling(self, data_dict=None, config=None):
if data_dict is None:
return partial(self.random_world_scaling, config=config)
gt_boxes, gt_mask, points = data_dict['object_bbx_center'], \
data_dict['object_bbx_mask'], \
data_dict['lidar_np']
gt_boxes_valid = gt_boxes[gt_mask == 1]
gt_boxes_valid, points = augment_utils.global_scaling(
gt_boxes_valid, points, config['WORLD_SCALE_RANGE']
)
gt_boxes[:gt_boxes_valid.shape[0], :] = gt_boxes_valid
data_dict['object_bbx_center'] = gt_boxes
data_dict['object_bbx_mask'] = gt_mask
data_dict['lidar_np'] = points
return data_dict
def forward(self, data_dict):
"""
Args:
data_dict:
points: (N, 3 + C_in)
gt_boxes: optional, (N, 7) [x, y, z, dx, dy, dz, heading]
gt_names: optional, (N), string
...
Returns:
"""
if self.train:
for cur_augmentor in self.data_augmentor_queue:
data_dict = cur_augmentor(data_dict=data_dict)
return data_dict
import numpy as np
from v2xvit.utils import common_utils
def random_flip_along_x(gt_boxes, points):
"""
Args:
gt_boxes: (N, 7 + C), [x, y, z, dx, dy, dz, heading, [vx], [vy]]
points: (M, 3 + C)
Returns:
"""
enable = np.random.choice([False, True], replace=False, p=[0.5, 0.5])
if enable:
gt_boxes[:, 1] = -gt_boxes[:, 1]
gt_boxes[:, 6] = -gt_boxes[:, 6]
points[:, 1] = -points[:, 1]
if gt_boxes.shape[1] > 7:
gt_boxes[:, 8] = -gt_boxes[:, 8]
return gt_boxes, points
def random_flip_along_y(gt_boxes, points):
"""
Args:
gt_boxes: (N, 7 + C), [x, y, z, dx, dy, dz, heading, [vx], [vy]]
points: (M, 3 + C)
Returns:
"""
enable = np.random.choice([False, True], replace=False, p=[0.5, 0.5])
if enable:
gt_boxes[:, 0] = -gt_boxes[:, 0]
gt_boxes[:, 6] = -(gt_boxes[:, 6] + np.pi)
points[:, 0] = -points[:, 0]
if gt_boxes.shape[1] > 7:
gt_boxes[:, 7] = -gt_boxes[:, 7]
return gt_boxes, points
def global_rotation(gt_boxes, points, rot_range):
"""
Args:
gt_boxes: (N, 7 + C), [x, y, z, dx, dy, dz, heading, [vx], [vy]]
points: (M, 3 + C),
rot_range: [min, max]
Returns:
"""
noise_rotation = np.random.uniform(rot_range[0],
rot_range[1])
points = common_utils.rotate_points_along_z(points[np.newaxis, :, :],
np.array([noise_rotation]))[0]
gt_boxes[:, 0:3] = \
common_utils.rotate_points_along_z(gt_boxes[np.newaxis, :, 0:3],
np.array([noise_rotation]))[0]
gt_boxes[:, 6] += noise_rotation
if gt_boxes.shape[1] > 7:
gt_boxes[:, 7:9] = common_utils.rotate_points_along_z(
np.hstack((gt_boxes[:, 7:9], np.zeros((gt_boxes.shape[0], 1))))[
np.newaxis, :, :],
np.array([noise_rotation]))[0][:, 0:2]
return gt_boxes, points
def global_scaling(gt_boxes, points, scale_range):
"""
Args:
gt_boxes: (N, 7), [x, y, z, dx, dy, dz, heading]
points: (M, 3 + C),
scale_range: [min, max]
Returns:
"""
if scale_range[1] - scale_range[0] < 1e-3:
return gt_boxes, points
noise_scale = np.random.uniform(scale_range[0], scale_range[1])
points[:, :3] *= noise_scale
gt_boxes[:, :6] *= noise_scale
return gt_boxes, points
import re
import yaml
import os
import math
import numpy as np
def load_yaml(file, opt=None):
"""
Load yaml file and return a dictionary.
Parameters
----------
file : string
yaml file path.
opt : argparser
Argparser.
Returns
-------
param : dict
A dictionary that contains defined parameters.
"""
if opt and opt.model_dir:
file = os.path.join(opt.model_dir, 'config.yaml')
stream = open(file, 'r')
loader = yaml.Loader
loader.add_implicit_resolver(
u'tag:yaml.org,2002:float',
re.compile(u'''^(?:
[-+]?(?:[0-9][0-9_]*)\\.[0-9_]*(?:[eE][-+]?[0-9]+)?
|[-+]?(?:[0-9][0-9_]*)(?:[eE][-+]?[0-9]+)
|\\.[0-9_]+(?:[eE][-+][0-9]+)?
|[-+]?[0-9][0-9_]*(?::[0-5]?[0-9])+\\.[0-9_]*
|[-+]?\\.(?:inf|Inf|INF)
|\\.(?:nan|NaN|NAN))$''', re.X),
list(u'-+0123456789.'))
param = yaml.load(stream, Loader=loader)
if "yaml_parser" in param:
param = eval(param["yaml_parser"])(param)
return param
def load_voxel_params(param):
"""
Based on the lidar range and resolution of voxel, calcuate the anchor box
and target resolution.
Parameters
----------
param : dict
Original loaded parameter dictionary.
Returns
-------
param : dict
Modified parameter dictionary with new attribute `anchor_args[W][H][L]`
"""
anchor_args = param['postprocess']['anchor_args']
cav_lidar_range = anchor_args['cav_lidar_range']
voxel_size = param['preprocess']['args']['voxel_size']
vw = voxel_size[0]
vh = voxel_size[1]
vd = voxel_size[2]
anchor_args['vw'] = vw
anchor_args['vh'] = vh
anchor_args['vd'] = vd
anchor_args['W'] = int((cav_lidar_range[3] - cav_lidar_range[0]) / vw)
anchor_args['H'] = int((cav_lidar_range[4] - cav_lidar_range[1]) / vh)
anchor_args['D'] = int((cav_lidar_range[5] - cav_lidar_range[2]) / vd)
param['postprocess'].update({'anchor_args': anchor_args})
# sometimes we just want to visualize the data without implementing model
if 'model' in param:
param['model']['args']['W'] = anchor_args['W']
param['model']['args']['H'] = anchor_args['H']
param['model']['args']['D'] = anchor_args['D']
return param
def load_point_pillar_params(param):
"""
Based on the lidar range and resolution of voxel, calcuate the anchor box
and target resolution.
Parameters
----------
param : dict
Original loaded parameter dictionary.
Returns
-------
param : dict
Modified parameter dictionary with new attribute.
"""
cav_lidar_range = param['preprocess']['cav_lidar_range']
voxel_size = param['preprocess']['args']['voxel_size']
grid_size = (np.array(cav_lidar_range[3:6]) - np.array(
cav_lidar_range[0:3])) / \
np.array(voxel_size)
grid_size = np.round(grid_size).astype(np.int64)
param['model']['args']['point_pillar_scatter']['grid_size'] = grid_size
anchor_args = param['postprocess']['anchor_args']
vw = voxel_size[0]
vh = voxel_size[1]
vd = voxel_size[2]
anchor_args['vw'] = vw
anchor_args['vh'] = vh
anchor_args['vd'] = vd
anchor_args['W'] = math.ceil((cav_lidar_range[3] - cav_lidar_range[0]) / vw)
anchor_args['H'] = math.ceil((cav_lidar_range[4] - cav_lidar_range[1]) / vh)
anchor_args['D'] = math.ceil((cav_lidar_range[5] - cav_lidar_range[2]) / vd)
param['postprocess'].update({'anchor_args': anchor_args})
return param
def load_second_params(param):
"""
Based on the lidar range and resolution of voxel, calcuate the anchor box
and target resolution.
Parameters
----------
param : dict
Original loaded parameter dictionary.
Returns
-------
param : dict
Modified parameter dictionary with new attribute.
"""
cav_lidar_range = param['preprocess']['cav_lidar_range']
voxel_size = param['preprocess']['args']['voxel_size']
grid_size = (np.array(cav_lidar_range[3:6]) - np.array(
cav_lidar_range[0:3])) / \
np.array(voxel_size)
grid_size = np.round(grid_size).astype(np.int64)
param['model']['args']['grid_size'] = grid_size
anchor_args = param['postprocess']['anchor_args']
vw = voxel_size[0]
vh = voxel_size[1]
vd = voxel_size[2]
anchor_args['vw'] = vw
anchor_args['vh'] = vh
anchor_args['vd'] = vd
anchor_args['W'] = math.ceil((cav_lidar_range[3] - cav_lidar_range[0]) / vw)
anchor_args['H'] = math.ceil((cav_lidar_range[4] - cav_lidar_range[1]) / vh)
anchor_args['D'] = math.ceil((cav_lidar_range[5] - cav_lidar_range[2]) / vd)
param['postprocess'].update({'anchor_args': anchor_args})
return param
def load_bev_params(param):
"""
Load bev related geometry parameters s.t. boundary, resolutions, input
shape, target shape etc.
Parameters
----------
param : dict
Original loaded parameter dictionary.
Returns
-------
param : dict
Modified parameter dictionary with new attribute `geometry_param`.
"""
res = param["preprocess"]["args"]["res"]
L1, W1, H1, L2, W2, H2 = param["preprocess"]["cav_lidar_range"]
downsample_rate = param["preprocess"]["args"]["downsample_rate"]
def f(low, high, r):
return int((high - low) / r)
input_shape = (
int((f(L1, L2, res))),
int((f(W1, W2, res))),
int((f(H1, H2, res)) + 1)
)
label_shape = (
int(input_shape[0] / downsample_rate),
int(input_shape[1] / downsample_rate),
7
)
geometry_param = {
'L1': L1,
'L2': L2,
'W1': W1,
'W2': W2,
'H1': H1,
'H2': H2,
"downsample_rate": downsample_rate,
"input_shape": input_shape,
"label_shape": label_shape,
"res": res
}
param["preprocess"]["geometry_param"] = geometry_param
param["postprocess"]["geometry_param"] = geometry_param
param["model"]["args"]["geometry_param"] = geometry_param
return param
def save_yaml(data, save_name):
"""
Save the dictionary into a yaml file.
Parameters
----------
data : dict
The dictionary contains all data.
save_name : string
Full path of the output yaml file.
"""
with open(save_name, 'w') as outfile:
yaml.dump(data, outfile, default_flow_style=False)choice A
This algorithm model takes into account the realistic factors of communication overload and solves the problem of excessive communication pressure.
choice B
This model takes into account real-world problems, which are time asynchrony and posture errors, and solves the problem of spatial alignment.
choice C
This algorithm model takes into account real-world issues such as time asynchrony and sensor heterogeneity, and solves the problem of time and spatial alignment.
choice D
The algorithm model takes into account realistic issues such as communication pressure overload and solves the problem of communication strategy
difficulty
easy
domain
Code Repository Understanding
length
short
sub domain
Code repo QA
Discussion
No discussion posts on this page yet. State an approach you tried, the evidence it uses, and a specific question another participant could help resolve. Use the posting template.
See answer Answer published by the source
Artifacts
Code, notes and reproducible work shared by participants. Files are served from a separate origin.
No artifacts on this page yet. Share reproducible code or notes in a contribution. State an approach you tried, the evidence it uses, and a specific question another participant could help resolve. Use the posting template.
Source and history
initial import