Added util3d_motion_estimation.h tests (2D->3D done)

This commit is contained in:
matlabbe
2025-07-05 20:42:32 -07:00
parent 7a7c33dd14
commit 3c5c00456b
4 changed files with 578 additions and 29 deletions
@@ -39,6 +39,41 @@ namespace rtabmap
namespace util3d
{
/**
* @brief Estimates a 6-DOF camera transform from 3D-2D point correspondences using PnP RANSAC.
*
* This function estimates the motion (transform) between two views by solving the Perspective-n-Point (PnP)
* problem using 3D points from one frame and their corresponding 2D keypoints in another frame.
* It optionally refines the result, computes covariance of the pose estimate, and handles degenerate cases.
*
* @param words3A 3D points in frame A, indexed by feature ID.
* @param words2B 2D keypoints in frame B, indexed by feature ID (shared with `words3A`).
* @param cameraModel Intrinsic and extrinsic parameters of the camera (must be valid).
* @param minInliers Minimum number of inliers required to accept the estimated transform. If the value is <4, it is set internally to 4.
* @param iterations Number of RANSAC iterations for PnP.
* @param reprojError Maximum allowed reprojection error (in pixels) to consider a point an inlier.
* @param flagsPnP Flags to control the `cv::solvePnPRansac` behavior (e.g., `cv::SOLVEPNP_ITERATIVE`).
* @param refineIterations Number of iterations for non-linear optimization (set to 0 to disable refinement).
* @param varianceMedianRatio Index divisor used to select the robust variance threshold from sorted error residuals (e.g., 4 → use the 25% percentile).
* @param maxVariance Maximum allowed median variance (linear error). Estimates with higher variance are rejected.
* @param guess Initial guess for the camera pose (must not be null). Typically from odometry or motion model.
* @param words3B Optional 3D points in frame B (if available). Used to better estimate 3D errors and variances.
* @param covariance Optional output pointer for the estimated 6x6 pose covariance matrix.
* The matrix contains linear variance in the top-left 3x3 and angular variance in the bottom-right 3x3.
* @param matchesOut Optional output vector of all matched IDs used (regardless of inlier status).
* @param inliersOut Optional output vector of matched IDs that were determined to be inliers.
* @param splitLinearCovarianceComponents Whether to split and compute variance for X, Y, Z components separately.
*
* @return The estimated transformation from frame B to frame A. If estimation fails or is rejected due to variance,
* a null transform is returned (i.e., `transform.isNull()` will be true).
*
* @note
* - If `words3B` is provided, 3D variance is computed by comparing reprojected points to actual transformed points.
* - If `words3B` is empty, variance is estimated using reprojection error only.
* - The function assumes the camera model's local transform is known and factored into the pose estimation.
*
* @see cv::solvePnPRansac
*/
Transform RTABMAP_CORE_EXPORT estimateMotion3DTo2D(
const std::map<int, cv::Point3f> & words3A,
const std::map<int, cv::KeyPoint> & words2B,
@@ -56,26 +91,41 @@ Transform RTABMAP_CORE_EXPORT estimateMotion3DTo2D(
std::vector<int> * matchesOut = 0,
std::vector<int> * inliersOut = 0,
bool splitLinearCovarianceComponents = false);
Transform RTABMAP_CORE_EXPORT estimateMotion3DTo2D(
const std::map<int, cv::Point3f> & words3A,
const std::map<int, cv::KeyPoint> & words2B,
const std::vector<CameraModel> & cameraModels,
unsigned int samplingPolicy = 0, // 0=AUTO, 1=ANY, 2=HOMOGENEOUS
int minInliers = 10,
int iterations = 100,
double reprojError = 5.,
int flagsPnP = 0,
int pnpRefineIterations = 1,
int varianceMedianRatio = 4,
float maxVariance = 0,
const Transform & guess = Transform::getIdentity(),
const std::map<int, cv::Point3f> & words3B = std::map<int, cv::Point3f>(),
cv::Mat * covariance = 0, // mean reproj error if words3B is not set
std::vector<int> * matchesOut = 0,
std::vector<int> * inliersOut = 0,
bool splitLinearCovarianceComponents = false);
/**
* @brief Estimates the 3D-to-2D motion (pose) transformation between a set of 3D points and their corresponding 2D keypoints using the OpenGV library.
*
* This function uses a robust multi-camera Perspective-n-Point (PnP) algorithm to estimate the transformation from a 3D point cloud (scene A)
* to a set of 2D keypoints (scene B) given the corresponding camera models and initial pose guess.
* The method supports multiple camera models and uses RANSAC with OpenGV for outlier rejection.
*
* @param words3A 3D points in the source frame (scene A), indexed by feature ID.
* @param words2B 2D keypoints in the destination frame (scene B), indexed by feature ID.
* @param cameraModels List of camera models (multi-camera rig setup) for the destination frame.
* @param samplingPolicy Sampling strategy (0 = auto, 1 = any, 2 = homogeneous multi-camera).
* @param minInliers Minimum number of inliers required to consider the estimated transform as valid.
* @param iterations Maximum number of RANSAC iterations.
* @param reprojError Reprojection error threshold used by RANSAC.
* @param flagsPnP PnP flags (not used internally).
* @param refineIterations Number of pose refinement iterations after RANSAC (not used internally).
* @param varianceMedianRatio Divider used to compute median-based variance from the error distribution.
* @param maxVariance Maximum allowed variance to accept the transform. Higher values permit more noisy estimates.
* @param guess Initial guess of the transformation.
* @param words3B Optional 3D points in destination frame (scene B) to evaluate the covariance using 3D
* correspondences, otherwise covariance is estimated from reprojection errors.
* @param covariance Optional output 6x6 covariance matrix of the estimated transform.
* @param matchesOut Optional output: matches grouped per camera.
* @param inliersOut Optional output: inliers grouped per camera.
* @param splitLinearCovarianceComponents If true, linear covariance is split into separate x/y/z components.
*
* @return The estimated Transform from scene A to scene B. Returns a null Transform if estimation fails or variance exceeds threshold.
*
* @note This function requires RTAB-Map to be built with OpenGV support.
*
* @warning The function assumes all camera models have the same image width and valid intrinsic parameters.
*
* @see https://github.com/laurentkneip/opengv
*/
Transform estimateMotion3DTo2D(
const std::map<int, cv::Point3f> & words3A,
const std::map<int, cv::KeyPoint> & words2B,
@@ -94,6 +144,28 @@ Transform estimateMotion3DTo2D(
std::vector<std::vector<int> > * matchesOut,
std::vector<std::vector<int> > * inliersOut,
bool splitLinearCovarianceComponents);
/**
* @brief Estimates the 3D-to-2D motion (pose) transformation between a set of 3D points and their corresponding 2D keypoints using the OpenGV library.
* @see estimateMotion3DTo2D(), the only difference is that output matches and inliers are combined in same vector instead of per camera
*/
Transform RTABMAP_CORE_EXPORT estimateMotion3DTo2D(
const std::map<int, cv::Point3f> & words3A,
const std::map<int, cv::KeyPoint> & words2B,
const std::vector<CameraModel> & cameraModels,
unsigned int samplingPolicy = 0, // 0=AUTO, 1=ANY, 2=HOMOGENEOUS
int minInliers = 10,
int iterations = 100,
double reprojError = 5.,
int flagsPnP = 0,
int pnpRefineIterations = 1,
int varianceMedianRatio = 4,
float maxVariance = 0,
const Transform & guess = Transform::getIdentity(),
const std::map<int, cv::Point3f> & words3B = std::map<int, cv::Point3f>(),
cv::Mat * covariance = 0, // mean reproj error if words3B is not set
std::vector<int> * matchesOut = 0,
std::vector<int> * inliersOut = 0,
bool splitLinearCovarianceComponents = false);
Transform RTABMAP_CORE_EXPORT estimateMotion3DTo3D(
const std::map<int, cv::Point3f> & words3A,
+14 -9
View File
@@ -224,11 +224,11 @@ Transform estimateMotion3DTo2D(
//divide by 4 instead of 2 to ignore very very far features (stereo)
double median_error_sqr_lin = 2.1981 * (double)errorSqrdDists[errorSqrdDists.size () / varianceMedianRatio];
UASSERT(uIsFinite(median_error_sqr_lin));
(*covariance)(cv::Range(0,3), cv::Range(0,3)) *= median_error_sqr_lin;
(*covariance)(cv::Range(0,3), cv::Range(0,3)) *= median_error_sqr_lin + 1e-6;
std::sort(errorSqrdAngles.begin(), errorSqrdAngles.end());
double median_error_sqr_ang = 2.1981 * (double)errorSqrdAngles[errorSqrdAngles.size () / varianceMedianRatio];
UASSERT(uIsFinite(median_error_sqr_ang));
(*covariance)(cv::Range(3,6), cv::Range(3,6)) *= median_error_sqr_ang;
(*covariance)(cv::Range(3,6), cv::Range(3,6)) *= median_error_sqr_ang + 1e-6;
if(splitLinearCovarianceComponents)
{
@@ -242,9 +242,9 @@ Transform estimateMotion3DTo2D(
UASSERT(uIsFinite(median_error_sqr_x));
UASSERT(uIsFinite(median_error_sqr_y));
UASSERT(uIsFinite(median_error_sqr_z));
covariance->at<double>(0,0) = median_error_sqr_x;
covariance->at<double>(1,1) = median_error_sqr_y;
covariance->at<double>(2,2) = median_error_sqr_z;
covariance->at<double>(0,0) = median_error_sqr_x + 1e-6;
covariance->at<double>(1,1) = median_error_sqr_y + 1e-6;
covariance->at<double>(2,2) = median_error_sqr_z + 1e-6;
median_error_sqr_lin = uMax3(median_error_sqr_x, median_error_sqr_y, median_error_sqr_z);
}
@@ -267,7 +267,7 @@ Transform estimateMotion3DTo2D(
err += uNormSquared(imagePoints.at(inliers[i]).x - imagePointsReproj.at(inliers[i]).x, imagePoints.at(inliers[i]).y - imagePointsReproj.at(inliers[i]).y);
}
UASSERT(uIsFinite(err));
*covariance *= std::sqrt(err/float(inliers.size()));
*covariance *= std::sqrt(err/float(inliers.size())) + 1e-6;
}
}
}
@@ -483,6 +483,7 @@ Transform estimateMotion3DTo2D(
// convert 2d-3d correspondences into bearing vectors
std::vector<std::shared_ptr<opengv::bearingVectors_t>> multiBearingVectors;
multiBearingVectors.resize(cameraModels.size());
std::vector<std::vector<int> > localIndexToGlobalIndex(cameraModels.size());
for(size_t i=0; i<cameraModels.size();++i)
{
multiPoints[i] = std::make_shared<opengv::points_t>();
@@ -497,6 +498,7 @@ Transform estimateMotion3DTo2D(
cameraModels[cameraIndex].project(imagePoints[i].x, imagePoints[i].y, 1, pt[0], pt[1], pt[2]);
pt = cv::normalize(pt);
multiBearingVectors[cameraIndex]->push_back(opengv::bearingVector_t(pt[0], pt[1], pt[2]));
localIndexToGlobalIndex[cameraIndex].push_back(i);
}
//create a non-central absolute multi adapter
@@ -527,9 +529,10 @@ Transform estimateMotion3DTo2D(
UDEBUG("Ransac result: %s", pnp.prettyPrint().c_str());
UDEBUG("Ransac iterations done: %d", ransac.iterations_);
for (size_t i=0; i < cameraModels.size(); ++i)
{
inliers.insert(inliers.end(), ransac.inliers_[i].begin(), ransac.inliers_[i].end());
for (size_t i=0; i < cameraModels.size(); ++i) {
for (size_t j=0; j < ransac.inliers_[i].size(); ++j) {
inliers.push_back(localIndexToGlobalIndex[i][ransac.inliers_[i][j]]);
}
}
}
else
@@ -705,6 +708,7 @@ Transform estimateMotion3DTo2D(
if(matchesOut)
{
matchesOut->clear();
matchesOut->resize(cameraModels.size());
UASSERT(matches.size() == cameraIndexes.size());
for(size_t i=0; i<matches.size(); ++i)
@@ -715,6 +719,7 @@ Transform estimateMotion3DTo2D(
}
if(inliersOut)
{
inliersOut->clear();
inliersOut->resize(cameraModels.size());
for(unsigned int i=0; i<inliers.size(); ++i)
{
+6 -1
View File
@@ -39,4 +39,9 @@ gtest_discover_tests(test_util3d_correspondences)
#util3d_mapping.h
add_executable(test_util3d_mapping test_util3d_mapping.cpp)
target_link_libraries(test_util3d_mapping gtest_main rtabmap_core)
gtest_discover_tests(test_util3d_mapping)
gtest_discover_tests(test_util3d_mapping)
#util3d_motion_estimation.h
add_executable(test_util3d_motion_estimation test_util3d_motion_estimation.cpp)
target_link_libraries(test_util3d_motion_estimation gtest_main rtabmap_core)
gtest_discover_tests(test_util3d_motion_estimation)
@@ -0,0 +1,467 @@
#include "gtest/gtest.h"
#include "rtabmap/core/util3d.h"
#include "rtabmap/core/util3d_motion_estimation.h"
#include "rtabmap/core/CameraModel.h"
#include "rtabmap/utilite/UException.h"
#include "rtabmap/utilite/UConversion.h"
#include "rtabmap/core/Version.h"
#include <pcl/io/pcd_io.h>
using namespace rtabmap;
float randomNoise(float max) {
return ((static_cast<float>(rand()) / RAND_MAX) * 2.0f - 1.0f) * max;
}
TEST(Util3dMotionEstimation, estimateMotion3DTo2DBasic) {
// Two triangles in front of the camera at two different depths, centered with the middle of the image frame
std::map<int, cv::Point3f> words3A = {
{0, cv::Point3f(1,0,1)},
{1, cv::Point3f(1,1,-1)},
{2, cv::Point3f(1,-1,-1)},
{3, cv::Point3f(2,0,0)},
{4, cv::Point3f(2,0.5,0)},
{5, cv::Point3f(3,-0.5,0)},
{6, cv::Point3f(2,0,10)} // outlier
};
CameraModel cam(200, 200, 320, 240, CameraModel::opticalRotation(), 0, cv::Size(640, 480));
std::map<int, cv::KeyPoint> words2B;
for(auto & pt: words3A) {
cv::Point3f ptt = util3d::transformPoint(pt.second, cam.localTransform().inverse());
float u,v;
cam.reproject(ptt.x,ptt.y,ptt.z, u, v);
if(cam.inFrame(u,v)) {
words2B.insert(std::make_pair(pt.first, cv::KeyPoint(u, v, 3)));
}
else {
words2B.insert(std::make_pair(pt.first, cv::KeyPoint(10, 10, 3)));
}
}
std::map<int, cv::Point3f> words3B; // leave empty
Transform guess = Transform::getIdentity(); // non-null identity
cv::Mat covariance;
std::vector<int> matchesOut, inliersOut;
// Test without image size set, so covariance is computed completely by
// reproj errors, which is expected to be zero here
Transform result = util3d::estimateMotion3DTo2D(
words3A, words2B, CameraModel(200, 200, 320, 240),
/*minInliers=*/4,
/*iterations=*/100,
/*reprojError=*/2.0,
/*flagsPnP=*/0,
/*refineIterations=*/1,
/*varianceMedianRatio=*/4,
/*maxVariance=*/0.0f,
guess,
words3B,
&covariance,
&matchesOut,
&inliersOut,
/*splitLinearCovarianceComponents=*/false
);
EXPECT_FALSE(result.isNull());
float x,y,z,roll,pitch,yaw;
result.getTranslationAndEulerAngles(x,y,z,roll,pitch,yaw);
EXPECT_NEAR(x, 0, 1e-6);
EXPECT_NEAR(y, 0, 1e-6);
EXPECT_NEAR(z, 0, 1e-6);
EXPECT_NEAR(roll, 0, 1e-6);
EXPECT_NEAR(pitch, 0, 1e-6);
EXPECT_NEAR(yaw, 0, 1e-6);
EXPECT_EQ(matchesOut.size(), 7u);
EXPECT_EQ(inliersOut.size(), 6u);
// covariance must be 6x6
EXPECT_EQ(covariance.rows, 6);
EXPECT_EQ(covariance.cols, 6);
EXPECT_NEAR(covariance.at<double>(0,0), 1e-6, 1e-6);
EXPECT_NEAR(covariance.at<double>(3,3), 1e-6, 1e-6);
// Test with image size set to compute covariance differently:
// 3D points of A reprojected in B frame with 10 % error. For the angle,
// it should still be close to zero (10% error is added to the ray, so if
// there was no error in angle, then result doesn't change).
result = util3d::estimateMotion3DTo2D(
words3A, words2B, cam,
/*minInliers=*/4,
/*iterations=*/100,
/*reprojError=*/2.0,
/*flagsPnP=*/0,
/*refineIterations=*/1,
/*varianceMedianRatio=*/4,
/*maxVariance=*/0.0f,
guess,
words3B,
&covariance,
&matchesOut,
&inliersOut,
/*splitLinearCovarianceComponents=*/false
);
EXPECT_FALSE(result.isNull());
result.getTranslationAndEulerAngles(x,y,z,roll,pitch,yaw);
EXPECT_NEAR(x, 0, 1e-6);
EXPECT_NEAR(y, 0, 1e-6);
EXPECT_NEAR(z, 0, 1e-6);
EXPECT_NEAR(roll, 0, 1e-6);
EXPECT_NEAR(pitch, 0, 1e-6);
EXPECT_NEAR(yaw, 0, 1e-6);
EXPECT_EQ(matchesOut.size(), 7u);
EXPECT_EQ(inliersOut.size(), 6u);
// covariance must be 6x6
EXPECT_EQ(covariance.rows, 6);
EXPECT_EQ(covariance.cols, 6);
EXPECT_NEAR(covariance.at<double>(0,0), 0.066, 1e-3);
EXPECT_NEAR(covariance.at<double>(3,3), 1e-6, 1e-6);
// Test with exact same 3D points, covariance in xyz expected to be close to 0 (or epsilon 1e-6)
result = util3d::estimateMotion3DTo2D(
words3A, words2B, cam,
/*minInliers=*/4,
/*iterations=*/100,
/*reprojError=*/2.0,
/*flagsPnP=*/0,
/*refineIterations=*/1,
/*varianceMedianRatio=*/4,
/*maxVariance=*/0.0f,
guess,
words3A,
&covariance,
&matchesOut,
&inliersOut,
/*splitLinearCovarianceComponents=*/false
);
EXPECT_FALSE(result.isNull());
result.getTranslationAndEulerAngles(x,y,z,roll,pitch,yaw);
EXPECT_NEAR(x, 0, 1e-6);
EXPECT_NEAR(y, 0, 1e-6);
EXPECT_NEAR(z, 0, 1e-6);
EXPECT_NEAR(roll, 0, 1e-6);
EXPECT_NEAR(pitch, 0, 1e-6);
EXPECT_NEAR(yaw, 0, 1e-6);
EXPECT_EQ(matchesOut.size(), 7u);
EXPECT_EQ(inliersOut.size(), 6u);
// covariance must be 6x6
EXPECT_EQ(covariance.rows, 6);
EXPECT_EQ(covariance.cols, 6);
EXPECT_NEAR(covariance.at<double>(0,0), 1e-6, 1e-6);
EXPECT_NEAR(covariance.at<double>(3,3), 1e-6, 1e-6);
}
// Same test than above, but with added noise on the points and pixels
TEST(Util3dMotionEstimation, estimateMotion3DTo2DWithNoise) {
// Two triangles in front of the camera at two different depths, centered with the middle of the image frame
std::map<int, cv::Point3f> words3A = {
{0, cv::Point3f(1,0,1)},
{1, cv::Point3f(1,1,-1)},
{2, cv::Point3f(1,-1,-1)},
{3, cv::Point3f(2,0,0)},
{4, cv::Point3f(2,0.5,0)},
{5, cv::Point3f(3,-0.5,0)},
{6, cv::Point3f(2,0,10)} // outlier
};
CameraModel cam(200, 200, 320, 240, CameraModel::opticalRotation(), 0, cv::Size(640, 480));
std::map<int, cv::KeyPoint> words2B;
std::map<int, cv::Point3f> words3B;
for(auto & pt: words3A) {
cv::Point3f ptt = util3d::transformPoint(pt.second, cam.localTransform().inverse());
float u,v;
cam.reproject(ptt.x,ptt.y,ptt.z, u, v);
if(cam.inFrame(u,v)) {
// Add +-5 pixels noise to 2D keypoints
words2B.insert(std::make_pair(pt.first, cv::KeyPoint(u+randomNoise(5.0f), v+randomNoise(5.0f), 3)));
}
else {
words2B.insert(std::make_pair(pt.first, cv::KeyPoint(10, 10, 3)));
}
// Add +-2 cm noise to 3D points
words3B.insert(std::make_pair(pt.first,
cv::Point3f(pt.second.x+randomNoise(0.02f), pt.second.y+randomNoise(0.02f), pt.second.z+randomNoise(0.02f))));
pt.second.x += randomNoise(0.02f);
pt.second.y += randomNoise(0.02f);
pt.second.z += randomNoise(0.02f);
}
Transform guess = Transform::getIdentity(); // non-null identity
cv::Mat covariance;
std::vector<int> matchesOut, inliersOut;
// Test without image size set, so covariance is computed completely by
// reproj errors, which is expected to be zero here
Transform result = util3d::estimateMotion3DTo2D(
words3A, words2B, cam,
/*minInliers=*/4,
/*iterations=*/100,
/*reprojError=*/5.0,
/*flagsPnP=*/0,
/*refineIterations=*/1,
/*varianceMedianRatio=*/4,
/*maxVariance=*/0.0f,
guess,
words3B,
&covariance,
&matchesOut,
&inliersOut,
/*splitLinearCovarianceComponents=*/true
);
EXPECT_FALSE(result.isNull());
float x,y,z,roll,pitch,yaw;
result.getTranslationAndEulerAngles(x,y,z,roll,pitch,yaw);
EXPECT_NEAR(x, 0, 3e-2);
EXPECT_NEAR(y, 0, 3e-2);
EXPECT_NEAR(z, 0, 3e-2);
EXPECT_NEAR(roll, 0, 1e-2);
EXPECT_NEAR(pitch, 0, 1e-2);
EXPECT_NEAR(yaw, 0, 1e-2);
EXPECT_EQ(matchesOut.size(), 7u);
EXPECT_EQ(inliersOut.size(), 6u);
// covariance must be 6x6
EXPECT_EQ(covariance.rows, 6);
EXPECT_EQ(covariance.cols, 6);
EXPECT_NEAR(covariance.at<double>(0,0), 5e-3, 1e-2);
EXPECT_NEAR(covariance.at<double>(1,1), 5e-3, 1e-2);
EXPECT_NEAR(covariance.at<double>(2,2), 5e-3, 1e-2);
EXPECT_NE(covariance.at<double>(0,0), covariance.at<double>(1,1));
EXPECT_NE(covariance.at<double>(1,1), covariance.at<double>(2,2));
EXPECT_NE(covariance.at<double>(0,0), covariance.at<double>(2,2));
EXPECT_NEAR(covariance.at<double>(3,3), 1e-2, 1e-2);
}
TEST(Util3dMotionEstimation, estimateMotion3DTo2DMultiCamBasic) {
// Two triangles in front of the camera at two different depths, centered with the middle of the image frame
std::map<int, cv::Point3f> words3A = {
{0, cv::Point3f(1,0,0.5)},
{1, cv::Point3f(1,0.5,-0.5)},
{2, cv::Point3f(1,-0.5,-0.5)},
{3, cv::Point3f(2,0,0)},
{4, cv::Point3f(2,0.25,0)},
{5, cv::Point3f(3,-0.25,0)},
{6, cv::Point3f(2,0,10)} // outlier
};
// Transform that point cloud for the left and right cameras
std::map<int, cv::Point3f> words3ALeftRight;
Transform leftT(0,0,M_PI/2);
Transform rightT(0,0,-M_PI/2);
int index = 7;
for(auto & pt: words3A) {
cv::Point3f ptT = util3d::transformPoint(pt.second, leftT);
words3ALeftRight.insert(std::make_pair(index++, ptT));
ptT = util3d::transformPoint(pt.second, rightT);
words3ALeftRight.insert(std::make_pair(index++, ptT));
}
words3A.insert(words3ALeftRight.begin(), words3ALeftRight.end());
float imageWidth = 640;
CameraModel camFront(200, 200, 320, 240, CameraModel::opticalRotation(), 0, cv::Size(imageWidth, 480));
CameraModel camLeft(200, 200, 320, 240, leftT*CameraModel::opticalRotation(), 0, cv::Size(imageWidth, 480));
CameraModel camRight(200, 200, 320, 240, rightT*CameraModel::opticalRotation(), 0, cv::Size(imageWidth, 480));
std::vector<CameraModel> models = {camFront, camLeft, camRight};
std::map<int, cv::KeyPoint> words2B;
for(auto & pt: words3A) {
for(size_t i=0; i<models.size(); ++i) {
cv::Point3f ptt = util3d::transformPoint(pt.second, models[i].localTransform().inverse());
float u,v;
if(ptt.z>0) {
models[i].reproject(ptt.x,ptt.y,ptt.z, u, v);
if(models[i].inFrame(u,v)) {
words2B.insert(std::make_pair(pt.first, cv::KeyPoint((i*imageWidth)+u, v, 3)));
break;
}
else if(pt.second.z > 9) {
words2B.insert(std::make_pair(pt.first, cv::KeyPoint((i*imageWidth)+10, 10, 3)));
break;
}
}
}
}
EXPECT_EQ(words3A.size(), words2B.size());
std::map<int, cv::Point3f> words3B; // leave empty
Transform guess = Transform::getIdentity(); // non-null identity
cv::Mat covariance;
std::vector<std::vector<int> > matchesOut, inliersOut;
// For the three approaches, the results should be the same
Transform result;
for(int i=0; i<3; ++i) {
result = util3d::estimateMotion3DTo2D(
words3A, words2B, models,
/*samplingPolicy*/i,
/*minInliers=*/4,
/*iterations=*/100,
/*reprojError=*/2.0,
/*flagsPnP=*/0,
/*refineIterations=*/1,
/*varianceMedianRatio=*/4,
/*maxVariance=*/0.0f,
guess,
words3B,
&covariance,
&matchesOut,
&inliersOut,
/*splitLinearCovarianceComponents=*/false
);
#ifdef RTABMAP_OPENGV
EXPECT_FALSE(result.isNull());
float x,y,z,roll,pitch,yaw;
result.getTranslationAndEulerAngles(x,y,z,roll,pitch,yaw);
EXPECT_NEAR(x, 0, 1e-2);
EXPECT_NEAR(y, 0, 1e-2);
EXPECT_NEAR(z, 0, 1e-2);
EXPECT_NEAR(roll, 0, 5e-3);
EXPECT_NEAR(pitch, 0, 5e-3);
EXPECT_NEAR(yaw, 0, 5e-3);
EXPECT_EQ(matchesOut.size(), 3u);
EXPECT_EQ(inliersOut.size(), 3u);
for(size_t i=0; i<matchesOut.size(); ++i) {
EXPECT_EQ(matchesOut[i].size(), 7u);
}
for(size_t i=0; i<inliersOut.size(); ++i) {
EXPECT_EQ(inliersOut[i].size(), 6u);
}
// covariance must be 6x6
EXPECT_EQ(covariance.rows, 6);
EXPECT_EQ(covariance.cols, 6);
EXPECT_NEAR(covariance.at<double>(0,0), 0.03, 1e-2);
EXPECT_NEAR(covariance.at<double>(3,3), 1e-3, 1e-3);
#else
EXPECT_TRUE(result.isNull());
#endif
}
}
// Same thing than above, but with noise
TEST(Util3dMotionEstimation, estimateMotion3DTo2DMultiCamWithNoise) {
// Two triangles in front of the camera at two different depths, centered with the middle of the image frame
std::map<int, cv::Point3f> words3A = {
{0, cv::Point3f(1,0,0.5)},
{1, cv::Point3f(1,0.5,-0.5)},
{2, cv::Point3f(1,-0.5,-0.5)},
{3, cv::Point3f(2,0,0)},
{4, cv::Point3f(2,0.25,0)},
{5, cv::Point3f(3,-0.25,0)},
{6, cv::Point3f(2,0,10)} // outlier
};
// Transform that point cloud for the left and right cameras
std::map<int, cv::Point3f> words3ALeftRight;
Transform leftT(0,0,M_PI/2);
Transform rightT(0,0,-M_PI/2);
int index = 7;
for(auto & pt: words3A) {
cv::Point3f ptT = util3d::transformPoint(pt.second, leftT);
words3ALeftRight.insert(std::make_pair(index++, ptT));
ptT = util3d::transformPoint(pt.second, rightT);
words3ALeftRight.insert(std::make_pair(index++, ptT));
}
words3A.insert(words3ALeftRight.begin(), words3ALeftRight.end());
float imageWidth = 640;
CameraModel camFront(200, 200, 320, 240, CameraModel::opticalRotation(), 0, cv::Size(imageWidth, 480));
CameraModel camLeft(200, 200, 320, 240, leftT*CameraModel::opticalRotation(), 0, cv::Size(imageWidth, 480));
CameraModel camRight(200, 200, 320, 240, rightT*CameraModel::opticalRotation(), 0, cv::Size(imageWidth, 480));
std::vector<CameraModel> models = {camFront, camLeft, camRight};
std::map<int, cv::KeyPoint> words2B;
std::map<int, cv::Point3f> words3B;
for(auto & pt: words3A) {
for(size_t i=0; i<models.size(); ++i) {
cv::Point3f ptt = util3d::transformPoint(pt.second, models[i].localTransform().inverse());
float u,v;
if(ptt.z>0) {
models[i].reproject(ptt.x,ptt.y,ptt.z, u, v);
if(models[i].inFrame(u,v)) {
// Add +-5 pixels noise to 2D keypoints
words2B.insert(std::make_pair(pt.first, cv::KeyPoint((i*imageWidth)+u+randomNoise(5.0f), v+randomNoise(5.0f), 3)));
break;
}
else if(pt.second.z > 9) {
words2B.insert(std::make_pair(pt.first, cv::KeyPoint((i*imageWidth)+10, 10, 3)));
break;
}
}
}
// Add +-2 cm noise to 3D points
words3B.insert(std::make_pair(pt.first,
cv::Point3f(pt.second.x+randomNoise(0.02f), pt.second.y+randomNoise(0.02f), pt.second.z+randomNoise(0.02f))));
pt.second.x += randomNoise(0.02f);
pt.second.y += randomNoise(0.02f);
pt.second.z += randomNoise(0.02f);
}
EXPECT_EQ(words3A.size(), words2B.size());
Transform guess = Transform::getIdentity(); // non-null identity
cv::Mat covariance;
std::vector<std::vector<int> > matchesOut, inliersOut;
// For the three approaches, the results should be the same
Transform result;
for(int i=0; i<3; ++i) {
result = util3d::estimateMotion3DTo2D(
words3A, words2B, models,
/*samplingPolicy*/i,
/*minInliers=*/4,
/*iterations=*/100,
/*reprojError=*/6.0,
/*flagsPnP=*/0,
/*refineIterations=*/1,
/*varianceMedianRatio=*/4,
/*maxVariance=*/0.0f,
guess,
words3B,
&covariance,
&matchesOut,
&inliersOut,
/*splitLinearCovarianceComponents=*/false
);
#ifdef RTABMAP_OPENGV
EXPECT_FALSE(result.isNull());
float x,y,z,roll,pitch,yaw;
result.getTranslationAndEulerAngles(x,y,z,roll,pitch,yaw);
EXPECT_NEAR(x, 0, 6e-2);
EXPECT_NEAR(y, 0, 6e-2);
EXPECT_NEAR(z, 0, 6e-2);
EXPECT_NEAR(roll, 0, 5e-2);
EXPECT_NEAR(pitch, 0, 5e-2);
EXPECT_NEAR(yaw, 0, 5e-2);
EXPECT_EQ(matchesOut.size(), 3u);
EXPECT_EQ(inliersOut.size(), 3u);
for(size_t i=0; i<matchesOut.size(); ++i) {
EXPECT_EQ(matchesOut[i].size(), 7u);
}
for(size_t i=0; i<inliersOut.size(); ++i) {
EXPECT_GE(inliersOut[i].size(), 2u);
}
// covariance must be 6x6
EXPECT_EQ(covariance.rows, 6);
EXPECT_EQ(covariance.cols, 6);
EXPECT_LT(covariance.at<double>(0,0), 0.008);
EXPECT_GT(covariance.at<double>(0,0), 1e-5);
EXPECT_LT(covariance.at<double>(3,3), 0.06);
EXPECT_GT(covariance.at<double>(3,3), 1e-5);
#else
EXPECT_TRUE(result.isNull());
#endif
}
}