mirror of
https://github.com/introlab/rtabmap.git
synced 2026-10-04 00:57:46 +08:00
Added util3d_motion_estimation.h tests (2D->3D done)
This commit is contained in:
@@ -39,6 +39,41 @@ namespace rtabmap
|
||||
namespace util3d
|
||||
{
|
||||
|
||||
/**
|
||||
* @brief Estimates a 6-DOF camera transform from 3D-2D point correspondences using PnP RANSAC.
|
||||
*
|
||||
* This function estimates the motion (transform) between two views by solving the Perspective-n-Point (PnP)
|
||||
* problem using 3D points from one frame and their corresponding 2D keypoints in another frame.
|
||||
* It optionally refines the result, computes covariance of the pose estimate, and handles degenerate cases.
|
||||
*
|
||||
* @param words3A 3D points in frame A, indexed by feature ID.
|
||||
* @param words2B 2D keypoints in frame B, indexed by feature ID (shared with `words3A`).
|
||||
* @param cameraModel Intrinsic and extrinsic parameters of the camera (must be valid).
|
||||
* @param minInliers Minimum number of inliers required to accept the estimated transform. If the value is <4, it is set internally to 4.
|
||||
* @param iterations Number of RANSAC iterations for PnP.
|
||||
* @param reprojError Maximum allowed reprojection error (in pixels) to consider a point an inlier.
|
||||
* @param flagsPnP Flags to control the `cv::solvePnPRansac` behavior (e.g., `cv::SOLVEPNP_ITERATIVE`).
|
||||
* @param refineIterations Number of iterations for non-linear optimization (set to 0 to disable refinement).
|
||||
* @param varianceMedianRatio Index divisor used to select the robust variance threshold from sorted error residuals (e.g., 4 → use the 25% percentile).
|
||||
* @param maxVariance Maximum allowed median variance (linear error). Estimates with higher variance are rejected.
|
||||
* @param guess Initial guess for the camera pose (must not be null). Typically from odometry or motion model.
|
||||
* @param words3B Optional 3D points in frame B (if available). Used to better estimate 3D errors and variances.
|
||||
* @param covariance Optional output pointer for the estimated 6x6 pose covariance matrix.
|
||||
* The matrix contains linear variance in the top-left 3x3 and angular variance in the bottom-right 3x3.
|
||||
* @param matchesOut Optional output vector of all matched IDs used (regardless of inlier status).
|
||||
* @param inliersOut Optional output vector of matched IDs that were determined to be inliers.
|
||||
* @param splitLinearCovarianceComponents Whether to split and compute variance for X, Y, Z components separately.
|
||||
*
|
||||
* @return The estimated transformation from frame B to frame A. If estimation fails or is rejected due to variance,
|
||||
* a null transform is returned (i.e., `transform.isNull()` will be true).
|
||||
*
|
||||
* @note
|
||||
* - If `words3B` is provided, 3D variance is computed by comparing reprojected points to actual transformed points.
|
||||
* - If `words3B` is empty, variance is estimated using reprojection error only.
|
||||
* - The function assumes the camera model's local transform is known and factored into the pose estimation.
|
||||
*
|
||||
* @see cv::solvePnPRansac
|
||||
*/
|
||||
Transform RTABMAP_CORE_EXPORT estimateMotion3DTo2D(
|
||||
const std::map<int, cv::Point3f> & words3A,
|
||||
const std::map<int, cv::KeyPoint> & words2B,
|
||||
@@ -56,26 +91,41 @@ Transform RTABMAP_CORE_EXPORT estimateMotion3DTo2D(
|
||||
std::vector<int> * matchesOut = 0,
|
||||
std::vector<int> * inliersOut = 0,
|
||||
bool splitLinearCovarianceComponents = false);
|
||||
|
||||
Transform RTABMAP_CORE_EXPORT estimateMotion3DTo2D(
|
||||
const std::map<int, cv::Point3f> & words3A,
|
||||
const std::map<int, cv::KeyPoint> & words2B,
|
||||
const std::vector<CameraModel> & cameraModels,
|
||||
unsigned int samplingPolicy = 0, // 0=AUTO, 1=ANY, 2=HOMOGENEOUS
|
||||
int minInliers = 10,
|
||||
int iterations = 100,
|
||||
double reprojError = 5.,
|
||||
int flagsPnP = 0,
|
||||
int pnpRefineIterations = 1,
|
||||
int varianceMedianRatio = 4,
|
||||
float maxVariance = 0,
|
||||
const Transform & guess = Transform::getIdentity(),
|
||||
const std::map<int, cv::Point3f> & words3B = std::map<int, cv::Point3f>(),
|
||||
cv::Mat * covariance = 0, // mean reproj error if words3B is not set
|
||||
std::vector<int> * matchesOut = 0,
|
||||
std::vector<int> * inliersOut = 0,
|
||||
bool splitLinearCovarianceComponents = false);
|
||||
|
||||
/**
|
||||
* @brief Estimates the 3D-to-2D motion (pose) transformation between a set of 3D points and their corresponding 2D keypoints using the OpenGV library.
|
||||
*
|
||||
* This function uses a robust multi-camera Perspective-n-Point (PnP) algorithm to estimate the transformation from a 3D point cloud (scene A)
|
||||
* to a set of 2D keypoints (scene B) given the corresponding camera models and initial pose guess.
|
||||
* The method supports multiple camera models and uses RANSAC with OpenGV for outlier rejection.
|
||||
*
|
||||
* @param words3A 3D points in the source frame (scene A), indexed by feature ID.
|
||||
* @param words2B 2D keypoints in the destination frame (scene B), indexed by feature ID.
|
||||
* @param cameraModels List of camera models (multi-camera rig setup) for the destination frame.
|
||||
* @param samplingPolicy Sampling strategy (0 = auto, 1 = any, 2 = homogeneous multi-camera).
|
||||
* @param minInliers Minimum number of inliers required to consider the estimated transform as valid.
|
||||
* @param iterations Maximum number of RANSAC iterations.
|
||||
* @param reprojError Reprojection error threshold used by RANSAC.
|
||||
* @param flagsPnP PnP flags (not used internally).
|
||||
* @param refineIterations Number of pose refinement iterations after RANSAC (not used internally).
|
||||
* @param varianceMedianRatio Divider used to compute median-based variance from the error distribution.
|
||||
* @param maxVariance Maximum allowed variance to accept the transform. Higher values permit more noisy estimates.
|
||||
* @param guess Initial guess of the transformation.
|
||||
* @param words3B Optional 3D points in destination frame (scene B) to evaluate the covariance using 3D
|
||||
* correspondences, otherwise covariance is estimated from reprojection errors.
|
||||
* @param covariance Optional output 6x6 covariance matrix of the estimated transform.
|
||||
* @param matchesOut Optional output: matches grouped per camera.
|
||||
* @param inliersOut Optional output: inliers grouped per camera.
|
||||
* @param splitLinearCovarianceComponents If true, linear covariance is split into separate x/y/z components.
|
||||
*
|
||||
* @return The estimated Transform from scene A to scene B. Returns a null Transform if estimation fails or variance exceeds threshold.
|
||||
*
|
||||
* @note This function requires RTAB-Map to be built with OpenGV support.
|
||||
*
|
||||
* @warning The function assumes all camera models have the same image width and valid intrinsic parameters.
|
||||
*
|
||||
* @see https://github.com/laurentkneip/opengv
|
||||
*/
|
||||
Transform estimateMotion3DTo2D(
|
||||
const std::map<int, cv::Point3f> & words3A,
|
||||
const std::map<int, cv::KeyPoint> & words2B,
|
||||
@@ -94,6 +144,28 @@ Transform estimateMotion3DTo2D(
|
||||
std::vector<std::vector<int> > * matchesOut,
|
||||
std::vector<std::vector<int> > * inliersOut,
|
||||
bool splitLinearCovarianceComponents);
|
||||
/**
|
||||
* @brief Estimates the 3D-to-2D motion (pose) transformation between a set of 3D points and their corresponding 2D keypoints using the OpenGV library.
|
||||
* @see estimateMotion3DTo2D(), the only difference is that output matches and inliers are combined in same vector instead of per camera
|
||||
*/
|
||||
Transform RTABMAP_CORE_EXPORT estimateMotion3DTo2D(
|
||||
const std::map<int, cv::Point3f> & words3A,
|
||||
const std::map<int, cv::KeyPoint> & words2B,
|
||||
const std::vector<CameraModel> & cameraModels,
|
||||
unsigned int samplingPolicy = 0, // 0=AUTO, 1=ANY, 2=HOMOGENEOUS
|
||||
int minInliers = 10,
|
||||
int iterations = 100,
|
||||
double reprojError = 5.,
|
||||
int flagsPnP = 0,
|
||||
int pnpRefineIterations = 1,
|
||||
int varianceMedianRatio = 4,
|
||||
float maxVariance = 0,
|
||||
const Transform & guess = Transform::getIdentity(),
|
||||
const std::map<int, cv::Point3f> & words3B = std::map<int, cv::Point3f>(),
|
||||
cv::Mat * covariance = 0, // mean reproj error if words3B is not set
|
||||
std::vector<int> * matchesOut = 0,
|
||||
std::vector<int> * inliersOut = 0,
|
||||
bool splitLinearCovarianceComponents = false);
|
||||
|
||||
Transform RTABMAP_CORE_EXPORT estimateMotion3DTo3D(
|
||||
const std::map<int, cv::Point3f> & words3A,
|
||||
|
||||
@@ -224,11 +224,11 @@ Transform estimateMotion3DTo2D(
|
||||
//divide by 4 instead of 2 to ignore very very far features (stereo)
|
||||
double median_error_sqr_lin = 2.1981 * (double)errorSqrdDists[errorSqrdDists.size () / varianceMedianRatio];
|
||||
UASSERT(uIsFinite(median_error_sqr_lin));
|
||||
(*covariance)(cv::Range(0,3), cv::Range(0,3)) *= median_error_sqr_lin;
|
||||
(*covariance)(cv::Range(0,3), cv::Range(0,3)) *= median_error_sqr_lin + 1e-6;
|
||||
std::sort(errorSqrdAngles.begin(), errorSqrdAngles.end());
|
||||
double median_error_sqr_ang = 2.1981 * (double)errorSqrdAngles[errorSqrdAngles.size () / varianceMedianRatio];
|
||||
UASSERT(uIsFinite(median_error_sqr_ang));
|
||||
(*covariance)(cv::Range(3,6), cv::Range(3,6)) *= median_error_sqr_ang;
|
||||
(*covariance)(cv::Range(3,6), cv::Range(3,6)) *= median_error_sqr_ang + 1e-6;
|
||||
|
||||
if(splitLinearCovarianceComponents)
|
||||
{
|
||||
@@ -242,9 +242,9 @@ Transform estimateMotion3DTo2D(
|
||||
UASSERT(uIsFinite(median_error_sqr_x));
|
||||
UASSERT(uIsFinite(median_error_sqr_y));
|
||||
UASSERT(uIsFinite(median_error_sqr_z));
|
||||
covariance->at<double>(0,0) = median_error_sqr_x;
|
||||
covariance->at<double>(1,1) = median_error_sqr_y;
|
||||
covariance->at<double>(2,2) = median_error_sqr_z;
|
||||
covariance->at<double>(0,0) = median_error_sqr_x + 1e-6;
|
||||
covariance->at<double>(1,1) = median_error_sqr_y + 1e-6;
|
||||
covariance->at<double>(2,2) = median_error_sqr_z + 1e-6;
|
||||
|
||||
median_error_sqr_lin = uMax3(median_error_sqr_x, median_error_sqr_y, median_error_sqr_z);
|
||||
}
|
||||
@@ -267,7 +267,7 @@ Transform estimateMotion3DTo2D(
|
||||
err += uNormSquared(imagePoints.at(inliers[i]).x - imagePointsReproj.at(inliers[i]).x, imagePoints.at(inliers[i]).y - imagePointsReproj.at(inliers[i]).y);
|
||||
}
|
||||
UASSERT(uIsFinite(err));
|
||||
*covariance *= std::sqrt(err/float(inliers.size()));
|
||||
*covariance *= std::sqrt(err/float(inliers.size())) + 1e-6;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -483,6 +483,7 @@ Transform estimateMotion3DTo2D(
|
||||
// convert 2d-3d correspondences into bearing vectors
|
||||
std::vector<std::shared_ptr<opengv::bearingVectors_t>> multiBearingVectors;
|
||||
multiBearingVectors.resize(cameraModels.size());
|
||||
std::vector<std::vector<int> > localIndexToGlobalIndex(cameraModels.size());
|
||||
for(size_t i=0; i<cameraModels.size();++i)
|
||||
{
|
||||
multiPoints[i] = std::make_shared<opengv::points_t>();
|
||||
@@ -497,6 +498,7 @@ Transform estimateMotion3DTo2D(
|
||||
cameraModels[cameraIndex].project(imagePoints[i].x, imagePoints[i].y, 1, pt[0], pt[1], pt[2]);
|
||||
pt = cv::normalize(pt);
|
||||
multiBearingVectors[cameraIndex]->push_back(opengv::bearingVector_t(pt[0], pt[1], pt[2]));
|
||||
localIndexToGlobalIndex[cameraIndex].push_back(i);
|
||||
}
|
||||
|
||||
//create a non-central absolute multi adapter
|
||||
@@ -527,9 +529,10 @@ Transform estimateMotion3DTo2D(
|
||||
|
||||
UDEBUG("Ransac result: %s", pnp.prettyPrint().c_str());
|
||||
UDEBUG("Ransac iterations done: %d", ransac.iterations_);
|
||||
for (size_t i=0; i < cameraModels.size(); ++i)
|
||||
{
|
||||
inliers.insert(inliers.end(), ransac.inliers_[i].begin(), ransac.inliers_[i].end());
|
||||
for (size_t i=0; i < cameraModels.size(); ++i) {
|
||||
for (size_t j=0; j < ransac.inliers_[i].size(); ++j) {
|
||||
inliers.push_back(localIndexToGlobalIndex[i][ransac.inliers_[i][j]]);
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -705,6 +708,7 @@ Transform estimateMotion3DTo2D(
|
||||
|
||||
if(matchesOut)
|
||||
{
|
||||
matchesOut->clear();
|
||||
matchesOut->resize(cameraModels.size());
|
||||
UASSERT(matches.size() == cameraIndexes.size());
|
||||
for(size_t i=0; i<matches.size(); ++i)
|
||||
@@ -715,6 +719,7 @@ Transform estimateMotion3DTo2D(
|
||||
}
|
||||
if(inliersOut)
|
||||
{
|
||||
inliersOut->clear();
|
||||
inliersOut->resize(cameraModels.size());
|
||||
for(unsigned int i=0; i<inliers.size(); ++i)
|
||||
{
|
||||
|
||||
@@ -39,4 +39,9 @@ gtest_discover_tests(test_util3d_correspondences)
|
||||
#util3d_mapping.h
|
||||
add_executable(test_util3d_mapping test_util3d_mapping.cpp)
|
||||
target_link_libraries(test_util3d_mapping gtest_main rtabmap_core)
|
||||
gtest_discover_tests(test_util3d_mapping)
|
||||
gtest_discover_tests(test_util3d_mapping)
|
||||
|
||||
#util3d_motion_estimation.h
|
||||
add_executable(test_util3d_motion_estimation test_util3d_motion_estimation.cpp)
|
||||
target_link_libraries(test_util3d_motion_estimation gtest_main rtabmap_core)
|
||||
gtest_discover_tests(test_util3d_motion_estimation)
|
||||
@@ -0,0 +1,467 @@
|
||||
#include "gtest/gtest.h"
|
||||
#include "rtabmap/core/util3d.h"
|
||||
#include "rtabmap/core/util3d_motion_estimation.h"
|
||||
#include "rtabmap/core/CameraModel.h"
|
||||
#include "rtabmap/utilite/UException.h"
|
||||
#include "rtabmap/utilite/UConversion.h"
|
||||
#include "rtabmap/core/Version.h"
|
||||
#include <pcl/io/pcd_io.h>
|
||||
|
||||
using namespace rtabmap;
|
||||
|
||||
float randomNoise(float max) {
|
||||
return ((static_cast<float>(rand()) / RAND_MAX) * 2.0f - 1.0f) * max;
|
||||
}
|
||||
|
||||
TEST(Util3dMotionEstimation, estimateMotion3DTo2DBasic) {
|
||||
|
||||
// Two triangles in front of the camera at two different depths, centered with the middle of the image frame
|
||||
std::map<int, cv::Point3f> words3A = {
|
||||
{0, cv::Point3f(1,0,1)},
|
||||
{1, cv::Point3f(1,1,-1)},
|
||||
{2, cv::Point3f(1,-1,-1)},
|
||||
{3, cv::Point3f(2,0,0)},
|
||||
{4, cv::Point3f(2,0.5,0)},
|
||||
{5, cv::Point3f(3,-0.5,0)},
|
||||
{6, cv::Point3f(2,0,10)} // outlier
|
||||
};
|
||||
|
||||
CameraModel cam(200, 200, 320, 240, CameraModel::opticalRotation(), 0, cv::Size(640, 480));
|
||||
std::map<int, cv::KeyPoint> words2B;
|
||||
for(auto & pt: words3A) {
|
||||
cv::Point3f ptt = util3d::transformPoint(pt.second, cam.localTransform().inverse());
|
||||
float u,v;
|
||||
cam.reproject(ptt.x,ptt.y,ptt.z, u, v);
|
||||
if(cam.inFrame(u,v)) {
|
||||
words2B.insert(std::make_pair(pt.first, cv::KeyPoint(u, v, 3)));
|
||||
}
|
||||
else {
|
||||
words2B.insert(std::make_pair(pt.first, cv::KeyPoint(10, 10, 3)));
|
||||
}
|
||||
}
|
||||
std::map<int, cv::Point3f> words3B; // leave empty
|
||||
|
||||
Transform guess = Transform::getIdentity(); // non-null identity
|
||||
cv::Mat covariance;
|
||||
std::vector<int> matchesOut, inliersOut;
|
||||
|
||||
// Test without image size set, so covariance is computed completely by
|
||||
// reproj errors, which is expected to be zero here
|
||||
Transform result = util3d::estimateMotion3DTo2D(
|
||||
words3A, words2B, CameraModel(200, 200, 320, 240),
|
||||
/*minInliers=*/4,
|
||||
/*iterations=*/100,
|
||||
/*reprojError=*/2.0,
|
||||
/*flagsPnP=*/0,
|
||||
/*refineIterations=*/1,
|
||||
/*varianceMedianRatio=*/4,
|
||||
/*maxVariance=*/0.0f,
|
||||
guess,
|
||||
words3B,
|
||||
&covariance,
|
||||
&matchesOut,
|
||||
&inliersOut,
|
||||
/*splitLinearCovarianceComponents=*/false
|
||||
);
|
||||
|
||||
EXPECT_FALSE(result.isNull());
|
||||
float x,y,z,roll,pitch,yaw;
|
||||
result.getTranslationAndEulerAngles(x,y,z,roll,pitch,yaw);
|
||||
EXPECT_NEAR(x, 0, 1e-6);
|
||||
EXPECT_NEAR(y, 0, 1e-6);
|
||||
EXPECT_NEAR(z, 0, 1e-6);
|
||||
EXPECT_NEAR(roll, 0, 1e-6);
|
||||
EXPECT_NEAR(pitch, 0, 1e-6);
|
||||
EXPECT_NEAR(yaw, 0, 1e-6);
|
||||
EXPECT_EQ(matchesOut.size(), 7u);
|
||||
EXPECT_EQ(inliersOut.size(), 6u);
|
||||
|
||||
// covariance must be 6x6
|
||||
EXPECT_EQ(covariance.rows, 6);
|
||||
EXPECT_EQ(covariance.cols, 6);
|
||||
EXPECT_NEAR(covariance.at<double>(0,0), 1e-6, 1e-6);
|
||||
EXPECT_NEAR(covariance.at<double>(3,3), 1e-6, 1e-6);
|
||||
|
||||
// Test with image size set to compute covariance differently:
|
||||
// 3D points of A reprojected in B frame with 10 % error. For the angle,
|
||||
// it should still be close to zero (10% error is added to the ray, so if
|
||||
// there was no error in angle, then result doesn't change).
|
||||
result = util3d::estimateMotion3DTo2D(
|
||||
words3A, words2B, cam,
|
||||
/*minInliers=*/4,
|
||||
/*iterations=*/100,
|
||||
/*reprojError=*/2.0,
|
||||
/*flagsPnP=*/0,
|
||||
/*refineIterations=*/1,
|
||||
/*varianceMedianRatio=*/4,
|
||||
/*maxVariance=*/0.0f,
|
||||
guess,
|
||||
words3B,
|
||||
&covariance,
|
||||
&matchesOut,
|
||||
&inliersOut,
|
||||
/*splitLinearCovarianceComponents=*/false
|
||||
);
|
||||
|
||||
EXPECT_FALSE(result.isNull());
|
||||
result.getTranslationAndEulerAngles(x,y,z,roll,pitch,yaw);
|
||||
EXPECT_NEAR(x, 0, 1e-6);
|
||||
EXPECT_NEAR(y, 0, 1e-6);
|
||||
EXPECT_NEAR(z, 0, 1e-6);
|
||||
EXPECT_NEAR(roll, 0, 1e-6);
|
||||
EXPECT_NEAR(pitch, 0, 1e-6);
|
||||
EXPECT_NEAR(yaw, 0, 1e-6);
|
||||
EXPECT_EQ(matchesOut.size(), 7u);
|
||||
EXPECT_EQ(inliersOut.size(), 6u);
|
||||
|
||||
// covariance must be 6x6
|
||||
EXPECT_EQ(covariance.rows, 6);
|
||||
EXPECT_EQ(covariance.cols, 6);
|
||||
EXPECT_NEAR(covariance.at<double>(0,0), 0.066, 1e-3);
|
||||
EXPECT_NEAR(covariance.at<double>(3,3), 1e-6, 1e-6);
|
||||
|
||||
// Test with exact same 3D points, covariance in xyz expected to be close to 0 (or epsilon 1e-6)
|
||||
result = util3d::estimateMotion3DTo2D(
|
||||
words3A, words2B, cam,
|
||||
/*minInliers=*/4,
|
||||
/*iterations=*/100,
|
||||
/*reprojError=*/2.0,
|
||||
/*flagsPnP=*/0,
|
||||
/*refineIterations=*/1,
|
||||
/*varianceMedianRatio=*/4,
|
||||
/*maxVariance=*/0.0f,
|
||||
guess,
|
||||
words3A,
|
||||
&covariance,
|
||||
&matchesOut,
|
||||
&inliersOut,
|
||||
/*splitLinearCovarianceComponents=*/false
|
||||
);
|
||||
|
||||
EXPECT_FALSE(result.isNull());
|
||||
result.getTranslationAndEulerAngles(x,y,z,roll,pitch,yaw);
|
||||
EXPECT_NEAR(x, 0, 1e-6);
|
||||
EXPECT_NEAR(y, 0, 1e-6);
|
||||
EXPECT_NEAR(z, 0, 1e-6);
|
||||
EXPECT_NEAR(roll, 0, 1e-6);
|
||||
EXPECT_NEAR(pitch, 0, 1e-6);
|
||||
EXPECT_NEAR(yaw, 0, 1e-6);
|
||||
EXPECT_EQ(matchesOut.size(), 7u);
|
||||
EXPECT_EQ(inliersOut.size(), 6u);
|
||||
|
||||
// covariance must be 6x6
|
||||
EXPECT_EQ(covariance.rows, 6);
|
||||
EXPECT_EQ(covariance.cols, 6);
|
||||
EXPECT_NEAR(covariance.at<double>(0,0), 1e-6, 1e-6);
|
||||
EXPECT_NEAR(covariance.at<double>(3,3), 1e-6, 1e-6);
|
||||
}
|
||||
|
||||
// Same test than above, but with added noise on the points and pixels
|
||||
TEST(Util3dMotionEstimation, estimateMotion3DTo2DWithNoise) {
|
||||
|
||||
// Two triangles in front of the camera at two different depths, centered with the middle of the image frame
|
||||
std::map<int, cv::Point3f> words3A = {
|
||||
{0, cv::Point3f(1,0,1)},
|
||||
{1, cv::Point3f(1,1,-1)},
|
||||
{2, cv::Point3f(1,-1,-1)},
|
||||
{3, cv::Point3f(2,0,0)},
|
||||
{4, cv::Point3f(2,0.5,0)},
|
||||
{5, cv::Point3f(3,-0.5,0)},
|
||||
{6, cv::Point3f(2,0,10)} // outlier
|
||||
};
|
||||
|
||||
CameraModel cam(200, 200, 320, 240, CameraModel::opticalRotation(), 0, cv::Size(640, 480));
|
||||
std::map<int, cv::KeyPoint> words2B;
|
||||
std::map<int, cv::Point3f> words3B;
|
||||
for(auto & pt: words3A) {
|
||||
cv::Point3f ptt = util3d::transformPoint(pt.second, cam.localTransform().inverse());
|
||||
float u,v;
|
||||
cam.reproject(ptt.x,ptt.y,ptt.z, u, v);
|
||||
if(cam.inFrame(u,v)) {
|
||||
// Add +-5 pixels noise to 2D keypoints
|
||||
words2B.insert(std::make_pair(pt.first, cv::KeyPoint(u+randomNoise(5.0f), v+randomNoise(5.0f), 3)));
|
||||
}
|
||||
else {
|
||||
words2B.insert(std::make_pair(pt.first, cv::KeyPoint(10, 10, 3)));
|
||||
}
|
||||
// Add +-2 cm noise to 3D points
|
||||
words3B.insert(std::make_pair(pt.first,
|
||||
cv::Point3f(pt.second.x+randomNoise(0.02f), pt.second.y+randomNoise(0.02f), pt.second.z+randomNoise(0.02f))));
|
||||
pt.second.x += randomNoise(0.02f);
|
||||
pt.second.y += randomNoise(0.02f);
|
||||
pt.second.z += randomNoise(0.02f);
|
||||
}
|
||||
|
||||
Transform guess = Transform::getIdentity(); // non-null identity
|
||||
cv::Mat covariance;
|
||||
std::vector<int> matchesOut, inliersOut;
|
||||
|
||||
// Test without image size set, so covariance is computed completely by
|
||||
// reproj errors, which is expected to be zero here
|
||||
Transform result = util3d::estimateMotion3DTo2D(
|
||||
words3A, words2B, cam,
|
||||
/*minInliers=*/4,
|
||||
/*iterations=*/100,
|
||||
/*reprojError=*/5.0,
|
||||
/*flagsPnP=*/0,
|
||||
/*refineIterations=*/1,
|
||||
/*varianceMedianRatio=*/4,
|
||||
/*maxVariance=*/0.0f,
|
||||
guess,
|
||||
words3B,
|
||||
&covariance,
|
||||
&matchesOut,
|
||||
&inliersOut,
|
||||
/*splitLinearCovarianceComponents=*/true
|
||||
);
|
||||
|
||||
EXPECT_FALSE(result.isNull());
|
||||
float x,y,z,roll,pitch,yaw;
|
||||
result.getTranslationAndEulerAngles(x,y,z,roll,pitch,yaw);
|
||||
EXPECT_NEAR(x, 0, 3e-2);
|
||||
EXPECT_NEAR(y, 0, 3e-2);
|
||||
EXPECT_NEAR(z, 0, 3e-2);
|
||||
EXPECT_NEAR(roll, 0, 1e-2);
|
||||
EXPECT_NEAR(pitch, 0, 1e-2);
|
||||
EXPECT_NEAR(yaw, 0, 1e-2);
|
||||
EXPECT_EQ(matchesOut.size(), 7u);
|
||||
EXPECT_EQ(inliersOut.size(), 6u);
|
||||
|
||||
// covariance must be 6x6
|
||||
EXPECT_EQ(covariance.rows, 6);
|
||||
EXPECT_EQ(covariance.cols, 6);
|
||||
EXPECT_NEAR(covariance.at<double>(0,0), 5e-3, 1e-2);
|
||||
EXPECT_NEAR(covariance.at<double>(1,1), 5e-3, 1e-2);
|
||||
EXPECT_NEAR(covariance.at<double>(2,2), 5e-3, 1e-2);
|
||||
EXPECT_NE(covariance.at<double>(0,0), covariance.at<double>(1,1));
|
||||
EXPECT_NE(covariance.at<double>(1,1), covariance.at<double>(2,2));
|
||||
EXPECT_NE(covariance.at<double>(0,0), covariance.at<double>(2,2));
|
||||
EXPECT_NEAR(covariance.at<double>(3,3), 1e-2, 1e-2);
|
||||
|
||||
}
|
||||
|
||||
TEST(Util3dMotionEstimation, estimateMotion3DTo2DMultiCamBasic) {
|
||||
|
||||
// Two triangles in front of the camera at two different depths, centered with the middle of the image frame
|
||||
std::map<int, cv::Point3f> words3A = {
|
||||
{0, cv::Point3f(1,0,0.5)},
|
||||
{1, cv::Point3f(1,0.5,-0.5)},
|
||||
{2, cv::Point3f(1,-0.5,-0.5)},
|
||||
{3, cv::Point3f(2,0,0)},
|
||||
{4, cv::Point3f(2,0.25,0)},
|
||||
{5, cv::Point3f(3,-0.25,0)},
|
||||
{6, cv::Point3f(2,0,10)} // outlier
|
||||
};
|
||||
// Transform that point cloud for the left and right cameras
|
||||
std::map<int, cv::Point3f> words3ALeftRight;
|
||||
Transform leftT(0,0,M_PI/2);
|
||||
Transform rightT(0,0,-M_PI/2);
|
||||
|
||||
int index = 7;
|
||||
for(auto & pt: words3A) {
|
||||
cv::Point3f ptT = util3d::transformPoint(pt.second, leftT);
|
||||
words3ALeftRight.insert(std::make_pair(index++, ptT));
|
||||
ptT = util3d::transformPoint(pt.second, rightT);
|
||||
words3ALeftRight.insert(std::make_pair(index++, ptT));
|
||||
}
|
||||
words3A.insert(words3ALeftRight.begin(), words3ALeftRight.end());
|
||||
|
||||
float imageWidth = 640;
|
||||
CameraModel camFront(200, 200, 320, 240, CameraModel::opticalRotation(), 0, cv::Size(imageWidth, 480));
|
||||
CameraModel camLeft(200, 200, 320, 240, leftT*CameraModel::opticalRotation(), 0, cv::Size(imageWidth, 480));
|
||||
CameraModel camRight(200, 200, 320, 240, rightT*CameraModel::opticalRotation(), 0, cv::Size(imageWidth, 480));
|
||||
std::vector<CameraModel> models = {camFront, camLeft, camRight};
|
||||
std::map<int, cv::KeyPoint> words2B;
|
||||
|
||||
for(auto & pt: words3A) {
|
||||
for(size_t i=0; i<models.size(); ++i) {
|
||||
cv::Point3f ptt = util3d::transformPoint(pt.second, models[i].localTransform().inverse());
|
||||
float u,v;
|
||||
if(ptt.z>0) {
|
||||
models[i].reproject(ptt.x,ptt.y,ptt.z, u, v);
|
||||
if(models[i].inFrame(u,v)) {
|
||||
words2B.insert(std::make_pair(pt.first, cv::KeyPoint((i*imageWidth)+u, v, 3)));
|
||||
break;
|
||||
}
|
||||
else if(pt.second.z > 9) {
|
||||
words2B.insert(std::make_pair(pt.first, cv::KeyPoint((i*imageWidth)+10, 10, 3)));
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
EXPECT_EQ(words3A.size(), words2B.size());
|
||||
std::map<int, cv::Point3f> words3B; // leave empty
|
||||
|
||||
Transform guess = Transform::getIdentity(); // non-null identity
|
||||
cv::Mat covariance;
|
||||
std::vector<std::vector<int> > matchesOut, inliersOut;
|
||||
|
||||
// For the three approaches, the results should be the same
|
||||
Transform result;
|
||||
for(int i=0; i<3; ++i) {
|
||||
result = util3d::estimateMotion3DTo2D(
|
||||
words3A, words2B, models,
|
||||
/*samplingPolicy*/i,
|
||||
/*minInliers=*/4,
|
||||
/*iterations=*/100,
|
||||
/*reprojError=*/2.0,
|
||||
/*flagsPnP=*/0,
|
||||
/*refineIterations=*/1,
|
||||
/*varianceMedianRatio=*/4,
|
||||
/*maxVariance=*/0.0f,
|
||||
guess,
|
||||
words3B,
|
||||
&covariance,
|
||||
&matchesOut,
|
||||
&inliersOut,
|
||||
/*splitLinearCovarianceComponents=*/false
|
||||
);
|
||||
|
||||
#ifdef RTABMAP_OPENGV
|
||||
EXPECT_FALSE(result.isNull());
|
||||
float x,y,z,roll,pitch,yaw;
|
||||
result.getTranslationAndEulerAngles(x,y,z,roll,pitch,yaw);
|
||||
EXPECT_NEAR(x, 0, 1e-2);
|
||||
EXPECT_NEAR(y, 0, 1e-2);
|
||||
EXPECT_NEAR(z, 0, 1e-2);
|
||||
EXPECT_NEAR(roll, 0, 5e-3);
|
||||
EXPECT_NEAR(pitch, 0, 5e-3);
|
||||
EXPECT_NEAR(yaw, 0, 5e-3);
|
||||
EXPECT_EQ(matchesOut.size(), 3u);
|
||||
EXPECT_EQ(inliersOut.size(), 3u);
|
||||
for(size_t i=0; i<matchesOut.size(); ++i) {
|
||||
EXPECT_EQ(matchesOut[i].size(), 7u);
|
||||
}
|
||||
for(size_t i=0; i<inliersOut.size(); ++i) {
|
||||
EXPECT_EQ(inliersOut[i].size(), 6u);
|
||||
}
|
||||
|
||||
// covariance must be 6x6
|
||||
EXPECT_EQ(covariance.rows, 6);
|
||||
EXPECT_EQ(covariance.cols, 6);
|
||||
EXPECT_NEAR(covariance.at<double>(0,0), 0.03, 1e-2);
|
||||
EXPECT_NEAR(covariance.at<double>(3,3), 1e-3, 1e-3);
|
||||
#else
|
||||
EXPECT_TRUE(result.isNull());
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
// Same thing than above, but with noise
|
||||
TEST(Util3dMotionEstimation, estimateMotion3DTo2DMultiCamWithNoise) {
|
||||
|
||||
// Two triangles in front of the camera at two different depths, centered with the middle of the image frame
|
||||
std::map<int, cv::Point3f> words3A = {
|
||||
{0, cv::Point3f(1,0,0.5)},
|
||||
{1, cv::Point3f(1,0.5,-0.5)},
|
||||
{2, cv::Point3f(1,-0.5,-0.5)},
|
||||
{3, cv::Point3f(2,0,0)},
|
||||
{4, cv::Point3f(2,0.25,0)},
|
||||
{5, cv::Point3f(3,-0.25,0)},
|
||||
{6, cv::Point3f(2,0,10)} // outlier
|
||||
};
|
||||
// Transform that point cloud for the left and right cameras
|
||||
std::map<int, cv::Point3f> words3ALeftRight;
|
||||
Transform leftT(0,0,M_PI/2);
|
||||
Transform rightT(0,0,-M_PI/2);
|
||||
|
||||
int index = 7;
|
||||
for(auto & pt: words3A) {
|
||||
cv::Point3f ptT = util3d::transformPoint(pt.second, leftT);
|
||||
words3ALeftRight.insert(std::make_pair(index++, ptT));
|
||||
ptT = util3d::transformPoint(pt.second, rightT);
|
||||
words3ALeftRight.insert(std::make_pair(index++, ptT));
|
||||
}
|
||||
words3A.insert(words3ALeftRight.begin(), words3ALeftRight.end());
|
||||
|
||||
float imageWidth = 640;
|
||||
CameraModel camFront(200, 200, 320, 240, CameraModel::opticalRotation(), 0, cv::Size(imageWidth, 480));
|
||||
CameraModel camLeft(200, 200, 320, 240, leftT*CameraModel::opticalRotation(), 0, cv::Size(imageWidth, 480));
|
||||
CameraModel camRight(200, 200, 320, 240, rightT*CameraModel::opticalRotation(), 0, cv::Size(imageWidth, 480));
|
||||
std::vector<CameraModel> models = {camFront, camLeft, camRight};
|
||||
std::map<int, cv::KeyPoint> words2B;
|
||||
std::map<int, cv::Point3f> words3B;
|
||||
|
||||
for(auto & pt: words3A) {
|
||||
for(size_t i=0; i<models.size(); ++i) {
|
||||
cv::Point3f ptt = util3d::transformPoint(pt.second, models[i].localTransform().inverse());
|
||||
float u,v;
|
||||
if(ptt.z>0) {
|
||||
models[i].reproject(ptt.x,ptt.y,ptt.z, u, v);
|
||||
if(models[i].inFrame(u,v)) {
|
||||
// Add +-5 pixels noise to 2D keypoints
|
||||
words2B.insert(std::make_pair(pt.first, cv::KeyPoint((i*imageWidth)+u+randomNoise(5.0f), v+randomNoise(5.0f), 3)));
|
||||
break;
|
||||
}
|
||||
else if(pt.second.z > 9) {
|
||||
words2B.insert(std::make_pair(pt.first, cv::KeyPoint((i*imageWidth)+10, 10, 3)));
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Add +-2 cm noise to 3D points
|
||||
words3B.insert(std::make_pair(pt.first,
|
||||
cv::Point3f(pt.second.x+randomNoise(0.02f), pt.second.y+randomNoise(0.02f), pt.second.z+randomNoise(0.02f))));
|
||||
pt.second.x += randomNoise(0.02f);
|
||||
pt.second.y += randomNoise(0.02f);
|
||||
pt.second.z += randomNoise(0.02f);
|
||||
}
|
||||
EXPECT_EQ(words3A.size(), words2B.size());
|
||||
|
||||
Transform guess = Transform::getIdentity(); // non-null identity
|
||||
cv::Mat covariance;
|
||||
std::vector<std::vector<int> > matchesOut, inliersOut;
|
||||
|
||||
// For the three approaches, the results should be the same
|
||||
Transform result;
|
||||
for(int i=0; i<3; ++i) {
|
||||
result = util3d::estimateMotion3DTo2D(
|
||||
words3A, words2B, models,
|
||||
/*samplingPolicy*/i,
|
||||
/*minInliers=*/4,
|
||||
/*iterations=*/100,
|
||||
/*reprojError=*/6.0,
|
||||
/*flagsPnP=*/0,
|
||||
/*refineIterations=*/1,
|
||||
/*varianceMedianRatio=*/4,
|
||||
/*maxVariance=*/0.0f,
|
||||
guess,
|
||||
words3B,
|
||||
&covariance,
|
||||
&matchesOut,
|
||||
&inliersOut,
|
||||
/*splitLinearCovarianceComponents=*/false
|
||||
);
|
||||
|
||||
#ifdef RTABMAP_OPENGV
|
||||
EXPECT_FALSE(result.isNull());
|
||||
float x,y,z,roll,pitch,yaw;
|
||||
result.getTranslationAndEulerAngles(x,y,z,roll,pitch,yaw);
|
||||
EXPECT_NEAR(x, 0, 6e-2);
|
||||
EXPECT_NEAR(y, 0, 6e-2);
|
||||
EXPECT_NEAR(z, 0, 6e-2);
|
||||
EXPECT_NEAR(roll, 0, 5e-2);
|
||||
EXPECT_NEAR(pitch, 0, 5e-2);
|
||||
EXPECT_NEAR(yaw, 0, 5e-2);
|
||||
EXPECT_EQ(matchesOut.size(), 3u);
|
||||
EXPECT_EQ(inliersOut.size(), 3u);
|
||||
for(size_t i=0; i<matchesOut.size(); ++i) {
|
||||
EXPECT_EQ(matchesOut[i].size(), 7u);
|
||||
}
|
||||
for(size_t i=0; i<inliersOut.size(); ++i) {
|
||||
EXPECT_GE(inliersOut[i].size(), 2u);
|
||||
}
|
||||
|
||||
// covariance must be 6x6
|
||||
EXPECT_EQ(covariance.rows, 6);
|
||||
EXPECT_EQ(covariance.cols, 6);
|
||||
EXPECT_LT(covariance.at<double>(0,0), 0.008);
|
||||
EXPECT_GT(covariance.at<double>(0,0), 1e-5);
|
||||
EXPECT_LT(covariance.at<double>(3,3), 0.06);
|
||||
EXPECT_GT(covariance.at<double>(3,3), 1e-5);
|
||||
#else
|
||||
EXPECT_TRUE(result.isNull());
|
||||
#endif
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user