Added util3d_features.h doc/tests

This commit is contained in:
matlabbe
2025-06-22 12:43:13 -07:00
parent cb6cdcae90
commit 3ca3c37ee7
5 changed files with 417 additions and 22 deletions
+105 -8
View File
@@ -43,7 +43,41 @@ namespace rtabmap
namespace util3d namespace util3d
{ {
/**
* @brief Projects 2D keypoints to 3D space using the provided depth image and camera models.
*
* This function takes a vector of 2D keypoints and projects them into 3D space by using
* depth values from a depth image and the associated camera models. It supports multi-camera setups
* by assuming the depth image is horizontally stacked with sub-images corresponding to each camera.
*
* If a depth value at a keypoint location is invalid or outside the specified depth range
* (`minDepth`, `maxDepth`), the output 3D point will be set to NaN.
*
* @param keypoints A vector of 2D keypoints (in image coordinates).
* @param depth The depth image (must be either `CV_32FC1` or `CV_16UC1`).
* For multiple cameras, the depth images should be horizontally concatenated.
* @param cameraModels A vector of camera models, one per camera. Each model must provide intrinsic
* parameters and optionally a local transform to apply to the resulting 3D point.
* @param minDepth Minimum valid depth value. If negative, no minimum is enforced.
* @param maxDepth Maximum valid depth value. If zero or negative, no maximum is enforced.
*
* @return A vector of 3D points (`cv::Point3f`) corresponding to the input keypoints.
* If the depth is invalid or outside the valid range, the point will contain NaNs.
*
* @throws Assertion failure if the depth image is empty or not of the expected type,
* or if the camera model vector is empty, or if camera index computation fails.
*/
std::vector<cv::Point3f> RTABMAP_CORE_EXPORT generateKeypoints3DDepth(
const std::vector<cv::KeyPoint> & keypoints,
const cv::Mat & depth,
const std::vector<CameraModel> & cameraModels,
float minDepth = 0,
float maxDepth = 0);
/**
* @brief Projects 2D keypoints to 3D space using the provided depth image and camera model.
*
* @see util3d::generateKeypoints3DDepth()
*/
std::vector<cv::Point3f> RTABMAP_CORE_EXPORT generateKeypoints3DDepth( std::vector<cv::Point3f> RTABMAP_CORE_EXPORT generateKeypoints3DDepth(
const std::vector<cv::KeyPoint> & keypoints, const std::vector<cv::KeyPoint> & keypoints,
const cv::Mat & depth, const cv::Mat & depth,
@@ -51,13 +85,29 @@ std::vector<cv::Point3f> RTABMAP_CORE_EXPORT generateKeypoints3DDepth(
float minDepth = 0, float minDepth = 0,
float maxDepth = 0); float maxDepth = 0);
std::vector<cv::Point3f> RTABMAP_CORE_EXPORT generateKeypoints3DDepth( /**
const std::vector<cv::KeyPoint> & keypoints, * @brief Projects 2D keypoints into 3D space using a disparity image and a stereo camera model.
const cv::Mat & depth, *
const std::vector<CameraModel> & cameraModels, * This function computes 3D coordinates for each input 2D keypoint by using the disparity image
float minDepth = 0, * and the stereo camera model. Invalid or out-of-range depth values result in 3D points with `NaN` components.
float maxDepth = 0); *
* The function applies the local transform of the left camera (from the stereo model) to each valid 3D point,
* if the transform is not null or identity.
*
* @param keypoints A vector of 2D keypoints (image coordinates) to be projected into 3D.
* @param disparity The disparity image (must be of type `CV_16SC1` or `CV_32F`).
* Disparity values should correspond to the keypoints' locations.
* @param stereoCameraModel A valid stereo camera model that provides projection parameters
* and an optional local transform.
* @param minDepth Minimum depth threshold. If negative, no minimum constraint is applied.
* @param maxDepth Maximum depth threshold. If zero or negative, no maximum constraint is applied.
*
* @return A vector of 3D points (`cv::Point3f`) corresponding to the input keypoints.
* Points with invalid or out-of-range depth are returned as `(NaN, NaN, NaN)`.
*
* @throws Assertion failure if the disparity image is empty or of incorrect type,
* or if the stereo camera model is not valid for projection.
*/
std::vector<cv::Point3f> RTABMAP_CORE_EXPORT generateKeypoints3DDisparity( std::vector<cv::Point3f> RTABMAP_CORE_EXPORT generateKeypoints3DDisparity(
const std::vector<cv::KeyPoint> & keypoints, const std::vector<cv::KeyPoint> & keypoints,
const cv::Mat & disparity, const cv::Mat & disparity,
@@ -65,6 +115,36 @@ std::vector<cv::Point3f> RTABMAP_CORE_EXPORT generateKeypoints3DDisparity(
float minDepth = 0, float minDepth = 0,
float maxDepth = 0); float maxDepth = 0);
/**
* @brief Computes 3D keypoints from corresponding 2D points in a stereo image pair.
*
* This function triangulates 3D points from pairs of corresponding 2D points
* (`leftCorners`, `rightCorners`) using a given stereo camera model. It optionally applies
* a validity mask and filters 3D points by depth range.
*
* For each point pair, the disparity is computed as the x-coordinate difference between
* left and right corners. Only positive disparities are considered valid. If a mask is
* provided, only entries with a non-zero value are processed.
*
* The resulting 3D points are optionally transformed using the stereo camera model's local transform,
* if one is defined and non-identity.
*
* Invalid or out-of-range points are set to `(NaN, NaN, NaN)`.
*
* @param leftCorners A vector of 2D points from the left stereo image.
* @param rightCorners A vector of corresponding 2D points from the right stereo image.
* @param model The stereo camera model containing intrinsic parameters and optional local transform.
* @param mask (Optional) A binary mask indicating which matches are valid (non-zero = valid).
* If empty, all matches are considered valid.
* @param minDepth Minimum allowed depth value. If negative, no minimum is applied.
* @param maxDepth Maximum allowed depth value. If zero or negative, no maximum is applied.
*
* @return A vector of 3D points (`cv::Point3f`) corresponding to valid stereo matches.
* Invalid points or those outside the depth range are returned as `(NaN, NaN, NaN)`.
*
* @throws Assertion failure if the input vectors are inconsistent in size,
* or if the stereo camera model is invalid (e.g., non-positive focal length or baseline).
*/
std::vector<cv::Point3f> RTABMAP_CORE_EXPORT generateKeypoints3DStereo( std::vector<cv::Point3f> RTABMAP_CORE_EXPORT generateKeypoints3DStereo(
const std::vector<cv::Point2f> & leftCorners, const std::vector<cv::Point2f> & leftCorners,
const std::vector<cv::Point2f> & rightCorners, const std::vector<cv::Point2f> & rightCorners,
@@ -84,6 +164,23 @@ std::map<int, cv::Point3f> RTABMAP_CORE_EXPORT generateWords3DMono(
double * variance = 0, double * variance = 0,
std::vector<int> * matchesOut = 0); std::vector<int> * matchesOut = 0);
/**
* @brief Aggregates word IDs and corresponding keypoints into a multimap.
*
* This function pairs each word ID from the input list with the corresponding keypoint
* from the input vector and stores them in a `std::multimap<int, cv::KeyPoint>`.
*
* It is assumed that the `wordIds` list and the `keypoints` vector are of the same length
* and ordered such that each word ID corresponds to the keypoint at the same index.
*
* @param wordIds A list of integer word IDs (e.g., visual word identifiers).
* @param keypoints A vector of keypoints associated with the word IDs.
*
* @return A multimap where each key is a word ID and the value is the corresponding `cv::KeyPoint`.
* Multiple keypoints can be associated with the same word ID.
*
* @throws Assertion failure if `wordIds.size() != keypoints.size()`.
*/
std::multimap<int, cv::KeyPoint> RTABMAP_CORE_EXPORT aggregate( std::multimap<int, cv::KeyPoint> RTABMAP_CORE_EXPORT aggregate(
const std::list<int> & wordIds, const std::list<int> & wordIds,
const std::vector<cv::KeyPoint> & keypoints); const std::vector<cv::KeyPoint> & keypoints);
+2
View File
@@ -2688,6 +2688,7 @@ cv::Point3f projectDisparityTo3D(
float disparity, float disparity,
const StereoCameraModel & model) const StereoCameraModel & model)
{ {
printf("disparity=%f\n", disparity);
if(disparity > 0.0f && model.baseline() > 0.0f && model.left().fx() > 0.0f) if(disparity > 0.0f && model.baseline() > 0.0f && model.left().fx() > 0.0f)
{ {
//Z = baseline * f / (d + cx1-cx0); //Z = baseline * f / (d + cx1-cx0);
@@ -2697,6 +2698,7 @@ cv::Point3f projectDisparityTo3D(
c = model.right().cx() - model.left().cx(); c = model.right().cx() - model.left().cx();
} }
float W = model.baseline()/(disparity + c); float W = model.baseline()/(disparity + c);
printf("W=%f\n", W);
return cv::Point3f((pt.x - model.left().cx())*W, return cv::Point3f((pt.x - model.left().cx())*W,
(pt.y - model.left().cy())*model.left().fx()/model.left().fy()*W, model.left().fx()*W); (pt.y - model.left().cy())*model.left().fx()/model.left().fy()*W, model.left().fx()*W);
} }
+14 -13
View File
@@ -51,19 +51,6 @@ namespace rtabmap
namespace util3d namespace util3d
{ {
std::vector<cv::Point3f> generateKeypoints3DDepth(
const std::vector<cv::KeyPoint> & keypoints,
const cv::Mat & depth,
const CameraModel & cameraModel,
float minDepth,
float maxDepth)
{
UASSERT(cameraModel.isValidForProjection());
std::vector<CameraModel> models;
models.push_back(cameraModel);
return generateKeypoints3DDepth(keypoints, depth, models, minDepth, maxDepth);
}
std::vector<cv::Point3f> generateKeypoints3DDepth( std::vector<cv::Point3f> generateKeypoints3DDepth(
const std::vector<cv::KeyPoint> & keypoints, const std::vector<cv::KeyPoint> & keypoints,
const cv::Mat & depth, const cv::Mat & depth,
@@ -119,6 +106,19 @@ std::vector<cv::Point3f> generateKeypoints3DDepth(
return keypoints3d; return keypoints3d;
} }
std::vector<cv::Point3f> generateKeypoints3DDepth(
const std::vector<cv::KeyPoint> & keypoints,
const cv::Mat & depth,
const CameraModel & cameraModel,
float minDepth,
float maxDepth)
{
UASSERT(cameraModel.isValidForProjection());
std::vector<CameraModel> models;
models.push_back(cameraModel);
return generateKeypoints3DDepth(keypoints, depth, models, minDepth, maxDepth);
}
std::vector<cv::Point3f> generateKeypoints3DDisparity( std::vector<cv::Point3f> generateKeypoints3DDisparity(
const std::vector<cv::KeyPoint> & keypoints, const std::vector<cv::KeyPoint> & keypoints,
const cv::Mat & disparity, const cv::Mat & disparity,
@@ -415,6 +415,7 @@ std::multimap<int, cv::KeyPoint> aggregate(
const std::list<int> & wordIds, const std::list<int> & wordIds,
const std::vector<cv::KeyPoint> & keypoints) const std::vector<cv::KeyPoint> & keypoints)
{ {
UASSERT(wordIds.size() == keypoints.size());
std::multimap<int, cv::KeyPoint> words; std::multimap<int, cv::KeyPoint> words;
std::vector<cv::KeyPoint>::const_iterator kpIter = keypoints.begin(); std::vector<cv::KeyPoint>::const_iterator kpIter = keypoints.begin();
for(std::list<int>::const_iterator iter=wordIds.begin(); iter!=wordIds.end(); ++iter) for(std::list<int>::const_iterator iter=wordIds.begin(); iter!=wordIds.end(); ++iter)
+5
View File
@@ -25,3 +25,8 @@ gtest_discover_tests(test_util3d_filtering)
add_executable(test_util3d_registration test_util3d_registration.cpp) add_executable(test_util3d_registration test_util3d_registration.cpp)
target_link_libraries(test_util3d_registration gtest_main rtabmap_core) target_link_libraries(test_util3d_registration gtest_main rtabmap_core)
gtest_discover_tests(test_util3d_registration) gtest_discover_tests(test_util3d_registration)
#util3d_features.h
add_executable(test_util3d_features test_util3d_features.cpp)
target_link_libraries(test_util3d_features gtest_main rtabmap_core)
gtest_discover_tests(test_util3d_features)
+290
View File
@@ -0,0 +1,290 @@
#include "gtest/gtest.h"
#include "rtabmap/core/util3d.h"
#include "rtabmap/core/util3d_features.h"
#include "rtabmap/core/CameraModel.h"
#include "rtabmap/utilite/UException.h"
#include "rtabmap/utilite/UConversion.h"
using namespace rtabmap;
TEST(Util3dFeatures, generateKeypoints3DDepthBasicProjection) {
// Arrange
std::vector<cv::KeyPoint> keypoints = {
cv::KeyPoint(10.0f, 10.0f, 1.0f),
cv::KeyPoint(20.0f, 20.0f, 1.0f)
};
// Create a 30x30 depth image with constant depth of 2.0
cv::Mat depth = cv::Mat::ones(30, 30, CV_32FC1) * 2.0f;
CameraModel model(100, 100, 15, 15, Transform::getIdentity(), 0, cv::Size(30,30));
auto keypoints3d = util3d::generateKeypoints3DDepth(
keypoints,
depth,
model,
0.5f,
5.0f
);
// Assert
ASSERT_EQ(keypoints3d.size(), keypoints.size());
for (const auto& pt : keypoints3d) {
EXPECT_TRUE(util3d::isFinite(pt)); // not NaN or Inf
EXPECT_NEAR(pt.z, 2.0f, 1e-5);
}
// invalid depth
keypoints3d = util3d::generateKeypoints3DDepth(
keypoints,
cv::Mat::zeros(30, 30, CV_32FC1),
model,
0.5f,
5.0f
);
// Assert
ASSERT_EQ(keypoints3d.size(), keypoints.size());
for (const auto& pt : keypoints3d) {
EXPECT_TRUE(!util3d::isFinite(pt)); // not NaN or Inf
}
}
TEST(Util3dFeatures, generateKeypoints3DDepthMultiCameras) {
std::vector<cv::KeyPoint> keypoints = {
cv::KeyPoint(15.0f, 15.0f, 1.0f),
cv::KeyPoint(45.0f, 15.0f, 1.0f),
cv::KeyPoint(75.0f, 15.0f, 1.0f),
cv::KeyPoint(105.0f, 15.0f, 1.0f)
};
cv::Mat depth = cv::Mat::ones(30, 120, CV_32FC1) * 2.0f;
std::vector<CameraModel> cameraModels = {
CameraModel(100, 100, 15, 15, Transform(0,0,M_PI/2)*CameraModel::opticalRotation(), 0, cv::Size(30,30)), // left
CameraModel(100, 100, 15, 15, CameraModel::opticalRotation(), 0, cv::Size(30,30)), // forward
CameraModel(100, 100, 15, 15, Transform(0,0,-M_PI/2)*CameraModel::opticalRotation(), 0, cv::Size(30,30)), // right
CameraModel(100, 100, 15, 15, Transform(0,0,M_PI)*CameraModel::opticalRotation(), 0, cv::Size(30,30)) // back
};
std::vector<cv::Point3f> keypoints3d = util3d::generateKeypoints3DDepth(
keypoints,
depth,
cameraModels,
0.5f,
5.0f
);
ASSERT_EQ(keypoints3d.size(), keypoints.size());
for (const auto& pt : keypoints3d) {
EXPECT_TRUE(util3d::isFinite(pt));
}
EXPECT_NEAR(keypoints3d[0].y, 2.0f, 1e-5);
EXPECT_NEAR(keypoints3d[1].x, 2.0f, 1e-5);
EXPECT_NEAR(keypoints3d[2].y, -2.0f, 1e-5);
EXPECT_NEAR(keypoints3d[3].x, -2.0f, 1e-5);
}
TEST(Util3dFeatures, generateKeypoints3DDisparityValidDisparity) {
std::vector<cv::KeyPoint> keypoints = {
cv::KeyPoint(5.0f, 5.0f, 1.0f),
cv::KeyPoint(10.0f, 10.0f, 1.0f)
};
StereoCameraModel stereoModel(100.0, 100.0, 10.0, 10.0, 0.075, Transform::getIdentity(), cv::Size(20,20));
// d = (baseline * f)/Z
cv::Mat disparity = cv::Mat::zeros(20, 20, CV_32F);
disparity.at<float>(5, 5) = stereoModel.baseline()*stereoModel.left().fx() / 2.0f; // Depth = 2.0
disparity.at<float>(10, 10) = stereoModel.baseline()*stereoModel.left().fx(); // Depth = 1.0
std::vector<cv::Point3f> keypoints3D = util3d::generateKeypoints3DDisparity(
keypoints, disparity, stereoModel, 0.1f, 5.0f
);
ASSERT_EQ(keypoints3D.size(), keypoints.size());
EXPECT_NEAR(keypoints3D[0].z, 2.0f, 1e-4);
EXPECT_NEAR(keypoints3D[1].z, 1.0f, 1e-4);
}
TEST(Util3dFeatures, generateKeypoints3DDisparityInvalidDisparityReturnsNaN) {
std::vector<cv::KeyPoint> keypoints = {
cv::KeyPoint(5.0f, 5.0f, 1.0f),
cv::KeyPoint(10.0f, 10.0f, 1.0f)
};
cv::Mat disparity = cv::Mat::zeros(20, 20, CV_32F);
disparity.at<float>(5, 5) = 0.0f; // Invalid disparity
disparity.at<float>(10, 10) = -1.0f; // Invalid disparity
StereoCameraModel stereoModel(100.0, 100.0, 10.0, 10.0, 0.075, Transform::getIdentity(), cv::Size(20,20));
std::vector<cv::Point3f> keypoints3D = util3d::generateKeypoints3DDisparity(
keypoints, disparity, stereoModel, 0.1f, 5.0f
);
for (const auto &pt : keypoints3D) {
EXPECT_FALSE(util3d::isFinite(pt));
}
}
TEST(Util3dFeatures, generateKeypoints3DDisparityDepthClipping) {
std::vector<cv::KeyPoint> keypoints = {
cv::KeyPoint(5.0f, 5.0f, 1.0f), // z = 2.0
cv::KeyPoint(10.0f, 10.0f, 1.0f) // z = 0.5
};
StereoCameraModel stereoModel(100.0, 100.0, 10.0, 10.0, 0.075, Transform::getIdentity(), cv::Size(20,20));
// d = (baseline * f)/Z
cv::Mat disparity = cv::Mat::zeros(20, 20, CV_32F);
disparity.at<float>(5, 5) = stereoModel.baseline()*stereoModel.left().fx() / 2.0f; // Depth = 2.0
disparity.at<float>(10, 10) = stereoModel.baseline()*stereoModel.left().fx() / 0.5f; // Depth = 0.5
// Clip to minDepth = 1.0
std::vector<cv::Point3f> keypoints3D = util3d::generateKeypoints3DDisparity(
keypoints, disparity, stereoModel, 1.0f, 3.0f
);
EXPECT_FALSE(util3d::isFinite(keypoints3D[1])); // z = 0.5, should be NaN
EXPECT_NEAR(keypoints3D[0].z, 2.0f, 1e-4); // z = 2.0, should be valid
}
TEST(Util3dFeatures, generateKeypoints3DStereoValidPointsWithDepthFilter) {
std::vector<cv::Point2f> leftCorners = {
{100.0f, 100.0f},
{120.0f, 120.0f}
};
std::vector<cv::Point2f> rightCorners = {
{90.0f, 100.0f}, // disparity = 10
{110.0f, 120.0f} // disparity = 10
};
StereoCameraModel model(500.0, 500.0, 100.0, 100.0, 0.075, Transform::getIdentity());
std::vector<unsigned char> mask; // Empty mask = all valid
float minDepth = 2.0f;
float maxDepth = 6.0f;
auto result = util3d::generateKeypoints3DStereo(leftCorners, rightCorners, model, mask, minDepth, maxDepth);
ASSERT_EQ(result.size(), leftCorners.size());
for (const auto& pt : result) {
EXPECT_TRUE(util3d::isFinite(pt));
EXPECT_NEAR(pt.z, 3.75f, 0.001);
}
}
TEST(Util3dFeatures, generateKeypoints3DStereoInvalidDisparityResultsInNaN) {
std::vector<cv::Point2f> leftCorners = {
{100.0f, 100.0f}
};
std::vector<cv::Point2f> rightCorners = {
{100.0f, 100.0f} // disparity = 0.0 (invalid)
};
StereoCameraModel model(500.0, 500.0, 100.0, 100.0, 0.075, Transform::getIdentity());
std::vector<unsigned char> mask;
auto result = util3d::generateKeypoints3DStereo(leftCorners, rightCorners, model, mask, 0.1f, 10.0f);
ASSERT_EQ(result.size(), 1);
EXPECT_TRUE(std::isnan(result[0].x));
EXPECT_TRUE(std::isnan(result[0].y));
EXPECT_TRUE(std::isnan(result[0].z));
}
TEST(Util3dFeatures, generateKeypoints3DStereoAppliesMaskCorrectly) {
std::vector<cv::Point2f> leftCorners = {
{100.0f, 100.0f},
{120.0f, 120.0f}
};
std::vector<cv::Point2f> rightCorners = {
{90.0f, 100.0f}, // disparity = 10
{110.0f, 120.0f} // disparity = 10
};
std::vector<unsigned char> mask = {0, 1}; // Only second is valid
StereoCameraModel model(500.0, 500.0, 100.0, 100.0, 0.075, Transform::getIdentity());
auto result = util3d::generateKeypoints3DStereo(leftCorners, rightCorners, model, mask, 0.1f, 10.0f);
ASSERT_EQ(result.size(), 2);
EXPECT_TRUE(std::isnan(result[0].x));
EXPECT_TRUE(util3d::isFinite(result[1]));
}
TEST(Util3dFeatures, generateKeypoints3DStereoOutOfRangeDepthResultsInNaN) {
std::vector<cv::Point2f> leftCorners = {
{100.0f, 100.0f}
};
std::vector<cv::Point2f> rightCorners = {
{99.5f, 100.0f} // very small disparity → large depth
};
StereoCameraModel model(500.0, 500.0, 100.0, 100.0, 0.075, Transform::getIdentity());
std::vector<unsigned char> mask;
auto result = util3d::generateKeypoints3DStereo(leftCorners, rightCorners, model, mask, 0.1f, 2.0f);
ASSERT_EQ(result.size(), 1);
EXPECT_TRUE(std::isnan(result[0].x));
}
TEST(Util3dFeatures, aggregateBasicAggregation) {
std::list<int> wordIds = {101, 102, 103};
std::vector<cv::KeyPoint> keypoints = {
cv::KeyPoint(10.0f, 20.0f, 1.0f),
cv::KeyPoint(30.0f, 40.0f, 1.0f),
cv::KeyPoint(50.0f, 60.0f, 1.0f)
};
auto result = util3d::aggregate(wordIds, keypoints);
ASSERT_EQ(result.size(), 3);
auto it = result.begin();
EXPECT_EQ(it->first, 101);
EXPECT_FLOAT_EQ(it->second.pt.x, 10.0f); ++it;
EXPECT_EQ(it->first, 102);
EXPECT_FLOAT_EQ(it->second.pt.y, 40.0f); ++it;
EXPECT_EQ(it->first, 103);
EXPECT_FLOAT_EQ(it->second.pt.x, 50.0f);
}
TEST(Util3dFeatures, aggregateHandlesDuplicateWordIds) {
std::list<int> wordIds = {200, 201, 200};
std::vector<cv::KeyPoint> keypoints = {
cv::KeyPoint(1.0f, 2.0f, 1.0f),
cv::KeyPoint(3.0f, 4.0f, 1.0f),
cv::KeyPoint(5.0f, 6.0f, 1.0f)
};
auto result = util3d::aggregate(wordIds, keypoints);
EXPECT_EQ(result.count(200), 2);
EXPECT_EQ(result.count(201), 1);
auto range = result.equal_range(200);
std::vector<float> x_values;
for (auto it = range.first; it != range.second; ++it) {
x_values.push_back(it->second.pt.x);
}
EXPECT_EQ(x_values[0], 1.0f);
EXPECT_EQ(x_values[1], 5.0f);
}
TEST(Util3dFeatures, aggregateEmptyInputReturnsEmptyMapOrInvalid) {
std::list<int> wordIds;
std::vector<cv::KeyPoint> keypoints;
auto result = util3d::aggregate(wordIds, keypoints);
EXPECT_TRUE(result.empty());
wordIds.push_back(1);
EXPECT_THROW(util3d::aggregate(wordIds, keypoints), UException);
}