Limit Max Features (#1614)

* limit max features internally inside pydetector and superpoint_rpautrat to  limit keypoint/desc data size and improve performance

* avoid full reset when max features is changed to work around the memory temprorary param update logic

* remove leftover comment

* try using image roi instead

* revert and regenerate

* unintended change

* fixing issue with pydetector, adding mask filtering

---------

Co-authored-by: Felix Toft <[email protected]>
This commit is contained in:
Felix Toft
2025-11-16 11:37:03 -08:00
committed by GitHub
co-authored by Felix Toft
parent e612d103bf
commit d1b5d62d21
5 changed files with 284 additions and 255 deletions
@@ -4,6 +4,7 @@
*/
#include "SuperpointRpautrat.h"
#include <rtabmap/core/Features2d.h>
#include <rtabmap/utilite/ULogger.h>
#include <rtabmap/utilite/UDirectory.h>
#include <rtabmap/utilite/UFile.h>
@@ -95,7 +96,7 @@ static std::string exportSuperPointTorchScript(
return output;
}
SPDetectorRpautrat::SPDetectorRpautrat(std::string superpointWeightsPath, std::string superpointModelPath, std::string outputDir, float threshold, bool nms, int minDistance, bool cuda) :
SPDetectorRpautrat::SPDetectorRpautrat(std::string superpointWeightsPath, std::string superpointModelPath, std::string outputDir, float threshold, bool nms, int minDistance, bool cuda, int maxFeatures, bool ssc) :
device_(torch::kCPU),
superpointWeightsPath_(superpointWeightsPath),
superpointModelPath_(superpointModelPath),
@@ -103,6 +104,8 @@ SPDetectorRpautrat::SPDetectorRpautrat(std::string superpointWeightsPath, std::s
threshold_(threshold),
nms_(nms),
minDistance_(minDistance),
maxFeatures_(maxFeatures),
ssc_(ssc),
detected_(false)
{
if(cuda && !torch::cuda::is_available())
@@ -136,12 +139,9 @@ cv::Mat SPDetectorRpautrat::compute(const std::vector<cv::KeyPoint> &keypoints)
}
// These should have the same size
UASSERT(static_cast<size_t>(desc_.size(0)) == keypoints.size());
UASSERT(static_cast<size_t>(desc_.rows) == keypoints.size());
// Move to CPU and return descriptors computed in the forward pass
torch::Tensor desc_cpu = desc_.to(torch::kCPU);
cv::Mat desc_mat(cv::Size(desc_cpu.size(1), desc_cpu.size(0)), CV_32FC1, desc_cpu.data_ptr<float>());
return desc_mat.clone();
return desc_;
}
std::vector<cv::KeyPoint> SPDetectorRpautrat::detect(const cv::Mat &img, const cv::Mat & mask)
@@ -189,12 +189,12 @@ std::vector<cv::KeyPoint> SPDetectorRpautrat::detect(const cv::Mat &img, const c
x = x.set_requires_grad(false).to(device_);
auto outputs = model_.forward({x}).toTuple();
keypoints_tensor_ = outputs->elements()[0].toTensor(); // [N, 2] keypoint coordinates
auto scores_tensor = outputs->elements()[1].toTensor(); // [N] keypoint scores
desc_ = outputs->elements()[2].toTensor(); // [N, 256] descriptors
auto kpts_tensor = outputs->elements()[0].toTensor(); // [N, 2] keypoint coordinates
auto scores_tensor = outputs->elements()[1].toTensor(); // [N] keypoint scores
torch::Tensor desc_tensor = outputs->elements()[2].toTensor(); // [N, 256] descriptors
// Convert to CPU for processing
auto keypoints_cpu = keypoints_tensor_.to(torch::kCPU);
auto keypoints_cpu = kpts_tensor.to(torch::kCPU);
auto scores_cpu = scores_tensor.to(torch::kCPU);
std::vector<cv::KeyPoint> filtered_keypoints;
@@ -213,16 +213,20 @@ std::vector<cv::KeyPoint> SPDetectorRpautrat::detect(const cv::Mat &img, const c
}
}
// Update the stored tensors to maintain correspondence
// This way if keypoints are re-ordered, we can still match kpts->descs in the compute step
// Filter descriptors based on mask
auto keep_indices = torch::from_blob(keep_indices_vec.data(), {(long int)keep_indices_vec.size()}, torch::kLong);
keep_indices = keep_indices.to(keypoints_tensor_.device());
auto filtered_keypoints_tensor = keypoints_tensor_.index_select(0, keep_indices);
auto filtered_descriptors = desc_.index_select(0, keep_indices);
keep_indices = keep_indices.to(desc_tensor.device());
auto filtered_descriptors = desc_tensor.index_select(0, keep_indices);
keypoints_tensor_ = filtered_keypoints_tensor;
desc_ = filtered_descriptors;
// Convert descriptors to cv::Mat
auto filtered_descriptors_cpu = filtered_descriptors.to(torch::kCPU);
cv::Mat descriptors_mat(filtered_descriptors_cpu.size(0), filtered_descriptors_cpu.size(1), CV_32FC1, filtered_descriptors_cpu.data_ptr<float>());
cv::Mat descriptors_clone = descriptors_mat.clone(); // Clone to own the memory
// Apply limitKeypoints to enforce maxFeatures and SSC
Feature2D::limitKeypoints(filtered_keypoints, descriptors_clone, maxFeatures_, cv::Size(img.cols, img.rows), ssc_);
desc_ = descriptors_clone;
detected_ = true;
return filtered_keypoints;
}
@@ -23,17 +23,22 @@ class SPDetectorRpautrat {
float threshold = 0.005f,
bool nms = true,
int nmsRadius = 4,
bool cuda = false
bool cuda = false,
int maxFeatures = 1000,
bool ssc = false
);
virtual ~SPDetectorRpautrat();
std::vector<cv::KeyPoint> detect(const cv::Mat &img, const cv::Mat & mask = cv::Mat());
cv::Mat compute(const std::vector<cv::KeyPoint> &keypoints);
// Setters for post-processing parameters that don't require model reinitialization
void setMaxFeatures(int maxFeatures) { maxFeatures_ = maxFeatures; }
void setSSC(bool ssc) { ssc_ = ssc; }
private:
torch::jit::script::Module model_;
torch::Device device_;
torch::Tensor desc_;
torch::Tensor keypoints_tensor_;
cv::Mat desc_;
std::string superpointWeightsPath_;
std::string superpointModelPath_;
@@ -42,6 +47,8 @@ class SPDetectorRpautrat {
bool nms_;
int minDistance_;
bool cuda_;
int maxFeatures_;
bool ssc_;
bool detected_;
};
@@ -59,7 +59,7 @@ def generate_model(
# Load SuperPoint model and weights
model = SuperPoint(
nms_radius=nms_radius,
threshold=threshold,
detection_threshold=threshold,
).eval().to(device)
# Load weights without forcing CPU location to allow CUDA usage