Add FlannIndex abstract interface and implement NanoFlannIndex subclass (#1744)

* Add FlannIndex abstract interface and implement NanoFlannIndex subclass

* Refactored: made NanoFlann a new NN type instead of inheriting FlannIndex. Added tests. Vendoring nanoflann.h directly in the repo. RegistrationVis now use NANOFLANN_INDEX_KDTREE_SINGLE (instead of FLANN_INDEX_KDTREE_SINGLE) flann index for 2d points matching.

* cleanup comments, added FlannIndex doxygen

* Fixing windows tests

* updating flaky test

* Simplified interface, added flann kdtree single approach selectable by parameters.

* RegVis: symmetry of nanoflann for two branches of guess feature matching

* cv::BFMatcher baseline

* Small cmake optimization FLANN_KDTREE_MEM_OPT only defined for FlannIndex

* Refactored where FLANN_KDTREE_MEM_OPT is defined

* fixed file name already exist

* cleanup

* fixup build

---------

Co-authored-by: matlabbe <[email protected]>
This commit is contained in:
Muhammad
2026-08-16 09:51:39 -07:00
committed by GitHub
co-authored by matlabbe
parent df52523a0c
commit f647014f54
28 changed files with 7537 additions and 179 deletions
+138 -11
View File
@@ -34,36 +34,117 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
namespace rtabmap {
class NanoFlannIndex;
/**
* @class FlannIndex
* @brief Nearest neighbor index over a set of features
*
* Wraps the search structures of the vendored rtflann and nanoflann libraries
* behind one interface, the structure being chosen with flann_algorithm_t at
* build time. Used for the visual word dictionary (VWDictionary) and for the
* 2D point searches of visual registration (RegistrationVis).
*
* The features are not copied: the index refers to the matrices it is given and
* keeps them alive, cv::Mat data being reference counted, so they must not be
* modified in place while it is in use. Every point it holds is
* designated by an index, assigned in the order the points were added and
* stable for the lifetime of the index: removePoint() leaves a hole rather
* than renumbering the points after it.
*/
class RTABMAP_CORE_EXPORT FlannIndex
{
public:
// A forward of the internal enum, indexes should match. See src/rtflann/defines.h
/**
* @enum flann_algorithm_t
* @brief The index structure built by buildIndex()
*
* The values under 8 are forwarded from rtflann's own enum and have to
* match it (see src/rtflann/defines.h); the nanoflann ones are
* rtabmap-specific and kept outside its range (0-7, 254, 255). A value is
* written in the serialized index header and checked back on load, so none
* of them may be renumbered.
*
* The nanoflann structures take float features only (nanoflann has no
* Hamming metric) and search exactly, ignoring "checks". That makes them
* the fastest ones for 2D and 3D points, and the wrong ones for
* descriptors: an exact search visits more and more of the tree as the
* dimension grows, down to being as slow as an exhaustive search. Prefer
* the approximate rtflann kd-trees for those.
*/
enum flann_algorithm_t
{
FLANN_INDEX_LINEAR = 0,
FLANN_INDEX_KDTREE = 1,
FLANN_INDEX_KDTREE_SINGLE = 4,
FLANN_INDEX_LSH = 6,
FLANN_INDEX_LINEAR = 0, ///< Exhaustive search
FLANN_INDEX_KDTREE = 1, ///< 4 randomized kd-trees, searched approximately
FLANN_INDEX_KDTREE_SINGLE = 4, ///< Single kd-tree, searched exactly
FLANN_INDEX_LSH = 6, ///< Locality-Sensitive Hashing (binary descriptors)
/// nanoflann kd-tree. With a rebalancing factor of 1 it is built once,
/// which is the cheapest to build and to search; over 1 it is the
/// weight-balanced tree accepting addPoints()/removePoint(), which
/// cannot be serialized while some of its points are removed.
NANOFLANN_INDEX_KDTREE_SINGLE = 100,
};
FlannIndex();
virtual ~FlannIndex();
/** @brief Drop the index and everything it holds, back to the state of a new one. */
void release();
/**
* @brief Serialize the index, to be given back to loadIndex()
* @param computeChecksum Add a checksum of the indexed features to the
* data, which loadIndex() compares against the features it is given
* @return The serialized index, empty when there is nothing to serialize or
* when the structure in use cannot be
*
* The format depends on the architecture and on the versions of the
* vendored libraries: loadIndex() refuses an index it cannot read, leaving
* it to be rebuilt.
*/
std::vector<unsigned char> serializeIndex(bool computeChecksum = true) const;
/** @return Number of indexed features, the removed ones excluded. */
size_t indexedFeatures() const;
// return Bytes
/**
* @return Bytes used by the index, the features themselves excluded as
* they are only referred to.
*/
size_t memoryUsed() const;
// Note that useDistanceL1 doesn't have any effect if LSH is used
/**
* @brief Build the index over the given features, releasing any previous one
* @param algorithm The structure to build
* @param features One feature per row, CV_32FC1 or, for the rtflann
* structures only, CV_8UC1 for binary descriptors (Hamming distance)
* @param useDistanceL1 Search with the L1 distance instead of L2, ignored
* by LSH and by the binary descriptors
* @param rebalancingFactor Fraction (factor-1)/factor of the index that can
* be left removed before it is rebuilt, e.g. half of it for 2. Set
* to 1 to never rebuild it.
*/
void buildIndex(
flann_algorithm_t algorithm,
const cv::Mat & features,
bool useDistanceL1 = false,
float rebalancingFactor = 2.0f);
// Return false if the indexData doesn't correspond to expected features used and parameters.
/**
* @brief Load an index serialized by serializeIndex(), releasing any previous one
* @param indexData The serialized index
* @param algorithm The structure it was built with
* @param features The very same features it was built with, in the same
* order: the index refers to them by their row
* @param useDistanceL1 The distance it was built with
* @param rebalancingFactor See buildIndex(). The serialized data carries the
* one the index was built with, which is deprecated and ignored:
* this one is used instead.
* @param errorMsg Filled with what didn't match when the index is refused
* @return False if the data doesn't correspond to the given features and
* parameters, in which case the index is left released
*/
bool loadIndex(
const std::vector<unsigned char> & indexData,
flann_algorithm_t algorithm,
@@ -71,6 +152,7 @@ public:
bool useDistanceL1 = false,
float rebalancingFactor = 2.0f,
std::string * errorMsg = NULL);
/** @brief Load an index from a raw buffer, see the overload above. */
bool loadIndex(
const unsigned char * indexData,
size_t indexDataSize,
@@ -80,16 +162,46 @@ public:
float rebalancingFactor = 2.0f,
std::string * errorMsg = NULL);
/** @return Whether an index has been built or loaded. */
bool isBuilt();
/** @return Type of the indexed features (CV_32FC1 or CV_8UC1). */
int featuresType() const {return featuresType_;}
/** @return Dimension of the indexed features. */
int featuresDim() const {return featuresDim_;}
/**
* @brief Add features to the index
* @param features One feature per row, of the type and dimension the index
* was built with
* @return The index assigned to each of them, empty when the structure
* doesn't accept points after it is built
*/
std::vector<unsigned int> addPoints(const cv::Mat & features);
/**
* @brief Remove an indexed feature, by the index addPoints() gave for it
*
* The feature is only marked as removed: it is skipped by the searches, but
* keeps taking memory until the index is rebuilt (see the rebalancing
* factor of buildIndex()). Not supported by every structure.
*/
void removePoint(unsigned int index);
// return squared distances (indices should be casted in size_t)
/**
* @brief Search the k nearest neighbors of each query
* @param query One feature per row, of the type and dimension the index was
* built with
* @param indices Neighbors found, one query per row, CV_32SC1. The
* neighbors that couldn't be found are set to -1.
* @param dists Their squared distances, CV_32FC1, or CV_32SC1 for the
* Hamming distances of binary descriptors
* @param knn Number of neighbors to search for
* @param checks Number of leaves an approximate search visits, the exact
* structures ignoring it
* @param eps Search for eps-approximate neighbors
* @param sorted Give the neighbors back by increasing distance
*/
void knnSearch(
const cv::Mat & query,
cv::Mat & indices,
@@ -99,7 +211,21 @@ public:
float eps = 0.0,
bool sorted = true) const;
// return squared distances
/**
* @brief Search the neighbors of each query within a radius
* @param query One feature per row, of the type and dimension the index was
* built with
* @param indices Neighbors found, one vector per query
* @param dists Their squared distances, one vector per query
* @param radius Search radius, squared internally: it is a distance, not a
* squared one
* @param maxNeighbors Maximum number of neighbors per query, the nearest
* ones being kept. 0 for all of them.
* @param checks Number of leaves an approximate search visits, the exact
* structures ignoring it
* @param eps Search for eps-approximate neighbors
* @param sorted Give the neighbors back by increasing distance
*/
void radiusSearch(
const cv::Mat & query,
std::vector<std::vector<size_t> > & indices,
@@ -111,7 +237,8 @@ public:
bool sorted = true) const;
private:
void * index_;
void * index_; // rtflann backend
NanoFlannIndex * nanoIndex_; // nanoflann backend, only one of the two is set
unsigned int nextIndex_;
int featuresType_;
int featuresDim_;
+4 -4
View File
@@ -252,10 +252,10 @@ class RTABMAP_CORE_EXPORT Parameters
RTABMAP_PARAM(Mem, RotateImagesUpsideUp, bool, false, "Rotate images so that upside is up if they are not already. This can be useful in case the robots don't have all same camera orientation but are using the same map, so that not rotation-invariant visual features can still be used across the fleet.");
// KeypointMemory (Keypoint-based)
RTABMAP_PARAM(Kp, NNStrategy, int, 1, "kNNFlannNaive=0, kNNFlannKdTree=1, kNNFlannLSH=2, kNNBruteForce=3, kNNBruteForceGPU=4");
RTABMAP_PARAM(Kp, NNStrategy, int, 1, "FLANN Linear=0, FLANN KdTree=1, FLANN LSH=2, Brute Force=3, Brute Force GPU=4, FLANN KdTree Single=5, NanoFLANN KdTree=6");
RTABMAP_PARAM(Kp, IncrementalDictionary, bool, true, "");
RTABMAP_PARAM(Kp, IncrementalFlann, bool, true, uFormat("When using FLANN based strategy, add/remove points to its index without always rebuilding the index (the index is built only when the dictionary increases of the factor \"%s\" in size).", kKpFlannRebalancingFactor().c_str()));
RTABMAP_PARAM(Kp, FlannRebalancingFactor, float, 2.0, uFormat("Factor used when rebuilding the incremental FLANN index (see \"%s\"). Set <=1 to disable.", kKpIncrementalFlann().c_str()));
RTABMAP_PARAM(Kp, IncrementalFlann, bool, true, uFormat("When using FLANN based strategy, add/remove points to its index without always rebuilding the index (the index is only rebuilt when too many of its features have been removed, see \"%s\").", kKpFlannRebalancingFactor().c_str()));
RTABMAP_PARAM(Kp, FlannRebalancingFactor, float, 2.0, uFormat("Rebuild the incremental FLANN index (see \"%s\") once the ratio (factor-1)/factor of its features has been removed, e.g. half of them for a factor of 2. Rebuilding frees the memory of the removed features and speeds up the searches. Features are mostly removed when memory management is enabled (\"%s\" or \"%s\"). Set to 1 to never rebuild, which also uses less memory as the features don't have to be referenced one by one.", kKpIncrementalFlann().c_str(), kRtabmapTimeThr().c_str(), kRtabmapMemoryThr().c_str()));
RTABMAP_PARAM(Kp, ByteToFloat, bool, false, uFormat("For %s=1, binary descriptors are converted to float by converting each byte to float instead of converting each bit to float. When converting bytes instead of bits, less memory is used and search is faster at the cost of slightly less accurate matching.", kKpNNStrategy().c_str()));
RTABMAP_PARAM(Kp, MaxDepth, float, 0, "Filter extracted keypoints by depth (0=inf).");
RTABMAP_PARAM(Kp, MinDepth, float, 0, "Filter extracted keypoints by depth.");
@@ -780,7 +780,7 @@ class RTABMAP_CORE_EXPORT Parameters
RTABMAP_PARAM(Vis, GridRows, int, 1, uFormat("Number of rows of the grid used to extract uniformly \"%s / grid cells\" features from each cell.", kVisMaxFeatures().c_str()));
RTABMAP_PARAM(Vis, GridCols, int, 1, uFormat("Number of columns of the grid used to extract uniformly \"%s / grid cells\" features from each cell.", kVisMaxFeatures().c_str()));
RTABMAP_PARAM(Vis, CorType, int, 0, "Correspondences computation approach: 0=Features Matching, 1=Optical Flow");
RTABMAP_PARAM(Vis, CorNNType, int, 1, uFormat("[%s=0] kNNFlannNaive=0, kNNFlannKdTree=1, kNNFlannLSH=2, kNNBruteForce=3, kNNBruteForceGPU=4, BruteForceCrossCheck=5, SuperGlue=6, GMS=7. Used for features matching approach.", kVisCorType().c_str()));
RTABMAP_PARAM(Vis, CorNNType, int, 1, uFormat("[%s=0] FLANN Linear=0, FLANN KdTree=1, FLANN LSH=2, Brute Force=3, Brute Force GPU=4, Brute Force Cross Check=5, SuperGlue=6, GMS=7, FLANN KdTree Single=8, NanoFLANN KdTree=9. Used for features matching approach.", kVisCorType().c_str()));
RTABMAP_PARAM(Vis, CorNNDR, float, 0.8, uFormat("[%s=0] NNDR: nearest neighbor distance ratio. Used for knn features matching approach.", kVisCorType().c_str()));
RTABMAP_PARAM(Vis, CorGuessWinSize, int, 40, uFormat("[%s=0] Matching window size (pixels) around projected points when a guess transform is provided to find correspondences. 0 means disabled.", kVisCorType().c_str()));
RTABMAP_PARAM(Vis, CorGuessMatchToProjection, bool, false, uFormat("[%s=0] Match frame's corners to source's projected points (when guess transform is provided) instead of projected points to frame's corners.", kVisCorType().c_str()));
@@ -75,6 +75,13 @@ public:
int getMinInliers() const {return _minInliers;}
/** @return **Vis/CorNNType** nearest-neighbor strategy. */
int getNNType() const {return _nnType;}
/** @return Name of the **Vis/CorNNType** nearest-neighbor strategy in use. */
std::string getNNTypeName() const {return getNNTypeName(_nnType);}
/**
* @brief Name of a Vis/CorNNType value
*/
static std::string getNNTypeName(int nnType);
/** @return **Vis/CorNNDR** ratio test threshold. */
float getNNDR() const {return _nndr;}
/** @return **Vis/EstimationType** (0: 3D→3D, 1: PnP, 2: epipolar). */
@@ -68,6 +68,11 @@ public:
/**
* @enum NNStrategy
* @brief Nearest neighbor search strategies for descriptor matching
*
* The values are those of the Kp/NNStrategy parameter, saved in user
* configurations and databases: append, never renumber. The Vis/CorNNType
* parameter has strategies of its own, its values are mapped to these ones
* by RegistrationVis::nnStrategyFromCorNNType().
*/
enum NNStrategy{
kNNFlannNaive, ///< FLANN naive search (exhaustive)
@@ -75,6 +80,8 @@ public:
kNNFlannLSH, ///< FLANN Locality-Sensitive Hashing (ideal for binary descriptors)
kNNBruteForce, ///< Brute force CPU search
kNNBruteForceGPU, ///< Brute force GPU-accelerated search (requires CUDA)
kNNFlannKdTreeSingle, ///< FLANN single exact kd-tree index (rebuilt whenever a word is added, for an index built once)
kNNNanoFlannKdTree, ///< nanoflann kd-tree index (float descriptors only, incremental)
kNNUndef ///< Undefined strategy
};
@@ -106,11 +113,17 @@ public:
return "BRUTE FORCE";
case kNNBruteForceGPU:
return "BRUTE FORCE GPU";
case kNNNanoFlannKdTree:
return "NANOFLANN KD-TREE";
case kNNFlannKdTreeSingle:
return "FLANN KD-TREE SINGLE";
default:
return "Unknown";
}
}
public:
/**
* @brief Constructor