mirror of
https://gitcode.com/gh_mirrors/ope/OpenFace.git
synced 2026-08-02 00:27:48 +00:00
Updating dlib and linking to it differently, increasing compilation speed and reducing dependency issues with it.
This commit is contained in:
738
lib/3rdParty/dlib/include/dlib/random_forest/random_forest_regression.h
vendored
Normal file
738
lib/3rdParty/dlib/include/dlib/random_forest/random_forest_regression.h
vendored
Normal file
@@ -0,0 +1,738 @@
|
||||
// Copyright (C) 2018 Davis E. King (davis@dlib.net)
|
||||
// License: Boost Software License See LICENSE.txt for the full license.
|
||||
#ifndef DLIB_RANdOM_FOREST_REGRESSION_H_
|
||||
#define DLIB_RANdOM_FOREST_REGRESSION_H_
|
||||
|
||||
#include "random_forest_regression_abstract.h"
|
||||
#include <vector>
|
||||
#include "../matrix.h"
|
||||
#include <algorithm>
|
||||
#include "../threads.h"
|
||||
|
||||
namespace dlib
|
||||
{
|
||||
|
||||
// ----------------------------------------------------------------------------------------
|
||||
|
||||
class dense_feature_extractor
|
||||
{
|
||||
|
||||
public:
|
||||
typedef uint32_t feature;
|
||||
typedef matrix<double,0,1> sample_type;
|
||||
|
||||
dense_feature_extractor(
|
||||
) = default;
|
||||
|
||||
void setup (
|
||||
const std::vector<sample_type>& x,
|
||||
const std::vector<double>& y
|
||||
)
|
||||
{
|
||||
DLIB_CASSERT(x.size() > 0);
|
||||
DLIB_CASSERT(x.size() == y.size());
|
||||
for (auto& el : x)
|
||||
DLIB_CASSERT(el.size() == x[0].size(), "All the vectors in a training set have to have the same dimensionality.");
|
||||
|
||||
DLIB_CASSERT(x[0].size() != 0, "The vectors can't be empty.");
|
||||
|
||||
num_feats = x[0].size();
|
||||
}
|
||||
|
||||
|
||||
void get_random_features (
|
||||
dlib::rand& rnd,
|
||||
size_t num,
|
||||
std::vector<feature>& feats
|
||||
) const
|
||||
{
|
||||
DLIB_ASSERT(max_num_feats() != 0);
|
||||
num = std::min(num, num_feats);
|
||||
|
||||
feats.clear();
|
||||
for (size_t i = 0; i < num_feats; ++i)
|
||||
feats.push_back(i);
|
||||
|
||||
// now pick num features at random
|
||||
for (size_t i = 0; i < num; ++i)
|
||||
{
|
||||
auto idx = rnd.get_integer_in_range(i,num_feats);
|
||||
std::swap(feats[i], feats[idx]);
|
||||
}
|
||||
feats.resize(num);
|
||||
}
|
||||
|
||||
double extract_feature_value (
|
||||
const sample_type& item,
|
||||
const feature& f
|
||||
) const
|
||||
{
|
||||
DLIB_ASSERT(max_num_feats() != 0);
|
||||
return item(f);
|
||||
}
|
||||
|
||||
size_t max_num_feats (
|
||||
) const
|
||||
{
|
||||
return num_feats;
|
||||
}
|
||||
|
||||
friend void serialize(const dense_feature_extractor& item, std::ostream& out)
|
||||
{
|
||||
serialize("dense_feature_extractor", out);
|
||||
serialize(item.num_feats, out);
|
||||
}
|
||||
|
||||
friend void deserialize(dense_feature_extractor& item, std::istream& in)
|
||||
{
|
||||
check_serialized_version("dense_feature_extractor", in);
|
||||
deserialize(item.num_feats, in);
|
||||
}
|
||||
|
||||
private:
|
||||
size_t num_feats = 0;
|
||||
};
|
||||
|
||||
// ----------------------------------------------------------------------------------------
|
||||
|
||||
template <
|
||||
typename feature_extractor
|
||||
>
|
||||
struct internal_tree_node
|
||||
{
|
||||
uint32_t left;
|
||||
uint32_t right;
|
||||
float split_threshold;
|
||||
typename feature_extractor::feature split_feature;
|
||||
};
|
||||
|
||||
template <typename feature_extractor>
|
||||
void serialize(const internal_tree_node<feature_extractor>& item, std::ostream& out)
|
||||
{
|
||||
serialize(item.left, out);
|
||||
serialize(item.right, out);
|
||||
serialize(item.split_threshold, out);
|
||||
serialize(item.split_feature, out);
|
||||
}
|
||||
|
||||
template <typename feature_extractor>
|
||||
void deserialize(internal_tree_node<feature_extractor>& item, std::istream& in)
|
||||
{
|
||||
deserialize(item.left, in);
|
||||
deserialize(item.right, in);
|
||||
deserialize(item.split_threshold, in);
|
||||
deserialize(item.split_feature, in);
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------------------
|
||||
|
||||
template <
|
||||
typename feature_extractor = dense_feature_extractor
|
||||
>
|
||||
class random_forest_regression_function
|
||||
{
|
||||
|
||||
public:
|
||||
|
||||
typedef feature_extractor feature_extractor_type;
|
||||
typedef typename feature_extractor::sample_type sample_type;
|
||||
|
||||
random_forest_regression_function(
|
||||
) = default;
|
||||
|
||||
random_forest_regression_function (
|
||||
feature_extractor_type&& fe_,
|
||||
std::vector<std::vector<internal_tree_node<feature_extractor>>>&& trees_,
|
||||
std::vector<std::vector<float>>&& leaves_
|
||||
) :
|
||||
fe(std::move(fe_)),
|
||||
trees(std::move(trees_)),
|
||||
leaves(std::move(leaves_))
|
||||
{
|
||||
DLIB_ASSERT(trees.size() > 0);
|
||||
DLIB_ASSERT(trees.size() == leaves.size(), "Every set of tree nodes has to have leaves");
|
||||
#ifdef ENABLE_ASSERTS
|
||||
for (size_t i = 0; i < trees.size(); ++i)
|
||||
{
|
||||
DLIB_ASSERT(trees[i].size() > 0, "A tree can't have 0 leaves.");
|
||||
for (auto& node : trees[i])
|
||||
{
|
||||
DLIB_ASSERT(trees[i].size()+leaves[i].size() > node.left, "left node index in tree is too big. There is no associated tree node or leaf.");
|
||||
DLIB_ASSERT(trees[i].size()+leaves[i].size() > node.right, "right node index in tree is too big. There is no associated tree node or leaf.");
|
||||
}
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
size_t get_num_trees(
|
||||
) const
|
||||
{
|
||||
return trees.size();
|
||||
}
|
||||
|
||||
const std::vector<std::vector<internal_tree_node<feature_extractor>>>& get_internal_tree_nodes (
|
||||
) const { return trees; }
|
||||
|
||||
const std::vector<std::vector<float>>& get_tree_leaves (
|
||||
) const { return leaves; }
|
||||
|
||||
const feature_extractor_type& get_feature_extractor (
|
||||
) const { return fe; }
|
||||
|
||||
double operator() (
|
||||
const sample_type& x
|
||||
) const
|
||||
{
|
||||
DLIB_ASSERT(get_num_trees() > 0);
|
||||
|
||||
double accum = 0;
|
||||
|
||||
for (size_t i = 0; i < trees.size(); ++i)
|
||||
{
|
||||
auto& tree = trees[i];
|
||||
// walk the tree to the leaf
|
||||
uint32_t idx = 0;
|
||||
while(idx < tree.size())
|
||||
{
|
||||
auto feature_value = fe.extract_feature_value(x, tree[idx].split_feature);
|
||||
if (feature_value < tree[idx].split_threshold)
|
||||
idx = tree[idx].left;
|
||||
else
|
||||
idx = tree[idx].right;
|
||||
}
|
||||
// compute leaf index
|
||||
accum += leaves[i][idx-tree.size()];
|
||||
}
|
||||
|
||||
return accum/trees.size();
|
||||
}
|
||||
|
||||
friend void serialize(const random_forest_regression_function& item, std::ostream& out)
|
||||
{
|
||||
serialize("random_forest_regression_function", out);
|
||||
serialize(item.fe, out);
|
||||
serialize(item.trees, out);
|
||||
serialize(item.leaves, out);
|
||||
}
|
||||
|
||||
friend void deserialize(random_forest_regression_function& item, std::istream& in)
|
||||
{
|
||||
check_serialized_version("random_forest_regression_function", in);
|
||||
deserialize(item.fe, in);
|
||||
deserialize(item.trees, in);
|
||||
deserialize(item.leaves, in);
|
||||
}
|
||||
|
||||
private:
|
||||
|
||||
/*!
|
||||
CONVENTION
|
||||
- trees.size() == leaves.size()
|
||||
- Any .left or .right index in trees that is larger than the number of
|
||||
nodes in the tree references a leaf. Moreover, the index of the leaf is
|
||||
computed by subtracting the number of nodes in the tree.
|
||||
!*/
|
||||
|
||||
feature_extractor_type fe;
|
||||
|
||||
// internal nodes of trees
|
||||
std::vector<std::vector<internal_tree_node<feature_extractor>>> trees;
|
||||
// leaves of trees
|
||||
std::vector<std::vector<float>> leaves;
|
||||
|
||||
};
|
||||
|
||||
// ----------------------------------------------------------------------------------------
|
||||
|
||||
template <
|
||||
typename feature_extractor = dense_feature_extractor
|
||||
>
|
||||
class random_forest_regression_trainer
|
||||
{
|
||||
public:
|
||||
typedef feature_extractor feature_extractor_type;
|
||||
typedef random_forest_regression_function<feature_extractor> trained_function_type;
|
||||
typedef typename feature_extractor::sample_type sample_type;
|
||||
|
||||
|
||||
random_forest_regression_trainer (
|
||||
) = default;
|
||||
|
||||
const feature_extractor_type& get_feature_extractor (
|
||||
) const
|
||||
{
|
||||
return fe_;
|
||||
}
|
||||
|
||||
void set_feature_extractor (
|
||||
const feature_extractor_type& feat_extractor
|
||||
)
|
||||
{
|
||||
fe_ = feat_extractor;
|
||||
}
|
||||
|
||||
void set_seed (
|
||||
const std::string& seed
|
||||
)
|
||||
{
|
||||
random_seed = seed;
|
||||
}
|
||||
|
||||
const std::string& get_random_seed (
|
||||
) const
|
||||
{
|
||||
return random_seed;
|
||||
}
|
||||
|
||||
size_t get_num_trees (
|
||||
) const
|
||||
{
|
||||
return num_trees;
|
||||
}
|
||||
|
||||
void set_num_trees (
|
||||
size_t num
|
||||
)
|
||||
{
|
||||
DLIB_CASSERT(num > 0);
|
||||
num_trees = num;
|
||||
}
|
||||
|
||||
void set_feature_subsampling_fraction (
|
||||
double frac
|
||||
)
|
||||
{
|
||||
DLIB_CASSERT(0 < frac && frac <= 1);
|
||||
feature_subsampling_frac = frac;
|
||||
}
|
||||
|
||||
double get_feature_subsampling_frac(
|
||||
) const
|
||||
{
|
||||
return feature_subsampling_frac;
|
||||
}
|
||||
|
||||
void set_min_samples_per_leaf (
|
||||
size_t num
|
||||
)
|
||||
{
|
||||
DLIB_ASSERT(num > 0);
|
||||
min_samples_per_leaf = num;
|
||||
}
|
||||
|
||||
size_t get_min_samples_per_leaf(
|
||||
) const
|
||||
{
|
||||
return min_samples_per_leaf;
|
||||
}
|
||||
|
||||
void be_verbose (
|
||||
)
|
||||
{
|
||||
verbose = true;
|
||||
}
|
||||
|
||||
void be_quiet (
|
||||
)
|
||||
{
|
||||
verbose = false;
|
||||
}
|
||||
|
||||
trained_function_type train (
|
||||
const std::vector<sample_type>& x,
|
||||
const std::vector<double>& y
|
||||
) const
|
||||
{
|
||||
std::vector<double> junk;
|
||||
return do_train(x,y,junk,false);
|
||||
}
|
||||
|
||||
trained_function_type train (
|
||||
const std::vector<sample_type>& x,
|
||||
const std::vector<double>& y,
|
||||
std::vector<double>& oob_values
|
||||
) const
|
||||
{
|
||||
return do_train(x,y,oob_values,true);
|
||||
}
|
||||
|
||||
private:
|
||||
|
||||
trained_function_type do_train (
|
||||
const std::vector<sample_type>& x,
|
||||
const std::vector<double>& y,
|
||||
std::vector<double>& oob_values,
|
||||
bool compute_oob_values
|
||||
) const
|
||||
{
|
||||
DLIB_CASSERT(x.size() == y.size());
|
||||
DLIB_CASSERT(x.size() > 0);
|
||||
|
||||
feature_extractor_type fe = fe_;
|
||||
fe.setup(x,y);
|
||||
|
||||
DLIB_CASSERT(fe.max_num_feats() != 0);
|
||||
|
||||
std::vector<std::vector<internal_tree_node<feature_extractor>>> all_trees(num_trees);
|
||||
std::vector<std::vector<float>> all_leaves(num_trees);
|
||||
|
||||
const double sumy = sum(mat(y));
|
||||
|
||||
const size_t feats_per_node = std::max(1.0,std::round(fe.max_num_feats()*feature_subsampling_frac));
|
||||
|
||||
// Each tree couldn't have more than this many interior nodes. It might
|
||||
// end up having less though. We need to know this value because the way
|
||||
// we mark a left or right pointer in a tree as pointing to a leaf is by
|
||||
// making its index larger than the number of interior nodes in the tree.
|
||||
// But we don't know the tree's size before we finish building it. So we
|
||||
// will use max_num_nodes as a proxy during tree construction and then go
|
||||
// back and fix it once a tree's size is known.
|
||||
const uint32_t max_num_nodes = y.size();
|
||||
|
||||
std::vector<uint32_t> oob_hits;
|
||||
|
||||
if (compute_oob_values)
|
||||
{
|
||||
oob_values.resize(y.size());
|
||||
oob_hits.resize(y.size());
|
||||
}
|
||||
|
||||
|
||||
std::mutex m;
|
||||
|
||||
// Calling build_tree(i) creates the ith tree and stores the results in
|
||||
// all_trees and all_leaves.
|
||||
auto build_tree = [&](long i)
|
||||
{
|
||||
dlib::rand rnd(random_seed + std::to_string(i));
|
||||
auto& tree = all_trees[i];
|
||||
auto& leaves = all_leaves[i];
|
||||
|
||||
// Check if there are fewer than min_samples_per_leaf and if so then
|
||||
// don't make any tree. Just average the things and be done.
|
||||
if (y.size() <= min_samples_per_leaf)
|
||||
{
|
||||
leaves.push_back(sumy/y.size());
|
||||
return;
|
||||
}
|
||||
|
||||
|
||||
// pick a random bootstrap of the data.
|
||||
std::vector<std::pair<float,uint32_t>> idxs(y.size());
|
||||
for (auto& idx : idxs)
|
||||
idx = std::make_pair(0,rnd.get_integer(y.size()));
|
||||
|
||||
// We are going to use ranges_to_process as a stack that tracks which
|
||||
// range of samples we are going to split next.
|
||||
std::vector<range_t> ranges_to_process;
|
||||
// start with the root of the tree, i.e. the entire range of training
|
||||
// samples.
|
||||
ranges_to_process.emplace_back(sumy,0,y.size());
|
||||
// push an unpopulated root node into the tree. We will populate it
|
||||
// when we process its corresponding range.
|
||||
tree.emplace_back();
|
||||
|
||||
std::vector<typename feature_extractor::feature> feats;
|
||||
|
||||
while(ranges_to_process.size() > 0)
|
||||
{
|
||||
// Grab the next range/node to process.
|
||||
const auto range = ranges_to_process.back();
|
||||
ranges_to_process.pop_back();
|
||||
|
||||
|
||||
// Get the split features we will consider at this node.
|
||||
fe.get_random_features(rnd, feats_per_node, feats);
|
||||
// Then find the best split
|
||||
auto best_split = find_best_split_among_feats(fe, range, feats, x, y, idxs);
|
||||
|
||||
range_t left_split(best_split.left_sum, range.begin, best_split.split_idx);
|
||||
range_t right_split(best_split.right_sum, best_split.split_idx, range.end);
|
||||
|
||||
DLIB_ASSERT(left_split.begin < left_split.end);
|
||||
DLIB_ASSERT(right_split.begin < right_split.end);
|
||||
|
||||
// Now that we know the split we can populate the parent node we popped
|
||||
// from ranges_to_process.
|
||||
tree[range.tree_idx].split_threshold = best_split.split_threshold;
|
||||
tree[range.tree_idx].split_feature = best_split.split_feature;
|
||||
|
||||
// If the left split is big enough to make a new interior leaf
|
||||
// node. We also stop splitting if all the samples went into this node.
|
||||
// This could happen if the features are all uniform so there just
|
||||
// isn't any way to split them anymore.
|
||||
if (left_split.size() > min_samples_per_leaf && right_split.size() != 0)
|
||||
{
|
||||
// allocate an interior leaf node for it.
|
||||
left_split.tree_idx = tree.size();
|
||||
tree.emplace_back();
|
||||
// set the pointer in the parent node to the newly allocated
|
||||
// node.
|
||||
tree[range.tree_idx].left = left_split.tree_idx;
|
||||
|
||||
ranges_to_process.emplace_back(left_split);
|
||||
}
|
||||
else
|
||||
{
|
||||
// Add to leaves. Don't forget to set the pointer in the
|
||||
// parent node to the newly allocated leaf node.
|
||||
tree[range.tree_idx].left = leaves.size() + max_num_nodes;
|
||||
leaves.emplace_back(left_split.avg());
|
||||
}
|
||||
|
||||
|
||||
// If the right split is big enough to make a new interior leaf
|
||||
// node. We also stop splitting if all the samples went into this node.
|
||||
// This could happen if the features are all uniform so there just
|
||||
// isn't any way to split them anymore.
|
||||
if (right_split.size() > min_samples_per_leaf && left_split.size() != 0)
|
||||
{
|
||||
// allocate an interior leaf node for it.
|
||||
right_split.tree_idx = tree.size();
|
||||
tree.emplace_back();
|
||||
// set the pointer in the parent node to the newly allocated
|
||||
// node.
|
||||
tree[range.tree_idx].right = right_split.tree_idx;
|
||||
|
||||
ranges_to_process.emplace_back(right_split);
|
||||
}
|
||||
else
|
||||
{
|
||||
// Add to leaves. Don't forget to set the pointer in the
|
||||
// parent node to the newly allocated leaf node.
|
||||
tree[range.tree_idx].right = leaves.size() + max_num_nodes;
|
||||
leaves.emplace_back(right_split.avg());
|
||||
}
|
||||
} // end while (still building tree)
|
||||
|
||||
// Fix the leaf pointers in the tree now that we know the correct
|
||||
// tree.size() value.
|
||||
DLIB_CASSERT(max_num_nodes >= tree.size());
|
||||
const auto offset = max_num_nodes - tree.size();
|
||||
for (auto& n : tree)
|
||||
{
|
||||
if (n.left >= max_num_nodes)
|
||||
n.left -= offset;
|
||||
if (n.right >= max_num_nodes)
|
||||
n.right -= offset;
|
||||
}
|
||||
|
||||
|
||||
if (compute_oob_values)
|
||||
{
|
||||
std::sort(idxs.begin(), idxs.end(),
|
||||
[](const std::pair<float,uint32_t>& a, const std::pair<float,uint32_t>& b) {return a.second<b.second; });
|
||||
|
||||
std::lock_guard<std::mutex> lock(m);
|
||||
|
||||
size_t j = 0;
|
||||
for (size_t i = 0; i < oob_values.size(); ++i)
|
||||
{
|
||||
// check if i is in idxs
|
||||
while(j < idxs.size() && i > idxs[j].second)
|
||||
++j;
|
||||
|
||||
// i isn't in idxs so it's an oob sample and we should process it.
|
||||
if (j == idxs.size() || idxs[j].second != i)
|
||||
{
|
||||
oob_hits[i]++;
|
||||
|
||||
// walk the tree to find the leaf value for this oob sample
|
||||
uint32_t idx = 0;
|
||||
while(idx < tree.size())
|
||||
{
|
||||
auto feature_value = fe.extract_feature_value(x[i], tree[idx].split_feature);
|
||||
if (feature_value < tree[idx].split_threshold)
|
||||
idx = tree[idx].left;
|
||||
else
|
||||
idx = tree[idx].right;
|
||||
}
|
||||
oob_values[i] += leaves[idx-tree.size()];
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if (verbose)
|
||||
parallel_for_verbose(0, num_trees, build_tree);
|
||||
else
|
||||
parallel_for(0, num_trees, build_tree);
|
||||
|
||||
|
||||
if (compute_oob_values)
|
||||
{
|
||||
double meanval = 0;
|
||||
double cnt = 0;
|
||||
for (size_t i = 0; i < oob_values.size(); ++i)
|
||||
{
|
||||
if (oob_hits[i] != 0)
|
||||
{
|
||||
oob_values[i] /= oob_hits[i];
|
||||
meanval += oob_values[i];
|
||||
++cnt;
|
||||
}
|
||||
}
|
||||
|
||||
// If there are some elements that didn't get hits, we set their oob values
|
||||
// to the mean oob value.
|
||||
if (cnt != 0)
|
||||
{
|
||||
const double typical_value = meanval/cnt;
|
||||
for (size_t i = 0; i < oob_values.size(); ++i)
|
||||
{
|
||||
if (oob_hits[i] == 0)
|
||||
oob_values[i] = typical_value;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return trained_function_type(std::move(fe), std::move(all_trees), std::move(all_leaves));
|
||||
}
|
||||
|
||||
struct range_t
|
||||
{
|
||||
range_t(
|
||||
double sumy,
|
||||
uint32_t begin,
|
||||
uint32_t end
|
||||
) : sumy(sumy), begin(begin), end(end), tree_idx(0) {}
|
||||
|
||||
double sumy;
|
||||
uint32_t begin;
|
||||
uint32_t end;
|
||||
|
||||
// Every range object corresponds to an entry in a tree. This tells you the
|
||||
// tree node that owns the range.
|
||||
uint32_t tree_idx;
|
||||
|
||||
uint32_t size() const { return end-begin; }
|
||||
double avg() const { return sumy/size(); }
|
||||
};
|
||||
|
||||
struct best_split_details
|
||||
{
|
||||
double score = -std::numeric_limits<double>::infinity();
|
||||
double left_sum;
|
||||
double right_sum;
|
||||
uint32_t split_idx;
|
||||
double split_threshold;
|
||||
typename feature_extractor::feature split_feature;
|
||||
|
||||
bool operator < (const best_split_details& rhs) const
|
||||
{
|
||||
return score < rhs.score;
|
||||
}
|
||||
};
|
||||
|
||||
static best_split_details find_best_split (
|
||||
const range_t& range,
|
||||
const std::vector<double>& y,
|
||||
const std::vector<std::pair<float,uint32_t>>& idxs
|
||||
)
|
||||
/*!
|
||||
requires
|
||||
- max(mat(idxs)) < y.size()
|
||||
- range.sumy == sum of y[idxs[j].second] for all valid j in range [range.begin, range.end).
|
||||
ensures
|
||||
- finds a threshold T such that there exists an i satisfying the following:
|
||||
- y[idxs[j].second] < T for all j <= i
|
||||
- y[idxs[j].second] > T for all j > i
|
||||
Therefore, the threshold T partitions the contents of y into two groups,
|
||||
relative to the ordering established by idxs. Moreover the partitioning
|
||||
of y values into two groups has the additional requirement that it is
|
||||
optimal in the sense that the sum of the squared deviations from each
|
||||
partition's mean is minimized.
|
||||
!*/
|
||||
{
|
||||
|
||||
size_t best_i = range.begin;
|
||||
double best_score = -1;
|
||||
double left_sum = 0;
|
||||
double best_left_sum = y[idxs[range.begin].second];
|
||||
const auto size = range.size();
|
||||
size_t left_size = 0;
|
||||
for (size_t i = range.begin; i+1 < range.end; ++i)
|
||||
{
|
||||
++left_size;
|
||||
left_sum += y[idxs[i].second];
|
||||
|
||||
// Don't split here because the next element has the same feature value so
|
||||
// we can't *really* split here.
|
||||
if (idxs[i].first==idxs[i+1].first)
|
||||
continue;
|
||||
|
||||
const double right_sum = range.sumy-left_sum;
|
||||
|
||||
const double score = left_sum*left_sum/left_size + right_sum*right_sum/(size-left_size);
|
||||
|
||||
if (score > best_score)
|
||||
{
|
||||
best_score = score;
|
||||
best_i = i;
|
||||
best_left_sum = left_sum;
|
||||
}
|
||||
}
|
||||
|
||||
best_split_details result;
|
||||
result.score = best_score;
|
||||
result.left_sum = best_left_sum;
|
||||
result.right_sum = range.sumy-best_left_sum;
|
||||
result.split_idx = best_i+1; // one past the end of the left range
|
||||
result.split_threshold = (idxs[best_i].first+idxs[best_i+1].first)/2;
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
|
||||
static best_split_details find_best_split_among_feats(
|
||||
const feature_extractor& fe,
|
||||
const range_t& range,
|
||||
const std::vector<typename feature_extractor::feature>& feats,
|
||||
const std::vector<sample_type>& x,
|
||||
const std::vector<double>& y,
|
||||
std::vector<std::pair<float,uint32_t>>& idxs
|
||||
)
|
||||
{
|
||||
auto compare_first = [](const std::pair<float,uint32_t>& a, const std::pair<float,uint32_t>& b) { return a.first<b.first; };
|
||||
best_split_details best;
|
||||
for (auto& feat : feats)
|
||||
{
|
||||
// Extract feature values for this feature and sort the indexes based on
|
||||
// that feature so we can then find the best split.
|
||||
for (auto i = range.begin; i < range.end; ++i)
|
||||
idxs[i].first = fe.extract_feature_value(x[idxs[i].second], feat);
|
||||
|
||||
std::sort(idxs.begin()+range.begin, idxs.begin()+range.end, compare_first);
|
||||
|
||||
auto split = find_best_split(range, y, idxs);
|
||||
|
||||
if (best < split)
|
||||
{
|
||||
best = split;
|
||||
best.split_feature = feat;
|
||||
}
|
||||
}
|
||||
|
||||
// resort idxs based on winning feat
|
||||
for (auto i = range.begin; i < range.end; ++i)
|
||||
idxs[i].first = fe.extract_feature_value(x[idxs[i].second], best.split_feature);
|
||||
std::sort(idxs.begin()+range.begin, idxs.begin()+range.end, compare_first);
|
||||
|
||||
return best;
|
||||
}
|
||||
|
||||
std::string random_seed;
|
||||
size_t num_trees = 1000;
|
||||
double feature_subsampling_frac = 1.0/3.0;
|
||||
size_t min_samples_per_leaf = 5;
|
||||
feature_extractor_type fe_;
|
||||
bool verbose = false;
|
||||
};
|
||||
|
||||
// ----------------------------------------------------------------------------------------
|
||||
|
||||
}
|
||||
|
||||
#endif // DLIB_RANdOM_FOREST_REGRESSION_H_
|
||||
|
||||
|
||||
460
lib/3rdParty/dlib/include/dlib/random_forest/random_forest_regression_abstract.h
vendored
Normal file
460
lib/3rdParty/dlib/include/dlib/random_forest/random_forest_regression_abstract.h
vendored
Normal file
@@ -0,0 +1,460 @@
|
||||
// Copyright (C) 2018 Davis E. King (davis@dlib.net)
|
||||
// License: Boost Software License See LICENSE.txt for the full license.
|
||||
#undef DLIB_RANdOM_FOREST_REGRESION_ABSTRACT_H_
|
||||
#ifdef DLIB_RANdOM_FOREST_REGRESION_ABSTRACT_H_
|
||||
|
||||
#include <vector>
|
||||
#include "../matrix.h"
|
||||
|
||||
namespace dlib
|
||||
{
|
||||
|
||||
// ----------------------------------------------------------------------------------------
|
||||
|
||||
class dense_feature_extractor
|
||||
{
|
||||
/*!
|
||||
WHAT THIS OBJECT REPRESENTS
|
||||
This object is a tool for extracting features from objects. In particular,
|
||||
it is designed to be used with the random forest regression tools discussed
|
||||
below.
|
||||
|
||||
This particular feature extract does almost nothing since it works on
|
||||
vectors in R^n and simply selects elements from each vector. However, the
|
||||
tools below are templated and allow you to design your own feature extractors
|
||||
that operate on whatever object types you create. So for example, maybe
|
||||
you want to perform regression on images rather than vectors. Moreover,
|
||||
your feature extraction could be more complex. Maybe you are selecting
|
||||
differences between pairs of pixels in an image or doing something
|
||||
involving geometric transforms during feature extraction. Any of these
|
||||
kinds of more complex feature extraction patterns can be realized with the
|
||||
random forest tools by implementing your own feature extractor object and
|
||||
using it with the random forest objects.
|
||||
|
||||
Therefore, you should consider this dense_feature_extractor as an example
|
||||
that documents the interface as well as the simple default extractor for
|
||||
use with dense vectors.
|
||||
|
||||
|
||||
THREAD SAFETY
|
||||
It is safe to call const members of this object from multiple threads. ANY
|
||||
USER DEFINED FEATURE EXTRACTORS MUST ALSO MEET THIS GUARONTEE AS WELL SINCE
|
||||
IT IS ASSUMED BY THE RANDOM FOREST TRAINING ROUTINES.
|
||||
!*/
|
||||
|
||||
public:
|
||||
typedef uint32_t feature;
|
||||
typedef matrix<double,0,1> sample_type;
|
||||
|
||||
dense_feature_extractor(
|
||||
);
|
||||
/*!
|
||||
ensures
|
||||
- #max_num_feats() == 0
|
||||
!*/
|
||||
|
||||
void setup (
|
||||
const std::vector<sample_type>& x,
|
||||
const std::vector<double>& y
|
||||
);
|
||||
/*!
|
||||
requires
|
||||
- x.size() == y.size()
|
||||
- x.size() > 0
|
||||
- x[0].size() > 0
|
||||
- all the vectors in x have the same dimensionality.
|
||||
ensures
|
||||
- Configures this feature extractor to work on the given training data.
|
||||
For dense feature extractors all we do is record the dimensionality of
|
||||
the training vectors.
|
||||
- #max_num_feats() == x[0].size()
|
||||
(In general, setup() sets max_num_feats() to some non-zero value so that
|
||||
the other methods of this object can then be called. The point of setup()
|
||||
is to allow a feature extractor to gather whatever statistics it needs from
|
||||
training data. That is, more complex feature extraction strategies my
|
||||
themselves be trained from data.)
|
||||
!*/
|
||||
|
||||
void get_random_features (
|
||||
dlib::rand& rnd,
|
||||
size_t num,
|
||||
std::vector<feature>& feats
|
||||
) const;
|
||||
/*!
|
||||
requires
|
||||
- max_num_feats() != 0
|
||||
ensures
|
||||
- #feats.size() == min(num, max_num_feats())
|
||||
- This function randomly identifies num features and stores them into feats.
|
||||
These feature objects can then be used with extract_feature_value() to
|
||||
obtain a value from any particular sample_type object. This value is the
|
||||
"feature value" used by a decision tree algorithm to deice how to split
|
||||
and traverse trees.
|
||||
- The above two conditions define the behavior of get_random_features() in
|
||||
general. For this specific implementation of the feature extraction interface
|
||||
this function selects num integer values from the range [0, max_num_feats()),
|
||||
without replacement. These values are stored into feats.
|
||||
!*/
|
||||
|
||||
double extract_feature_value (
|
||||
const sample_type& item,
|
||||
const feature& f
|
||||
) const;
|
||||
/*!
|
||||
requires
|
||||
- #max_num_feats() != 0
|
||||
- f was produced from a call to get_random_features().
|
||||
ensures
|
||||
- Extracts the feature value corresponding to f. For this simple dense
|
||||
feature extractor this simply means returning item(f). But in general
|
||||
you can design feature extractors that do something more complex.
|
||||
!*/
|
||||
|
||||
size_t max_num_feats (
|
||||
) const;
|
||||
/*!
|
||||
ensures
|
||||
- returns the number of distinct features this object might extract. That is,
|
||||
a feature extractor essentially defines a mapping from sample_type objects to
|
||||
vectors in R^max_num_feats().
|
||||
!*/
|
||||
};
|
||||
|
||||
void serialize(const dense_feature_extractor& item, std::ostream& out);
|
||||
void deserialize(dense_feature_extractor& item, std::istream& in);
|
||||
/*!
|
||||
provides serialization support
|
||||
!*/
|
||||
|
||||
// ----------------------------------------------------------------------------------------
|
||||
|
||||
template <
|
||||
typename feature_extractor
|
||||
>
|
||||
struct internal_tree_node
|
||||
{
|
||||
/*!
|
||||
WHAT THIS OBJECT REPRESENTS
|
||||
This object is an internal node in a regression tree. See the code of
|
||||
random_forest_regression_function to see how it is used to create a tree.
|
||||
!*/
|
||||
|
||||
uint32_t left;
|
||||
uint32_t right;
|
||||
float split_threshold;
|
||||
typename feature_extractor::feature split_feature;
|
||||
};
|
||||
|
||||
template <typename feature_extractor>
|
||||
void serialize(const internal_tree_node<feature_extractor>& item, std::ostream& out);
|
||||
template <typename feature_extractor>
|
||||
void deserialize(internal_tree_node<feature_extractor>& item, std::istream& in);
|
||||
/*!
|
||||
provides serialization support
|
||||
!*/
|
||||
|
||||
// ----------------------------------------------------------------------------------------
|
||||
|
||||
template <
|
||||
typename feature_extractor = dense_feature_extractor
|
||||
>
|
||||
class random_forest_regression_function
|
||||
{
|
||||
/*!
|
||||
REQUIREMENTS ON feature_extractor
|
||||
feature_extractor must be dense_feature_extractor or a type with a
|
||||
compatible interface.
|
||||
|
||||
WHAT THIS OBJECT REPRESENTS
|
||||
This object represents a regression forest. This is a collection of
|
||||
decision trees that take an object as input and each vote on a real value
|
||||
to associate with the object. The final real value output is the average
|
||||
of all the votes from each of the trees.
|
||||
!*/
|
||||
|
||||
public:
|
||||
|
||||
typedef feature_extractor feature_extractor_type;
|
||||
typedef typename feature_extractor::sample_type sample_type;
|
||||
|
||||
random_forest_regression_function(
|
||||
);
|
||||
/*!
|
||||
ensures
|
||||
- #num_trees() == 0
|
||||
!*/
|
||||
|
||||
random_forest_regression_function (
|
||||
feature_extractor_type&& fe_,
|
||||
std::vector<std::vector<internal_tree_node<feature_extractor>>>&& trees_,
|
||||
std::vector<std::vector<float>>&& leaves_
|
||||
);
|
||||
/*!
|
||||
requires
|
||||
- trees.size() > 0
|
||||
- trees.size() = leaves.size()
|
||||
- for all valid i:
|
||||
- leaves[i].size() > 0
|
||||
- trees[i].size()+leaves[i].size() > the maximal left or right index values in trees[i].
|
||||
(i.e. each left or right value must index to some existing internal tree node or leaf node).
|
||||
ensures
|
||||
- #get_internal_tree_nodes() == trees_
|
||||
- #get_tree_leaves() == leaves_
|
||||
- #get_feature_extractor() == fe_
|
||||
!*/
|
||||
|
||||
size_t get_num_trees(
|
||||
) const;
|
||||
/*!
|
||||
ensures
|
||||
- returns the number of trees in this regression forest.
|
||||
!*/
|
||||
|
||||
const std::vector<std::vector<internal_tree_node<feature_extractor>>>& get_internal_tree_nodes (
|
||||
) const;
|
||||
/*!
|
||||
ensures
|
||||
- returns the internal tree nodes that define the regression trees.
|
||||
- get_internal_tree_nodes().size() == get_num_trees()
|
||||
!*/
|
||||
|
||||
const std::vector<std::vector<float>>& get_tree_leaves (
|
||||
) const;
|
||||
/*!
|
||||
ensures
|
||||
- returns the tree leaves that define the regression trees.
|
||||
- get_tree_leaves().size() == get_num_trees()
|
||||
!*/
|
||||
|
||||
const feature_extractor_type& get_feature_extractor (
|
||||
) const;
|
||||
/*!
|
||||
ensures
|
||||
- returns the feature extractor used by the trees.
|
||||
!*/
|
||||
|
||||
double operator() (
|
||||
const sample_type& x
|
||||
) const;
|
||||
/*!
|
||||
requires
|
||||
- get_num_trees() > 0
|
||||
ensures
|
||||
- Maps x to a real value and returns the value. To do this, we find the
|
||||
get_num_trees() leaf values associated with x and then return the average
|
||||
of these leaf values.
|
||||
!*/
|
||||
};
|
||||
|
||||
void serialize(const random_forest_regression_function& item, std::ostream& out);
|
||||
void deserialize(random_forest_regression_function& item, std::istream& in);
|
||||
/*!
|
||||
provides serialization support
|
||||
!*/
|
||||
|
||||
// ----------------------------------------------------------------------------------------
|
||||
|
||||
template <
|
||||
typename feature_extractor = dense_feature_extractor
|
||||
>
|
||||
class random_forest_regression_trainer
|
||||
{
|
||||
/*!
|
||||
REQUIREMENTS ON feature_extractor
|
||||
feature_extractor must be dense_feature_extractor or a type with a
|
||||
compatible interface.
|
||||
|
||||
WHAT THIS OBJECT REPRESENTS
|
||||
This object implements Breiman's classic random forest regression
|
||||
algorithm. The algorithm learns to map objects, nominally vectors in R^n,
|
||||
into the reals. It essentially optimizes the mean squared error by fitting
|
||||
a bunch of decision trees, each of which vote on the output value of the
|
||||
regressor. The final prediction is obtained by averaging all the
|
||||
predictions.
|
||||
|
||||
For more information on the algorithm see:
|
||||
Breiman, Leo. "Random forests." Machine learning 45.1 (2001): 5-32.
|
||||
!*/
|
||||
|
||||
public:
|
||||
typedef feature_extractor feature_extractor_type;
|
||||
typedef random_forest_regression_function<feature_extractor> trained_function_type;
|
||||
typedef typename feature_extractor::sample_type sample_type;
|
||||
|
||||
|
||||
random_forest_regression_trainer (
|
||||
);
|
||||
/*!
|
||||
ensures
|
||||
- #get_min_samples_per_leaf() == 5
|
||||
- #get_num_trees() == 1000
|
||||
- #get_feature_subsampling_frac() == 1.0/3.0
|
||||
- #get_feature_extractor() == a default initialized feature extractor.
|
||||
- #get_random_seed() == ""
|
||||
- this object is not verbose.
|
||||
!*/
|
||||
|
||||
const feature_extractor_type& get_feature_extractor (
|
||||
) const;
|
||||
/*!
|
||||
ensures
|
||||
- returns the feature extractor used when train() is invoked.
|
||||
!*/
|
||||
|
||||
void set_feature_extractor (
|
||||
const feature_extractor_type& feat_extractor
|
||||
);
|
||||
/*!
|
||||
ensures
|
||||
- #get_feature_extractor() == feat_extractor
|
||||
!*/
|
||||
|
||||
void set_seed (
|
||||
const std::string& seed
|
||||
);
|
||||
/*!
|
||||
ensures
|
||||
- #get_random_seed() == seed
|
||||
!*/
|
||||
|
||||
const std::string& get_random_seed (
|
||||
) const;
|
||||
/*!
|
||||
ensures
|
||||
- A central part of this algorithm is random selection of both training
|
||||
samples and features. This function returns the seed used to initialized
|
||||
the random number generator used for these random selections.
|
||||
!*/
|
||||
|
||||
size_t get_num_trees (
|
||||
) const;
|
||||
/*!
|
||||
ensures
|
||||
- Random forests built by this object will contain get_num_trees() trees.
|
||||
!*/
|
||||
|
||||
void set_num_trees (
|
||||
size_t num
|
||||
);
|
||||
/*!
|
||||
requires
|
||||
- num > 0
|
||||
ensures
|
||||
- #get_num_trees() == num
|
||||
!*/
|
||||
|
||||
void set_feature_subsampling_fraction (
|
||||
double frac
|
||||
);
|
||||
/*!
|
||||
requires
|
||||
- 0 < frac <= 1
|
||||
ensures
|
||||
- #get_feature_subsampling_frac() == frac
|
||||
!*/
|
||||
|
||||
double get_feature_subsampling_frac(
|
||||
) const;
|
||||
/*!
|
||||
ensures
|
||||
- When we build trees, at each node we don't look at all the available
|
||||
features. We consider only get_feature_subsampling_frac() fraction of
|
||||
them, selected at random.
|
||||
!*/
|
||||
|
||||
void set_min_samples_per_leaf (
|
||||
size_t num
|
||||
);
|
||||
/*!
|
||||
requires
|
||||
- num > 0
|
||||
ensures
|
||||
- #get_min_samples_per_leaf() == num
|
||||
!*/
|
||||
|
||||
size_t get_min_samples_per_leaf(
|
||||
) const;
|
||||
/*!
|
||||
ensures
|
||||
- When building trees, each leaf node in a tree will contain at least
|
||||
get_min_samples_per_leaf() samples. This means that the output votes of
|
||||
each tree are averages of at least get_min_samples_per_leaf() y values.
|
||||
!*/
|
||||
|
||||
void be_verbose (
|
||||
);
|
||||
/*!
|
||||
ensures
|
||||
- This object will print status messages to standard out so that the
|
||||
progress of training can be tracked..
|
||||
!*/
|
||||
|
||||
void be_quiet (
|
||||
);
|
||||
/*!
|
||||
ensures
|
||||
- this object will not print anything to standard out
|
||||
!*/
|
||||
|
||||
random_forest_regression_function<feature_extractor> train (
|
||||
const std::vector<sample_type>& x,
|
||||
const std::vector<double>& y,
|
||||
std::vector<double>& oob_values
|
||||
) const;
|
||||
/*!
|
||||
requires
|
||||
- x.size() == y.size()
|
||||
- x.size() > 0
|
||||
- Running following code:
|
||||
auto fe = get_feature_extractor()
|
||||
fe.setup(x,y);
|
||||
Must be valid and result in fe.max_num_feats() != 0
|
||||
ensures
|
||||
- This function fits a regression forest to the given training data. The
|
||||
goal being to regress x to y in the mean squared sense. It therefore
|
||||
fits regression trees and returns the resulting random_forest_regression_function
|
||||
RF, which will have the following properties:
|
||||
- RF.get_num_trees() == get_num_trees()
|
||||
- for all valid i:
|
||||
- RF(x[i]) should output a value close to y[i]
|
||||
- RF.get_feature_extractor() will be a copy of this->get_feature_extractor()
|
||||
that has been configured by a call the feature extractor's setup() routine.
|
||||
To run the algorithm we need to use a feature extractor. We obtain a
|
||||
valid feature extractor by making a copy of get_feature_extractor(), then
|
||||
invoking setup(x,y) on it. This feature extractor is what is used to fit
|
||||
the trees and is also the feature extractor stored in the returned random
|
||||
forest.
|
||||
- #oob_values.size() == y.size()
|
||||
- for all valid i:
|
||||
- #oob_values[i] == the "out of bag" prediction for y[i]. It is
|
||||
calculated by computing the average output from trees not trained on
|
||||
y[i]. This is similar to a leave-one-out cross-validation prediction
|
||||
of y[i] and can be used to estimate the generalization error of the
|
||||
regression forest.
|
||||
- Training uses all the available CPU cores.
|
||||
!*/
|
||||
|
||||
random_forest_regression_function<feature_extractor> train (
|
||||
const std::vector<sample_type>& x,
|
||||
const std::vector<double>& y
|
||||
) const;
|
||||
/*!
|
||||
requires
|
||||
- x.size() == y.size()
|
||||
- x.size() > 0
|
||||
- Running following code:
|
||||
auto fe = get_feature_extractor()
|
||||
fe.setup(x,y);
|
||||
Must be valid and result in fe.max_num_feats() != 0
|
||||
ensures
|
||||
- This function is identical to train(x,y,oob_values) except that the
|
||||
oob_values are not calculated.
|
||||
!*/
|
||||
};
|
||||
|
||||
// ----------------------------------------------------------------------------------------
|
||||
|
||||
}
|
||||
|
||||
#endif // DLIB_RANdOM_FOREST_REGRESION_ABSTRACT_H_
|
||||
|
||||
Reference in New Issue
Block a user