Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion include/maths/analytics/CTreeShapFeatureImportance.h
Original file line number Diff line number Diff line change
Expand Up @@ -73,7 +73,10 @@ class MATHS_ANALYTICS_EXPORT CTreeShapFeatureImportance {

//! Compute inner node values as weighted average of the children (leaf) values.
//!
//! The weights are the number of rows of \p frame reaching each node.
//! The weights are the number of rows of \p frame reaching each node. A node
//! below the root which no rows reach uses equal weights instead, matching the
//! even split used when computing SHAP values. If no rows reach the root its
//! value is NaN.
static void computeInternalNodeValues(TTreeVec& forest);

//! Get the maximum depth of any tree in \p forest.
Expand Down
14 changes: 12 additions & 2 deletions lib/maths/analytics/CBoostedTreeImpl.cc
Original file line number Diff line number Diff line change
Expand Up @@ -1517,9 +1517,16 @@ void CBoostedTreeImpl::initializeFixedCandidateSplits(core::CDataFrame& frame) {
set.reserve(m_NumberSplitsPerFeature + 2);
}
for (auto row = beginRows; row != endRows; ++row) {
auto encodedRow = m_Encoder->encode(*row);
for (std::size_t i = 0; i < features.size(); ++i) {
if (state[i].size() <= m_NumberSplitsPerFeature + 1) {
state[i].insert(m_Encoder->encode(*row)[features[i]]);
// Missing values are NaN, which never compares equal to
// itself, so each one would otherwise count as a distinct
// value and corrupt the sorted candidate splits.
double value{encodedRow[features[i]]};
if (core::CDataFrame::isMissing(value) == false) {
state[i].insert(value);
}
}
}
}
Expand All @@ -1538,7 +1545,10 @@ void CBoostedTreeImpl::initializeFixedCandidateSplits(core::CDataFrame& frame) {

TDoubleVec values;
for (std::size_t i = 0; i < features.size(); ++i) {
if (uniques[i].size() <= m_NumberSplitsPerFeature) {
// We need at least two distinct values to split. Since missing values are
// skipped there may be none, for example if the encoding was computed on a
// different data set.
if (uniques[i].size() > 1 && uniques[i].size() <= m_NumberSplitsPerFeature) {
values.assign(uniques[i].begin(), uniques[i].end());
std::sort(values.begin(), values.end());
auto& featureCandidateSplits = m_FixedCandidateSplits[features[i]];
Expand Down
27 changes: 21 additions & 6 deletions lib/maths/analytics/CTreeShapFeatureImportance.cc
Original file line number Diff line number Diff line change
Expand Up @@ -169,6 +169,12 @@ void CTreeShapFeatureImportance::computeInternalNodeValues(TTree& tree, std::siz
auto& rightChild = tree[node.rightChildIndex()];
double leftWeight{static_cast<double>(leftChild.numberSamples())};
double rightWeight{static_cast<double>(rightChild.numberSamples())};
if (node.numberSamples() == 0 && nodeIndex != 0) {
// No counted rows reach this node so its parent gives it zero weight
// and its value doesn't change the expected value. Use the unweighted
// mean, which matches the even split shapRecursive uses for it.
leftWeight = rightWeight = 1.0;
}
node.value((leftWeight * leftChild.value() + rightWeight * rightChild.value()) /
(leftWeight + rightWeight));
}
Expand Down Expand Up @@ -249,12 +255,21 @@ void CTreeShapFeatureImportance::shapRecursive(const TTree& tree,
unwindPath(splitPath, pathIndex, nextIndex);
}

double hotFractionZero{incomingFractionZero *
static_cast<double>(tree[hotIndex].numberSamples()) /
static_cast<double>(tree[nodeIndex].numberSamples())};
double coldFractionZero{incomingFractionZero *
static_cast<double>(tree[coldIndex].numberSamples()) /
static_cast<double>(tree[nodeIndex].numberSamples())};
// A node below the root which no counted rows reach leaves the child
// fractions as 0 / 0. An even split keeps them summing to one and doesn't
// change the expected value, since the fraction leading into the node is
// zero. It does affect the attributions of any row which some subset of its
// features routes into the node, for which it is the neutral choice. The
// root only has no samples if no rows were counted at all, which mustn't be
// masked. TreeInferenceModel in Elasticsearch must use the same rule.
double numberSamples{static_cast<double>(tree[nodeIndex].numberSamples())};
bool evenSplit{numberSamples == 0.0 && nodeIndex != 0};
double hotFractionZero{
incomingFractionZero *
(evenSplit ? 0.5 : static_cast<double>(tree[hotIndex].numberSamples()) / numberSamples)};
double coldFractionZero{
incomingFractionZero *
(evenSplit ? 0.5 : static_cast<double>(tree[coldIndex].numberSamples()) / numberSamples)};
this->shapRecursive(tree, encodedRow, hotIndex, hotFractionZero, incomingFractionOne,
splitFeature, splitPath, nextIndex, shap);
this->shapRecursive(tree, encodedRow, coldIndex, coldFractionZero, 0.0,
Expand Down
191 changes: 191 additions & 0 deletions lib/maths/analytics/unittest/CBoostedTreeTest.cc
Original file line number Diff line number Diff line change
Expand Up @@ -62,6 +62,7 @@ class CBoostedTreeImplForTest {
using TLossFunctionUPtr = maths::analytics::CBoostedTreeImpl::TLossFunctionUPtr;
using TSizeVec = maths::analytics::CBoostedTreeImpl::TSizeVec;
using TDoubleVec = maths::analytics::CBoostedTreeImpl::TDoubleVec;
using TFloatVecVec = maths::analytics::CBoostedTreeImpl::TFloatVecVec;
using TBoostedTreeUPtr = std::unique_ptr<maths::analytics::CBoostedTree>;

public:
Expand All @@ -79,6 +80,10 @@ class CBoostedTreeImplForTest {
return m_TreeImpl.featureSampleProbabilities();
}

const TFloatVecVec& fixedCandidateSplits() const {
return m_TreeImpl.m_FixedCandidateSplits;
}

void treeFeatureBag(TDoubleVec& probabilities, TSizeVec& treeFeatureBag) const {
m_TreeImpl.treeFeatureBag(probabilities, treeFeatureBag);
}
Expand Down Expand Up @@ -914,6 +919,192 @@ BOOST_AUTO_TEST_CASE(testLowCardinalityFeatures) {
BOOST_TEST_REQUIRE(rSquared > 0.94);
}

BOOST_AUTO_TEST_CASE(testLowCardinalityFeaturesFixedCandidateSplitsWithMissingValues) {

// Test that missing values don't leak into the fixed candidate splits of low
// cardinality features. Missing values are NaN, which is never equal to itself,
// so they must not be counted as extra distinct values.

std::size_t rows{500};
std::size_t cols{6};

test::CRandomNumbers rng;
TDoubleVecVec x(cols - 1);
for (std::size_t i = 0; i < cols - 1; ++i) {
rng.generateUniformSamples(0.0, 10.0, rows, x[i]);
for (auto& xj : x[i]) {
xj = std::floor(xj);
}
}

// Make 2% of each feature's values missing. The target uses the complete values
// so that no training rows are dropped.
TDoubleVecVec xMissing{x};
for (std::size_t i = 0; i < cols - 1; ++i) {
for (std::size_t j = i; j < rows; j += 50) {
xMissing[i][j] = core::CDataFrame::valueOfMissing();
}
}
auto target = [&](const TRowRef& row) {
double result{0.0};
for (std::size_t i = 0; i < cols - 1; ++i) {
result += static_cast<double>(i + 1) * x[i][row.index()];
}
return result;
};

auto frame = core::makeMainStorageDataFrame(cols, rows).first;
fillDataFrame(rows, 0, cols, xMissing, TDoubleVec(rows, 0.0), target, *frame);

auto regression = maths::analytics::CBoostedTreeFactory::constructFromParameters(
1, std::make_unique<maths::analytics::boosted_tree::CMse>())
.buildForTrain(*frame, cols - 1);

// Each feature takes the values 0, 1, ..., 9 so its fixed candidate splits
// should be the midpoints 0.5, 1.5, ..., 8.5.
CBoostedTreeImplForTest treeImpl{regression->impl()};
std::size_t numberFixed{0};
for (const auto& splits : treeImpl.fixedCandidateSplits()) {
if (splits.empty() == false) {
++numberFixed;
BOOST_REQUIRE_EQUAL(std::size_t{9}, splits.size());
for (std::size_t j = 0; j < splits.size(); ++j) {
BOOST_REQUIRE_EQUAL(static_cast<double>(j) + 0.5,
static_cast<double>(splits[j]));
}
}
}
BOOST_REQUIRE_EQUAL(cols - 1, numberFixed);
}

BOOST_AUTO_TEST_CASE(testLowCardinalityFeaturesWithMissingValues) {

// Test training a linear model on low cardinality features with a few missing
// values. Corrupted candidate splits produce splits which no training rows reach
// and degrade the model.

std::size_t trainRows{500};
std::size_t testRows{200};
std::size_t rows{trainRows + testRows};
double noiseVariance{4.0};
std::size_t cols{6};

test::CRandomNumbers rng;
TDoubleVecVec x(cols - 1);
for (std::size_t i = 0; i < cols - 1; ++i) {
rng.generateUniformSamples(0.0, 10.0, rows, x[i]);
for (auto& xj : x[i]) {
xj = std::floor(xj);
}
}

// Make 2% of each feature's values missing. The target uses the complete values
// so that no training rows are dropped.
TDoubleVecVec xMissing{x};
for (std::size_t i = 0; i < cols - 1; ++i) {
for (std::size_t j = i; j < rows; j += 50) {
xMissing[i][j] = core::CDataFrame::valueOfMissing();
}
}
auto target = [&](const TRowRef& row) {
double result{0.0};
for (std::size_t i = 0; i < cols - 1; ++i) {
result += static_cast<double>(i + 1) * x[i][row.index()];
}
return result;
};

TDoubleVec noise;
rng.generateNormalSamples(0.0, noiseVariance, rows, noise);

auto frame = core::makeMainStorageDataFrame(cols, rows).first;
fillDataFrame(trainRows, testRows, cols, xMissing, noise, target, *frame);

auto regression = maths::analytics::CBoostedTreeFactory::constructFromParameters(
1, std::make_unique<maths::analytics::boosted_tree::CMse>())
.buildForTrain(*frame, cols - 1);

regression->train();
regression->predict();

// Every node should be reached by some of the rows its sample counts are
// computed from.
std::size_t numberZeroSampleNodes{0};
for (const auto& tree : regression->trainedModel()) {
for (const auto& node : tree) {
if (node.numberSamples() == 0) {
++numberZeroSampleNodes;
}
}
}
BOOST_REQUIRE_EQUAL(std::size_t{0}, numberZeroSampleNodes);

double bias;
double rSquared;
std::tie(bias, rSquared) = computeEvaluationMetrics(
*frame, trainRows, rows,
[&](const TRowRef& row) { return regression->prediction(row)[0]; },
target, noiseVariance / static_cast<double>(rows));
LOG_DEBUG(<< "bias = " << bias << ", rSquared = " << rSquared);

BOOST_TEST_REQUIRE(rSquared > 0.9);
}

BOOST_AUTO_TEST_CASE(testLowCardinalityFeatureAllMissingAfterSeparateEncoding) {

// Test that we can train when a low cardinality feature selected by encoding
// on one data set has no values in the data set we train on. Skipping missing
// values then leaves no distinct values for its fixed candidate splits.

std::size_t rows{300};
std::size_t cols{4};

test::CRandomNumbers rng;
TDoubleVecVec x(cols - 1);
for (std::size_t i = 0; i < cols - 1; ++i) {
rng.generateUniformSamples(0.0, 10.0, rows, x[i]);
for (auto& xj : x[i]) {
xj = std::floor(xj);
}
}
auto target = [&](const TRowRef& row) {
double result{0.0};
for (std::size_t i = 0; i < cols - 1; ++i) {
result += static_cast<double>(i + 1) * x[i][row.index()];
}
return result;
};

auto encodingFrame = core::makeMainStorageDataFrame(cols, rows).first;
fillDataFrame(rows, 0, cols, x, TDoubleVec(rows, 0.0), target, *encodingFrame);

std::stringstream persistState;
{
auto encoded = maths::analytics::CBoostedTreeFactory::constructFromParameters(
1, std::make_unique<maths::analytics::boosted_tree::CMse>())
.buildForEncode(*encodingFrame, cols - 1);
core::CJsonStatePersistInserter inserter(persistState);
encoded->acceptPersistInserter(inserter);
persistState.flush();
}

TDoubleVecVec xMissing{x};
std::fill(xMissing[0].begin(), xMissing[0].end(), core::CDataFrame::valueOfMissing());
auto trainingFrame = core::makeMainStorageDataFrame(cols, rows).first;
fillDataFrame(rows, 0, cols, xMissing, TDoubleVec(rows, 0.0), target, *trainingFrame);

auto regression = maths::analytics::CBoostedTreeFactory::constructFromString(persistState)
.restoreFor(*trainingFrame, cols - 1);
regression->train();
regression->predict();

trainingFrame->readRows(1, [&](const TRowItr& beginRows, const TRowItr& endRows) {
for (auto row = beginRows; row != endRows; ++row) {
BOOST_REQUIRE(std::isfinite(regression->prediction(*row)[0]));
}
});
}

BOOST_AUTO_TEST_CASE(testLowTrainFractionPerFold) {

// Test regression using a very low train fraction per fold. This should
Expand Down
Loading