Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions pybrush/BrushEstimator.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,7 @@
from sklearn.metrics import average_precision_score, mean_squared_error

from pybrush import Parameters, Dataset, SearchSpace, brush_rng, individual
from pybrush._brush import set_random_state as set_brush_random_state
from pybrush.EstimatorInterface import EstimatorInterface
from pybrush import RegressorEngine, ClassifierEngine, MultiClassifierEngine

Expand Down Expand Up @@ -63,6 +64,12 @@ def fit(self, X, y):
1-d array of (boolean) target values.
"""

# Dataset construction can optionally shuffle a split, so seed the
# native generator before constructing any Brush object, not only when
# Engine::init runs later in this method.
if isinstance(self.random_state, int):
set_brush_random_state(self.random_state)

self.feature_names_ = []
self.feature_types_ = []
if isinstance(X, pd.DataFrame):
Expand Down
9 changes: 3 additions & 6 deletions pybrush/EstimatorInterface.py
Original file line number Diff line number Diff line change
Expand Up @@ -174,12 +174,9 @@ class EstimatorInterface():
If specified, spits statistics into a logfile. "" means don't log.
random_state: int or None, default None
If int, then the value is used to seed the c++ random generator; if None,
then a seed will be generated using a non-deterministic generator. It is
important to notice that, even if the random state is fixed, it is
unlikely that running brush using multiple threads will have the same
results. This happens because the Operating System's scheduler is
responsible to choose which thread will run at any given time, thus
reproductibility is not guaranteed.
then a seed will be generated using a non-deterministic generator. A
fixed integer makes repeated runs reproducible for the same Brush build,
data, and parameter configuration, including multi-island runs.
"""

def __init__(self,
Expand Down
28 changes: 19 additions & 9 deletions src/engine.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -516,11 +516,15 @@ void Engine<T>::run(Dataset &data)
// heavily inspired in https://github.com/heal-research/operon/blob/main/source/algorithms/nsga2.cpp
auto [init, cond, body, back, done] = taskflow.emplace(
[&](tf::Subflow& subflow) {
auto fit_init_pop = subflow.for_each_index(0, this->params.num_islands, 1, [&](int island) {
// Evaluate the individuals at least once
// Set validation loss before calling update best

evaluator.update_fitness(this->pop, island, data, params, true, true);
// Fitting an island mutates program weights. Keep that phase in
// a stable order as well, so the first selection sees identical
// fitness values for repeated seeded runs.
auto fit_init_pop = subflow.emplace([&]() {
for (int island = 0; island < this->params.num_islands; ++island) {
// Evaluate the individuals at least once. Set validation
// loss before calling update_best.
evaluator.update_fitness(this->pop, island, data, params, true, true);
}
});
auto find_init_best = subflow.emplace([&]() {
// Make sure we initialize it. We do this update here because we need to
Expand All @@ -541,7 +545,13 @@ void Engine<T>::run(Dataset &data)
batch = data.get_batch(); // will return the original dataset if it is set to dont use batch
}).name("prepare generation");// set generation in params, get batch

auto run_generation = subflow.for_each_index(0, this->params.num_islands, 1, [&](int island) {
// Selection consumes the process-wide random stream. Keep island
// selection ordered: Taskflow workers are not OpenMP workers, so
// using their scheduling order would otherwise make a fixed seed
// nondeterministic. Initial fitness evaluation is ordered too,
// so the complete seeded evolutionary path is scheduler-neutral.
auto run_generation = subflow.emplace([&]() {
for (int island = 0; island < this->params.num_islands; ++island) {

evaluator.update_fitness(this->pop, island, data, params, false, false); // fit the weights with all training data

Expand All @@ -556,8 +566,8 @@ void Engine<T>::run(Dataset &data)
}

this->pop.add_offspring_indexes(island);

}).name("runs one generation at each island in parallel");
}
}).name("runs one generation at each island in deterministic order");

auto update_pop = subflow.emplace([&]() { // sync point
// Variation is not thread safe.
Expand Down Expand Up @@ -692,4 +702,4 @@ void Engine<T>::run(Dataset &data)
template class Brush::Engine<Brush::ProgramType::Regressor>;
template class Brush::Engine<Brush::ProgramType::BinaryClassifier>;
template class Brush::Engine<Brush::ProgramType::MulticlassClassifier>;
template class Brush::Engine<Brush::ProgramType::Representer>;
template class Brush::Engine<Brush::ProgramType::Representer>;
8 changes: 4 additions & 4 deletions src/program/node.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -121,7 +121,8 @@ void to_json(json& j, const Node& p)
{
j = json{
{"name", p.name},
{"center_op", p.center_op},
// center_op is always set to false --> not implemented yet
{"center_op", false},
{"node_is_fixed", p.node_is_fixed},
{"weight_is_fixed", p.weight_is_fixed},
{"prob_change", p.prob_change},
Expand Down Expand Up @@ -319,9 +320,6 @@ void from_json(const json &j, Node& p)
else
p.name = NodeTypeName[p.node_type];

if (j.contains("center_op"))
j.at("center_op").get_to(p.center_op);

// used in split nodes
if (j.contains("feature"))
{
Expand Down Expand Up @@ -361,6 +359,8 @@ void from_json(const json &j, Node& p)

// after this point we set attributes that are modified in init
p.init();
// center_op is intentionally ignored
p.center_op = false;

// these below needs to be set after init(), since `init` sets these values
if (j.contains("node_is_fixed"))
Expand Down
6 changes: 5 additions & 1 deletion src/program/node.h
Original file line number Diff line number Diff line change
Expand Up @@ -122,7 +122,9 @@ struct Node {
float W;

/// whether to center the operator in pretty printing
bool center_op; // TODO: use center_op in printing
// Retained for backwards-compatible program JSON. This has not being implemented yet
// While doing some random_state tests, i realized that this feature was being set to true sometimes, undesirable. So we make sure to initialize it as false always
bool center_op = false;

// /// @brief a node hash / unique ID for the node, except weights
// size_t node_hash;
Expand Down Expand Up @@ -171,6 +173,8 @@ struct Node {
}

void init(){
center_op = false;

// starting weights with neutral element of the operation. offsetsum
// is the only node that does not multiply the weight --- instead, it adds it,
// so we need to handle it differently
Expand Down
46 changes: 27 additions & 19 deletions src/util/rnd.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -63,6 +63,9 @@ namespace Brush { namespace Util{
for (size_t i = 0; i < rg.size(); ++i) {
rg[i].seed(seeds[i]);
}

has_spare_normal = false;
spare_normal = 0.0f;
}

int Rnd::rnd_int( int lowerLimit, int upperLimit )
Expand Down Expand Up @@ -119,28 +122,33 @@ namespace Brush { namespace Util{
float Rnd::gasdev()
//Returns a normally distributed deviate with zero mean and unit variance
{
float ran = rnd_flt(-1,1);
static int iset=0;
static float gset;
float fac,rsq,v1,v2;
if (iset == 0) {// We don't have an extra deviate handy, so
do{
v1=float(2.0*rnd_flt(-1,1)-1.0); //pick two uniform numbers in the square ex
v2=float(2.0*rnd_flt(-1,1)-1.0); //tending from -1 to +1 in each direction,
rsq=v1*v1+v2*v2; //see if they are in the unit circle,
} while (rsq >= 1.0 || rsq == 0.0); //and if they are not, try again.
fac=float(sqrt(-2.0*log(rsq)/rsq));
// Box–Muller / polar implementation
// Returns a normally distributed deviate with zero mean
// and unit variance.

if (has_spare_normal) {
has_spare_normal = false;
return spare_normal;
}

// float ran = rnd_flt(-1,1);

float rsq,v1,v2;

do{
v1=float(rnd_flt(-1.0f,1.0f)); //pick two uniform numbers in the square ex
v2=float(rnd_flt(-1.0f,1.0f)); //tending from -1 to +1 in each direction,
rsq=v1*v1+v2*v2; //see if they are in the unit circle,
} while (rsq >= 1.0 || rsq == 0.0); //and if they are not, try again.

const float fac = float(sqrt(-2.0*log(rsq)/rsq));

//Now make the Box-Muller transformation to get two normal deviates. Return one and
//save the other for next time.
gset=v1*fac;
iset=1; //Set flag.
spare_normal=v1*fac;
has_spare_normal=true;

return v2*fac;
}
else
{ //We have an extra deviate handy,
iset=0; //so unset the flag,
return gset; //and return it.
}
}

/// returns a shuffled index vector of length n
Expand Down
9 changes: 9 additions & 0 deletions src/util/rnd.h
Original file line number Diff line number Diff line change
Expand Up @@ -163,6 +163,15 @@ namespace Brush { namespace Util{

// Vector of pseudo-random number generators, one for each thread
vector<std::mt19937> rg;

// Cached half of the Box-Muller transform used by gasdev(). This
// state is part of the generator and must be reset with its seed.

// Gaussian random numbers are generated with box-muller --> that
// process generates two independent distributed random normal numbers.
// The idea here is to cache the second one and use it, instead of discarding.
bool has_spare_normal = false;
float spare_normal = 0.0f;

// private static attribute used by every instance of the class.
// All threads share common static members of the class
Expand Down
1 change: 1 addition & 0 deletions tests/cpp/gtest.cpp
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
#include "testsHeader.h"

int main(int argc, char **argv) {
Brush::Util::r.set_seed(42);
testing::InitGoogleTest(&argc, argv);
return RUN_ALL_TESTS();
}
40 changes: 40 additions & 0 deletions tests/cpp/test_brush.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,7 @@ TEST(Engine, EngineWorks)
SearchSpace ss(data);

Parameters params;
params.set_random_state(42);
params.set_pop_size(100);
params.set_max_gens(10);
params.set_mig_prob(0.0);
Expand Down Expand Up @@ -124,6 +125,38 @@ TEST(Engine, EngineWorks)
// TODO: validation loss
}

TEST(Engine, FixedSeedProducesIdenticalRuns)
{
MatrixXf X(12, 2);
ArrayXf y(12);
X << 0.0f, 1.0f, 1.0f, 0.0f, 2.0f, 1.0f, 3.0f, 0.0f,
4.0f, 1.0f, 5.0f, 0.0f, 6.0f, 1.0f, 7.0f, 0.0f,
8.0f, 1.0f, 9.0f, 0.0f, 10.0f, 1.0f, 11.0f, 0.0f;
y << 1.0f, 2.0f, 2.5f, 3.5f, 4.0f, 5.0f, 5.5f, 6.5f,
7.0f, 8.0f, 8.5f, 9.5f;

Dataset data(X, y);
SearchSpace first_space(data);
SearchSpace second_space(data);
Parameters params;
params.set_random_state(42);
params.set_pop_size(12);
params.set_max_gens(4);
params.set_num_islands(2);
params.set_n_jobs(2);
params.set_mig_prob(0.0f);
params.set_verbosity(0);

Brush::RegressorEngine first(params, first_space);
first.run(data);
Brush::RegressorEngine second(params, second_space);
second.run(data);

ASSERT_EQ(first.best_ind.program.get_model(), second.best_ind.program.get_model());
ASSERT_EQ(first.best_ind.fitness.get_wvalues(), second.best_ind.fitness.get_wvalues());
ASSERT_EQ(first.get_population_as_json(), second.get_population_as_json());
}

#include <vector>
#include <string>

Expand All @@ -139,6 +172,7 @@ TEST_P(EngineTest, ClassificationEngineWorks)
ASSERT_TRUE(data.classification);

Parameters params;
params.set_random_state(42);
params.set_pop_size(10);

// TODO: this set_class_weights is not working properly. check that
Expand Down Expand Up @@ -212,6 +246,7 @@ TEST(Engine, SavingLoadingFixedNodes)
SearchSpace ss(data);

Parameters params;
params.set_random_state(42);
params.set_verbosity(2);
params.set_scorer("log");
params.set_cx_prob(0.0);
Expand Down Expand Up @@ -239,6 +274,7 @@ TEST(Engine, SavingLoadingFixedNodes)
// TODO: why if I set cx_prob to 0.0 it does not work? (maybe because Im using the same params object for the two engines? do i need to remove save_pop file first?)

Parameters params2;
params2.set_random_state(42);
params2.set_verbosity(2);
params2.set_scorer("log");
params2.set_load_population("./tests/cpp/__pop_clf.json");
Expand Down Expand Up @@ -281,6 +317,7 @@ TEST(Engine, MaxStall)
SearchSpace ss(data);

Parameters params;
params.set_random_state(42);
params.set_pop_size(100);
params.set_max_gens(10000000);
params.set_mig_prob(0.0);
Expand Down Expand Up @@ -316,6 +353,7 @@ TEST(Engine, engine_save_load_pop_works)
};

Parameters params_save;
params_save.set_random_state(42);
params_save.set_functions(f);
params_save.set_pop_size(200);
params_save.set_max_gens(10);
Expand All @@ -324,6 +362,7 @@ TEST(Engine, engine_save_load_pop_works)
params_save.set_save_population("./tests/cpp/__pop_analcatdata_aids.json");

Parameters params_load;
params_load.set_random_state(42);
params_load.set_functions(f);
params_load.set_pop_size(200);
params_load.set_max_gens(10);
Expand Down Expand Up @@ -355,6 +394,7 @@ TEST(Engine, DEnc)
std::cout << "Running bandit: " << bandit << std::endl;
for (int run = 0; run < 2; ++run) {
Parameters params;
params.set_random_state(42);
params.set_pop_size(100);
params.set_max_gens(50);
params.set_max_stall(100); // avoid early stopping
Expand Down
1 change: 1 addition & 0 deletions tests/cpp/test_data.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -26,6 +26,7 @@ TEST(Data, ErrorHandling)
TEST(Data, MixedVariableTypes)
{
Parameters params;
params.set_random_state(42);

MatrixXf X(5,3);
X << 0 , 1, 0 , // binary with integer values
Expand Down
1 change: 1 addition & 0 deletions tests/cpp/test_evaluation.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -131,6 +131,7 @@ TEST(Evaluation, MulticlassSoftmaxHasOneOutputPerClass)
Dataset data(X, y, {}, {}, {}, true);
SearchSpace search_space(data);
Parameters params;
params.set_random_state(42);
params.classification = true;
params.set_n_classes(y);
params.max_depth = 3;
Expand Down
4 changes: 3 additions & 1 deletion tests/cpp/test_individuals.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,7 @@ TEST(Individual, FitAndPredictRegression)
// We must have a SearchSpace reference, so the operator ret-type checks dont
// fail even when feature names look right --- node metadata is consistent.
Parameters params;
params.set_random_state(42);
RegressorProgram prg = ss.make_regressor(0, 0, params);
Individual<PT::Regressor> ind(prg);

Expand Down Expand Up @@ -52,6 +53,7 @@ TEST(Individual, PredictProbaBinaryClassifier)
SearchSpace ss(data);

Parameters params;
params.set_random_state(42);
params.set_n_classes(data.y);
params.set_sample_weights(data.y);

Expand Down Expand Up @@ -79,4 +81,4 @@ TEST(Individual, ParentIdAndId)
ASSERT_EQ(child.parent_id.size(), 2u);
ASSERT_EQ(child.parent_id.at(0), 3u);
ASSERT_EQ(child.parent_id.at(1), 7u);
}
}
1 change: 1 addition & 0 deletions tests/cpp/test_params.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@ TEST(Params, ParamsTests)
std::cout << "unsigned long long: min=" << std::numeric_limits<unsigned long long>::min() << " max=" << std::numeric_limits<unsigned long long>::max() << "\n";

Parameters params;
params.set_random_state(42);

params.set_max_size(12);
ASSERT_EQ(params.max_size, 12);
Expand Down
3 changes: 2 additions & 1 deletion tests/cpp/test_population.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,7 @@ TEST(Population, PopulationTests)
SS.init(data);

Parameters params;
params.set_random_state(42);
params.pop_size = 20; // small pop just for tests
Population pop = Population<ProgramType::Regressor>();

Expand Down Expand Up @@ -147,4 +148,4 @@ TEST(Population, PopulationTests)
// testing that we can save and load the population
pop.save("./tests/cpp/__pop_save_100_gen.json");
pop.load("./tests/cpp/__pop_save_100_gen.json");
}
}
Loading
Loading