This commit is contained in:
2021-10-14 13:47:35 +02:00
commit 6625a8dfaa
4026 changed files with 844291 additions and 0 deletions
@@ -0,0 +1,334 @@
#include "CMRMR.hpp"
#include <cstdlib>
#include <cmath>
#include <map>
#include <numeric>
#include <algorithm>
#include <iostream>
#include <fstream>
#include <iomanip>
///-------------------- Public Functions --------------------
///-------------------------------------------------------------------------------------------------
void CMRMR::reset()
{
m_nFeatures = 0;
m_nSamples = 0;
//if (!m_classes.empty()) { for (auto& c : m_classes) { c.second.clear(); } } // useless
//if (!m_datas.empty()) { for (auto& d : m_datas) { d.clear(); } } // useless
//if (!m_discretized.empty()) { for (auto& d : m_discretized) { d.clear(); } } // useless
m_classes.clear();
m_datas.clear();
m_discretized.clear();
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
std::string CMRMR::print() const
{
std::stringstream ss;
ss << "Datas contain " << m_nFeatures << " Features, with " << m_nSamples << " samples, for " << m_classes.size() << " Classes (";
size_t i = 0;
for (auto& it : m_classes)
{
if (i != 0) { ss << ", "; }
ss << "Class " << i++ << " : " << "id(" << it.first << "), " << it.second.size() << " samples";
}
ss << ")." << std::endl;
ss << "Datas are " << (m_discretized.empty() ? "not " : "") << "z-scored and discretize." << std::endl;
return ss.str();
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
bool CMRMR::readCSV(const std::string& filename)
{
reset();
std::ifstream file;
file.open(filename);
if (!file.is_open())
{
std::cerr << "File cannot be opened." << std::endl;
return false;
}
std::string header;
if (!getline(file, header))
{
std::cerr << "Header can not be read." << std::endl;
return false;
}
m_nFeatures = std::count(header.begin(), header.end(), ',');
m_nSamples = 0;
std::string line;
while (getline(file, line))
{
if (std::count(line.begin(), line.end(), ',') != m_nFeatures)
{
std::cerr << "not same number of features : " << std::count(line.begin(), line.end(), ',') << "expected : " << m_nFeatures << std::endl;
return false;
}
std::vector<double> sample;
sample.reserve(m_nFeatures);
int classId;
double value;
char sep;
std::stringstream ss(line);
ss >> classId;
while (ss >> sep >> value) { sample.push_back(value); }
if (sample.size() != m_nFeatures)
{
std::cerr << "not found good number of features : " << sample.size() << "expected : " << m_nFeatures << std::endl;
return false;
}
m_datas.push_back(std::move(sample));
// Update class list
auto it = m_classes.find(classId);
if (it == m_classes.end())
{
m_classes.emplace(classId, std::vector<size_t>());
it = m_classes.find(classId);
}
it->second.push_back(m_nSamples++);
}
return true;
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
bool CMRMR::setDatas(const std::vector<std::vector<double>>& datas, const std::vector<int>& classes)
{
reset();
return addDatas(datas, classes);
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
bool CMRMR::addDatas(const std::vector<std::vector<double>>& datas, const std::vector<int>& classes)
{
const size_t n = datas.size();
if (n != classes.size())
{
std::cerr << "not same number of sample between datas and classes : " << n << " VS " << classes.size() << std::endl;
return false;
}
for (size_t i = 0; i < n; ++i) { if (!addSample(datas[i], classes[i])) { return false; } }
return true;
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
bool CMRMR::addSample(const std::vector<double>& sample, const int classId)
{
if (m_nSamples == 0) { m_nFeatures = sample.size(); }
else if (m_nFeatures != sample.size())
{
std::cerr << "not same number of features between sample and previous samples : " << sample.size() << " VS " << m_nFeatures << std::endl;
return false;
}
// Update datas
m_datas.push_back(sample);
// Update class list
auto it = m_classes.find(classId);
if (it == m_classes.end())
{
m_classes.emplace(classId, std::vector<size_t>());
it = m_classes.find(classId);
}
it->second.push_back(m_nSamples++);
return true;
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
std::vector<size_t> CMRMR::process(const double threshold, const size_t nFeatures, const EMRMRMethod method)
{
if (m_nSamples == 0 || m_nFeatures == 0) { return std::vector<size_t>(); }
m_discretized.reserve(m_nSamples);
for (auto&& v : m_datas) { m_discretized.emplace_back(std::begin(v), std::end(v)); }
if (threshold != std::numeric_limits<double>::infinity())
{
m_discretized.resize(m_nSamples);
for (auto& d : m_discretized) { d.resize(m_nFeatures); }
zScore(); // Compute zScore
discretize(threshold); // Compute discretization
}
return mRMR(nFeatures, method); // Apply mRMR Algorithm
}
///-------------------------------------------------------------------------------------------------
///-------------------- Private Functions --------------------
///-------------------------------------------------------------------------------------------------
std::vector<int> CMRMR::class2IdxVector(size_t& n) const
{
std::vector<int> res;
res.resize(m_nSamples); // Features from all samples
int min = std::numeric_limits<int>::max();
int max = std::numeric_limits<int>::min();
for (const auto& m : m_classes)
{
const int tmp = m.first;
if (min > tmp) { min = tmp; }
if (max < tmp) { max = tmp; }
for (const auto& sampleIdx : m.second) { res[sampleIdx] = tmp; }
}
for (auto& v : res) { v -= min; } // transform to 0 to n Indexes
n = size_t(max - min + 1);
return res;
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
std::vector<int> CMRMR::discret2IdxVector(const size_t feature, size_t& n) const
{
std::vector<int> res;
res.reserve(m_nSamples); // Features from all samples
int min = m_discretized[0][feature];
int max = m_discretized[0][feature];
for (const auto& d : m_discretized)
{
const int tmp = int(round(d[feature]));
res.push_back(tmp);
if (min > tmp) { min = tmp; }
if (max < tmp) { max = tmp; }
}
for (auto& id : res) { id -= min; } // transform to 0 to n Indexes
n = size_t(max - min + 1);
return res;
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
void CMRMR::zScore()
{
// We z-score by feature (so column by column)
for (size_t j = 0; j < m_nFeatures; ++j)
{
double sum = 0.0;
for (const auto& d : m_datas) { sum += d[j]; }
const double mean = sum / double(m_nSamples);
sum = 0.0;
for (const auto& d : m_datas)
{
const double tmp = d[j] - mean;
sum += tmp * tmp;
}
const double std = (m_nSamples == 1) ? 0 : sqrt(sum / double(m_nSamples - 1)); //m_nSamples - 1 is an unbiased version for Gaussian
for (size_t i = 0; i < m_nSamples; ++i) { m_discretized[i][j] = (m_datas[i][j] - mean) / std; }
}
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
void CMRMR::discretize(const double threshold)
{
for (auto& sample : m_discretized)
{
for (auto& feature : sample)
{
if (feature > threshold) { feature = 1; }
else if (feature < -threshold) { feature = -1; }
else { feature = 0; }
}
}
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
double CMRMR::mutualInfo(const size_t feature1, const size_t feature2) const
{
if ((feature1 != size_t(-1) && feature1 >= m_nFeatures) || (feature2 != size_t(-1) && feature2 >= m_nFeatures)) { return -1; }
// Copy Datas in int vector of size n_sample
size_t nstate1, nstate2;
std::vector<int> v1 = (feature1 != size_t(-1)) ? discret2IdxVector(feature1, nstate1) : class2IdxVector(nstate1);
std::vector<int> v2 = (feature2 != size_t(-1)) ? discret2IdxVector(feature2, nstate2) : class2IdxVector(nstate2);
// Joint Probabilities
std::vector<std::vector<double>> jointProba(nstate1, std::vector<double>(nstate2, 0.0));
for (size_t i = 0; i < m_nSamples; ++i) { jointProba[v1[i]][v2[i]]++; }
for (auto& row : jointProba) { for (auto& cell : row) { cell /= m_nSamples; } }
// Mutual Information
std::vector<double> proba1(nstate1, 0.0), proba2(nstate2, 0.0);
for (size_t i = 0; i < nstate1; ++i)
{
for (size_t j = 0; j < nstate2; ++j)
{
proba1[i] += jointProba[i][j];
proba2[j] += jointProba[i][j];
}
}
double res = 0.0;
for (size_t i = 0; i < nstate1; ++i)
{
for (size_t j = 0; j < nstate2; ++j)
{
if (jointProba[i][j] != 0 && proba1[i] != 0 && proba2[j] != 0) { res += jointProba[i][j] * log(jointProba[i][j] / proba1[i] / proba2[j]); }
}
}
res /= log(2);
return res;
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
std::vector<size_t> CMRMR::mRMR(const size_t nFeatures, const EMRMRMethod method) const
{
if (nFeatures == 0) { return std::vector<size_t>(); }
const size_t n = ((nFeatures < m_nFeatures) ? nFeatures : m_nFeatures);
std::vector<size_t> res(n);
// Initialize selection
std::vector<double> mutualInfos;
std::vector<bool> mask(m_nFeatures, true);
mutualInfos.reserve(m_nFeatures);
for (size_t i = 0; i < m_nFeatures; ++i) { mutualInfos.push_back(mutualInfo(size_t(-1), i)); } // Compute Mutual infos with classId
//const double entropy = mutualInfo(size_t(-1), size_t(-1)); // the entropy of target classification variable
// Sort in Descending Order
std::vector<size_t> indexes(m_nFeatures);
iota(indexes.begin(), indexes.end(), 0);
stable_sort(indexes.begin(), indexes.end(), [&mutualInfos](const size_t i1, const size_t i2) { return mutualInfos[i1] > mutualInfos[i2]; });
//stable_sort(mutualInfos.begin(), mutualInfos.end(), greater<double>()); // Useless
//mRMR selection
res[0] = indexes[0]; // We have the first Feature
mask[indexes[0]] = false; // After selection, no longer consider this feature
for (size_t i = 1; i < n; ++i) //the first one, res[0] has been determined already
{
double score = std::numeric_limits<double>::min();
for (const auto& id : indexes)
{
if (!mask[id]) { continue; } // We skeep this Id
const double relevance = mutualInfos[id];
double redundancy = 0;
for (size_t j = 0; j < i; ++j) { redundancy += mutualInfo(res[j], id); }
redundancy /= double(i);
// If more methods, a switch is preferable
const double tmp = (method == EMRMRMethod::MID) ? relevance - redundancy : relevance / (redundancy + 0.0001);
if (score < tmp) //update the best feature found and the score
{
score = tmp;
res[i] = id;
}
}
// Remove from the id list the final selection with the mask, we can remove index from indexes vector instead of use mask
// (a test can be make to find wich method of mask and vector modification is faster)
mask[res[i]] = false;
}
return res;
}
///-------------------------------------------------------------------------------------------------
@@ -0,0 +1,148 @@
///-------------------------------------------------------------------------------------------------
///
/// \file CMRMR.hpp
/// \brief mRMR (minimum Redundancy Maximum Relevance) Feature Selection Class.
/// \author Thibaut Monseigne (Inria).
/// \version 1.0.
/// \date 03/02/2020.
/// \copyright <a href="https://choosealicense.com/licenses/agpl-3.0/">GNU Affero General Public License v3.0</a>.
/// \remarks
/// - This Algorithm is the same principle as <a href="http://home.penglab.com/proj/mRMR/">Hanchuan Peng et al.</a> (<a href="http://home.penglab.com/proj/mRMR/FAQ_mrmr.htm#Q1.8">License</a>). The difference is in the implementation in C++.
/// - Paper : Feature selection based on mutual information: criteria of max-dependency, max-relevance, and min-redundancy, <a href="https://ieeexplore.ieee.org/document/1453511">here</a> \n
/// Hanchuan Peng, Fuhui Long, and Chris Ding, IEEE Transactions on Pattern Analysis and Machine Intelligence, Vol. 27, No. 8, pp.1226-1238, 2005.
///
///-------------------------------------------------------------------------------------------------
#pragma once
#include <vector>
#include <sstream>
#include <map>
#include <limits>
enum class EMRMRMethod { MID, MIQ };
inline std::string toString(const EMRMRMethod& e)
{
switch (e)
{
case EMRMRMethod::MIQ: return "MIQ";
case EMRMRMethod::MID: return "MID";
default: return "MID";
}
}
class CMRMR
{
public:
/// <summary> Initializes a new instance of the <see cref="CMRMR"/> class. </summary>
CMRMR() = default;
/// <summary> Finalizes an instance of the <see cref="CMRMR"/> class. </summary>
~CMRMR() { reset(); }
/// <summary> Reset all member of this object. </summary>
void reset();
/// <summary> Prints some informations.</summary>
//// <returns> the string with some informations. </returns>
std::string print() const;
/// <summary>Reads the CSV file (with example format in http://home.penglab.com/proj/mRMR/ website). </summary>
/// <param name="filename">The filename.</param>
/// <returns> True if Succes, False if Fail. </returns>
bool readCSV(const std::string& filename);
/// <summary> Reset previous datas and set this datas. </summary>
/// <param name="datas">The datas.</param>
/// <param name="classes">The classes of each sample.</param>
/// <returns> True if success, False if fail (not same number of sample on two parameters or not same feature on each datas). </returns>
bool setDatas(const std::vector<std::vector<double>>& datas, const std::vector<int>& classes);
/// <summary> add this datas to actual. </summary>
/// <param name="datas">The datas.</param>
/// <param name="classes">The classes of each sample.</param>
/// <returns> True if success, False if fail (not same number of sample on two parameters or not same number of feature on each datas). </returns>
bool addDatas(const std::vector<std::vector<double>>& datas, const std::vector<int>& classes);
/// <summary> add this sample to datas. </summary>
/// <param name="sample">The sample.</param>
/// <param name="classId">The classes of this sample.</param>
/// <returns> True if success, False if fail (not same number feature than previous datas). </returns>
bool addSample(const std::vector<double>& sample, const int classId);
/// <summary> Gets the datas. </summary>
/// <returns> datas (std::vector<std::vector<double>>) </returns>
std::vector<std::vector<double>> getDatas() const { return m_datas; }
/// <summary> Gets the discretized datas. </summary>
/// <returns> discretized datas (std::vector<std::vector<double>>) </returns>
std::vector<std::vector<double>> getDiscretized() const { return m_discretized; }
/// <summary> Gets the classes list with for each classes the Ids of sample concerned. </summary>
/// <returns> classes (std::map<int, std::vector<size_t>>) </returns>
std::map<int, std::vector<size_t>> getClasses() const { return m_classes; }
size_t getSampleCount() const { return m_nSamples; }
size_t getFeaturesCount() const { return m_nFeatures; }
/// <summary> Apply the mRMR algorithm after compute zscore and discretisation. </summary>
/// <param name="threshold">The threshold for discretization (if infinity we only use z-score).</param>
/// <param name="nFeatures"> The number of features to keep. Default 500 is the original theorical max of the method (not needed but a wink to the origin). </param>
/// <param name="method"> Method Used for mRMR. </param>
/// <returns></returns>
std::vector<size_t> process(const double threshold = std::numeric_limits<double>::infinity(), const size_t nFeatures = 500,
const EMRMRMethod method = EMRMRMethod::MID);
/// <summary> Override the ostream operator. </summary>
/// <param name="os"> The ostream. </param>
/// <param name="obj"> The object. </param>
/// <returns> Return the modified ostream. </returns>
friend std::ostream& operator <<(std::ostream& os, const CMRMR& obj) { return (os << obj.print()); }
private:
size_t m_nFeatures = 0; // Number of Features
size_t m_nSamples = 0; // Number of Samples
std::map<int, std::vector<size_t>> m_classes; // Datas in the format class -> vector id sample
std::vector<std::vector<double>> m_datas; // Datas in the format sample -> features
std::vector<std::vector<double>> m_discretized; // Datas in the format sample -> features discretized (z-score or z-score + discretization)
std::vector<int> class2IdxVector(size_t& n) const;
std::vector<int> discret2IdxVector(const size_t feature, size_t& n) const;
/// <summary> Compute the z-score. </summary>
/// z-score is compute for each feature separatly.\n
/// -# Compute Mean of feature \f$ \mu = \frac{1}{n}\sum_{i=1}^{n}(x_{i}) \f$ \n
/// -# Compute The Unbiased estimation of standard deviation \f$ \sigma = \sqrt{\frac{\sum_{i=1}^{n}(x_{i} - \mu)^{2}}{n - 1}} \f$\n
/// -# Compute The z-score \f$ z_i = \frac{x_{i} - \mu}{\sigma} \f$
void zScore();
/// <summary> Discretizes the datas with the specified threshold (value became in set \f$ \{-1,0,1\} \f$). </summary>
/// <param name="threshold">The threshold for discretization.</param>
void discretize(const double threshold);
/// <summary> Mutuals the information. </summary>
/// The mutual Information is the a measure of the mutual dependence between the two features.\n
/// -# We get the value of the feature (with z-score the value is in set \f$ \{-1,0,1\} \f$ and became in set \f$ \{0,1,2\} \f$ for the next step.\n
/// -# We make a matrix with the probability of each value sample by sample.\n
/// with \f$ s_i \f$ the sample \f$ i \f$, \f$ v^1_i \f$ the value of the first feature for \f$ s_i \f$ and \f$ v^2_i \f$ the value of the second feature for \f$ s_i \f$\n
/// \f$ m \f$ is the count of the common values of each sample (\f$\text{count}\left(s_i\left(v^1_i,v^2_i\right)\right)\text{, for each } i \in n\f$ \n
/// -# We commpute the mutal information.\n
/// With \f$ n_1 \f$ the number of state for feature 1, \f$ n_2 \f$ the number of state for feature 2. \f$ p \f$ the probability.\n
/// \f[ mi = \sum_{i\in n_1, j \in n_2}{p_{i,j} * \log\left(\frac{p_{i,j}}{p_{i} \times p_{j}}\right)} \f]
/// <param name="feature1">The feature number 1 (default (-1) is for use the classification target instead of feature).</param>
/// <param name="feature2">The feature number 2 (default (-1) is for use the classification target instead of feature).</param>
/// <returns> the mutal information. </returns>
double mutualInfo(const size_t feature1 = size_t(-1), const size_t feature2 = size_t(-1)) const;
/// <summary> mRMR (minimum Redundancy Maximum Relevance Feature Selection) algorithm. </summary>
/// We use the method describe in the <a href="https://ieeexplore.ieee.org/document/1453511">paper</a>.\n
/// -# We Compute the relevance of each feature with the mutal info with the classification target.\n
/// -# We sort all in descending value and take the first feature.\n
/// -# We loop on all nexted features and compute for each the redundancy of the feature with previous selected features.\n
/// The feature with the best score, the score is compute with two method (<see cref="EMRMRMethod"/>)\n
/// \f[ \text{MID} = \text{relevance} - \text{redundancy}\text{, }\quad\text{MIQ} = \frac{\text{relevance}}{\text{redundancy} + 10^{-3}}\f]
/// <param name="nFeatures"> The number of features to keep. Default 500 is the original theorical max of the method (not needed but a wink to the origin). </param>
/// <param name="method"> Method Used for mRMR. </param>
/// <returns> The selected indexes. </returns>
std::vector<size_t> mRMR(const size_t nFeatures = 500, const EMRMRMethod method = EMRMRMethod::MID) const;
};
@@ -0,0 +1,174 @@
#include "CBoxAlgorithmFeaturesSelection.hpp"
#include <fstream>
namespace OpenViBE {
namespace Plugins {
namespace FeaturesSelection {
///-------------------------------------------------------------------------------------------------
bool CBoxAlgorithmFeaturesSelection::initialize()
{
// Stimulations
m_stimDecoder.initialize(*this, 0);
m_iStim = m_stimDecoder.getOutputStimulationSet();
m_stimEncoder.initialize(*this, 0);
m_oStim = m_stimEncoder.getInputStimulationSet();
// Classes
//m_featureDecoders.initialize(*this, 1);
const Kernel::IBox& boxCtx = this->getStaticBoxContext();
m_nbClass = boxCtx.getInputCount() - 1;
m_featuresDecoders.resize(m_nbClass);
m_iFeatures.resize(m_nbClass);
for (size_t k = 0; k < m_nbClass; ++k)
{
m_featuresDecoders[k].initialize(*this, k + 1);
m_iFeatures[k] = m_featuresDecoders[k].getOutputMatrix();
}
auto& ctx = *this->getBoxAlgorithmContext();
size_t idx = 0;
m_logLevel = Kernel::ELogLevel(uint64_t(FSettingValueAutoCast(ctx, idx++)));
m_stimName = FSettingValueAutoCast(ctx, idx++);
m_filename = CString(FSettingValueAutoCast(ctx, idx++)).toASCIIString();
m_method = EFeatureSelection(uint64_t(FSettingValueAutoCast(ctx, idx++)));
m_nFinalFeatures = uint64_t(FSettingValueAutoCast(ctx, idx++));
m_doDiscretization = FSettingValueAutoCast(ctx, idx++);
m_threshold = FSettingValueAutoCast(ctx, idx++);
m_mRMRMethod = EMRMRMethod(uint64_t(FSettingValueAutoCast(ctx, idx)));
m_selector.reset();
if (m_logLevel != Kernel::LogLevel_None)
{
getLogManager() << m_logLevel << "Trainer Initialized : \n\t" << m_nbClass << " Classes, Features Selection method is "
<< toString(m_method) << " with " << m_nFinalFeatures << " Features to select,";
if (m_doDiscretization) { getLogManager() << " with discretization (threshold = " << m_threshold << ")."; }
else { getLogManager() << " without discretization."; }
getLogManager() << "\n\tMethod used for mRMR is " << toString(m_mRMRMethod) << ".\n\t";
if (m_filename.empty()) { getLogManager() << "Config not saved.\n"; }
else { getLogManager() << "Config saved in file\'" << m_filename << "\'.\n"; }
}
return true;
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
bool CBoxAlgorithmFeaturesSelection::uninitialize()
{
m_stimDecoder.uninitialize();
m_stimEncoder.uninitialize();
for (auto& codec : m_featuresDecoders) { codec.uninitialize(); }
m_featuresDecoders.clear();
m_iFeatures.clear();
if (m_logLevel != Kernel::LogLevel_None) { getLogManager() << m_logLevel << "Trainer Uninitialized, selector infos : \n" << m_selector.print(); }
m_selector.reset();
return true;
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
bool CBoxAlgorithmFeaturesSelection::processInput(const size_t /*index*/)
{
getBoxAlgorithmContext()->markAlgorithmAsReadyToProcess();
return true;
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
bool CBoxAlgorithmFeaturesSelection::process()
{
if (m_isTrain) { return true; } // If train is made don't do process
Kernel::IBoxIO& boxCtx = this->getDynamicBoxContext();
//***** Stimulations (input 0) *****
for (size_t i = 0; i < boxCtx.getInputChunkCount(0); ++i)
{
m_stimDecoder.decode(i); // Decode the chunk
const uint64_t start = boxCtx.getInputChunkStartTime(0, i), end = boxCtx.getInputChunkEndTime(0, i); // Time Code
if (m_stimDecoder.isHeaderReceived()) // Header received
{
m_stimEncoder.encodeHeader();
boxCtx.markOutputAsReadyToSend(0, 0, 0);
}
if (m_stimDecoder.isBufferReceived()) // Buffer received
{
for (size_t j = 0; j < m_iStim->getStimulationCount(); ++j)
{
if (m_iStim->getStimulationIdentifier(j) == m_stimName)
{
// Process
getLogManager() << m_logLevel << "Train Flag Received, selector infos : \n" << m_selector.print();
m_result = m_selector.process((m_doDiscretization ? m_threshold : std::numeric_limits<double>::infinity()), m_nFinalFeatures, m_mRMRMethod);
getLogManager() << m_logLevel << "Features selected :";
for (const auto& r : m_result) { getLogManager() << " " << r; }
getLogManager() << "\n";
// Save File
if (!m_filename.empty()) { OV_ERROR_UNLESS_KRF(writeConfig(), "Error During File writing.", Kernel::ErrorType::BadFileWrite); }
// Send Stimulation
const uint64_t stim = this->getTypeManager().getEnumerationEntryValueFromName(OV_TypeId_Stimulation, "OVTK_StimulationId_TrainCompleted");
m_oStim->appendStimulation(stim, m_iStim->getStimulationDate(j), 0);
m_isTrain = true;
}
}
m_stimEncoder.encodeBuffer();
boxCtx.markOutputAsReadyToSend(0, start, end);
}
if (m_stimDecoder.isEndReceived()) // End received
{
m_stimEncoder.encodeEnd();
boxCtx.markOutputAsReadyToSend(0, start, end);
}
}
//***** Features (Input 1 to N) *****
for (size_t k = 0; k < m_nbClass; ++k)
{
for (size_t i = 0; i < boxCtx.getInputChunkCount(k + 1); ++i)
{
m_featuresDecoders[k].decode(i); // Decode the chunk
OV_ERROR_UNLESS_KRF(m_iFeatures[k]->getDimensionCount() == 1, "Invalid Input Signal.", Kernel::ErrorType::BadInput);
if (m_featuresDecoders[k].isBufferReceived()) // Buffer received
{
const size_t n = m_iFeatures[k]->getBufferElementCount();
double* buffer = m_iFeatures[k]->getBuffer();
//const std::vector<double> sample(buffer, buffer + n);
m_selector.addSample(std::vector<double>(buffer, buffer + n), int(k));
//getLogManager() << m_logLevel << " Sample received for class " << k << "\n";
}
}
}
return true;
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
bool CBoxAlgorithmFeaturesSelection::writeConfig()
{
std::ofstream file;
file.open(m_filename);
OV_ERROR_UNLESS_KRF(file.is_open(), "File can't be opened.", Kernel::ErrorType::BadFileWrite);
std::string sep;
file << "<OpenViBE-SettingsOverride>\n\t<SettingValue>";
for (const auto& r : m_result)
{
file << sep << r;
sep = ";";
}
file << "</SettingValue>\n</OpenViBE-SettingsOverride>";
file.close();
return true;
}
///-------------------------------------------------------------------------------------------------
} // namespace FeaturesSelection
} // namespace Plugins
} // namespace OpenViBE
@@ -0,0 +1,133 @@
///-------------------------------------------------------------------------------------------------
///
/// \file CBoxAlgorithmFeaturesSelection.hpp
/// \brief Classes of the Box Features Selection Trainer.
/// \author Thibaut Monseigne (Inria).
/// \version 1.0.
/// \date 12/02/2020.
/// \copyright <a href="https://choosealicense.com/licenses/agpl-3.0/">GNU Affero General Public License v3.0</a>.
///
///-------------------------------------------------------------------------------------------------
#pragma once
#include "ovp_defines.h"
#include <openvibe/ov_all.h>
#include <toolkit/ovtk_all.h>
#include "algorithm/CMRMR.hpp"
#include <vector>
#define OV_AttributeId_Box_FlagIsUnstable CIdentifier(0x666FFFFF, 0x666FFFFF)
namespace OpenViBE {
namespace Plugins {
namespace FeaturesSelection {
/// <summary> The class CBoxAlgorithmFeaturesSelection describes the box Features Selection Trainer. </summary>
class CBoxAlgorithmFeaturesSelection final : virtual public Toolkit::TBoxAlgorithm<IBoxAlgorithm>
{
public:
void release() override { delete this; }
bool initialize() override;
bool uninitialize() override;
bool processInput(const size_t index) override;
bool process() override;
_IsDerivedFromClass_Final_(Toolkit::TBoxAlgorithm<IBoxAlgorithm>, OVP_ClassId_BoxAlgorithm_FeaturesSelection)
protected:
//***** Codecs *****
Toolkit::TStimulationDecoder<CBoxAlgorithmFeaturesSelection> m_stimDecoder;
Toolkit::TStimulationEncoder<CBoxAlgorithmFeaturesSelection> m_stimEncoder;
std::vector<Toolkit::TFeatureVectorDecoder<CBoxAlgorithmFeaturesSelection>> m_featuresDecoders;
std::vector<CMatrix*> m_iFeatures; // Input Matrix pointer
IStimulationSet *m_iStim = nullptr, *m_oStim = nullptr; // Stimulation receiver/sender
//***** Settings *****
Kernel::ELogLevel m_logLevel = Kernel::LogLevel_Info; // Log Level
uint64_t m_stimName = OVTK_StimulationId_Train; // Name of stimulation to check for train lunch
std::string m_filename;
EFeatureSelection m_method = EFeatureSelection::MRMR;
size_t m_nbClass = 2; // Number of input classes
bool m_isTrain = false;
// mRMR Settings
CMRMR m_selector;
bool m_doDiscretization = true; // Check if we make Discretization
double m_threshold = 0.0; // Threshold for Discretisation
size_t m_nFinalFeatures = size_t(-1); // Number of Features in output
EMRMRMethod m_mRMRMethod = EMRMRMethod::MID; // mRMR Method
std::vector<size_t> m_result; // mRMR Result
bool writeConfig();
};
/// <summary> Listener of the box Features Selection Trainer. </summary>
class CBoxAlgorithmFeaturesSelectionListener final : public Toolkit::TBoxListener<IBoxListener>
{
public:
bool onInputAdded(Kernel::IBox& box, const size_t index) override
{
box.setInputType(index, OV_TypeId_FeatureVector);
box.setInputName(index, ("Class " + std::to_string(index)).c_str());
return true;
}
bool onInputRemoved(Kernel::IBox& box, const size_t index) override { return true; }
_IsDerivedFromClass_Final_(Toolkit::TBoxListener<IBoxListener>, CIdentifier::undefined())
};
/// <summary> Descriptor of the box Features Selection Trainer. </summary>
class CBoxAlgorithmFeaturesSelectionDesc final : virtual public IBoxAlgorithmDesc
{
public:
void release() override { }
CString getName() const override { return CString("Features Selection Trainer"); }
CString getAuthorName() const override { return CString("Thibaut Monseigne"); }
CString getAuthorCompanyName() const override { return CString("Inria"); }
CString getShortDescription() const override { return CString("Apply a Features Selection Algorithm"); }
CString getDetailedDescription() const override { return CString(""); }
CString getCategory() const override { return CString("Features Selection"); }
CString getVersion() const override { return CString("1.0"); }
CString getStockItemName() const override { return CString("gtk-execute"); }
CIdentifier getCreatedClass() const override { return OVP_ClassId_BoxAlgorithm_FeaturesSelection; }
IPluginObject* create() override { return new CBoxAlgorithmFeaturesSelection; }
IBoxListener* createBoxListener() const override { return new CBoxAlgorithmFeaturesSelectionListener; }
void releaseBoxListener(IBoxListener* listener) const override { delete listener; }
bool getBoxPrototype(Kernel::IBoxProto& prototype) const override
{
prototype.addInput("Train-Start Flag",OV_TypeId_Stimulations);
prototype.addInput("Class 1",OV_TypeId_FeatureVector);
prototype.addInput("Class 2",OV_TypeId_FeatureVector);
prototype.addFlag(Kernel::BoxFlag_CanAddInput);
prototype.addOutput("Train-Completed Flag",OV_TypeId_Stimulations);
prototype.addSetting("Log Level", OV_TypeId_LogLevel, "Information");
prototype.addSetting("Train trigger", OV_TypeId_Stimulation, "OVTK_StimulationId_Train");
prototype.addSetting("Filename to save Feature Selection", OV_TypeId_Filename, "${Player_ScenarioDirectory}/features-selected.xml");
prototype.addSetting("Method", OVP_TypeId_Features_Selection_Method, toString(EFeatureSelection::MRMR).c_str());
prototype.addSetting("Number of features to select", OV_TypeId_Integer, "2");
prototype.addSetting("Discretisation", OV_TypeId_Boolean, "true");
prototype.addSetting("Threshold", OV_TypeId_Float, "0.0");
prototype.addSetting("mRMR Method", OVP_TypeId_mRMR_Method, "MID");
return true;
}
_IsDerivedFromClass_Final_(IBoxAlgorithmDesc, OVP_ClassId_BoxAlgorithm_FeaturesSelectionDesc)
};
} // namespace FeaturesSelection
} // namespace Plugins
} // namespace OpenViBE
@@ -0,0 +1,116 @@
#include "CBoxAlgorithmFeaturesSelector.hpp"
namespace OpenViBE {
namespace Plugins {
namespace FeaturesSelection {
///-------------------------------------------------------------------------------------------------
bool CBoxAlgorithmFeaturesSelector::initialize()
{
m_decoder.initialize(*this, 0);
m_iMatrix = m_decoder.getOutputMatrix();
m_encoder.initialize(*this, 0);
m_oMatrix = m_encoder.getInputMatrix();
m_lookup.clear();
return true;
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
bool CBoxAlgorithmFeaturesSelector::uninitialize()
{
m_decoder.uninitialize();
m_encoder.uninitialize();
m_lookup.clear();
return true;
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
bool CBoxAlgorithmFeaturesSelector::processInput(const size_t /*index*/)
{
getBoxAlgorithmContext()->markAlgorithmAsReadyToProcess();
return true;
}
///-------------------------------------------------------------------------------------------------
///-------------------------------------------------------------------------------------------------
bool CBoxAlgorithmFeaturesSelector::process()
{
Kernel::IBoxIO& boxCtx = this->getDynamicBoxContext();
for (size_t i = 0; i < boxCtx.getInputChunkCount(0); ++i)
{
m_decoder.decode(i); // Decode the chunk
if (m_decoder.isHeaderReceived()) // Header Received
{
// Parse Setting
const std::string setting = CString(FSettingValueAutoCast(*this->getBoxAlgorithmContext(), 0)).toASCIIString();
OV_ERROR_UNLESS_KRF(parseSetting(setting), "Parsing has failed", Kernel::ErrorType::BadSetting);
OV_ERROR_UNLESS_KRF(!m_lookup.empty(), "No channel selected", Kernel::ErrorType::BadConfig);
getLogManager() << Kernel::LogLevel_Debug << "Features selected :";
for (const auto& r : m_lookup) { getLogManager() << " " << r; }
getLogManager() << "\n";
// Initialize Output Matrix
m_oMatrix->resize(m_lookup.size());
m_oMatrix->resetBuffer();
for (size_t j = 0; j < m_lookup.size(); ++j) { m_oMatrix->setDimensionLabel(0, j, m_iMatrix->getDimensionLabel(0, m_lookup[j])); }
m_encoder.encodeHeader();
}
if (m_decoder.isBufferReceived()) // Buffer Received
{
const double* iBuffer = m_iMatrix->getBuffer();
double* oBuffer = m_oMatrix->getBuffer();
for (size_t j = 0; j < m_lookup.size(); ++j) { memcpy(oBuffer + j, iBuffer + m_lookup[j], sizeof(double)); }
m_encoder.encodeBuffer();
}
if (m_decoder.isEndReceived()) { m_encoder.encodeEnd(); } // End Received
boxCtx.markOutputAsReadyToSend(0, boxCtx.getInputChunkStartTime(0, i), boxCtx.getInputChunkEndTime(0, i));
}
return true;
}
bool CBoxAlgorithmFeaturesSelector::parseSetting(const std::string& setting)
{
OV_ERROR_UNLESS_KRF(!setting.empty(), "Empty Setting is forbidden", Kernel::ErrorType::BadSetting);
std::stringstream ss(setting);
std::string token;
std::vector<std::string> list;
while (std::getline(ss, token, OV_Value_EnumeratedStringSeparator)) { list.push_back(token); }
// Define Min Max
const size_t startIdx = 0, endIdx = m_iMatrix->getDimensionSize(0); // never update in case of duplication or changing order
size_t idx1, idx2;
char sep;
// For each token
for (const auto& tmp : list)
{
ss.clear();
ss.str(tmp);
// Check if it's range
const std::size_t pos = tmp.find(OV_Value_RangeStringSeparator);
if (pos != std::string::npos)
{
if (tmp[0] == OV_Value_RangeStringSeparator) { idx1 = startIdx; }
else { ss >> idx1; }
ss >> sep;
if (tmp[tmp.length() - 1] == OV_Value_RangeStringSeparator) { idx2 = endIdx; }
else { ss >> idx2; }
OV_ERROR_UNLESS_KRF((idx1 < idx2 && idx2 <= endIdx), "Invalid Range String.", Kernel::ErrorType::BadSetting);
for (size_t i = idx1; i < idx2; ++i) { m_lookup.push_back(i); }
}
else
{
ss >> idx1;
OV_ERROR_UNLESS_KRF(idx1 <= endIdx, "Invalid Index.", Kernel::ErrorType::BadSetting);
m_lookup.push_back(idx1);
}
}
return true;
}
///-------------------------------------------------------------------------------------------------
} // namespace FeaturesSelection
} // namespace Plugins
} // namespace OpenViBE
@@ -0,0 +1,84 @@
///-------------------------------------------------------------------------------------------------
///
/// \file CBoxAlgorithmFeaturesSelector.hpp
/// \brief Classes of the Box Features Selector.
/// \author Thibaut Monseigne (Inria).
/// \version 1.0.
/// \date 12/02/2020.
/// \copyright <a href="https://choosealicense.com/licenses/agpl-3.0/">GNU Affero General Public License v3.0</a>.
///
///-------------------------------------------------------------------------------------------------
#pragma once
#include "ovp_defines.h"
#include <openvibe/ov_all.h>
#include <toolkit/ovtk_all.h>
#include "algorithm/CMRMR.hpp"
#include <vector>
namespace OpenViBE {
namespace Plugins {
namespace FeaturesSelection {
/// <summary> The class CBoxAlgorithmFeaturesSelector describes the box Features Selector. </summary>
class CBoxAlgorithmFeaturesSelector final : virtual public Toolkit::TBoxAlgorithm<IBoxAlgorithm>
{
public:
void release() override { delete this; }
bool initialize() override;
bool uninitialize() override;
bool processInput(const size_t index) override;
bool process() override;
_IsDerivedFromClass_Final_(Toolkit::TBoxAlgorithm<IBoxAlgorithm>, OVP_ClassId_BoxAlgorithm_FeaturesSelector)
protected:
//***** Codecs *****
Toolkit::TFeatureVectorDecoder<CBoxAlgorithmFeaturesSelector> m_decoder;
Toolkit::TFeatureVectorEncoder<CBoxAlgorithmFeaturesSelector> m_encoder;
CMatrix *m_iMatrix = nullptr, *m_oMatrix = nullptr;
//***** Settings *****
std::vector<size_t> m_lookup;
/// <summary> Parse the setting. </summary>
/// <param name="setting"> Setting to parse. </param>
/// <returns> True if Setting is correctly parsed.</returns>
bool parseSetting(const std::string& setting);
};
/// <summary> Descriptor of the box Features Selector. </summary>
class CBoxAlgorithmFeaturesSelectorDesc final : virtual public IBoxAlgorithmDesc
{
public:
void release() override { }
CString getName() const override { return CString("Features Selector"); }
CString getAuthorName() const override { return CString("Thibaut Monseigne"); }
CString getAuthorCompanyName() const override { return CString("Inria"); }
CString getShortDescription() const override { return CString("Select a subset of features vector."); }
CString getDetailedDescription() const override { return CString("Select Features with index starting from 0."); }
CString getCategory() const override { return CString("Features Selection"); }
CString getVersion() const override { return CString("1.0"); }
CString getStockItemName() const override { return CString("gtk-sort-ascending"); }
CIdentifier getCreatedClass() const override { return OVP_ClassId_BoxAlgorithm_FeaturesSelector; }
IPluginObject* create() override { return new CBoxAlgorithmFeaturesSelector; }
bool getBoxPrototype(Kernel::IBoxProto& prototype) const override
{
prototype.addInput("Input", OV_TypeId_FeatureVector);
prototype.addOutput("Output", OV_TypeId_FeatureVector);
prototype.addSetting("Features List", OV_TypeId_String, ":");
return true;
}
_IsDerivedFromClass_Final_(IBoxAlgorithmDesc, OVP_ClassId_BoxAlgorithm_FeaturesSelectorDesc)
};
} // namespace FeaturesSelection
} // namespace Plugins
} // namespace OpenViBE
@@ -0,0 +1,30 @@
///-------------------------------------------------------------------------------------------------
///
/// \file ovp_defines.h
/// \brief Defines list for Setting, Shortcut Macro and const.
/// \author Thibaut Monseigne (Inria).
/// \version 1.0.
/// \date 12/02/2020.
/// \copyright <a href="https://choosealicense.com/licenses/agpl-3.0/">GNU Affero General Public License v3.0</a>.
///
///-------------------------------------------------------------------------------------------------
#pragma once
#include <string>
// Boxes
//---------------------------------------------------------------------------------------------------
#define OVP_ClassId_BoxAlgorithm_FeaturesSelection OpenViBE::CIdentifier(0xee36249f, 0x22a32e6e)
#define OVP_ClassId_BoxAlgorithm_FeaturesSelectionDesc OpenViBE::CIdentifier(0xee36249f, 0x22a32e6f)
#define OVP_ClassId_BoxAlgorithm_FeaturesSelector OpenViBE::CIdentifier(0xee36249f, 0x22a32e7e)
#define OVP_ClassId_BoxAlgorithm_FeaturesSelectorDesc OpenViBE::CIdentifier(0xee36249f, 0x22a32e7f)
// Types Lists
//---------------------------------------------------------------------------------------------------
#define OVP_TypeId_Features_Selection_Method OpenViBE::CIdentifier(0x5261636B, 0x46534D45)
#define OVP_TypeId_mRMR_Method OpenViBE::CIdentifier(0x5261636B, 0x6D524D52)
enum class EFeatureSelection { MRMR };
//inline std::string toString(const EFeatureSelection& e) { switch (e) { default: return "mRMR (minimum Redundancy Maximum Relevance)"; } }
inline std::string toString(const EFeatureSelection& /*e*/) { return "mRMR (minimum Redundancy Maximum Relevance)"; }
@@ -0,0 +1,30 @@
#include <openvibe/ov_all.h>
#include "ovp_defines.h"
// Boxes Includes
#include "boxes/CBoxAlgorithmFeaturesSelection.hpp"
#include "boxes/CBoxAlgorithmFeaturesSelector.hpp"
namespace OpenViBE {
namespace Plugins {
namespace FeaturesSelection {
OVP_Declare_Begin()
// Register boxes
OVP_Declare_New(CBoxAlgorithmFeaturesSelectionDesc);
OVP_Declare_New(CBoxAlgorithmFeaturesSelectorDesc);
// Enumeration Feature Selection Method
context.getTypeManager().registerEnumerationType(OVP_TypeId_Features_Selection_Method, "Features Selection Method");
context.getTypeManager().registerEnumerationEntry(OVP_TypeId_Features_Selection_Method, toString(EFeatureSelection::MRMR).c_str(),
size_t(EFeatureSelection::MRMR));
// Enumeration mRMR Method
context.getTypeManager().registerEnumerationType(OVP_TypeId_mRMR_Method, "mRMR Method");
context.getTypeManager().registerEnumerationEntry(OVP_TypeId_mRMR_Method, "MID", size_t(EMRMRMethod::MID));
context.getTypeManager().registerEnumerationEntry(OVP_TypeId_mRMR_Method, "MIQ", size_t(EMRMRMethod::MIQ));
OVP_Declare_End()
} // namespace FeaturesSelection
} // namespace Plugins
} // namespace OpenViBE