aGrUM 3.2.0
a C++ library for (probabilistic) graphical models
scoreBIC.cpp
Go to the documentation of this file.
1/****************************************************************************
2 * This file is part of the aGrUM/pyAgrum library. *
3 * *
4 * Copyright (c) 2005-2026 by *
5 * - Pierre-Henri WUILLEMIN(_at_LIP6) *
6 * - Christophe GONZALES(_at_AMU) *
7 * *
8 * The aGrUM/pyAgrum library is free software; you can redistribute it *
9 * and/or modify it under the terms of either : *
10 * *
11 * - the GNU Lesser General Public License as published by *
12 * the Free Software Foundation, either version 3 of the License, *
13 * or (at your option) any later version, *
14 * - the MIT license (MIT), *
15 * - or both in dual license, as here. *
16 * *
17 * (see https://agrum.gitlab.io/articles/dual-licenses-lgplv3mit.html) *
18 * *
19 * This aGrUM/pyAgrum library is distributed in the hope that it will be *
20 * useful, but WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, *
21 * INCLUDING BUT NOT LIMITED TO THE WARRANTIES MERCHANTABILITY or FITNESS *
22 * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE *
23 * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER *
24 * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, *
25 * ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR *
26 * OTHER DEALINGS IN THE SOFTWARE. *
27 * *
28 * See LICENCES for more details. *
29 * *
30 * SPDX-FileCopyrightText: Copyright 2005-2026 *
31 * - Pierre-Henri WUILLEMIN(_at_LIP6) *
32 * - Christophe GONZALES(_at_AMU) *
33 * SPDX-License-Identifier: LGPL-3.0-or-later OR MIT *
34 * *
35 * Contact : info_at_agrum_dot_org *
36 * homepage : http://agrum.gitlab.io *
37 * gitlab : https://gitlab.com/agrumery/agrum *
38 * *
39 ****************************************************************************/
40
41
48
50
51#ifndef DOXYGEN_SHOULD_SKIP_THIS
52
54# ifdef GUM_NO_INLINE
56# endif /* GUM_NO_INLINE */
57
58namespace gum {
59
60 namespace learning {
61
62 // Constructors and destructor are defined out-of-line (not INLINE) on
63 // purpose: see the comment in score.cpp for the MSVC LNK2005 rationale.
64
67 const Prior& prior,
68 const std::vector< std::pair< std::size_t, std::size_t > >& ranges,
69 const Bijection< NodeId, std::size_t >& nodeId2columns) :
70 Score(parser, prior, ranges, nodeId2columns),
71 _internal_prior_(parser.database(), nodeId2columns) {
72 GUM_CONSTRUCTOR(ScoreBIC);
73 }
74
76 ScoreBIC::ScoreBIC(const DBRowGeneratorParser& parser,
77 const Prior& prior,
78 const Bijection< NodeId, std::size_t >& nodeId2columns) :
79 Score(parser, prior, nodeId2columns), _internal_prior_(parser.database(), nodeId2columns) {
80 GUM_CONSTRUCTOR(ScoreBIC);
81 }
82
84 ScoreBIC::ScoreBIC(const ScoreBIC& from) :
85 Score(from), _internal_prior_(from._internal_prior_) {
86 GUM_CONS_CPY(ScoreBIC);
87 }
88
90 ScoreBIC::ScoreBIC(ScoreBIC&& from) :
91 Score(std::move(from)), _internal_prior_(std::move(from._internal_prior_)) {
92 GUM_CONS_MOV(ScoreBIC);
93 }
94
96 ScoreBIC::~ScoreBIC() { GUM_DESTRUCTOR(ScoreBIC); }
97
99 ScoreBIC& ScoreBIC::operator=(const ScoreBIC& from) {
100 if (this != &from) {
101 Score::operator=(from);
102 _internal_prior_ = from._internal_prior_;
103 }
104 return *this;
105 }
106
108 ScoreBIC& ScoreBIC::operator=(ScoreBIC&& from) {
109 if (this != &from) {
110 Score::operator=(std::move(from));
111 _internal_prior_ = std::move(from._internal_prior_);
112 }
113 return *this;
114 }
115
117 std::string ScoreBIC::isPriorCompatible(PriorType prior_type, double weight) {
118 // check that the prior is compatible with the score
119 if ((prior_type == PriorType::DirichletPriorType)
120 || (prior_type == PriorType::SmoothingPriorType)
121 || (prior_type == PriorType::NoPriorType)) {
122 return "";
123 }
124
125 // prior types unsupported by the type checker
126 return std::format("The prior '{}' is not yet compatible with the score 'BIC'.",
127 priorTypeToString(prior_type));
128 }
129
131 double ScoreBIC::score_(const IdCondSet& idset) {
132 // get the counts for all the nodes in the idset and add the prior
133 std::vector< double > N_ijk(this->counter_.counts(idset, true));
134 const bool informative_external_prior = this->prior_->isInformative();
135 if (informative_external_prior) this->prior_->addJointPseudoCount(idset, N_ijk);
136 const std::size_t all_size = N_ijk.size();
137
138 // here, we distinguish idsets with conditioning nodes from those
139 // without conditioning nodes
140 if (idset.hasConditioningSet()) {
141 // get the counts for the conditioning nodes
142 std::vector< double > N_ij(this->marginalize_(idset[0], N_ijk));
143 const std::size_t conditioning_size = N_ij.size();
144
145 // initialize the score: this should be the penalty of the BIC score,
146 // i.e., -(ri-1 ) * qi * .5 * log ( N + N' )
147 const std::size_t target_domsize = all_size / conditioning_size;
148 const double penalty = conditioning_size * double(target_domsize - std::size_t(1));
149
150 // compute the score: it remains to compute the log likelihood, i.e.,
151 // sum_k=1^r_i sum_j=1^q_i N_ijk log (N_ijk / N_ij), which is also
152 // equivalent to:
153 // sum_j=1^q_i sum_k=1^r_i N_ijk log N_ijk - sum_j=1^q_i N_ij log N_ij
154 double score = 0.0;
155 for (const auto n_ijk: N_ijk) {
156 if (n_ijk) { score += n_ijk * std::log(n_ijk); }
157 }
158 double N = 0;
159 for (const auto n_ij: N_ij) {
160 if (n_ij) {
161 score -= n_ij * std::log(n_ij);
162 N += n_ij;
163 }
164 }
165
166 // finally, remove the penalty
167 score -= penalty * std::log(N) * 0.5;
168
169 // divide by log(2), since the log likelihood uses log_2
170 score *= this->one_log2_;
171
172 return score;
173 } else {
174 // here, there are no conditioning nodes
175
176 // initialize the score: this should be the penalty of the BIC score,
177 // i.e., -(ri-1 )
178 const double penalty = double(all_size - std::size_t(1));
179
180 // compute the score: it remains to compute the log likelihood, i.e.,
181 // sum_k=1^r_i N_ijk log (N_ijk / N), which is also
182 // equivalent to:
183 // sum_j=1^q_i sum_k=1^r_i N_ijk log N_ijk - N log N
184 double N = 0.0;
185 double score = 0.0;
186 for (const auto n_ijk: N_ijk) {
187 if (n_ijk) {
188 score += n_ijk * std::log(n_ijk);
189 N += n_ijk;
190 }
191 }
192 score -= N * std::log(N);
193
194 // finally, remove the penalty
195 score -= penalty * std::log(N) * 0.5;
196
197 // divide by log(2), since the log likelihood uses log_2
198 score *= this->one_log2_;
199
200 return score;
201 }
202 }
203
205 double ScoreBIC::N(const IdCondSet& idset) {
206 // get the counts for all the nodes in the idset and add the prior
207 std::vector< double > N_ijk(this->counter_.counts(idset, true));
208 if (this->prior_->isInformative()) this->prior_->addJointPseudoCount(idset, N_ijk);
209
210 double N = 0;
211 for (const auto n_ijk: N_ijk) {
212 N += n_ijk;
213 }
214
215 return N;
216 }
217
218 } /* namespace learning */
219
220} /* namespace gum */
221
222#endif /* DOXYGEN_SHOULD_SKIP_THIS */
the class used to read a row in the database and to transform it into a set of DBRow instances that c...
the base class for all a priori
Definition prior.h:84
ScoreBIC(const DBRowGeneratorParser &parser, const Prior &prior, const std::vector< std::pair< std::size_t, std::size_t > > &ranges, const Bijection< NodeId, std::size_t > &nodeId2columns=Bijection< NodeId, std::size_t >())
default constructor
The base class for all the scores used for learning (BIC, BDeu, etc).
Definition score.h:68
include the inlined functions if necessary
Definition CSVParser.h:55
constexpr const char * priorTypeToString(PriorType e) noexcept
Definition prior.h:69
gum is the global namespace for all aGrUM entities
Definition agrum.h:46
STL namespace.
the class for computing BIC scores
the class for computing BIC scores