14from dataclasses
import dataclass
16import matplotlib.pyplot
as plt
20from pandas.errors
import PerformanceWarning
23 "This module will soon be deprecated and eventually removed in a future release. "
24 "Its functionality is being taken over by the standalone SysVar package: "
25 "https://gitlab.desy.de/belle2/software/sysvar",
31A module that adds corrections to analysis dataframe.
32It adds weight variations according to the total uncertainty for easier error propagation.
35_weight_cols = [
'data_MC_ratio',
36 'data_MC_uncertainty_stat_dn',
37 'data_MC_uncertainty_stat_up',
38 'data_MC_uncertainty_sys_dn',
39 'data_MC_uncertainty_sys_up'
42_correction_types = [
'PID',
'FEI']
43_fei_mode_col =
'dec_mode'
47_seed_key_cols = [
'PDG',
'mcPDG',
'variable',
'threshold']
50def _table_key(table: pd.DataFrame) -> str:
52 Returns a stable identifier for a weight table, independent of the particle prefix.
54 cols = [col
for col
in _seed_key_cols
if col
in table.columns]
57 f
'Cannot derive a seed from the weight table: it has none of the identifying '
58 f
'columns {_seed_key_cols}, so distinct tables could not be told apart and would '
59 f
'share their variations.'
61 ident = table[cols].drop_duplicates().sort_values(cols)
62 digest = pd.util.hash_pandas_object(ident, index=
False).values.tobytes()
63 return hashlib.sha256(digest).hexdigest()
66def _derive_seed(base_seed: int, key: str) -> int:
68 Derives a reproducible 32-bit seed from a base seed and a key.
70 return int.from_bytes(hashlib.sha256(f
'{base_seed}:{key}'.encode()).digest()[:4],
'big')
73def _check_seeds(sys_seed: int, seed: int) ->
None:
75 Rejects ambiguous seed arguments and warns about the behaviour of sys_seed.
80 raise ValueError(
'Pass either seed or sys_seed, not both.')
82 'sys_seed only seeds the systematic variations; the statistical variations are '
83 'drawn anew on every call, so applying the same weight table to several dataframes '
84 'separately underestimates their contribution. Sharing sys_seed between weight '
85 'tables also correlates the systematic variations of tables with the same number of '
86 'rows. Use seed instead, which makes both components reproducible and independent '
96 Class that stores the information of a particle.
106 merged_table: pd.DataFrame
114 variable_aliases: dict
121 column_names: list =
None
127 cov: np.ndarray =
None
133 coverage: float =
None
136 plot_values: dict =
None
143 Returns the variable name with the prefix and use alias if defined.
150 return f
'{self.prefix}{name}'
154 Returns the list of variables that are used for the binning
156 variables = set(sum([list(d.keys())
for d
in self.
pdg_binning.values()], []))
157 return [f
'{self.get_varname(var)}' for var
in variables]
161 Returns the list of variables that are used for the PDG codes
166 pdg_vars += [
'mcPDG']
167 return [f
'{self.get_varname(var)}' for var
in pdg_vars]
171 rho_sys: np.ndarray =
None,
172 rho_stat: np.ndarray =
None) ->
None:
174 Generates variations of weights according to the uncertainties
177 "data_MC_uncertainty_stat_dn"]].max(axis=1)
179 "data_MC_uncertainty_sys_dn"]].max(axis=1)
183 self.
column_names = [f
"{self.weight_name}_{i}" for i
in range(n_variations)]
185 weights = cov + means
192 Returns the seed for one component of the variations: 'sys', 'stat', or 'total'
193 for the combined draw when cov is set.
195 Derived from seed, the table identity and the component, so a table gives the
196 same variations in every dataframe it is applied to, while different tables and
197 components are independent.
199 if self.
seed is None:
201 return _derive_seed(self.
seed, f
'{component}:{_table_key(self.merged_table)}')
205 rho_sys: np.ndarray =
None,
206 rho_stat: np.ndarray =
None) -> np.ndarray:
208 Returns the covariance matrix of the weights
211 zeros = np.zeros(len_means)
215 rho_sys = np.ones((len_means, len_means))
217 rho_sys = np.identity(len_means)
219 rho_stat = np.identity(len_means)
223 stat_cov = np.matmul(
226 if self.
seed is not None:
228 sys = np.random.default_rng(self.
get_seed(
'sys')).multivariate_normal(zeros, sys_cov, n_variations)
229 stat = np.random.default_rng(self.
get_seed(
'stat')).multivariate_normal(zeros, stat_cov, n_variations)
233 sys = np.random.multivariate_normal(zeros, sys_cov, n_variations)
235 stat = np.random.multivariate_normal(zeros, stat_cov, n_variations)
238 if self.
seed is not None:
239 return np.random.default_rng(self.
get_seed(
'total')).multivariate_normal(zeros, self.
cov, n_variations)
240 errors = np.random.multivariate_normal(zeros, self.
cov, n_variations)
245 Converts the object to a string.
247 separator =
'------------------'
248 title =
'ReweighterParticle'
249 prefix_str = f
'Type: {self.type} Prefix: {self.prefix}'
250 columns = _weight_cols
251 merged_table_str = f
'Merged table:\n{self.merged_table[columns].describe()}'
252 pdg_binning_str =
'PDG binning:\n'
254 pdg_binning_str += f
'{pdgs}: {self.pdg_binning[pdgs]}\n'
255 return '\n'.join([separator, title, prefix_str, merged_table_str, pdg_binning_str]) + separator
259 Plots the coverage of the ntuple.
263 vars = set(sum([list(d.keys())
for d
in self.
plot_values.values()], []))
265 fig, axs = plt.subplots(len(self.
plot_values), len(vars), figsize=(5 * len(vars), 3 * len(self.
plot_values)), dpi=120)
267 if len(axs.shape) < 2:
268 axs = axs.reshape(len(self.
plot_values), len(vars))
269 bin_plt = {
'linewidth': 3,
'linestyle':
'--',
'color':
'0.5'}
270 fig.suptitle(f
'{self.type} particle {self.prefix.strip("_")}')
271 for (reco_pdg, mc_pdg), ax_row
in zip(self.
plot_values, axs):
272 for var, ax
in zip(self.
plot_values[(reco_pdg, mc_pdg)], ax_row):
274 ymax = self.
plot_values[(reco_pdg, mc_pdg)][var][1].max() * 1.1
276 if self.
type ==
'PID':
277 ax.vlines(self.
pdg_binning[(reco_pdg, mc_pdg)][var], ymin, ymax,
281 elif self.
type ==
'FEI':
282 values = np.array([int(val[4:])
for val
in self.
pdg_binning[(reco_pdg, mc_pdg)][var]])
284 np.ones(len(values)) * ymax,
289 rest = np.setdiff1d(self.
plot_values[(reco_pdg, mc_pdg)][var][0], values)
291 np.ones(len(rest)) * ymax,
294 label=
'Rest category',
297 widths = (self.
plot_values[(reco_pdg, mc_pdg)][var][0][1:] - self.
plot_values[(reco_pdg, mc_pdg)][var][0][:-1])
298 centers = self.
plot_values[(reco_pdg, mc_pdg)][var][0][:-1] + widths / 2
304 ax.set_title(f
'True {pdg.to_name(mc_pdg)} to reco {pdg.to_name(reco_pdg)} coverage')
313 Class that reweights the dataframe.
316 n_variations (int): Number of weight variations to generate.
317 weight_name (str): Name of the weight column.
318 evaluate_plots (bool): Flag to indicate if the plots should be evaluated.
319 nbins (int): Number of bins for the plots.
323 n_variations: int = 100,
324 weight_name: str =
"Weight",
325 evaluate_plots: bool =
True,
327 fillna: float = 1.0) ->
None:
329 Initializes the Reweighter class.
350 Returns the kinematic bin columns of the dataframe.
352 return [col
for col
in weight_df.columns
if col.endswith(
'_min')
or col.endswith(
'_max')]
356 Returns the kinematic binning of the dataframe.
359 var_names = {
'_'.join(col.split(
'_')[:-1])
for col
in columns}
361 for var_name
in var_names:
362 bin_dict[var_name] = []
364 if col.startswith(var_name):
365 bin_dict[var_name] += list(weight_df[col].values)
366 bin_dict[var_name] = np.array(sorted(set(bin_dict[var_name])))
371 Returns the irregular binning of the dataframe.
373 return {_fei_mode_col: weight_df.loc[weight_df[_fei_mode_col].str.startswith(
'mode'),
374 _fei_mode_col].value_counts().index.to_list()}
377 ntuple_df: pd.DataFrame,
378 particle: ReweighterParticle) ->
None:
380 Checks if the variables are in the ntuple and returns them.
383 ntuple_df (pandas.DataFrame): Dataframe containing the analysis ntuple.
384 particle (ReweighterParticle): Particle object containing the necessary variables.
386 ntuple_variables = particle.get_binning_variables()
387 ntuple_variables += particle.get_pdg_variables()
388 for var
in ntuple_variables:
389 if var
not in ntuple_df.columns:
390 raise ValueError(f
'Variable {var} is not in the ntuple! Required variables are {ntuple_variables}')
391 return ntuple_variables
395 pdg_pid_variable_dict: dict) -> pd.DataFrame:
397 Merges the efficiency and fake rate weight tables.
400 weights_dict (dict): Dictionary containing the weight tables.
401 pdg_pid_variable_dict (dict): Dictionary containing the PDG codes and variable names.
404 for reco_pdg, mc_pdg
in weights_dict:
405 if reco_pdg
not in pdg_pid_variable_dict:
406 raise ValueError(f
'Reconstructed PDG code {reco_pdg} not found in thresholds!')
407 weight_df = weights_dict[(reco_pdg, mc_pdg)]
408 weight_df[
'mcPDG'] = mc_pdg
409 weight_df[
'PDG'] = reco_pdg
411 if 'charge' in weight_df.columns:
412 charge_dict = {
'+': [0, 2],
'-': [-2, 0]}
413 weight_df[[
'charge_min',
'charge_max']] = [charge_dict[val]
for val
in weight_df[
'charge'].values]
414 weight_df = weight_df.drop(columns=[
'charge'])
416 if 'iso_score_min' in weight_df.columns
and len(weight_df[
'iso_score_min'].unique()) == 1:
417 weight_df = weight_df.drop(columns=[
'iso_score_min',
'iso_score_max'])
418 pid_variable_name = pdg_pid_variable_dict[reco_pdg][0]
419 threshold = pdg_pid_variable_dict[reco_pdg][1]
420 selected_weights = weight_df.query(f
'variable == "{pid_variable_name}" and threshold == {threshold}')
421 if len(selected_weights) == 0:
422 available_variables = weight_df[
'variable'].unique()
423 available_thresholds = weight_df[
'threshold'].unique()
424 raise ValueError(f
'No weights found for PDG code {reco_pdg}, mcPDG {mc_pdg},'
425 f
' variable {pid_variable_name} and threshold {threshold}!\n'
426 f
' Available variables: {available_variables}\n'
427 f
' Available thresholds: {available_thresholds}')
428 weight_dfs.append(selected_weights)
429 return pd.concat(weight_dfs, ignore_index=
True)
432 ntuple_df: pd.DataFrame,
433 particle: ReweighterParticle) ->
None:
435 Adds a weight and uncertainty columns to the dataframe.
438 ntuple_df (pandas.DataFrame): Dataframe containing the analysis ntuple.
439 particle (ReweighterParticle): Particle object.
442 binning_df = pd.DataFrame(index=ntuple_df.index)
444 binning_df[
'mcPDG'] = ntuple_df[f
'{particle.get_varname("mcPDG")}'].abs()
445 binning_df[
'PDG'] = ntuple_df[f
'{particle.get_varname("PDG")}'].abs()
447 for reco_pdg, mc_pdg
in particle.pdg_binning:
448 ntuple_cut = f
'abs({particle.get_varname("mcPDG")}) == {mc_pdg} and abs({particle.get_varname("PDG")}) == {reco_pdg}'
449 if ntuple_df.query(ntuple_cut).empty:
451 plot_values[(reco_pdg, mc_pdg)] = {}
452 for var
in particle.pdg_binning[(reco_pdg, mc_pdg)]:
453 labels = [(particle.pdg_binning[(reco_pdg, mc_pdg)][var][i - 1], particle.pdg_binning[(reco_pdg, mc_pdg)][var][i])
454 for i
in range(1, len(particle.pdg_binning[(reco_pdg, mc_pdg)][var]))]
455 binning_df.loc[(binning_df[
'mcPDG'] == mc_pdg) & (binning_df[
'PDG'] == reco_pdg), var] = pd.cut(ntuple_df.query(
456 ntuple_cut)[f
'{particle.get_varname(var)}'],
457 particle.pdg_binning[(reco_pdg, mc_pdg)][var], labels=labels)
458 binning_df.loc[(binning_df[
'mcPDG'] == mc_pdg) & (binning_df[
'PDG'] == reco_pdg),
459 f
'{var}_min'] = binning_df.loc[(binning_df[
'mcPDG'] == mc_pdg) & (binning_df[
'PDG'] == reco_pdg),
461 binning_df.loc[(binning_df[
'mcPDG'] == mc_pdg) & (binning_df[
'PDG'] == reco_pdg),
462 f
'{var}_max'] = binning_df.loc[(binning_df[
'mcPDG'] == mc_pdg) & (binning_df[
'PDG'] == reco_pdg),
464 binning_df.drop(var, axis=1, inplace=
True)
466 values = ntuple_df.query(ntuple_cut)[f
'{particle.get_varname(var)}']
467 if len(values.unique()) < 2:
468 print(f
'Skip {var} for plotting!')
470 x_range = np.linspace(values.min(), values.max(), self.
nbins)
471 plot_values[(reco_pdg, mc_pdg)][var] = x_range, np.histogram(values, bins=x_range, density=
True)[0]
473 weight_cols = _weight_cols
474 if particle.column_names:
475 weight_cols = particle.column_names
476 binning_df = binning_df.merge(particle.merged_table[weight_cols + binning_df.columns.tolist()],
477 on=binning_df.columns.tolist(), how=
'left')
478 binning_df.index = ntuple_df.index
479 particle.coverage = 1 - binning_df[weight_cols[0]].isna().sum() / len(binning_df)
480 particle.plot_values = plot_values
481 for col
in weight_cols:
482 ntuple_df[f
'{particle.get_varname(col)}'] = binning_df[col]
483 ntuple_df[f
'{particle.get_varname(col)}'] = ntuple_df[f
'{particle.get_varname(col)}'].
fillna(self.
fillna)
488 pdg_pid_variable_dict: dict,
489 variable_aliases: dict =
None,
490 sys_seed: int =
None,
491 syscorr: bool =
True,
492 seed: int =
None) ->
None:
494 Adds weight variations according to the total uncertainty for easier error propagation.
497 prefix (str): Prefix for the new columns.
498 weights_dict (pandas.DataFrame): Dataframe containing the efficiency weights.
499 pdg_pid_variable_dict (dict): Dictionary containing the PID variables and thresholds.
500 variable_aliases (dict): Dictionary containing variable aliases.
501 sys_seed (int): Seed for the systematic variations only. Prefer seed.
502 syscorr (bool): When true assume systematics are 100% correlated defaults to
503 true. Note this is overridden by provision of a None value rho_sys
504 seed (int): Base seed for the systematic and statistical variations. Makes
505 them reproducible for a given weight table and independent between tables.
507 _check_seeds(sys_seed, seed)
512 if prefix
and not prefix.endswith(
'_'):
515 raise ValueError(f
"Particle with prefix '{prefix}' already exists!")
516 if variable_aliases
is None:
517 variable_aliases = {}
519 pdg_binning = {(reco_pdg, mc_pdg): self.
get_binning(merged_weight_df.query(f
'PDG == {reco_pdg} and mcPDG == {mc_pdg}'))
520 for reco_pdg, mc_pdg
in merged_weight_df[[
'PDG',
'mcPDG']].value_counts().index.to_list()}
523 merged_table=merged_weight_df,
524 pdg_binning=pdg_binning,
525 variable_aliases=variable_aliases,
534 Get a particle by its prefix.
536 cands = [particle
for particle
in self.
particles if particle.prefix.strip(
'_') == prefix.strip(
'_')]
543 Checks if the tables are provided in a legacy format and converts them to the standard format.
546 str_to_pdg = {
'B+': 521,
'B-': 521,
'B0': 511}
547 if 'cal' in table.columns:
548 result = pd.DataFrame(index=table.index)
549 result[
'data_MC_ratio'] = table[
'cal']
550 result[
'PDG'] = table[
'Btag'].apply(
lambda x: str_to_pdg.get(x))
552 result[
'mcPDG'] = result[
'PDG']
553 result[
'threshold'] = table[
'sig_prob_threshold']
554 result[_fei_mode_col] = table[_fei_mode_col]
555 result[
'data_MC_uncertainty_stat_dn'] = table[
'cal_stat_error']
556 result[
'data_MC_uncertainty_stat_up'] = table[
'cal_stat_error']
557 result[
'data_MC_uncertainty_sys_dn'] = table[
'cal_sys_error']
558 result[
'data_MC_uncertainty_sys_up'] = table[
'cal_sys_error']
559 elif 'cal factor' in table.columns:
560 result = pd.DataFrame(index=table.index)
561 result[
'data_MC_ratio'] = table[
'cal factor']
562 result[
'PDG'] = table[
'Btype'].apply(
lambda x: str_to_pdg.get(x))
563 result[
'mcPDG'] = result[
'PDG']
564 result[
'threshold'] = table[
'sig prob cut']
566 result[
'data_MC_uncertainty_stat_dn'] = table[
'error']
567 result[
'data_MC_uncertainty_stat_up'] = table[
'error']
568 result[
'data_MC_uncertainty_sys_dn'] = 0
569 result[
'data_MC_uncertainty_sys_up'] = 0
570 result[_fei_mode_col] = table[
'mode']
571 elif 'dmID' in table.columns:
572 result = pd.DataFrame(index=table.index)
573 result[
'data_MC_ratio'] = table[
'central_value']
574 result[
'PDG'] = table[
'PDG'].apply(ast.literal_eval).str[0].abs()
575 result[
'mcPDG'] = result[
'PDG']
576 result[
'threshold'] = table[
'sigProb']
578 result[
'data_MC_uncertainty_stat_dn'] = table[
'total_error']
579 result[
'data_MC_uncertainty_stat_up'] = table[
'total_error']
580 result[
'data_MC_uncertainty_sys_dn'] = 0
581 result[
'data_MC_uncertainty_sys_up'] = 0
582 result[_fei_mode_col] = (
'mode' + table[
'dmID'].astype(str)).replace(
'mode999',
'rest')
585 result = result.query(f
'threshold == {threshold}')
587 raise ValueError(f
'No weights found for threshold {threshold}!')
593 cov: np.ndarray =
None,
594 variable_aliases: dict =
None,
598 Adds weight variations according to the total uncertainty for easier error propagation.
601 prefix (str): Prefix for the new columns.
602 table (pandas.DataFrame): Dataframe containing the efficiency weights.
603 threshold (float): Threshold for the efficiency weights.
604 cov (numpy.ndarray): Covariance matrix for the efficiency weights.
605 variable_aliases (dict): Dictionary containing variable aliases.
606 seed (int): Base seed for the variations, see :meth:`add_pid_particle`.
611 if prefix
and not prefix.endswith(
'_'):
614 raise ValueError(f
"Particle with prefix '{prefix}' already exists!")
615 if variable_aliases
is None:
616 variable_aliases = {}
617 if table
is None or len(table) == 0:
618 raise ValueError(
'No weights provided!')
620 pdg_binning = {(reco_pdg, mc_pdg): self.
get_fei_binning(converted_table.query(f
'PDG == {reco_pdg} and mcPDG == {mc_pdg}'))
621 for reco_pdg, mc_pdg
in converted_table[[
'PDG',
'mcPDG']].value_counts().index.to_list()}
624 merged_table=converted_table,
625 pdg_binning=pdg_binning,
626 variable_aliases=variable_aliases,
634 Adds weight columns according to the FEI calibration tables
637 particle.merged_table[_fei_mode_col]
639 binning_df = pd.DataFrame(index=ntuple_df.index)
641 binning_df[
'PDG'] = ntuple_df[f
'{particle.get_varname("PDG")}'].abs()
643 binning_df[
'num_mode'] = ntuple_df[particle.get_varname(_fei_mode_col)].astype(int)
646 binning_df[_fei_mode_col] = pd.Series(np.nan, index=binning_df.index, dtype=
'object')
648 for reco_pdg, mc_pdg
in particle.pdg_binning:
649 plot_values[(reco_pdg, mc_pdg)] = {}
650 binning_df.loc[binning_df[
'PDG'] == reco_pdg, _fei_mode_col] = particle.merged_table.query(
651 f
'PDG == {reco_pdg} and {_fei_mode_col}.str.lower() == "{rest_str}"')[_fei_mode_col].values[0]
652 for mode
in particle.pdg_binning[(reco_pdg, mc_pdg)][_fei_mode_col]:
653 binning_df.loc[(binning_df[
'PDG'] == reco_pdg) & (binning_df[
'num_mode'] == int(mode[4:])), _fei_mode_col] = mode
655 values = ntuple_df[f
'{particle.get_varname(_fei_mode_col)}']
656 x_range = np.linspace(values.min(), values.max(), int(values.max()) + 1)
657 plot_values[(reco_pdg, mc_pdg)][_fei_mode_col] = x_range, np.histogram(values, bins=x_range, density=
True)[0]
660 weight_cols = _weight_cols
661 if particle.column_names:
662 weight_cols = particle.column_names
663 binning_df = binning_df.merge(particle.merged_table[weight_cols + [
'PDG', _fei_mode_col]],
664 on=[
'PDG', _fei_mode_col], how=
'left')
665 binning_df.index = ntuple_df.index
666 particle.coverage = 1 - binning_df[weight_cols[0]].isna().sum() / len(binning_df)
667 particle.plot_values = plot_values
668 for col
in weight_cols:
669 ntuple_df[f
'{particle.get_varname(col)}'] = binning_df[col]
673 generate_variations: bool =
True):
675 Reweights the dataframe according to the weight tables.
678 df (pandas.DataFrame): Dataframe containing the analysis ntuple.
679 generate_variations (bool): When true generate weight variations.
682 if particle.type
not in _correction_types:
683 raise ValueError(f
'Particle type {particle.type} not supported!')
684 print(f
'Required variables: {self.get_ntuple_variables(df, particle)}')
685 if generate_variations:
686 particle.generate_variations(n_variations=self.
n_variations)
687 if particle.type ==
'PID':
689 elif particle.type ==
'FEI':
695 Prints the coverage of each particle.
699 print(f
'{particle.type} {particle.prefix.strip("_")}: {particle.coverage*100:0.1f}%')
703 Plots the coverage of each particle.
706 particle.plot_coverage()
709def add_weights_to_dataframe(prefix: str,
712 custom_tables: dict =
None,
713 custom_thresholds: dict =
None,
714 **kw_args) -> pd.DataFrame:
716 Helper method that adds weights to a dataframe.
719 prefix (str): Prefix for the new columns.
720 df (pandas.DataFrame): Dataframe containing the analysis ntuple.
721 systematic (str): Type of the systematic corrections, options: "custom_PID" and "custom_FEI".
722 MC_production (str): Name of the MC production.
723 custom_tables (dict): Dictionary containing the custom efficiency weights.
724 custom_thresholds (dict): Dictionary containing the custom thresholds for the custom efficiency weights.
725 n_variations (int): Number of variations to generate.
726 generate_variations (bool): When true generate weight variations.
727 weight_name (str): Name of the weight column.
728 show_plots (bool): When true show the coverage plots.
729 variable_aliases (dict): Dictionary containing variable aliases.
730 cov_matrix (numpy.ndarray): Covariance matrix for the custom efficiency weights.
731 fillna (int): Value to fill NaN values with.
732 sys_seed (int): Seed for the systematic variations only, custom_PID only. Prefer seed.
733 seed (int): Base seed for the variations, reproducible per weight table and independent between tables.
734 syscorr (bool): When true assume systematics are 100% correlated defaults to true.
735 **kw_args: Additional arguments for the Reweighter class.
737 generate_variations = kw_args.get(
'generate_variations',
True)
738 n_variations = kw_args.get(
'n_variations', 100)
739 weight_name = kw_args.get(
'weight_name',
"Weight")
740 fillna = kw_args.get(
'fillna', 1.0)
742 with warnings.catch_warnings():
743 warnings.simplefilter(
"ignore", category=PerformanceWarning)
744 reweighter =
Reweighter(n_variations=n_variations,
745 weight_name=weight_name,
747 variable_aliases = kw_args.get(
'variable_aliases')
748 seed = kw_args.get(
'seed')
749 if systematic.lower() ==
'custom_fei':
750 if kw_args.get(
'sys_seed')
is not None:
751 warnings.warn(
'sys_seed has no effect for custom_FEI. Use seed instead.', UserWarning, stacklevel=2)
752 cov_matrix = kw_args.get(
'cov_matrix')
753 reweighter.add_fei_particle(prefix=prefix,
755 threshold=custom_thresholds,
756 variable_aliases=variable_aliases,
760 elif systematic.lower() ==
'custom_pid':
761 sys_seed = kw_args.get(
'sys_seed')
762 syscorr = kw_args.get(
'syscorr')
765 reweighter.add_pid_particle(prefix=prefix,
766 weights_dict=custom_tables,
767 pdg_pid_variable_dict=custom_thresholds,
768 variable_aliases=variable_aliases,
774 raise ValueError(f
'Systematic {systematic} is not supported!')
776 result = reweighter.reweight(df, generate_variations=generate_variations)
777 if kw_args.get(
'show_plots'):
778 reweighter.print_coverage()
779 reweighter.plot_coverage()
str type
Add the mcPDG code requirement for PID particle.
variable_aliases
Variable aliases of the weight table.
pd merged_table
Type of the particle (PID or FEI)
int sys_seed
Random seed for systematics only (legacy, prefer seed)
str get_varname(self, str varname)
weight_name
Weight column name that will be added to the ntuple.
np cov
Covariance matrix corresponds to the total uncertainty.
plot_coverage(self, fig=None, axs=None)
list get_pdg_variables(self)
list get_binning_variables(self)
pdg_binning
Kinematic binning of the weight table per particle.
np.ndarray get_covariance(self, int n_variations, np.ndarray rho_sys=None, np.ndarray rho_stat=None)
int seed
Base seed for all variations, see get_seed.
bool syscorr
When true assume systematics are 100% correlated.
int get_seed(self, str component)
prefix
Prefix of the particle in the ntuple.
dict plot_values
Values for the plots.
None generate_variations(self, int n_variations, np.ndarray rho_sys=None, np.ndarray rho_stat=None)
list column_names
Internal list of the names of the weight columns.
n_variations
Number of weight variations to generate.
pd.DataFrame merge_pid_weight_tables(self, dict weights_dict, dict pdg_pid_variable_dict)
nbins
Number of bins for the plots.
list particles
List of particles.
add_fei_weight_columns(self, pd.DataFrame ntuple_df, ReweighterParticle particle)
None __init__(self, int n_variations=100, str weight_name="Weight", bool evaluate_plots=True, int nbins=50, float fillna=1.0)
None add_pid_weight_columns(self, pd.DataFrame ntuple_df, ReweighterParticle particle)
weight_name
Name of the weight column.
None get_ntuple_variables(self, pd.DataFrame ntuple_df, ReweighterParticle particle)
None add_fei_particle(self, str prefix, pd.DataFrame table, float threshold, np.ndarray cov=None, dict variable_aliases=None, int seed=None)
None add_pid_particle(self, str prefix, dict weights_dict, dict pdg_pid_variable_dict, dict variable_aliases=None, int sys_seed=None, bool syscorr=True, int seed=None)
fillna
Value to fill NaN values.
dict get_fei_binning(self, weight_df)
convert_fei_table(self, pd.DataFrame table, float threshold)
bool weights_generated
Flag to indicate if the weights have been generated.
dict get_binning(self, weight_df)
list get_bin_columns(self, weight_df)
reweight(self, pd.DataFrame df, bool generate_variations=True)
ReweighterParticle get_particle(self, str prefix)
list correlations
Correlations between the particles.
evaluate_plots
Flag to indicate if the plots should be evaluated.