diff --git a/law.cfg b/law.cfg index 8e98fcc..dd72711 100644 --- a/law.cfg +++ b/law.cfg @@ -22,6 +22,7 @@ topsf.tasks.inference_v2.post_fit_shapes topsf.tasks.inference_v2.impacts topsf.tasks.inference_v2.plot_impacts topsf.tasks.inference_v2.plot_shapes +topsf.tasks.corrections [logging] @@ -35,7 +36,7 @@ columnflow.columnar_util-perf: INFO default_analysis: topsf.config.run3.analysis_sf.analysis_sf default_config: run3_sf_2022_postEE_nano_v12 -default_dataset: tt_fh_powheg +default_dataset: tt_sl_powheg run3_analysis: topsf.config.run3.analysis_sf.analysis_sf run3_config: run3_sf_2022_preEE_nano_v12 @@ -44,6 +45,7 @@ default_keep_reduced_events: True production_modules: columnflow.production.{categories,normalization,mc_weight,pileup,processes,seeds}, columnflow.production.cms.{btag,electron,mc_weight,muon,pdf,pileup,scale,seeds}, topsf.production.{default,gen_top} calibration_modules: columnflow.calibration.cms.{jets,met}, topsf.calibration.{default,skip_jec} selection_modules: columnflow.selection.cms.{json_filter,met_filters}, topsf.selection.{default,categories,jet,bjet,fatjet,lepton,wp} +hist_production_modules: topsf.weights.default ml_modules: columnflow.ml inference_modules: columnflow.inference, topsf.inference.{default,uhh2} @@ -58,7 +60,7 @@ skip_ensure_proxy: False # some remote workflow parameter defaults htcondor_flavor: $CF_HTCONDOR_FLAVOR -htcondor_share_software: False +htcondor_share_software: True slurm_flavor: $CF_SLURM_FLAVOR slurm_partition: $CF_SLURM_PARTITION @@ -81,10 +83,11 @@ log_array_function_runtime: False [outputs] # list of all used file systems -wlcg_file_systems: wlcg_fs, wlcg_fs_cernbox, wlcg_fs_infn_redirector, wlcg_fs_global_redirector +wlcg_file_systems: wlcg_fs, wlcg_fs_desy, wlcg_fs_cernbox, wlcg_fs_desy_store, wlcg_fs_infn_redirector, wlcg_fs_global_redirector # list of file systems used by columnflow.tasks.external.GetDatasetLFNs.iter_nano_files to # look for the correct fs per nano input file (in that order) +# lfn_sources: local_desy_dcache, wlcg_fs_desy_store, wlcg_fs_infn_redirector, wlcg_fs_global_redirector lfn_sources: local_desy_dcache, wlcg_fs_infn_redirector, wlcg_fs_global_redirector # output locations per task family @@ -122,6 +125,7 @@ task_cf.CreateDatacards: local job_file_dir: $CF_JOB_BASE job_file_dir_cleanup: False +job_file_dir_mkdtemp: sub_{{task_id}}_XXX [local_fs] @@ -139,6 +143,7 @@ base: /pnfs/desy.de/cms/tier2 # set this to your desired location, e.g.: # base: root://eosuser.cern.ch/eos/user/$CF_CERN_USER_FIRSTCHAR/$CF_CERN_USER/$CF_STORE_NAME base: &::wlcg_fs_desy::base +base_mkdir_rec: &::wlcg_fs_desy::webdav_base create_file_dir: True use_cache: $CF_WLCG_USE_CACHE cache_root: $CF_WLCG_CACHE_ROOT @@ -152,6 +157,23 @@ xrootd_base: root://dcache-cms-xrootd.desy.de:1094/pnfs/desy.de/cms/tier2/store/ gsiftp_base: gsiftp://dcache-cms-gridftp.desy.de/pnfs/desy.de/cms/tier2/store/user/$CF_CERN_USER/$CF_STORE_NAME webdav_base: davs://dcache-cms-webdav-wan.desy.de:2880/pnfs/desy.de/cms/tier2/store/user/$CF_CERN_USER/$CF_STORE_NAME base: &::webdav_base +base_filecopy: &::webdav_base +base_stat: &::webdav_base + + +[wlcg_fs_desy_store] + +xrootd_base: root://dcache-cms-xrootd.desy.de:1094/pnfs/desy.de/cms/tier2 +gsiftp_base: gsiftp://dcache-door-cms04.desy.de:2811/pnfs/desy.de/cms/tier2 +webdav_base: davs://dcache-cms-webdav-wan.desy.de:2880/pnfs/desy.de/cms/tier2 +base: &::webdav_base +use_cache: $CF_WLCG_USE_CACHE +cache_root: $CF_WLCG_CACHE_ROOT +cache_cleanup: $CF_WLCG_CACHE_CLEANUP +cache_max_size: 15GB +cache_global_lock: True +cache_mtime_patience: -1 +rucio_report_access: T2_DE_DESY [wlcg_fs_cernbox] @@ -170,6 +192,7 @@ cache_cleanup: $CF_WLCG_CACHE_CLEANUP cache_max_size: 15GB cache_global_lock: True cache_mtime_patience: -1 +rucio_report_access: True [wlcg_fs_global_redirector] @@ -181,6 +204,7 @@ cache_cleanup: $CF_WLCG_CACHE_CLEANUP cache_max_size: 15GB cache_global_lock: True cache_mtime_patience: -1 +rucio_report_access: True [versions] @@ -213,7 +237,7 @@ cache_mtime_patience: -1 # for quick changes of the scheduler local_scheduler: false -scheduler_host: naf-cms12.desy.de +scheduler_host: naf-cms14.desy.de scheduler_port: 8082 [luigi_resources] diff --git a/modules/cmsdb b/modules/cmsdb index 147067b..c4e3c96 160000 --- a/modules/cmsdb +++ b/modules/cmsdb @@ -1 +1 @@ -Subproject commit 147067bb529feed40b5119759dff749697892348 +Subproject commit c4e3c9640ca7945e96dfc9fbb81f919237b954c1 diff --git a/modules/columnflow b/modules/columnflow index 0100f25..f4544ff 160000 --- a/modules/columnflow +++ b/modules/columnflow @@ -1 +1 @@ -Subproject commit 0100f25d6170b4fc985712823e702679b46e2fb7 +Subproject commit f4544ff5b5b5e41752aee1d77c6af431083356d7 diff --git a/topsf/calibration/default.py b/topsf/calibration/default.py index 7fc1d8e..cec07da 100644 --- a/topsf/calibration/default.py +++ b/topsf/calibration/default.py @@ -4,82 +4,218 @@ Calibration methods. """ import functools +import law from columnflow.calibration import Calibrator, calibrator -from columnflow.calibration.cms.jets import jets_ak4, jets_ak8 +from columnflow.calibration.cms.jets import jec_ak4, jer_ak4, jec_ak8, jer_ak8 +from columnflow.calibration.cms.met import met_phi from columnflow.production.cms.mc_weight import mc_weight from columnflow.production.cms.jet import msoftdrop from columnflow.production.cms.seeds import deterministic_seeds from columnflow.util import maybe_import from columnflow.columnar_util import set_ak_column +from columnflow.production.cms.jet import jet_id, fatjet_id from topsf.calibration.jets import ( jet_lepton_cleaner, jec_subjets, - # jer_subjets + jer_subjets ) +from topsf.util import has_tag, record_calls + +logger = law.logger.get_logger(__name__) np = maybe_import("numpy") ak = maybe_import("awkward") set_ak_column_f32 = functools.partial(set_ak_column, value_type=np.float32) +jec_ak4_Puppi = jec_ak4.derive( + "jec_ak4_Puppi", + cls_dict={ + "met_name": "PuppiMET", + "raw_met_name": "RawPuppiMET", + } +) +jer_ak4_Puppi = jer_ak4.derive( + "jer_ak4_Puppi", + cls_dict={ + "met_name": "PuppiMET", + "raw_met_name": "RawPuppiMET", + } +) + +jec_ak8_Puppi = jec_ak8.derive( + "jec_ak8_Puppi", + cls_dict={ + "propagate_met": False, + "met_name": "DO_NOT_USE", + "raw_met_name": "DO_NOT_USE", + }, +) + +jer_ak8_Puppi = jer_ak8.derive( + "jer_ak8_Puppi", + cls_dict={ + "propagate_met": False, + "met_name": "DO_NOT_USE", + "raw_met_name": "DO_NOT_USE", + } +) + +jec_subjets_Puppi = jec_subjets.derive( + "jec_subjets_Puppi", + cls_dict={ + "propagate_met": False, + "met_name": "DO_NOT_USE", + "raw_met_name": "DO_NOT_USE", + } +) + +jer_subjets_Puppi = jer_subjets.derive( + "jer_subjets_Puppi", + cls_dict={ + "propagate_met": False, + "met_name": "DO_NOT_USE", + "raw_met_name": "DO_NOT_USE", + } +) + @calibrator( uses={ mc_weight, deterministic_seeds, jet_lepton_cleaner, - jets_ak4, - jets_ak8, - jec_subjets, - # jer_subjets, msoftdrop, + "Muon.pt", "Muon.tunepRelPt", }, produces={ mc_weight, deterministic_seeds, jet_lepton_cleaner, - jets_ak4, - jets_ak8, - jec_subjets, - # jer_subjets, msoftdrop, + "Muon.pt", "Muon.rawPt", }, ) def default(self: Calibrator, events: ak.Array, **kwargs) -> ak.Array: - events = self[jet_lepton_cleaner](events, **kwargs) # set Jet.pt to raw Pt -> run before jets (JER) calibrator - if self.dataset_inst.is_mc: - events = self[mc_weight](events, **kwargs) + run_list = [] + with record_calls(self, run_list): + # highPt muons: use tuneP pT + events = set_ak_column_f32(events, "Muon.rawPt", events.Muon.pt) + events = set_ak_column_f32(events, "Muon.pt", events.Muon.tunepRelPt * events.Muon.pt) + logger.info_once( + "Finished recalculating muon pt with tuneP for highPt muons. Stored original pt in Muon.rawPt." + ) + if self.dataset_inst.is_mc: + events = self[mc_weight](events, **kwargs) + events = self[deterministic_seeds](events, **kwargs) + + events = self[jet_lepton_cleaner](events, **kwargs) # set Jet.pt to raw Pt -> run before jets (JER) calibrator + # run JEC calibrators for AK4 and AK8 jets + # fake subjet area column by setting it to an array with the same structure as the subjet pt column containing 0.5 + # (needed to be able to use same code as for top-level AK4/AK8 jets, as the producer formally requires an `area` + # column, despite not actually using it) + events = set_ak_column_f32(events, "SubJet.area", 0.5 * ak.ones_like(events.SubJet.pt)) + if self.config_inst.x.year == 2024: + events = self[jec_ak4_Puppi](events, **kwargs) + events = self[jec_ak8_Puppi](events, **kwargs) + events = self[jec_subjets_Puppi](events, **kwargs) + if self.dataset_inst.is_mc: + events = self[jer_ak4_Puppi](events, **kwargs) + events = self[jer_ak8_Puppi](events, **kwargs) + # events = self[jer_subjets_Puppi](events, **kwargs) + else: + events = self[jec_ak4](events, **kwargs) + events = self[jec_ak8](events, **kwargs) + events = self[jec_subjets](events, **kwargs) + if self.dataset_inst.is_mc: + events = self[jer_ak4](events, **kwargs) + events = self[jer_ak8](events, **kwargs) + # events = self[jer_subjets](events, **kwargs) + + events = self[msoftdrop](events, **kwargs) + if self.config_inst.x.year in {2022, 2023}: + events = self[met_phi](events, **kwargs) + elif self.config_inst.x.year == 2024: + logger.warning_once("met_phi calibrator not run for 2024 config, as it is not yet available.") + else: + raise ValueError(f"Unsupported year {self.config_inst.x.year} in default calibrator") + + if not has_tag("skip_jet_ids", self.config_inst, self.dataset_inst, operator=any): + logger.debug("Recalulating (fat)jet IDs.") + events = self[jet_id](events, **kwargs) + events = self[fatjet_id](events, **kwargs) + + logger.info_once( + "Finished default calibration steps:\n" + + "\n".join(run_list) + ) - events = self[jets_ak4](events, **kwargs) # call_force ? - events = self[jets_ak8](events, **kwargs) # call_force ? - - events = self[deterministic_seeds](events, **kwargs) - - # fake subjet area column by setting it to an array with the same structure as the subjet pt column containing 0.5 - # (needed to be able to use same code as for top-level AK4/AK8 jets, as the producer formally requires an `area` - # column, despite not actually using it) - events = set_ak_column_f32(events, "SubJet.area", 0.5 * ak.ones_like(events.SubJet.pt)) + return events - events = self[jec_subjets](events, **kwargs) - # if self.dataset_inst.is_mc: - # events = self[jer_subjets](events, **kwargs) - events = self[msoftdrop](events, **kwargs) - return events +@default.init +def default_init(self: Calibrator) -> None: + # add met_phi only for 2022/23 configs + if self.config_inst.x.year in {2022, 2023}: + self.uses |= { + met_phi, + jec_ak4, + jec_ak8, + jer_ak4, + jer_ak8, + jec_subjets, + jer_subjets, + } + self.produces |= { + met_phi, + jec_ak4, + jec_ak8, + jer_ak4, + jer_ak8, + jec_subjets, + jer_subjets, + } + elif self.config_inst.x.year == 2024: + self.uses |= { + jec_ak4_Puppi, + jec_ak8_Puppi, + jer_ak4_Puppi, + jer_ak8_Puppi, + jec_subjets_Puppi, + jer_subjets_Puppi, + } + self.produces |= { + jec_ak4_Puppi, + jec_ak8_Puppi, + jer_ak4_Puppi, + jer_ak8_Puppi, + jec_subjets_Puppi, + jer_subjets_Puppi, + } + + if not has_tag("skip_jet_ids", self.config_inst, self.dataset_inst, operator=any): + self.uses |= { + jet_id, + fatjet_id, + } + self.produces |= { + jet_id, + fatjet_id, + } @calibrator( - uses={mc_weight, deterministic_seeds, jets_ak4, jets_ak8}, - produces={mc_weight, deterministic_seeds, jets_ak4, jets_ak8}, + uses={mc_weight, deterministic_seeds}, + produces={mc_weight, deterministic_seeds}, ) def no_jet_cleaning(self: Calibrator, events: ak.Array, **kwargs) -> ak.Array: if self.dataset_inst.is_mc: events = self[mc_weight](events, **kwargs) - events = self[jets_ak4](events, **kwargs) - events = self[jets_ak8](events, **kwargs) # call_force ? + # events = self[jets_ak4](events, **kwargs) + # events = self[jets_ak8](events, **kwargs) # call_force ? events = self[deterministic_seeds](events, **kwargs) return events diff --git a/topsf/calibration/jets.py b/topsf/calibration/jets.py index b7e503e..f3d57af 100644 --- a/topsf/calibration/jets.py +++ b/topsf/calibration/jets.py @@ -44,8 +44,10 @@ def jet_lepton_cleaner(self: Calibrator, events: ak.Array, **kwargs) -> ak.Array # revert JEC for jet pt and jet mass, # set correction factor to 0 - events = set_ak_column(events, "Jet.pt", events.Jet.pt * (1 - events.Jet.rawFactor)) - events = set_ak_column(events, "Jet.mass", events.Jet.mass * (1 - events.Jet.rawFactor)) + raw_pt = events.Jet.pt * (1 - events.Jet.rawFactor) + raw_mass = events.Jet.mass * (1 - events.Jet.rawFactor) + events = set_ak_column(events, "Jet.pt", raw_pt) + events = set_ak_column(events, "Jet.mass", raw_mass) events = set_ak_column(events, "Jet.rawFactor", 0) # build jet lorentz vectors @@ -158,6 +160,10 @@ def jet_lepton_cleaner(self: Calibrator, events: ak.Array, **kwargs) -> ak.Array value = ak.fill_none(ak.nan_to_none(getattr(jet_lv, var)), 0.0) events = set_ak_column(events, f"Jet.{var}", value) + # do not set to updated raw factor as out dated JECs are already reverted + # raw_factor = ak.nan_to_num(1 - raw_pt / events.Jet.pt, nan=0.0) + # events = set_ak_column(events, "Jet.rawFactor", raw_factor) + return events diff --git a/topsf/config/categories.py b/topsf/config/categories.py index 167ca5d..b3c5971 100644 --- a/topsf/config/categories.py +++ b/topsf/config/categories.py @@ -184,20 +184,34 @@ def sel_pt( "loose", "very_loose", ] + tau32_bin_names = [ + "below_very_tight", + "very_tight_to_tight", + "tight_to_medium", + "medium_to_loose", + "loose_to_very_loose", + "above_very_loose", + ] + # tau32_bins = [0] + [ + # config.x.toptag_working_points["tau32"][wp] + # for wp in tau32_wps + # ] + [1] tau32_bins = [0] + [ - config.x.toptag_working_points["tau32"][wp] + config.x.toptag_working_points[wp] for wp in tau32_wps ] + [1] tau32_categories = [] - for cat_idx, (tau32_min, tau32_max) in enumerate( - zip(tau32_bins[:-1], tau32_bins[1:]), + for cat_idx, ((tau32_min, tau32_max), bin_name) in enumerate( + # zip(tau32_bins[:-1], tau32_bins[1:]), + zip(zip(tau32_bins[:-1], tau32_bins[1:]), tau32_bin_names) ): tau32_min_repr = f"{int(tau32_min*100):03d}" tau32_max_repr = f"{int(tau32_max*100):03d}" cat_label = rf"{tau32_min} $\leq$ $\tau_{{3}}/\tau_{{2}}$ < {tau32_max}" - cat_name = f"tau32_{tau32_min_repr}_{tau32_max_repr}" + # cat_name = f"tau32_{tau32_min_repr}_{tau32_max_repr}" + cat_name = f"tau32_{bin_name}" sel_name = f"sel_{cat_name}" @categorizer( diff --git a/topsf/config/datasets.py b/topsf/config/datasets.py new file mode 100644 index 0000000..d3b5de8 --- /dev/null +++ b/topsf/config/datasets.py @@ -0,0 +1,98 @@ +# coding: utf-8 + +""" +Configuration of datasets for the m(ttbar) analysis. +Reads data from a YAML file and provides it in a structured way to the analysis configs. +""" + +from __future__ import annotations + +import yaml + +DATASETS_FILE = "/data/dust/user/matthiej/topsf/topsf/config/datasets.yaml" + + +def load_yaml(path: str) -> dict: + with open(path, "r") as f: + return yaml.safe_load(f) + + +DATASETS = load_yaml(DATASETS_FILE) + + +def register_datasets( + config, + names, + tags, + limit_dataset_files=None, +): + + for name in names: + + ds = config.add_dataset( + config.campaign.get_dataset(name) + ) + + if tags: + ds.add_tag(tags) + + if limit_dataset_files: + for info in ds.info.values(): + info.n_files = min(info.n_files, limit_dataset_files) + + +def add_datasets_from_yaml( + config, + limit_dataset_files=None, + dataset_types=None, + log=False, +): + """ + Add datasets defined in DATASETS (loaded from YAML) to config. + + Parameters: + config: configuration object + limit_dataset_files: optional int to limit files per dataset + dataset_types: optional iterable of top-level dataset keys (e.g. ["data","tt","qcd"]) + If provided, only those dataset types are added. + log: if True, print a summary + + Returns: + list of added dataset names + """ + tag = config.x.cpn_tag + if tag == "2024": + tag = "2024full" + + # normalize requested types to a set for fast membership tests + if dataset_types is None: + requested = None + else: + requested = set(dataset_types) + + total = 0 + added = [] + + for sample_type, info in DATASETS.items(): + if requested is not None and sample_type not in requested: + continue + + try: + names = info["eras"][tag] + except KeyError: + continue + + register_datasets( + config, + names, + tags=set(info.get("tags", [])), + limit_dataset_files=limit_dataset_files, + ) + + total += len(names) + added.extend(names) + + if log: + print(f"Added {total} datasets (types={', '.join(sorted(requested)) if requested else 'all'})") + + return added diff --git a/topsf/config/run3/analysis_sf.py b/topsf/config/run3/analysis_sf.py index 7fd2ab2..1052dbd 100644 --- a/topsf/config/run3/analysis_sf.py +++ b/topsf/config/run3/analysis_sf.py @@ -41,7 +41,7 @@ # (used in cf.HTCondorWorkflow) ana.x.cmssw_sandboxes = [ # "$CF_BASE/sandboxes/cmssw_default.sh", - "$TOPSF_BASE/sandboxes/combine_cmssw.sh", + # "$TOPSF_BASE/sandboxes/combine_cmssw.sh", ] # clear the list when cmssw bundling is disabled @@ -57,11 +57,13 @@ # set up configs # -from topsf.config.run3.config_sf import add_config +from topsf.config.run3.config_sf_new import add_new_config + import cmsdb.campaigns.run3_2022_preEE_nano_v12 import cmsdb.campaigns.run3_2022_postEE_nano_v12 -# import cmsdb.campaigns.run3_2023_preBPix_nano_v12 -# import cmsdb.campaigns.run3_2023_postBPix_nano_v12 +import cmsdb.campaigns.run3_2023_preBPix_nano_v12 +import cmsdb.campaigns.run3_2023_postBPix_nano_v12 +from cmsdb.campaigns.run3_2024_nano_v15 import campaign_run3_2024_nano_v15 as campaign_run3_2024_nano_v15 # noqa campaign_run3_2022_preEE_nano_v12 = cmsdb.campaigns.run3_2022_preEE_nano_v12.campaign_run3_2022_preEE_nano_v12 campaign_run3_2022_preEE_nano_v12.x.EE = "pre" @@ -69,87 +71,126 @@ campaign_run3_2022_postEE_nano_v12 = cmsdb.campaigns.run3_2022_postEE_nano_v12.campaign_run3_2022_postEE_nano_v12 campaign_run3_2022_postEE_nano_v12.x.EE = "post" -# campaign_run3_2023_preBPix_nano_v12 = cmsdb.campaigns.run3_2023_preBPix_nano_v12.campaign_run3_2023_preBPix_nano_v12 -# campaign_run3_2023_preBPix_nano_v12.x.BPix = "pre" +campaign_run3_2023_preBPix_nano_v12 = cmsdb.campaigns.run3_2023_preBPix_nano_v12.campaign_run3_2023_preBPix_nano_v12 +campaign_run3_2023_preBPix_nano_v12.x.BPix = "pre" -# campaign_run3_2023_postBPix_nano_v12 = cmsdb.campaigns.run3_2023_postBPix_nano_v12.campaign_run3_2023_postBPix_nano_v12 # noqa -# campaign_run3_2023_postBPix_nano_v12.x.BPix = "post" +campaign_run3_2023_postBPix_nano_v12 = cmsdb.campaigns.run3_2023_postBPix_nano_v12.campaign_run3_2023_postBPix_nano_v12 # noqa +campaign_run3_2023_postBPix_nano_v12.x.BPix = "post" # default config -config_2022_preEE = add_config( +config_2022_preEE = add_new_config( analysis_sf, campaign_run3_2022_preEE_nano_v12.copy(), config_name="run3_sf_2022_preEE_nano_v12", config_id=1_03_22_11, # 1: SF 03: Run3 22: year 1: full stat 1: pre EE ) -config_2022_postEE = add_config( +config_2022_postEE = add_new_config( analysis_sf, campaign_run3_2022_postEE_nano_v12.copy(), config_name="run3_sf_2022_postEE_nano_v12", config_id=1_03_22_12, # 1: SF 03: Run3 22: year 1: full stat 2: post EE ) -# config_2023_preBPix = add_config( -# analysis_sf, -# campaign_run3_2023_preBPix_nano_v12.copy(), -# config_name="run3_sf_2023_preBPix_nano_v12", -# config_id=1_03_23_11, # 1: SF 03: Run3 23: year 1: full stat 1: pre BPix -# ) - -# config_2023_postBPix = add_config( -# analysis_sf, -# campaign_run3_2023_postBPix_nano_v12.copy(), -# config_name="run3_sf_2023_postBPix_nano_v12", -# config_id=1_03_23_12, # 1: SF 03: Run3 23: year 1: full stat 2: post BPix -# ) +config_2023_preBPix = add_new_config( + analysis_sf, + campaign_run3_2023_preBPix_nano_v12.copy(), + config_name="run3_sf_2023_preBPix_nano_v12", + config_id=1_03_23_11, # 1: SF 03: Run3 23: year 1: full stat 1: pre BPix +) -# config with limited number of files -config_2022_preEE_limited = add_config( +config_2023_postBPix = add_new_config( analysis_sf, - campaign_run3_2022_preEE_nano_v12.copy(), - config_name="run3_sf_2022_preEE_nano_v12_limited", - config_id=1_03_22_21, # 1: SF 03: Run3 22: year 2: limited stat 1: pre EE - limit_dataset_files=1, + campaign_run3_2023_postBPix_nano_v12.copy(), + config_name="run3_sf_2023_postBPix_nano_v12", + config_id=1_03_23_12, # 1: SF 03: Run3 23: year 1: full stat 2: post BPix ) -config_2022_postEE_limited = add_config( +config_2024 = add_new_config( analysis_sf, - campaign_run3_2022_postEE_nano_v12.copy(), - config_name="run3_sf_2022_postEE_nano_v12_limited", - config_id=1_03_22_22, # 1: SF 03: Run3 22: year 2: limited stat 2: post EE - limit_dataset_files=1, + campaign_run3_2024_nano_v15.copy(), + config_name="run3_sf_2024_nano_v15", + config_id=1_03_24_11, # 1: SF 03: Run3 24: year 1: full stat 1: full campaign ) -# config_2023_preBPix_limited = add_config( +# # config with limited number of files +# config_2022_preEE_limited = add_new_config( +# analysis_sf, +# campaign_run3_2022_preEE_nano_v12.copy(), +# config_name="run3_sf_2022_preEE_nano_v12_limited", +# config_id=1_03_22_21, # 1: SF 03: Run3 22: year 2: limited stat 1: pre EE +# limit_dataset_files=2, +# ) + +# config_2022_postEE_limited = add_new_config( +# analysis_sf, +# campaign_run3_2022_postEE_nano_v12.copy(), +# config_name="run3_sf_2022_postEE_nano_v12_limited", +# config_id=1_03_22_22, # 1: SF 03: Run3 22: year 2: limited stat 2: post EE +# limit_dataset_files=2, +# ) + +# config_2023_preBPix_limited = add_new_config( # analysis_sf, # campaign_run3_2023_preBPix_nano_v12.copy(), # config_name="run3_sf_2023_preBPix_nano_v12_limited", # config_id=1_03_23_21, # 1: SF 03: Run3 23: year 2: limited stat 1: pre BPix -# limit_dataset_files=1, +# limit_dataset_files=2, # ) -# config_2023_postBPix_limited = add_config( +# config_2023_postBPix_limited = add_new_config( # analysis_sf, # campaign_run3_2023_postBPix_nano_v12.copy(), # config_name="run3_sf_2023_postBPix_nano_v12_limited", # config_id=1_03_23_22, # 1: SF 03: Run3 23: year 2: limited stat 2: post BPix -# limit_dataset_files=1, +# limit_dataset_files=2, # ) -# config with limited number of files -config_2022_preEE_medium_limited = add_config( - analysis_sf, - campaign_run3_2022_preEE_nano_v12.copy(), - config_name="run3_sf_2022_preEE_nano_v12_medium_limited", - config_id=1_03_22_31, - limit_dataset_files=10, -) +# config_2024_limited = add_new_config( +# analysis_sf, +# campaign_run3_2024_nano_v15.copy(), +# config_name="run3_sf_2024_nano_v15_limited", +# config_id=1_03_24_21, # 1: SF 03: Run3 24: year 2: limited stat 1: full campaign +# limit_dataset_files=2, +# ) -config_2022_postEE_medium_limited = add_config( - analysis_sf, - campaign_run3_2022_postEE_nano_v12.copy(), - config_name="run3_sf_2022_postEE_nano_v12_medium_limited", - config_id=1_03_22_32, - limit_dataset_files=10, -) +# # config with medium limited number of files +# config_2022_preEE_medium_limited = add_new_config( +# analysis_sf, +# campaign_run3_2022_preEE_nano_v12.copy(), +# config_name="run3_sf_2022_preEE_nano_v12_medium_limited", +# config_id=1_03_22_31, +# limit_dataset_files=10, +# ) + +# config_2022_postEE_medium_limited = add_new_config( +# analysis_sf, +# campaign_run3_2022_postEE_nano_v12.copy(), +# config_name="run3_sf_2022_postEE_nano_v12_medium_limited", +# config_id=1_03_22_32, +# limit_dataset_files=10, +# ) + +# config_2023_preBPix_medium_limited = add_new_config( +# analysis_sf, +# campaign_run3_2023_preBPix_nano_v12.copy(), +# config_name="run3_sf_2023_preBPix_nano_v12_medium_limited", +# config_id=1_03_23_31, +# limit_dataset_files=10, +# ) + +# config_2023_postBPix_medium_limited = add_new_config( +# analysis_sf, +# campaign_run3_2023_postBPix_nano_v12.copy(), +# config_name="run3_sf_2023_postBPix_nano_v12_medium_limited", +# config_id=1_03_23_32, +# limit_dataset_files=10, +# ) + +# config_2024_medium_limited = add_new_config( +# analysis_sf, +# campaign_run3_2024_nano_v15.copy(), +# config_name="run3_sf_2024_nano_v15_medium_limited", +# config_id=1_03_24_31, +# limit_dataset_files=10, +# ) diff --git a/topsf/config/run3/analysis_wp.py b/topsf/config/run3/analysis_wp.py index f43f3df..6512e07 100644 --- a/topsf/config/run3/analysis_wp.py +++ b/topsf/config/run3/analysis_wp.py @@ -53,9 +53,12 @@ # set up configs # -from topsf.config.run3.config_wp import add_config +from topsf.config.run3.config_wp_new import add_new_config import cmsdb.campaigns.run3_2022_preEE_nano_v12 import cmsdb.campaigns.run3_2022_postEE_nano_v12 +import cmsdb.campaigns.run3_2023_preBPix_nano_v12 +import cmsdb.campaigns.run3_2023_postBPix_nano_v12 +from cmsdb.campaigns.run3_2024_nano_v15 import campaign_run3_2024_nano_v15 as campaign_run3_2024_nano_v15 # noqa campaign_run3_2022_preEE_nano_v12 = cmsdb.campaigns.run3_2022_preEE_nano_v12.campaign_run3_2022_preEE_nano_v12 campaign_run3_2022_preEE_nano_v12.x.EE = "pre" @@ -63,51 +66,126 @@ campaign_run3_2022_postEE_nano_v12 = cmsdb.campaigns.run3_2022_postEE_nano_v12.campaign_run3_2022_postEE_nano_v12 campaign_run3_2022_postEE_nano_v12.x.EE = "post" -# default config -config_2022_preEE = add_config( +campaign_run3_2023_preBPix_nano_v12 = cmsdb.campaigns.run3_2023_preBPix_nano_v12.campaign_run3_2023_preBPix_nano_v12 +campaign_run3_2023_preBPix_nano_v12.x.BPix = "pre" + +campaign_run3_2023_postBPix_nano_v12 = cmsdb.campaigns.run3_2023_postBPix_nano_v12.campaign_run3_2023_postBPix_nano_v12 # noqa +campaign_run3_2023_postBPix_nano_v12.x.BPix = "post" + +# full stats config +config_2022_preEE = add_new_config( analysis_wp, campaign_run3_2022_preEE_nano_v12.copy(), config_name="run3_wp_2022_preEE_nano_v12", config_id=2_03_22_11, ) -config_2022_postEE = add_config( +config_2022_postEE = add_new_config( analysis_wp, campaign_run3_2022_postEE_nano_v12.copy(), config_name="run3_wp_2022_postEE_nano_v12", config_id=2_03_22_12, ) -# config with limited number of files -config_2022_preEE_limited = add_config( +config_2023_preBPix = add_new_config( analysis_wp, - campaign_run3_2022_preEE_nano_v12.copy(), - config_name="run3_wp_2022_preEE_nano_v12_limited", - config_id=2_03_22_21, - limit_dataset_files=1, + campaign_run3_2023_preBPix_nano_v12.copy(), + config_name="run3_wp_2023_preBPix_nano_v12", + config_id=2_03_23_11, ) -config_2022_postEE_limited = add_config( +config_2023_postBPix = add_new_config( analysis_wp, - campaign_run3_2022_postEE_nano_v12.copy(), - config_name="run3_wp_2022_postEE_nano_v12_limited", - config_id=2_03_22_22, - limit_dataset_files=1, + campaign_run3_2023_postBPix_nano_v12.copy(), + config_name="run3_wp_2023_postBPix_nano_v12", + config_id=2_03_23_12, ) -# config with limited number of files -config_2022_preEE_medium_limited = add_config( +config_2024 = add_new_config( analysis_wp, - campaign_run3_2022_preEE_nano_v12.copy(), - config_name="run3_wp_2022_preEE_nano_v12_medium_limited", - config_id=2_03_22_31, - limit_dataset_files=10, + campaign_run3_2024_nano_v15.copy(), + config_name="run3_wp_2024_nano_v15", + config_id=2_03_24_11, ) -config_2022_postEE_medium_limited = add_config( - analysis_wp, - campaign_run3_2022_postEE_nano_v12.copy(), - config_name="run3_wp_2022_postEE_nano_v12_medium_limited", - config_id=2_03_22_32, - limit_dataset_files=10, -) +# # config with limited number of files +# config_2022_preEE_limited = add_new_config( +# analysis_wp, +# campaign_run3_2022_preEE_nano_v12.copy(), +# config_name="run3_wp_2022_preEE_nano_v12_limited", +# config_id=2_03_22_21, +# limit_dataset_files=2, +# ) + +# config_2022_postEE_limited = add_new_config( +# analysis_wp, +# campaign_run3_2022_postEE_nano_v12.copy(), +# config_name="run3_wp_2022_postEE_nano_v12_limited", +# config_id=2_03_22_22, +# limit_dataset_files=2, +# ) + +# config_2023_preBPix_limited = add_new_config( +# analysis_wp, +# campaign_run3_2023_preBPix_nano_v12.copy(), +# config_name="run3_wp_2023_preBPix_nano_v12_limited", +# config_id=2_03_23_21, +# limit_dataset_files=2, +# ) + +# config_2023_postBPix_limited = add_new_config( +# analysis_wp, +# campaign_run3_2023_postBPix_nano_v12.copy(), +# config_name="run3_wp_2023_postBPix_nano_v12_limited", +# config_id=2_03_23_22, +# limit_dataset_files=2, +# ) + +# config_2024_limited = add_new_config( +# analysis_wp, +# campaign_run3_2024_nano_v15.copy(), +# config_name="run3_wp_2024_nano_v15_limited", +# config_id=2_03_24_21, +# limit_dataset_files=2, +# ) + +# # config with medium limited number of files +# config_2022_preEE_medium_limited = add_new_config( +# analysis_wp, +# campaign_run3_2022_preEE_nano_v12.copy(), +# config_name="run3_wp_2022_preEE_nano_v12_medium_limited", +# config_id=2_03_22_31, +# limit_dataset_files=10, +# ) + +# config_2022_postEE_medium_limited = add_new_config( +# analysis_wp, +# campaign_run3_2022_postEE_nano_v12.copy(), +# config_name="run3_wp_2022_postEE_nano_v12_medium_limited", +# config_id=2_03_22_32, +# limit_dataset_files=10, +# ) + +# config_2023_preBPix_medium_limited = add_new_config( +# analysis_wp, +# campaign_run3_2023_preBPix_nano_v12.copy(), +# config_name="run3_wp_2023_preBPix_nano_v12_medium_limited", +# config_id=2_03_23_31, +# limit_dataset_files=10, +# ) + +# config_2023_postBPix_medium_limited = add_new_config( +# analysis_wp, +# campaign_run3_2023_postBPix_nano_v12.copy(), +# config_name="run3_wp_2023_postBPix_nano_v12_medium_limited", +# config_id=2_03_23_32, +# limit_dataset_files=10, +# ) + +# config_2024_medium_limited = add_new_config( +# analysis_wp, +# campaign_run3_2024_nano_v15.copy(), +# config_name="run3_wp_2024_nano_v15_medium_limited", +# config_id=2_03_24_31, +# limit_dataset_files=10, +# ) diff --git a/topsf/config/run3/config_sf.py b/topsf/config/run3/config_sf.py index 2259f52..4025e92 100644 --- a/topsf/config/run3/config_sf.py +++ b/topsf/config/run3/config_sf.py @@ -58,7 +58,7 @@ def add_config( elif year == 2023: corr_postfix = f"{campaign.x.BPix}BPix" - implemented_years = [2022] + implemented_years = [2022, 2023] if year not in implemented_years: raise NotImplementedError("For now, only 2022 campaign is fully implemented") @@ -287,23 +287,23 @@ def add_subprocesses(proc, color_key): "st_twchannel_t_sl_powheg", "st_twchannel_tbar_sl_powheg", "st_twchannel_t_dl_powheg", - # "st_twchannel_tbar_dl_powheg", # FIXME one file missing at any remote fs in preEE + "st_twchannel_tbar_dl_powheg", # FIXME one file missing at any remote fs in preEE "st_twchannel_t_fh_powheg", "st_twchannel_tbar_fh_powheg", # DY 2022 v12 datasets - # "dy_m4to50_ht40to70_madgraph", # FIXME AssertionError for preEE/postEE - # "dy_m4to50_ht70to100_madgraph", # FIXME AssertionError for preEE/postEE + "dy_m4to50_ht40to70_madgraph", # FIXME AssertionError for preEE/postEE + "dy_m4to50_ht70to100_madgraph", # FIXME AssertionError for preEE/postEE "dy_m4to50_ht100to400_madgraph", "dy_m4to50_ht400to800_madgraph", "dy_m4to50_ht800to1500_madgraph", "dy_m4to50_ht1500to2500_madgraph", "dy_m4to50_ht2500toinf_madgraph", - # "dy_m50to120_ht40to70_madgraph", # FIXME AssertionError for preEE - # "dy_m50to120_ht70to100_madgraph", # FIXME AssertionError for postEE + "dy_m50to120_ht40to70_madgraph", # FIXME AssertionError for preEE + "dy_m50to120_ht70to100_madgraph", # FIXME AssertionError for postEE "dy_m50to120_ht100to400_madgraph", "dy_m50to120_ht400to800_madgraph", # WJets 2022 v12 datasets - # "w_lnu_mlnu0to120_ht100to400_madgraph", # FIXME AssertionError for postEE + "w_lnu_mlnu0to120_ht100to400_madgraph", # FIXME AssertionError for postEE "w_lnu_mlnu0to120_ht400to800_madgraph", "w_lnu_mlnu0to120_ht800to1500_madgraph", "w_lnu_mlnu0to120_ht1500to2500_madgraph", @@ -329,7 +329,7 @@ def add_subprocesses(proc, color_key): # "qcd_mu_pt30to50_pythia", # remove low pt qcd datasets due to issues with pileup jets (27.01.25) # "qcd_mu_pt50to80_pythia", # remove low pt qcd datasets due to issues with pileup jets (27.01.25) # "qcd_mu_pt80to120_pythia", # remove low pt qcd datasets due to issues with pileup jets (27.01.25) - # "qcd_mu_pt120to170_pythia", # FIXME AssertionError for preEE + "qcd_mu_pt120to170_pythia", # FIXME AssertionError for preEE "qcd_mu_pt170to300_pythia", "qcd_mu_pt300to470_pythia", "qcd_mu_pt470to600_pythia", @@ -341,21 +341,18 @@ def add_subprocesses(proc, color_key): # "qcd_em_pt30to50_pythia", # remove low pt qcd datasets due to issues with pileup jets (27.01.25) # "qcd_em_pt50to80_pythia", # remove low pt qcd datasets due to issues with pileup jets (27.01.25) # "qcd_em_pt80to120_pythia", # remove low pt qcd datasets due to issues with pileup jets (27.01.25) - # "qcd_em_pt120to170_pythia", # FIXME AssertionError for preEE/postEE - # "qcd_em_pt170to300_pythia", # FIXME AssertionError for postEE + "qcd_em_pt120to170_pythia", # FIXME AssertionError for preEE/postEE + "qcd_em_pt170to300_pythia", # FIXME AssertionError for postEE "qcd_em_pt300toinf_pythia", ] - if campaign.x.EE == "pre": + if cfg.x.cpn_tag == "2022preEE": dataset_names += [ "data_egamma_c", "data_egamma_d", "data_mu_c", "data_mu_d", - "dy_m50to120_ht70to100_madgraph", - "w_lnu_mlnu0to120_ht100to400_madgraph", - "qcd_em_pt170to300_pythia", ] - if campaign.x.EE == "post": + elif cfg.x.cpn_tag == "2022postEE": dataset_names += [ "data_egamma_e", "data_egamma_f", @@ -363,9 +360,16 @@ def add_subprocesses(proc, color_key): "data_mu_e", "data_mu_f", "data_mu_g", - "st_twchannel_tbar_dl_powheg", - "dy_m50to120_ht40to70_madgraph", - "qcd_mu_pt120to170_pythia", + ] + elif cfg.x.cpn_tag == "2023preBPix": + dataset_names += [ + "data_egamma_c", + "data_mu_c", + ] + elif cfg.x.cpn_tag == "2023postBPix": + dataset_names += [ + "data_egamma_d", + "data_mu_d", ] for dataset_name in dataset_names: @@ -465,10 +469,14 @@ def add_subprocesses(proc, color_key): # "qcd_em_pt120to170_pythia", ], } - if campaign.x.EE == "pre": + if cfg.x.cpn_tag == "2022preEE": cfg.x.dataset_groups["testing"] += ["data_egamma_c", "data_mu_c"] - if campaign.x.EE == "post": + elif cfg.x.cpn_tag == "2022postEE": cfg.x.dataset_groups["testing"] += ["data_egamma_f", "data_mu_f"] + elif cfg.x.cpn_tag == "2023preBPix": + cfg.x.dataset_groups["testing"] += ["data_egamma_c", "data_mu_c"] + elif cfg.x.cpn_tag == "2023postBPix": + cfg.x.dataset_groups["testing"] += ["data_egamma_d", "data_mu_d"] # category groups for conveniently looping over certain categories # (used during plotting) @@ -715,6 +723,17 @@ def add_subprocesses(proc, color_key): "lumi_13TeV_2022": 0.01j, "lumi_13TeV_correlated": 0.006j, }) + elif year == 2023: + if campaign.has_tag("preBPix"): + cfg.x.luminosity = Number(17794, { + "lumi_13TeV_2023": 0.01j, + "lumi_13TeV_correlated": 0.006j, + }) + elif campaign.has_tag("postBPix"): + cfg.x.luminosity = Number(9451, { + "lumi_13TeV_2023": 0.01j, + "lumi_13TeV_correlated": 0.006j, + }) else: raise NotImplementedError(f"Luminosity for year {year} is not defined.") @@ -784,18 +803,27 @@ def add_subprocesses(proc, color_key): # jec configuration taken from HBW # https://github.com/uhh-cms/hh2bbww/blob/master/hbw/config/config_run2.py#L138C5-L269C1 # https://twiki.cern.ch/twiki/bin/view/CMS/JECDataMC?rev=201 - jerc_postfix = "" - if year == 2022 and campaign.x.EE == "post": - jerc_postfix = "EE" + jerc_postfix = campaign.x.postfix + if jerc_postfix not in ("", "EE", "BPix"): + raise ValueError(f"Invalid JERC postfix '{jerc_postfix}' for campaign {campaign.name}.") + if year == 2022: + jer_campaign = jec_campaign = f"Summer{year2}{jerc_postfix}_22Sep2023" + elif year == 2023: + era = "Cv1234" if campaign.has_tag("preBPix") else "D" + jer_campaign = f"Summer{year2}{jerc_postfix}Prompt{year2}_Run{era}" + jec_campaign = f"Summer{year2}{jerc_postfix}Prompt{year2}" - jerc_campaign = f"Summer{year2}{jerc_postfix}_22Sep2023" jet_type = "AK4PFPuppi" fatjet_type = "AK8PFPuppi" + jec_ak4_version = jec_ak8_version = { + 2022: "V2", + 2023: "V2" if jerc_postfix == "" else "V3", + }[year] cfg.x.jec = DotDict.wrap({ "Jet": { - "campaign": jerc_campaign, - "version": {2016: "V7", 2017: "V5", 2018: "V5", 2022: "V2"}[year], + "campaign": jec_campaign, + "version": jec_ak4_version, "jet_type": jet_type, "levels": ["L1FastJet", "L2Relative", "L2L3Residual", "L3Absolute"], "levels_for_type1_met": ["L1FastJet"], @@ -857,11 +885,11 @@ def add_subprocesses(proc, color_key): # "CorrelationGroupFlavor", # "CorrelationGroupUncorrelated", ], - "data_per_era": True, + "data_per_era": False if year == 2023 else True, }, "FatJet": { - "campaign": jerc_campaign, - "version": {2016: "V7", 2017: "V5", 2018: "V5", 2022: "V2"}[year], + "campaign": jec_campaign, + "version": jec_ak8_version, "jet_type": fatjet_type, "levels": ["L1FastJet", "L2Relative", "L2L3Residual", "L3Absolute"], "levels_for_type1_met": ["L1FastJet"], @@ -923,11 +951,11 @@ def add_subprocesses(proc, color_key): # "CorrelationGroupFlavor", # "CorrelationGroupUncorrelated", ], - "data_per_era": True, + "data_per_era": False if year == 2023 else True, }, "SubJet": { - "campaign": jerc_campaign, - "version": {2016: "V7", 2017: "V5", 2018: "V5", 2022: "V2"}[year], + "campaign": jec_campaign, + "version": jec_ak4_version, "jet_type": jet_type, "levels": ["L1FastJet", "L2Relative", "L2L3Residual", "L3Absolute"], "levels_for_type1_met": ["L1FastJet"], @@ -989,7 +1017,7 @@ def add_subprocesses(proc, color_key): # "CorrelationGroupFlavor", # "CorrelationGroupUncorrelated", ], - "data_per_era": True, + "data_per_era": False if year == 2023 else True, }, }) @@ -998,18 +1026,18 @@ def add_subprocesses(proc, color_key): # TODO: get jerc working for Run3 cfg.x.jer = DotDict.wrap({ "Jet": { - "campaign": jerc_campaign, - "version": {2022: "JRV1"}[year], + "campaign": jer_campaign, + "version": {2022: "JRV1", 2023: "JRV1"}[year], "jet_type": jet_type, }, "FatJet": { - "campaign": jerc_campaign, - "version": {2022: "JRV1"}[year], + "campaign": jer_campaign, + "version": {2022: "JRV1", 2023: "JRV1"}[year], "jet_type": fatjet_type, }, "SubJet": { - "campaign": jerc_campaign, - "version": {2022: "JRV1"}[year], + "campaign": jer_campaign, + "version": {2022: "JRV1", 2023: "JRV1"}[year], "jet_type": jet_type, }, }) @@ -1056,6 +1084,32 @@ def add_subprocesses(proc, color_key): "TimePtEta", ] + if cfg.x.run == 2: + cfg.x.met_phi_correction_set = "{variable}_metphicorr_pfmet_{data_source}" + else: + cfg.x.met_phi_correction_set = "met_xy_corrections" + cfg.x.met_phi_correction = { + "met_name": "PuppiMET", + "correction_set": "met_xy_corrections", + "keep_uncorrected": False, + "variable_config": { + "pt": ( + "pt", + "pt_stat_yup", + "pt_stat_ydn", + "pt_stat_xup", + "pt_stat_xdn", + ), + "phi": ( + "phi", + "phi_stat_yup", + "phi_stat_ydn", + "phi_stat_xup", + "phi_stat_xdn", + ), + }, + } + # # tagger working points # @@ -1064,35 +1118,35 @@ def add_subprocesses(proc, color_key): # https://btv-wiki.docs.cern.ch/ScaleFactors/Run3Summer22/ # https://btv-wiki.docs.cern.ch/ScaleFactors/Run3Summer22EE/ # TODO: add correct 2022 + 2022preEE WP for deepcsv if needed - btag_key = f"2022{campaign.x.EE}EE" if year == 2022 else year + # TODO: use PNet? + btag_key = cfg.x.cpn_tag cfg.x.btag_working_points = DotDict.wrap({ "deepjet": { "loose": { - "2022preEE": 0.0583, "2022postEE": 0.0614, + "2022preEE": 0.0583, "2022postEE": 0.0614, "2023preBPix": 0.0479, "2023postBPix": 0.048, }[btag_key], "medium": { - "2022preEE": 0.3086, "2022postEE": 0.3196, + "2022preEE": 0.3086, "2022postEE": 0.3196, "2023preBPix": 0.2431, "2023postBPix": 0.2435, }[btag_key], "tight": { - "2022preEE": 0.7183, "2022postEE": 0.7300, + "2022preEE": 0.7183, "2022postEE": 0.7300, "2023preBPix": 0.6553, "2023postBPix": 0.6563, }[btag_key], }, "deepcsv": { "loose": { - "2022preEE": 0.1208, "2022postEE": 0.1208, + "2022preEE": 0.1208, "2022postEE": 0.1208, "2023preBPix": 0.1208, "2023postBPix": 0.1208, }[btag_key], "medium": { - "2022preEE": 0.4168, "2022postEE": 0.4168, + "2022preEE": 0.4168, "2022postEE": 0.4168, "2023preBPix": 0.4168, "2023postBPix": 0.4168, }[btag_key], "tight": { - "2022preEE": 0.7665, "2022postEE": 0.7665, + "2022preEE": 0.7665, "2022postEE": 0.7665, "2023preBPix": 0.7665, "2023postBPix": 0.7665, }[btag_key], }, }) # top-tag working points # https://twiki.cern.ch/twiki/bin/view/CMS/JetTopTagging?rev=41 - # FIXME use my own WPs here? cfg.x.toptag_working_points = DotDict.wrap({ "tau32": { "very_loose": 0.69, @@ -1167,12 +1221,14 @@ def add_subprocesses(proc, color_key): "min_pt": 300, "max_abseta": 2.5, "msoftdrop_range": (105, 210), + # https://twiki.cern.ch/twiki/bin/view/CMS/JetID13p6TeV + "jetId": 2, # bit2 (2): pass tight ID, fail tightLepVeto, bit3 (6): pass tight and tightLepVeto ID # probe jet pt bins (used by category builder) "pt_bins": [300, 400, 480, 600, None], # parameters for b-tagged subjets "subjet_column": "SubJet", "subjet_btag": "btagDeepB", - "subjet_btag_wp": cfg.x.btag_working_points.deepcsv.loose, + "subjet_btag_wp": cfg.x.btag_working_points.deepcsv.loose, # FIXME: use DeepJet or PNet? }, # TODO: implement (requires custom nano) "hotvr": { @@ -1189,6 +1245,8 @@ def add_subprocesses(proc, color_key): }, "ak4": { "column": "Jet", + # https://twiki.cern.ch/twiki/bin/view/CMS/JetID13p6TeV + "jetId": 2, # bit2 (2): pass tight ID, fail tightLepVeto, bit3 (6): pass tight and tightLepVeto ID "min_pt": 15, # TODO: check UHH2 "max_abseta": 2.5, # TODO: check UHH2 "btag_column": "btagDeepFlavB", # nano v9: "DeepJet b+bb+lepb tag discriminator" @@ -1197,9 +1255,11 @@ def add_subprocesses(proc, color_key): }) # MET selection parameters + # FIXME: use PuppiMET for Run 3? What's the difference? It's better? + # https://twiki.cern.ch/twiki/bin/viewauth/CMS/MissingETRun2Corrections?rev=79#xy_Shift_Correction_MET_phi_modu cfg.x.met_selection = DotDict.wrap({ "default": { - "column": "MET", + "column": "PuppiMET" if cfg.x.run == 3 else "MET", "min_pt": 50, }, }) @@ -1208,28 +1268,29 @@ def add_subprocesses(proc, color_key): # producer configurations # - if cfg.x.run == 3: - # TODO: check that everyting is setup as intended - - # btag weight configuration - cfg.x.btag_sf = ("deepJet_shape", cfg.x.btag_sf_jec_sources) + # btag weight configuration + cfg.x.btag_sf = ("deepJet_shape", cfg.x.btag_sf_jec_sources) - # lepton sf taken from - # https://github.com/uhh-cms/hh2bbww/blob/master/hbw/config/config_run2.py#L338C1-L352C85 - # names of electron correction sets and working points - # (used in the electron_sf producer) - if cfg.x.cpn_tag == "2022postEE": - # TODO: we need to use different SFs for control regions - cfg.x.electron_sf_names = ("Electron-ID-SF", "2022Re-recoE+PromptFG", "Tight") - elif cfg.x.cpn_tag == "2022preEE": - cfg.x.electron_sf_names = ("Electron-ID-SF", "2022Re-recoBCD", "Tight") - - # names of muon correction sets and working points - # (used in the muon producer) + # lepton sf taken from + # https://github.com/uhh-cms/hh2bbww/blob/master/hbw/config/config_run2.py#L338C1-L352C85 + # names of electron correction sets and working points + # (used in the electron_sf producer) + if cfg.x.cpn_tag == "2022postEE": # TODO: we need to use different SFs for control regions - cfg.x.muon_sf_names = ("NUM_TightPFIso_DEN_TightID", f"{cfg.x.cpn_tag}") - cfg.x.muon_id_sf_names = ("NUM_TightID_DEN_TrackerMuons", f"{cfg.x.cpn_tag}") - cfg.x.muon_iso_sf_names = ("NUM_TightPFIso_DEN_TightID", f"{cfg.x.cpn_tag}") + cfg.x.electron_sf_names = ("Electron-ID-SF", "2022Re-recoE+PromptFG", "Tight") + elif cfg.x.cpn_tag == "2022preEE": + cfg.x.electron_sf_names = ("Electron-ID-SF", "2022Re-recoBCD", "Tight") + elif cfg.x.cpn_tag == "2023preBPix": + cfg.x.electron_sf_names = ("Electron-ID-SF", "2023PromptC", "Tight") + elif cfg.x.cpn_tag == "2023BPix": + cfg.x.electron_sf_names = ("Electron-ID-SF", "2023PromptD", "Tight") + + # names of muon correction sets and working points + # (used in the muon producer) + # TODO: we need to use different SFs for control regions + cfg.x.muon_sf_names = ("NUM_TightPFIso_DEN_TightID", f"{cfg.x.cpn_tag}") + cfg.x.muon_id_sf_names = ("NUM_TightID_DEN_TrackerMuons", f"{cfg.x.cpn_tag}") + cfg.x.muon_iso_sf_names = ("NUM_TightPFIso_DEN_TightID", f"{cfg.x.cpn_tag}") # top pt reweighting parameters # https://twiki.cern.ch/twiki/bin/viewauth/CMS/TopPtReweighting#TOP_PAG_corrections_based_on_dat?rev=31 @@ -1395,11 +1456,13 @@ def add_shifts(cfg): # # external files - json_mirror = "/afs/cern.ch/user/j/jmatthie/public/mirrors/jsonpog-integration-b7a48c75" + json_mirror = "/afs/cern.ch/user/j/jmatthie/public/mirrors/jsonpog-integration-406118ec" # updated 31.07.25 local_repo = "/data/dust/user/matthiej/topsf" # TODO: avoid hardcoding path - if cfg.x.run == 3: + if cfg.x.cpn_tag == "2022preEE" or cfg.x.cpn_tag == "2022postEE": corr_tag = f"{year}_Summer22{jerc_postfix}" + elif cfg.x.cpn_tag == "2023preBPix" or cfg.x.cpn_tag == "2023postBPix": + corr_tag = f"{year}_Summer23{jerc_postfix}" cfg.x.external_files = DotDict.wrap({ # pileup weight corrections @@ -1423,60 +1486,51 @@ def add_shifts(cfg): # btag scale factor "btag_sf_corr": (f"{json_mirror}/POG/BTV/{corr_tag}/btagging.json.gz", "v1"), - # met phi corrector - "met_phi_corr": (f"{json_mirror}/POG/JME/{corr_tag}/met.json.gz", "v1"), - # V+jets reweighting "vjets_reweighting": f"{local_repo}/data/json/vjets_reweighting.json.gz", }) - # temporary fix due to missing corrections in run 3 - if cfg.x.run == 3: - # cfg.add_tag("skip_electron_weights") - # cfg.add_tag("skip_muon_weights") - cfg.x.external_files.pop("met_phi_corr") + if cfg.x.run == 2: + cfg.x.external_files.update(DotDict.wrap({ + "met_phi_corr": (f"{json_mirror}/POG/JME/{corr_tag}/met.json.gz", "v1"), + })) + elif cfg.x.run == 3: + met_corr_tag = f"{year}_{year}{jerc_postfix}" + cfg.x.external_files.update(DotDict.wrap({ + # met phi corrector + "met_phi_corr": (f"{json_mirror}/POG/JME/{corr_tag}/met_xyCorrections_{met_corr_tag}.json.gz", "v1"), + })) - if year == 2022 and campaign.x.EE == "pre": + if cfg.x.cpn_tag == "2022preEE": cfg.x.external_files.update(DotDict.wrap({ # files from https://twiki.cern.ch/twiki/bin/view/CMSPublic/SWGuideGoodLumiSectionsJSONFile "lumi": { "golden": ("https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/Cert_Collisions2022_355100_362760_Golden.json", "v1"), # noqa "normtag": ("/afs/cern.ch/user/l/lumipro/public/Normtags/normtag_PHYSICS.json", "v1"), }, - "pu": { - # "json": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/BCD/pileup_JSON.txt", "v1"), # noqa - "json": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/BCDEFG/pileup_JSON.txt", "v1"), # noqa - "mc_profile": ("https://raw.githubusercontent.com/cms-sw/cmssw/bb525104a7ddb93685f8ced6fed1ab793b2d2103/SimGeneral/MixingModule/python/Run3_2022_LHC_Simulation_10h_2h_cfi.py", "v1"), # noqa - "data_profile": { - # "nominal": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/BCD/pileupHistogram-Cert_Collisions2022_355100_357900_eraBCD_GoldenJson-13p6TeV-69200ub-99bins.root", "v1"), # noqa - # "minbias_xs_up": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/BCD/pileupHistogram-Cert_Collisions2022_355100_357900_eraBCD_GoldenJson-13p6TeV-72400ub-99bins.root", "v1"), # noqa - # "minbias_xs_down": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/BCD/pileupHistogram-Cert_Collisions2022_355100_357900_eraBCD_GoldenJson-13p6TeV-66000ub-99bins.root", "v1"), # noqa - "nominal": (f"/afs/cern.ch/user/a/anhaddad/public/Collisions22/pileupHistogram-Cert_Collisions2022_355100_362760_GoldenJson-13p6TeV-69200ub-100bins.root", "v1"), # noqa - "minbias_xs_up": (f"/afs/cern.ch/user/a/anhaddad/public/Collisions22/pileupHistogram-Cert_Collisions2022_355100_362760_GoldenJson-13p6TeV-72400ub-100bins.root", "v1"), # noqa - "minbias_xs_down": (f"/afs/cern.ch/user/a/anhaddad/public/Collisions22/pileupHistogram-Cert_Collisions2022_355100_362760_GoldenJson-13p6TeV-66000ub-100bins.root", "v1"), # noqa - }, - }, })) - elif year == 2022 and campaign.x.EE == "post": + elif cfg.x.cpn_tag == "2022postEE": cfg.x.external_files.update(DotDict.wrap({ # files from https://twiki.cern.ch/twiki/bin/view/CMSPublic/SWGuideGoodLumiSectionsJSONFile "lumi": { "golden": ("https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/Cert_Collisions2022_355100_362760_Golden.json", "v1"), # noqa "normtag": ("/afs/cern.ch/user/l/lumipro/public/Normtags/normtag_PHYSICS.json", "v1"), }, - "pu": { - # "json": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/EFG/pileup_JSON.txt", "v1"), # noqa - "json": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/BCDEFG/pileup_JSON.txt", "v1"), # noqa - "mc_profile": ("https://raw.githubusercontent.com/cms-sw/cmssw/bb525104a7ddb93685f8ced6fed1ab793b2d2103/SimGeneral/MixingModule/python/Run3_2022_LHC_Simulation_10h_2h_cfi.py", "v1"), # noqa - "data_profile": { - # data profiles were produced with 99 bins instead of 100 --> use custom produced data profiles instead # noqa - # "nominal": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/EFG/pileupHistogram-Cert_Collisions2022_359022_362760_eraEFG_GoldenJson-13p6TeV-69200ub-99bins.root", "v1"), # noqa - # "minbias_xs_up": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/EFG/pileupHistogram-Cert_Collisions2022_359022_362760_eraEFG_GoldenJson-13p6TeV-72400ub-99bins.root", "v1"), # noqa - # "minbias_xs_down": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/EFG/pileupHistogram-Cert_Collisions2022_359022_362760_eraEFG_GoldenJson-13p6TeV-66000ub-99bins.root", "v1"), # noqa - "nominal": (f"/afs/cern.ch/user/a/anhaddad/public/Collisions22/pileupHistogram-Cert_Collisions2022_355100_362760_GoldenJson-13p6TeV-69200ub-100bins.root", "v1"), # noqa - "minbias_xs_up": (f"/afs/cern.ch/user/a/anhaddad/public/Collisions22/pileupHistogram-Cert_Collisions2022_355100_362760_GoldenJson-13p6TeV-72400ub-100bins.root", "v1"), # noqa - "minbias_xs_down": (f"/afs/cern.ch/user/a/anhaddad/public/Collisions22/pileupHistogram-Cert_Collisions2022_355100_362760_GoldenJson-13p6TeV-66000ub-100bins.root", "v1"), # noqa - }, + })) + elif cfg.x.cpn_tag == "2023preBPix": + cfg.x.external_files.update(DotDict.wrap({ + # files from https://twiki.cern.ch/twiki/bin/view/CMSPublic/SWGuideGoodLumiSectionsJSONFile + "lumi": { + "golden": ("https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions23/Cert_Collisions2023_366442_370790_Golden.json", "v1"), # noqa + "normtag": ("/afs/cern.ch/user/l/lumipro/public/Normtags/normtag_PHYSICS.json", "v1"), + }, + })) + elif cfg.x.cpn_tag == "2023postBPix": + cfg.x.external_files.update(DotDict.wrap({ + # files from https://twiki.cern.ch/twiki/bin/view/CMSPublic/SWGuideGoodLumiSectionsJSONFile + "lumi": { + "golden": ("https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions23/Cert_Collisions2023_366442_370790_Golden.json", "v1"), # noqa + "normtag": ("/afs/cern.ch/user/l/lumipro/public/Normtags/normtag_PHYSICS.json", "v1"), }, })) else: @@ -1578,9 +1632,11 @@ def add_shifts(cfg): # missing transverse momentum "MET.pt", "MET.phi", "MET.significance", "MET.covXX", "MET.covXY", "MET.covYY", + "PuppiMET.phi", "PuppiMET.pt", # number of primary vertices "PV.npvs", + "PV.npvsGood", # average number of pileup interactions "Pileup.nTrueInt", diff --git a/topsf/config/run3/config_sf_new.py b/topsf/config/run3/config_sf_new.py new file mode 100644 index 0000000..af9609a --- /dev/null +++ b/topsf/config/run3/config_sf_new.py @@ -0,0 +1,2151 @@ +# coding: utf-8 + +""" +++++++++++ WIP ++++++++++ +Configuration creation for top-tagging scale factor +derivation using Run3 samples. +++++++++++ WIP ++++++++++ +""" + +from __future__ import annotations + +import functools +import os +import law + +import order as od +import yaml + +from scinum import Number + +from columnflow.util import DotDict +from columnflow.cms_util import CATInfo, CATSnapshot +from columnflow.config_util import ( + add_shift_aliases, + get_root_processes_from_campaign, + get_shifts_from_sources, + verify_config_processes, +) +from columnflow.selection.cms.btag import BTagWPCountConfig +from columnflow.production.cms.btag import BTagSFConfig, BTagWPSFConfig +from columnflow.production.cms.electron import ElectronSFConfig +from columnflow.production.cms.muon import MuonSFConfig +from columnflow.production.cms.jet import JetIdConfig + +from topsf.config.variables import add_variables +from topsf.config.categories import add_categories +from topsf.config.datasets import add_datasets_from_yaml +from topsf.config.taggers import btag_wps, toptag_wps +from topsf.util import has_tag + + +thisdir = os.path.dirname(os.path.abspath(__file__)) +logger = law.logger.get_logger(__name__) + + +def add_new_config( + analysis: od.Analysis, + campaign: od.Campaign, + config_name: str | None = None, + config_id: int | None = None, + limit_dataset_files: int | None = None, +) -> od.Config: + """ + Configurable function for creating a config for a run3 analysis given + a base *analysis* object and a *campaign* (i.e. set of datasets). + """ + # validation + assert campaign.x.year in [2022, 2023, 2024] + if campaign.x.year == 2022: + assert campaign.x.EE in ["pre", "post"] + elif campaign.x.year == 2023: + assert campaign.x.BPix in ["pre", "post"] + + # gather campaign data + year = campaign.x.year + year2 = year % 100 + corr_postfix = "" + if year == 2022: + corr_postfix = f"{campaign.x.EE}EE" + elif year == 2023: + corr_postfix = f"{campaign.x.BPix}BPix" + + implemented_years = [2022, 2023, 2024] + + if year not in implemented_years: + raise NotImplementedError("For now, only 2022, 2023, and 2024 campaigns are fully implemented") + + # create a config by passing the campaign + # (if id and name are not set they will be taken from the campaign) + cfg = analysis.add_config(campaign, name=config_name, id=config_id) + + # add tags to config + cfg.x.run = 3 + cfg.x.cpn_tag = f"{year}{corr_postfix}" + cfg.x.year = year + vnano = campaign.x.version + logger.info(f"Creating config '{cfg.name}' for campaign '{campaign.name}' with year {year} and version {vnano}") + cfg.add_tag("skip_kfactor_weights") # FIXME temporary, remove when kfactors are available for all processes + logger.warning_once( + "K factor reweighting for v+jets datasets currently disabled for all configs, as k factors are not available." + ) + + # + # configure processes + # + + # get all root processes + procs = get_root_processes_from_campaign(campaign) + + # create parent processes for w_lnu, dy and qcd + for i_proc, (proc_name, proc_label, child_procs) in enumerate([ + ("vx", "V+jets, VV", ["dy", "w_lnu", "vv"]), # FIXME Why does dy not work anymore? + ("mj", "Multijet", ["qcd"]), + ]): + proc = od.Process( + name=proc_name, + id=int(1e7 + (i_proc + 1)), + label=proc_label, + ) + for child_proc in child_procs: + procs.n(child_proc).add_parent_process(proc) + + # get all root processes (including newly added ones) + procs = get_root_processes_from_campaign(campaign) + + # create sub-processes for st, tt + # (defined via cuts on gen-level objects; will be normalized + # to xs of parent process) + top_subprocess_cfg = DotDict.wrap({ + "0o1q": { + "index": 1, + "label": "not merged (0q or 1q)", + "colors": { + "tt": "#A80068", + "st": "#FF9300", + }, + }, + "2q": { + "index": 2, + "label": "semi-merged (2q)", + "colors": { + "tt": "#FF58D0", + "st": "#FFFF00", + }, + }, + "3q": { + "index": 3, + "label": "fully merged (3q)", + "colors": { + "tt": "#FF0064", + "st": "#FFC900", + }, + }, + "bkg": { + "index": 4, + "label": "background", + "colors": { + "tt": "#700034", + "st": "#A62800", + }, + }, + }) + + # helper function for adding subprocesses + def add_subprocesses(proc, color_key): + """Add subprocesses to an existing process.""" + subprocs = {} + for subproc_name, subproc_cfg in top_subprocess_cfg.items(): + subprocs[subproc_name] = subproc = proc.add_process( + name=f"{proc.name}_{subproc_name}", + id=int(proc.id + 1e6 * (subproc_cfg.index)), + label=f"{proc.label}, {subproc_cfg.label}", + color=subproc_cfg.colors[color_key], + aux={ + "subprocess_name": subproc_name, + }, + ) + subproc.add_tag("is_subprocess") + + # mark process as signal (used by inference model) + if subproc_name != "bkg": + subproc.add_tag("is_topsf_signal") + + proc.add_tag("has_subprocesses") + return subprocs + + # add subprocesses to processes with top quarks + for root_proc in ("st", "tt"): + root_proc_inst = getattr(procs.n, root_proc) + subprocs = {} # [depth][subproc_key] -> od.Process + for proc, depth, children in root_proc_inst.walk_processes( + algo="bfs", + include_self=True, + ): + # add subprocesses to top-level process (tt, st) + subprocs[depth] = add_subprocesses(proc, color_key=root_proc) + + # mark subprocesses as children of parent subprocesses + parent_subprocs = subprocs.get(depth - 1, {}) + if not parent_subprocs: + continue + for subproc_name, subproc_inst in subprocs[depth].items(): + parent_subprocs[subproc_name].add_process(subproc_inst) + + # set color of some processes + colors = { + "data": "#000000", # black + "tt": "#E04F21", # red + "qcd": "#5E8FFC", # blue + "w_lnu": "#82FF28", # green + "st": "#3E00FB", # dark purple + "dy": "#FBFF36", # yellow + "vv": "#B900FC", # pink + "other": "#999999", # grey + # christopher's color scheme + "vx": "#00FF00", + "mj": "#00D0FF", + } + + # add processes we are interested in + # remove processes we don't need from list and following dict! + process_names = [ + "data", + "tt", + "st", + # "dy", + # "w_lnu", + # "vv", + # "qcd", + "vx", + "mj", + ] + + cfg.x.process_rates = { + "tt": 1.05, + "st": 1.5, + "vx": 1.2, + "mj": 2.0, + } + + cfg.x.inference_processes = [ + f"{base_proc}_{subproc_suffix}" + for base_proc in ("tt", "st") + for subproc_suffix in ("3q", "2q", "0o1q", "bkg") + ] + [ + "vx", + "mj", + ] + + # setup for fit + # TODO make configurable from CLI (as params of inference model) + cfg.x.fit_setup = { + "channels": [ + "1m", + "1e", + ], + "pt_bins": [ + "pt_300_400", + "pt_400_480", + "pt_480_600", + "pt_600_inf", + ], + "wp_names": [ + "very_tight", + "tight", + "medium", + "loose", + "very_loose", + ], + "fit_vars": [ + # "probejet_msoftdrop_inf_rebin", # use inference mSD + # "probejet_msoftdrop_widebins", + "probejet_msoftdrop_inf_rebin_fix", + ], + "shape_unc": [ + "fsr", + "isr", + # "electron", + "electron_reco", + "electron_id_iso", + "electron_trigger", + # "muon", + "muon_reco", + "muon_id", + "muon_iso", + "muon_trigger", + "minbias_xs", + # "top_pt", + "jec_Total", + "mur", + "muf", + # "btag_bc", + # "btag_light", + ], + } + if year == 2024: + cfg.x.fit_setup["shape_unc"] += [ + "btag_bc", + "btag_light", + ] + else: + cfg.x.fit_setup["shape_unc"] += [ + "btag_hf", + "btag_lf", + ] + + for process_name in process_names: + # add the process + proc = cfg.add_process(procs.get(process_name)) + + # mark the presence of a top quark + if any(proc.name.startswith(s) for s in ("tt", "st")): + proc.add_tag("has_top") + + # mark ttbar processes (needed for top pt reweighting) + if proc.name.startswith("tt"): + proc.add_tag("is_ttbar") + + # configuration of colors, labels, etc. can happen here + proc.color = colors.get(proc.name, "#aaaaaa") + + # + # datasets + # + dataset_names = add_datasets_from_yaml( + cfg, + limit_dataset_files=limit_dataset_files, + dataset_types=[ + "data", + "tt", + "st", + "dy", + "w_lnu", + "vv", + "qcd", + ], + log=False, + ) + + for dataset in cfg.datasets: + # update JECera information + if dataset.is_data and (dataset.name.endswith("c") or dataset.name.endswith("d")): + dataset.x.jec_era = "RunCD" + if "twchannel" in dataset.name: + dataset.add_tag("has_top_associated_w") + + # verify that the root processes of each dataset (or one of their + # ancestor processes) are registered in the config + verify_config_processes(cfg, warn=True) + logger.info(f"Added {len(cfg.processes)} processes and {len(cfg.datasets)} datasets to config '{cfg.name}'") + + # + # defaults + # + + # default objects, such as calibrator, selector, producer, + # ml model, inference model, etc + cfg.x.default_calibrator = "default" + cfg.x.default_selector = "default" + cfg.x.default_reducer = "cf_default" + cfg.x.default_producer = "default" + cfg.x.default_hist_producer = "default" + cfg.x.default_ml_model = None + cfg.x.default_inference_model = "default" # "uhh2" + cfg.x.default_categories = ("incl",) + cfg.x.default_variables = ( + "probejet_pt", + "probejet_mass", + "probejet_msoftdrop_widebins", + "probejet_tau32", + "probejet_max_subjet_btag_score_btagDeepB", + "probejet_msoftdrop_inf_rebin_fix", + ) + + # + # parameter groups + # + + # process groups for conveniently looping over certain processs + # (used in wrapper_factory and during plotting) + cfg.x.process_groups = { + "all": process_names, + "all_subprocs": cfg.x.inference_processes, + } + + # dataset groups for conveniently looping over certain datasets + # (used in wrapper_factory and during plotting) + cfg.x.dataset_groups = { + "all": dataset_names, + "data": ["data_*"], + "dy": ["dy*"], + "w_lnu": ["w_lnu*"], + "qcd_mu": ["qcd_mu*"], + "qcd_em": ["qcd_em*"], + "qcd": ["qcd*"], + "st": ["st*"], + "tt": ["tt*"], + "vv": ["ww_pythia", "wz_pythia", "zz_pythia"], + "vx": ["w_lnu*", "dy*", "ww_pythia", "wz_pythia", "zz_pythia"], + "mj": ["qcd*"], + "mc": ["dy*", "w_lnu*", "ww_pythia", "wz_pythia", "zz_pythia", "st*", "tt*", "qcd*"], + "testing": [ + "tt_sl_powheg", + "st_tchannel_t_4f_powheg", + "dy_m4to50_ht800to1500_madgraph", + "w_lnu_mlnu0to120_ht1500to2500_madgraph", + "ww_pythia", + "qcd_mu_pt600to800_pythia", + # "qcd_em_pt120to170_pythia", + ], + } + if cfg.x.cpn_tag == "2022preEE": + cfg.x.dataset_groups["testing"] += ["data_egamma_c", "data_mu_c"] + elif cfg.x.cpn_tag == "2022postEE": + cfg.x.dataset_groups["testing"] += ["data_egamma_f", "data_mu_f"] + elif cfg.x.cpn_tag == "2023preBPix": + cfg.x.dataset_groups["testing"] += ["data_egamma_c", "data_mu_c"] + elif cfg.x.cpn_tag == "2023postBPix": + cfg.x.dataset_groups["testing"] += ["data_egamma_d", "data_mu_d"] + elif cfg.x.cpn_tag == "2024": + cfg.x.dataset_groups["testing"] += ["data_e_c", "data_mu_c"] + + # category groups for conveniently looping over certain categories + # (used during plotting) + cfg.x.category_groups = { + "default": ["1m"], + "1m_wp_very_tight_pass": [ + "1m__pt_300_400__tau32_wp_very_tight_pass", + "1m__pt_400_480__tau32_wp_very_tight_pass", + "1m__pt_480_600__tau32_wp_very_tight_pass", + "1m__pt_600_inf__tau32_wp_very_tight_pass", + ], + "1m_wp_very_tight_fail": [ + "1m__pt_300_400__tau32_wp_very_tight_fail", + "1m__pt_400_480__tau32_wp_very_tight_fail", + "1m__pt_480_600__tau32_wp_very_tight_fail", + "1m__pt_600_inf__tau32_wp_very_tight_fail", + ], + "1m_wp_tight_pass": [ + "1m__pt_300_400__tau32_wp_tight_pass", + "1m__pt_400_480__tau32_wp_tight_pass", + "1m__pt_480_600__tau32_wp_tight_pass", + "1m__pt_600_inf__tau32_wp_tight_pass", + ], + "1m_wp_tight_fail": [ + "1m__pt_300_400__tau32_wp_tight_fail", + "1m__pt_400_480__tau32_wp_tight_fail", + "1m__pt_480_600__tau32_wp_tight_fail", + "1m__pt_600_inf__tau32_wp_tight_fail", + ], + "1m_wp_medium_pass": [ + "1m__pt_300_400__tau32_wp_medium_pass", + "1m__pt_400_480__tau32_wp_medium_pass", + "1m__pt_480_600__tau32_wp_medium_pass", + "1m__pt_600_inf__tau32_wp_medium_pass", + ], + "1m_wp_medium_fail": [ + "1m__pt_300_400__tau32_wp_medium_fail", + "1m__pt_400_480__tau32_wp_medium_fail", + "1m__pt_480_600__tau32_wp_medium_fail", + "1m__pt_600_inf__tau32_wp_medium_fail", + ], + "1m_wp_loose_pass": [ + "1m__pt_300_400__tau32_wp_loose_pass", + "1m__pt_400_480__tau32_wp_loose_pass", + "1m__pt_480_600__tau32_wp_loose_pass", + "1m__pt_600_inf__tau32_wp_loose_pass", + ], + "1m_wp_loose_fail": [ + "1m__pt_300_400__tau32_wp_loose_fail", + "1m__pt_400_480__tau32_wp_loose_fail", + "1m__pt_480_600__tau32_wp_loose_fail", + "1m__pt_600_inf__tau32_wp_loose_fail", + ], + "1e_wp_loose_pass": [ + "1e__pt_300_400__tau32_wp_loose_pass", + "1e__pt_400_480__tau32_wp_loose_pass", + "1e__pt_480_600__tau32_wp_loose_pass", + "1e__pt_600_inf__tau32_wp_loose_pass", + ], + "1e_wp_loose_fail": [ + "1e__pt_300_400__tau32_wp_loose_fail", + "1e__pt_400_480__tau32_wp_loose_fail", + "1e__pt_480_600__tau32_wp_loose_fail", + "1e__pt_600_inf__tau32_wp_loose_fail", + ], + "1m_wp_very_loose_pass": [ + "1m__pt_300_400__tau32_wp_very_loose_pass", + "1m__pt_400_480__tau32_wp_very_loose_pass", + "1m__pt_480_600__tau32_wp_very_loose_pass", + "1m__pt_600_inf__tau32_wp_very_loose_pass", + ], + "1m_wp_very_loose_fail": [ + "1m__pt_300_400__tau32_wp_very_loose_fail", + "1m__pt_400_480__tau32_wp_very_loose_fail", + "1m__pt_480_600__tau32_wp_very_loose_fail", + "1m__pt_600_inf__tau32_wp_very_loose_fail", + ], + "1m_all_pt_wp_pass_fail": [ + "1m__pt_300_400__tau32_wp_very_tight_pass", + "1m__pt_400_480__tau32_wp_very_tight_pass", + "1m__pt_480_600__tau32_wp_very_tight_pass", + "1m__pt_600_inf__tau32_wp_very_tight_pass", + "1m__pt_300_400__tau32_wp_very_tight_fail", + "1m__pt_400_480__tau32_wp_very_tight_fail", + "1m__pt_480_600__tau32_wp_very_tight_fail", + "1m__pt_600_inf__tau32_wp_very_tight_fail", + "1m__pt_300_400__tau32_wp_tight_pass", + "1m__pt_400_480__tau32_wp_tight_pass", + "1m__pt_480_600__tau32_wp_tight_pass", + "1m__pt_600_inf__tau32_wp_tight_pass", + "1m__pt_300_400__tau32_wp_tight_fail", + "1m__pt_400_480__tau32_wp_tight_fail", + "1m__pt_480_600__tau32_wp_tight_fail", + "1m__pt_600_inf__tau32_wp_tight_fail", + "1m__pt_300_400__tau32_wp_medium_pass", + "1m__pt_400_480__tau32_wp_medium_pass", + "1m__pt_480_600__tau32_wp_medium_pass", + "1m__pt_600_inf__tau32_wp_medium_pass", + "1m__pt_300_400__tau32_wp_medium_fail", + "1m__pt_400_480__tau32_wp_medium_fail", + "1m__pt_480_600__tau32_wp_medium_fail", + "1m__pt_600_inf__tau32_wp_medium_fail", + "1m__pt_300_400__tau32_wp_loose_pass", + "1m__pt_400_480__tau32_wp_loose_pass", + "1m__pt_480_600__tau32_wp_loose_pass", + "1m__pt_600_inf__tau32_wp_loose_pass", + "1m__pt_300_400__tau32_wp_loose_fail", + "1m__pt_400_480__tau32_wp_loose_fail", + "1m__pt_480_600__tau32_wp_loose_fail", + "1m__pt_600_inf__tau32_wp_loose_fail", + "1m__pt_300_400__tau32_wp_very_loose_pass", + "1m__pt_400_480__tau32_wp_very_loose_pass", + "1m__pt_480_600__tau32_wp_very_loose_pass", + "1m__pt_600_inf__tau32_wp_very_loose_pass", + "1m__pt_300_400__tau32_wp_very_loose_fail", + "1m__pt_400_480__tau32_wp_very_loose_fail", + "1m__pt_480_600__tau32_wp_very_loose_fail", + "1m__pt_600_inf__tau32_wp_very_loose_fail", + ], + "1e_all_pt_wp_pass_fail": [ + "1e__pt_300_400__tau32_wp_very_tight_pass", + "1e__pt_400_480__tau32_wp_very_tight_pass", + "1e__pt_480_600__tau32_wp_very_tight_pass", + "1e__pt_600_inf__tau32_wp_very_tight_pass", + "1e__pt_300_400__tau32_wp_very_tight_fail", + "1e__pt_400_480__tau32_wp_very_tight_fail", + "1e__pt_480_600__tau32_wp_very_tight_fail", + "1e__pt_600_inf__tau32_wp_very_tight_fail", + "1e__pt_300_400__tau32_wp_tight_pass", + "1e__pt_400_480__tau32_wp_tight_pass", + "1e__pt_480_600__tau32_wp_tight_pass", + "1e__pt_600_inf__tau32_wp_tight_pass", + "1e__pt_300_400__tau32_wp_tight_fail", + "1e__pt_400_480__tau32_wp_tight_fail", + "1e__pt_480_600__tau32_wp_tight_fail", + "1e__pt_600_inf__tau32_wp_tight_fail", + "1e__pt_300_400__tau32_wp_medium_pass", + "1e__pt_400_480__tau32_wp_medium_pass", + "1e__pt_480_600__tau32_wp_medium_pass", + "1e__pt_600_inf__tau32_wp_medium_pass", + "1e__pt_300_400__tau32_wp_medium_fail", + "1e__pt_400_480__tau32_wp_medium_fail", + "1e__pt_480_600__tau32_wp_medium_fail", + "1e__pt_600_inf__tau32_wp_medium_fail", + "1e__pt_300_400__tau32_wp_loose_pass", + "1e__pt_400_480__tau32_wp_loose_pass", + "1e__pt_480_600__tau32_wp_loose_pass", + "1e__pt_600_inf__tau32_wp_loose_pass", + "1e__pt_300_400__tau32_wp_loose_fail", + "1e__pt_400_480__tau32_wp_loose_fail", + "1e__pt_480_600__tau32_wp_loose_fail", + "1e__pt_600_inf__tau32_wp_loose_fail", + "1e__pt_300_400__tau32_wp_very_loose_pass", + "1e__pt_400_480__tau32_wp_very_loose_pass", + "1e__pt_480_600__tau32_wp_very_loose_pass", + "1e__pt_600_inf__tau32_wp_very_loose_pass", + "1e__pt_300_400__tau32_wp_very_loose_fail", + "1e__pt_400_480__tau32_wp_very_loose_fail", + "1e__pt_480_600__tau32_wp_very_loose_fail", + "1e__pt_600_inf__tau32_wp_very_loose_fail", + ], + "1m_all_pt": [ + "1m__pt_300_400", + "1m__pt_400_480", + "1m__pt_480_600", + "1m__pt_600_inf", + ], + "1e_all_pt": [ + "1e__pt_300_400", + "1e__pt_400_480", + "1e__pt_480_600", + "1e__pt_600_inf", + ], + "1m_all_wp": [ + "1m__tau32_wp_very_tight_pass", + "1m__tau32_wp_very_tight_fail", + "1m__tau32_wp_tight_pass", + "1m__tau32_wp_tight_fail", + "1m__tau32_wp_medium_pass", + "1m__tau32_wp_medium_fail", + "1m__tau32_wp_loose_pass", + "1m__tau32_wp_loose_fail", + "1m__tau32_wp_very_loose_pass", + "1m__tau32_wp_very_loose_fail", + ], + "1e_all_wp": [ + "1e__tau32_wp_very_tight_pass", + "1e__tau32_wp_very_tight_fail", + "1e__tau32_wp_tight_pass", + "1e__tau32_wp_tight_fail", + "1e__tau32_wp_medium_pass", + "1e__tau32_wp_medium_fail", + "1e__tau32_wp_loose_pass", + "1e__tau32_wp_loose_fail", + "1e__tau32_wp_very_loose_pass", + "1e__tau32_wp_very_loose_fail", + ], + } + + # variable groups for conveniently looping over certain variables + # (used during plotting) + cfg.x.variable_groups = {} + + # shift groups for conveniently looping over certain shifts + # (used during plotting) + cfg.x.shift_groups = {} + + # selector step groups for conveniently looping over certain steps + # (used in cutflow tasks) + cfg.x.selector_step_groups = { + "default": [ + "LeptonTrigger", "Lepton", "AddLeptonVeto", "MET", "BJet", "METFilters", + ], + } + + # Exception: no weight producer configured for task. cf.MergeShiftedHistograms. + # As of 02.05.2024, it is required to pass a weight_producer for tasks creating histograms. + # You can add a 'default_weight_producer' to your config or directly add the weight_producer + # on command line via the '--weight_producer' parameter. To reproduce results from before this date, + # you can use the 'all_weights' weight_producer defined in columnflow.weight.all_weights: + # With cf 0.3.x, the 'weight_producer' has been renamed to 'hist_producer'. + + # custom labels for selector steps + cfg.x.selector_step_labels = {} + + # plotting settings groups + cfg.x.general_settings_groups = {} + cfg.x.process_settings_groups = {} + cfg.x.variable_settings_groups = {} + + # + # dataset customization + # + + # custom method and sandbox for determining dataset lfns + cfg.x.get_dataset_lfns = None + cfg.x.get_dataset_lfns_sandbox = None + + # whether to validate the number of obtained LFNs in GetDatasetLFNs + cfg.x.validate_dataset_lfns = limit_dataset_files is None + + # + # tagger working points + # + + # full b-tag working points dict + cfg.x.btag_working_points = btag_wps(cfg, full=True) + # store only upart wps for fixed wp sf producer + cfg.x.btag_working_points.btagUParTAK4B.fixed_wp = btag_wps(cfg, full=False) + + # top-tag working points + # https://twiki.cern.ch/twiki/bin/view/CMS/JetTopTagging?rev=41 + cfg.x.toptag_working_points = toptag_wps(era=cfg.x.cpn_tag) + + # + # selector configurations + # FIXME: adapt for Run3? + # + + # lepton selection parameters + cfg.x.lepton_selection = DotDict.wrap({ + "mu": { + "column": "Muon", + "min_pt": 55, + "max_abseta": 2.4, + "triggers": { + # # FIXME: adapt for UL17 + # "IsoMu24", + # "IsoTkMu24", + # as mttbar: + # "IsoMu27", + # updated Run3 trigger recommendations + "Mu50", + "HighPtTkMu100", + "CascadeMu100", + }, + "id": { + # "column": "tightId", + # "value": True, + # Run 3 high pT muon ID + "column": "highPtId", + "value": 2, # (1 = tracker high pT, 2 = global high pT, which includes tracker high pT) + }, + # "rel_iso": "pfRelIso03_all", + # "max_rel_iso": 1.5, + # Run 3 high pt muon iso + "rel_iso": "tkRelIso", + "max_rel_iso": 0.05, + # veto events with additional leptons passing looser cuts + "min_pt_addveto": 30, + "id_addveto": { + "column": "looseId", + "value": True, + }, + }, + "e": { + "column": "Electron", + "min_pt": 55, + "max_abseta": 2.4, + "triggers": { + # FIXME: adapt for UL17 + # "Ele27_WPTight_Gsf", + # "Ele115_CaloIdVT_GsfTrkIdT", + # as mttbar: + # "Ele35_WPTight_Gsf", + # updated Run3 trigger recommendations + "Ele30_WPTight_Gsf", + }, + #"id": "mvaFall17V2Iso_WP90", # noqa + "id": { + "column": "cutBased", + "value": 4, + }, + # veto events with additional leptons passing looser cuts + "min_pt_addveto": 30, + #"id_addveto": "mvaFall17V2Iso_WPL", # noqa + "id_addveto": { + "column": "cutBased", + "value": 1, + }, + }, + }) + + # jet selection parameters + subjet_wp_key = "btagUParTAK4B" if year == 2024 else "deepcsv" + subjet_btag_wp = getattr(cfg.x.btag_working_points, subjet_wp_key).loose + if subjet_btag_wp < 0: + raise ValueError(f"Invalid subjet b-tag working point for year {year}. Please check the configuration.") + + ak4_btag_wp_key = "btagUParTAK4B" if year == 2024 else "deepjet" + ak4_btag_wp = getattr(cfg.x.btag_working_points, ak4_btag_wp_key).medium + if ak4_btag_wp < 0: + raise ValueError(f"Invalid AK4 b-tag working point for year {year}. Please check the configuration.") + + cfg.x.jet_selection = DotDict.wrap({ + "ak8": { + "column": "FatJet", + "min_pt": 300, + "max_abseta": 2.5, + "msoftdrop_range": (105, 210), + # https://twiki.cern.ch/twiki/bin/view/CMS/JetID13p6TeV + "jetId": 2, # bit2 (2): pass tight ID, fail tightLepVeto, bit3 (6): pass tight and tightLepVeto ID + # probe jet pt bins (used by category builder) + "pt_bins": [300, 400, 480, 600, None], + # parameters for b-tagged subjets + "subjet_column": "SubJet", + "subjet_btag": "btagDeepB" if not year == 2024 else "btagUParTAK4B", + "subjet_btag_wp": subjet_btag_wp, + }, + # TODO: implement (requires custom nano) + "hotvr": { + "column": "HOTVRJetForTopTagging", + "min_pt": 200, + "max_abseta": 2.5, + # clustering parameters (not needed for analysis, added for reference) + # https://twiki.cern.ch/twiki/bin/view/CMS/JetTopTagging?rev=41 + "r_min_max": (0.1, 1.5), + "rho": 600, # GeV + "mu": 30, # Gev, mass jump threshold + "theta": 0.7, # mass jump strength + "min_pt_subjet": 30, # min pt of subject + }, + "ak4": { + "column": "Jet", + # https://twiki.cern.ch/twiki/bin/view/CMS/JetID13p6TeV + "jetId": 2, # bit2 (2): pass tight ID, fail tightLepVeto, bit3 (6): pass tight and tightLepVeto ID + "min_pt": 15, # TODO: check UHH2 + "max_abseta": 2.5, # TODO: check UHH2 + "btag_column": "btagDeepFlavB" if not year == 2024 else "btagUParTAK4B", + "btag_wp": ak4_btag_wp, + }, + }) + + # MET selection parameters + # https://twiki.cern.ch/twiki/bin/viewauth/CMS/MissingETRun2Corrections?rev=79#xy_Shift_Correction_MET_phi_modu + cfg.x.met_selection = DotDict.wrap({ + "default": { + "column": "PuppiMET", + "min_pt": 50, + }, + }) + + # + # luminosity + # + + # lumi values in inverse pb + # https://twiki.cern.ch/twiki/bin/viewauth/CMS/PdmVRun3Analysis + if year == 2022: + if campaign.x.EE == "pre": + cfg.x.luminosity = Number(7971, { + "lumi_13TeV_2022": 0.01j, + "lumi_13TeV_correlated": 0.006j, + }) + elif campaign.x.EE == "post": + cfg.x.luminosity = Number(26337, { + "lumi_13TeV_2022": 0.01j, + "lumi_13TeV_correlated": 0.006j, + }) + elif year == 2023: + if campaign.has_tag("preBPix"): + cfg.x.luminosity = Number(17794, { + "lumi_13TeV_2023": 0.01j, + "lumi_13TeV_correlated": 0.006j, + }) + elif campaign.has_tag("postBPix"): + cfg.x.luminosity = Number(9451, { + "lumi_13TeV_2023": 0.01j, + "lumi_13TeV_correlated": 0.006j, + }) + elif year == 2024: + # Total - EraB + cfg.x.luminosity = Number(109_950.0 - 130.0, { + "lumi_13p6TeV_2024": 0.016j, # CERN-CMS-DP-2026-003 + }) + else: + raise NotImplementedError(f"Luminosity for year {year} is not defined.") + + # + # cross sections + # + + # cross sections for diboson samples; taken from: + # - ww (NNLO): https://arxiv.org/abs/1408.5243 + # - wz (NLO): https://arxiv.org/abs/1105.0020 + # - zz (NNLO): https://www.sciencedirect.com/science/article/pii/S0370269314004614?via%3Dihub + diboson_xsecs_13 = { + "ww": Number(118.7, {"scale": (0.025j, 0.022j)}), + "wz": Number(46.74, {"scale": (0.041j, 0.033j)}), + # "wz": Number(28.55, {"scale": (0.041j, 0.032j)}) + Number(18.19, {"scale": (0.041j, 0.033j)}), # (W+Z) + (W-Z) # noqa + "zz": Number(16.99, {"scale": (0.032j, 0.024j)}), + } + # TODO Use 14 TeV xs for Run 3? + diboson_xsecs_14 = { + "ww": Number(131.1, {"scale": (0.026j, 0.022j)}), + "wz": Number(67.06, {"scale": (0.039j, 0.031j)}), + # "wz": Number(31.50, {"scale": (0.039j, 0.030j)}) + Number(20.32, {"scale": (0.039j, 0.031j)}), # (W+Z) + (W-Z) # noqa + "zz": Number(18.77, {"scale": (0.032j, 0.024j)}), + } + + # linear interpolation between 13 and 14 TeV + diboson_xsecs_13_6 = { + ds: diboson_xsecs_13[ds] + (13.6 - 13.0) * (diboson_xsecs_14[ds] - diboson_xsecs_13[ds]) / (14.0 - 13.0) + for ds in diboson_xsecs_13.keys() # ww: 125.8 wz: 58.932 zz: 18.058 noqa + } + + for ds in diboson_xsecs_14: + procs.n(ds).set_xsec(13.6, diboson_xsecs_13_6[ds]) + + # + # MET filters + # + + # https://twiki.cern.ch/twiki/bin/viewauth/CMS/MissingETOptionalFiltersRun2#Run_3_recommendations + cfg.x.met_filters = { + "Flag.goodVertices", + "Flag.globalSuperTightHalo2016Filter", + "Flag.EcalDeadCellTriggerPrimitiveFilter", + "Flag.BadPFMuonFilter", + "Flag.BadPFMuonDzFilter", + "Flag.eeBadScFilter", + "Flag.ecalBadCalibFilter", + } + if year == 2024: + cfg.x.met_filters.add("Flag.hfNoisyHitsFilter") + + # + # JEC & JER # FIXME: Taken from HBW + # https://github.com/uhh-cms/hh2bbww/blob/master/hbw/config/config_run2.py#L138C5-L269C1 + # + + # jec configuration + # https://twiki.cern.ch/twiki/bin/view/CMS/JECDataMC?rev=2017#Jet_Energy_Corrections_in_Run2 + + # jec configuration taken from HBW + # https://github.com/uhh-cms/hh2bbww/blob/master/hbw/config/config_run2.py#L138C5-L269C1 + # https://twiki.cern.ch/twiki/bin/view/CMS/JECDataMC?rev=201 + jerc_postfix = campaign.x.postfix + if jerc_postfix not in ("", "EE", "BPix"): + raise ValueError(f"Invalid JERC postfix '{jerc_postfix}' for campaign {campaign.name}.") + if year == 2022: + jer_campaign = jec_campaign = f"Summer{year2}{jerc_postfix}_22Sep2023" + elif year == 2023: + era = "Cv1234" if campaign.has_tag("preBPix") else "D" + jer_campaign = f"Summer{year2}{jerc_postfix}Prompt{year2}_Run{era}" + jec_campaign = f"Summer{year2}{jerc_postfix}Prompt{year2}" + elif year == 2024: + jec_campaign = "Summer24Prompt24" + jer_campaign = "Summer23BPixPrompt23_RunD" # no 2024 JER yet, use 2023 BPix: https://cms-jerc.web.cern.ch/Recommendations/#2024_1 # noqa + else: + raise NotImplementedError(f"JEC/JER configuration for year {year} is not defined.") + + jet_type = "AK4PFPuppi" + fatjet_type = "AK8PFPuppi" + jec_ak4_version = jec_ak8_version = { + 2022: "V3", + 2023: "V2" if not year == 2023 else "V3", + 2024: "V2", + }[year] + + cfg.x.jec = DotDict.wrap({ + "Jet": { + "campaign": jec_campaign, + "version": jec_ak4_version, + "jet_type": jet_type, + "levels": ["L1FastJet", "L2Relative", "L2L3Residual", "L3Absolute"], + "levels_for_type1_met": ["L1FastJet"], + "uncertainty_sources": [ + # "AbsoluteStat", + # "AbsoluteScale", + # "AbsoluteSample", + # "AbsoluteFlavMap", + # "AbsoluteMPFBias", + # "Fragmentation", + # "SinglePionECAL", + # "SinglePionHCAL", + # "FlavorQCD", + # "TimePtEta", + # "RelativeJEREC1", + # "RelativeJEREC2", + # "RelativeJERHF", + # "RelativePtBB", + # "RelativePtEC1", + # "RelativePtEC2", + # "RelativePtHF", + # "RelativeBal", + # "RelativeSample", + # "RelativeFSR", + # "RelativeStatFSR", + # "RelativeStatEC", + # "RelativeStatHF", + # "PileUpDataMC", + # "PileUpPtRef", + # "PileUpPtBB", + # "PileUpPtEC1", + # "PileUpPtEC2", + # "PileUpPtHF", + # "PileUpMuZero", + # "PileUpEnvelope", + # "SubTotalPileUp", + # "SubTotalRelative", + # "SubTotalPt", + # "SubTotalScale", + # "SubTotalAbsolute", + # "SubTotalMC", + "Total", + # "TotalNoFlavor", + # "TotalNoTime", + # "TotalNoFlavorNoTime", + # "FlavorZJet", + # "FlavorPhotonJet", + # "FlavorPureGluon", + # "FlavorPureQuark", + # "FlavorPureCharm", + # "FlavorPureBottom", + # "TimeRunA", + # "TimeRunB", + # "TimeRunC", + # "TimeRunD", + # "CorrelationGroupMPFInSitu", + # "CorrelationGroupIntercalibration", + # "CorrelationGroupbJES", + # "CorrelationGroupFlavor", + # "CorrelationGroupUncorrelated", + ], + # "data_per_era": True if year == 2022 else False, + "data_per_era": False, + }, + "FatJet": { + "campaign": jec_campaign, + "version": jec_ak8_version, + "jet_type": fatjet_type, + "levels": ["L1FastJet", "L2Relative", "L2L3Residual", "L3Absolute"], + "levels_for_type1_met": ["L1FastJet"], + "uncertainty_sources": [ + # "AbsoluteStat", + # "AbsoluteScale", + # "AbsoluteSample", + # "AbsoluteFlavMap", + # "AbsoluteMPFBias", + # "Fragmentation", + # "SinglePionECAL", + # "SinglePionHCAL", + # "FlavorQCD", + # "TimePtEta", + # "RelativeJEREC1", + # "RelativeJEREC2", + # "RelativeJERHF", + # "RelativePtBB", + # "RelativePtEC1", + # "RelativePtEC2", + # "RelativePtHF", + # "RelativeBal", + # "RelativeSample", + # "RelativeFSR", + # "RelativeStatFSR", + # "RelativeStatEC", + # "RelativeStatHF", + # "PileUpDataMC", + # "PileUpPtRef", + # "PileUpPtBB", + # "PileUpPtEC1", + # "PileUpPtEC2", + # "PileUpPtHF", + # "PileUpMuZero", + # "PileUpEnvelope", + # "SubTotalPileUp", + # "SubTotalRelative", + # "SubTotalPt", + # "SubTotalScale", + # "SubTotalAbsolute", + # "SubTotalMC", + "Total", + # "TotalNoFlavor", + # "TotalNoTime", + # "TotalNoFlavorNoTime", + # "FlavorZJet", + # "FlavorPhotonJet", + # "FlavorPureGluon", + # "FlavorPureQuark", + # "FlavorPureCharm", + # "FlavorPureBottom", + # "TimeRunA", + # "TimeRunB", + # "TimeRunC", + # "TimeRunD", + # "CorrelationGroupMPFInSitu", + # "CorrelationGroupIntercalibration", + # "CorrelationGroupbJES", + # "CorrelationGroupFlavor", + # "CorrelationGroupUncorrelated", + ], + # "data_per_era": True if year == 2022 else False, + "data_per_era": False, + }, + "SubJet": { + "campaign": jec_campaign, + "version": jec_ak4_version, + "jet_type": jet_type, + "levels": ["L1FastJet", "L2Relative", "L2L3Residual", "L3Absolute"], + "levels_for_type1_met": ["L1FastJet"], + "uncertainty_sources": [ + # "AbsoluteStat", + # "AbsoluteScale", + # "AbsoluteSample", + # "AbsoluteFlavMap", + # "AbsoluteMPFBias", + # "Fragmentation", + # "SinglePionECAL", + # "SinglePionHCAL", + # "FlavorQCD", + # "TimePtEta", + # "RelativeJEREC1", + # "RelativeJEREC2", + # "RelativeJERHF", + # "RelativePtBB", + # "RelativePtEC1", + # "RelativePtEC2", + # "RelativePtHF", + # "RelativeBal", + # "RelativeSample", + # "RelativeFSR", + # "RelativeStatFSR", + # "RelativeStatEC", + # "RelativeStatHF", + # "PileUpDataMC", + # "PileUpPtRef", + # "PileUpPtBB", + # "PileUpPtEC1", + # "PileUpPtEC2", + # "PileUpPtHF", + # "PileUpMuZero", + # "PileUpEnvelope", + # "SubTotalPileUp", + # "SubTotalRelative", + # "SubTotalPt", + # "SubTotalScale", + # "SubTotalAbsolute", + # "SubTotalMC", + "Total", + # "TotalNoFlavor", + # "TotalNoTime", + # "TotalNoFlavorNoTime", + # "FlavorZJet", + # "FlavorPhotonJet", + # "FlavorPureGluon", + # "FlavorPureQuark", + # "FlavorPureCharm", + # "FlavorPureBottom", + # "TimeRunA", + # "TimeRunB", + # "TimeRunC", + # "TimeRunD", + # "CorrelationGroupMPFInSitu", + # "CorrelationGroupIntercalibration", + # "CorrelationGroupbJES", + # "CorrelationGroupFlavor", + # "CorrelationGroupUncorrelated", + ], + # "data_per_era": True if year == 2022 else False, + "data_per_era": False, + }, + }) + + # JER + # https://twiki.cern.ch/twiki/bin/view/CMS/JetResolution?rev=107 + cfg.x.jer = DotDict.wrap({ + "Jet": { + "campaign": jer_campaign, + "version": {2022: "JRV1", 2023: "JRV1", 2024: "JRV1"}[year], + "jet_type": jet_type, + }, + "FatJet": { + "campaign": jer_campaign, + "version": {2022: "JRV1", 2023: "JRV1", 2024: "JRV1"}[year], + "jet_type": fatjet_type, + }, + "SubJet": { + "campaign": jer_campaign, + "version": {2022: "JRV1", 2023: "JRV1", 2024: "JRV1"}[year], + "jet_type": jet_type, + }, + }) + + # JEC uncertainty sources propagated to btag scale factors + # (names derived from contents in BTV correctionlib file) + cfg.x.btag_sf_jec_sources = [ + "", # total + "Absolute", + "AbsoluteMPFBias", + "AbsoluteScale", + "AbsoluteStat", + f"Absolute_{year}", + "BBEC1", + f"BBEC1_{year}", + "EC2", + f"EC2_{year}", + "FlavorQCD", + "Fragmentation", + "HF", + f"HF_{year}", + "PileUpDataMC", + "PileUpPtBB", + "PileUpPtEC1", + "PileUpPtEC2", + "PileUpPtHF", + "PileUpPtRef", + "RelativeBal", + "RelativeFSR", + "RelativeJEREC1", + "RelativeJEREC2", + "RelativeJERHF", + "RelativePtBB", + "RelativePtEC1", + "RelativePtEC2", + "RelativePtHF", + "RelativeSample", + f"RelativeSample_{year}", + "RelativeStatEC", + "RelativeStatFSR", + "RelativeStatHF", + "SinglePionECAL", + "SinglePionHCAL", + "TimePtEta", + ] + + if cfg.x.run == 2: + cfg.x.met_phi_correction_set = "{variable}_metphicorr_pfmet_{data_source}" + else: + from columnflow.calibration.cms.met import METPhiConfig + met_column = cfg.x.met_selection.default.column + cfg.x.met_phi_correction = METPhiConfig( + met_name=met_column, + met_type=met_column, + correction_set="met_xy_corrections", + keep_uncorrected=True, # TODO do we need this? + pt_phi_variations={ + "stat_xdn": "metphi_statx_down", + "stat_xup": "metphi_statx_up", + "stat_ydn": "metphi_staty_down", + "stat_yup": "metphi_staty_up", + }, + variations={ + "pu_dn": "minbias_xs_down", + "pu_up": "minbias_xs_up", + }, + ) + + # + # producer configurations + # + + # lepton sf taken from + # https://github.com/uhh-cms/hh2bbww/blob/master/hbw/config/config_run2.py#L338C1-L352C85 + # names of electron correction sets and working points + # (used in the electron_sf producer) + if cfg.x.cpn_tag == "2022postEE": + sf_campaign = "2022Re-recoE+PromptFG" + # TODO: we need to use different SFs for control regions + elif cfg.x.cpn_tag == "2022preEE": + sf_campaign = "2022Re-recoBCD" + elif cfg.x.cpn_tag == "2023preBPix": + sf_campaign = "2023PromptC" + elif cfg.x.cpn_tag == "2023postBPix": + sf_campaign = "2023PromptD" + elif cfg.x.cpn_tag == "2024": + sf_campaign = "2024Prompt" + else: + raise ValueError(f"Invalid campaign tag '{cfg.x.cpn_tag}' for electron SF configuration.") + + cfg.x.electron_reco_sf_config = ElectronSFConfig( + correction="Electron-ID-SF", + campaign=sf_campaign, + working_point={ + "RecoBelow20": (lambda variables: variables["pt"] < 20), + "Reco20to75": (lambda variables: (variables["pt"] >= 20) & (variables["pt"] < 75.0)), + "RecoAbove75": (lambda variables: variables["pt"] >= 75.0), + }, + ) + cfg.x.electron_id_iso_sf_config = ElectronSFConfig( + correction="Electron-ID-SF", + campaign=sf_campaign, + # working_point={ + # "wp80iso": (lambda variables: variables["pt"] > 10), # NOTE: probably the wrong SF + # }, + working_point="Tight", + ) + cfg.x.electron_trigger_sf_config = ElectronSFConfig( + correction="Electron-HLT-SF", + campaign=sf_campaign, + hlt_path="HLT_SF_Ele30_TightID", + ) + + # names of muon correction sets and working points + # (used in the muon producer) + # TODO: we need to use different SFs for control regions + cfg.x.muon_reco_sf_config = MuonSFConfig( + correction="NUM_GlobalMuons_DEN_TrackerMuonProbes", + ) + cfg.x.muon_id_sf_config = MuonSFConfig( + correction="NUM_HighPtID_DEN_GlobalMuonProbes", + ) + cfg.x.muon_iso_sf_config = MuonSFConfig( + correction="NUM_probe_TightRelTkIso_DEN_HighPtProbes", + ) + cfg.x.muon_trigger_sf_config = MuonSFConfig( + correction="NUM_HLT_DEN_TrkHighPtTightRelIsoProbes", + ) + + if year == 2024: + cfg.x.jet_id = JetIdConfig( + corrections={"AK4PUPPI_Tight": 2, "AK4PUPPI_TightLeptonVeto": 3}, + ) + cfg.x.fatjet_id = JetIdConfig( + corrections={"AK8PUPPI_Tight": 2, "AK8PUPPI_TightLeptonVeto": 3}, + ) + else: + logger.debug(f"(Fat)Jet ID recalculation not configured for {cfg.x.cpn_tag} campaign. Will be skipped.") + cfg.add_tag("skip_jet_ids") + + # b tagging SF configuration + discr = "btagDeepFlavB" if year != 2024 else "btagUParTAK4B" + + btag_uncs = { + # combined(?) uncertainties + # uncertainties to b/c jets + "down_bc": "bc_down", + "up_bc": "bc_up", + # uncertainties to light jets + "down_light": "light_down", + "up_light": "light_up", + # split uncertainties(?) (all needed?) + # uncertainties to b/c jets + "up_correlated_bc": "correlated_bc_up", + "up_uncorrelated_bc": "uncorrelated_bc_up", + "up_bfragmentation_bc": "bfragmentation_bc_up", + "up_fsrdef_bc": "fsrdef_bc_up", + "up_hdamp_bc": "hdamp_bc_up", + "up_isrdef_bc": "isrdef_bc_up", + "up_jer_bc": "jer_bc_up", + "up_jes_bc": "jes_bc_up", + "up_muf_bc": "muf_bc_up", + "up_mur_bc": "mur_bc_up", + "up_pdfas_bc": "pdfas_bc_up", + "up_pileup_bc": "pileup_bc_up", + "up_statistic_bc": "statistic_bc_up", # to be decorrelated between years + "up_topmass_bc": "topmass_bc_up", + "up_type3_bc": "type3_bc_up", + "down_correlated_bc": "correlated_bc_down", + "down_uncorrelated_bc": "uncorrelated_bc_down", + "down_bfragmentation_bc": "bfragmentation_bc_down", + "down_fsrdef_bc": "fsrdef_bc_down", + "down_hdamp_bc": "hdamp_bc_down", + "down_isrdef_bc": "isrdef_bc_down", + "down_jer_bc": "jer_bc_down", + "down_jes_bc": "jes_bc_down", + "down_muf_bc": "muf_bc_down", + "down_mur_bc": "mur_bc_down", + "down_pdfas_bc": "pdfas_bc_down", + "down_pileup_bc": "pileup_bc_down", + "down_statistic_bc": "statistic_bc_down", # to be decorrelated between years + "down_topmass_bc": "topmass_bc_down", + "down_type3_bc": "type3_bc_down", + # uncertainties to light jets + "down_correlated_light": "correlated_light_down", + "up_correlated_light": "correlated_light_up", + "down_uncorrelated_light": "uncorrelated_light_down", + "up_uncorrelated_light": "uncorrelated_light_up", + } + if year == 2024: + cfg.add_tag("skip_btag_weights") + logger.debug("Setting up fixed WP based btag SFs for 2024, as shape based SFs are not yet available. Please switch to shape based SFs as soon as they are available.") # noqa + # NOTE: switch to shape based SF also for 2024 as soon as they are available + cfg.x.btag_sf = BTagSFConfig( + correction_set="Dummy", + jec_sources=cfg.x.btag_sf_jec_sources, + discriminator=discr, + ) + # implementation from hbt analysis: + # https://github.com/uhh-cms/hh2bbtautau/blob/4b2f1bc57a9c2ada18776e5ac6f0372269e1e26c/hbt/config/configs_hbt.py#L1410 # noqa + cfg.x.btag_wp_count_config = BTagWPCountConfig( + jet_name="Jet", + btag_column=discr, + btag_wps=cfg.x.btag_working_points.btagUParTAK4B.fixed_wp, + pt_edges=(0, 20, 30, 50, 70, 100, 140, 200, 300, 600, 10_000), + abs_eta_edges=(0.0, 1.0, 1.5, 2.0, 5.0), + ) + + def dataset_groups(dataset_inst: od.Dataset) -> list[od.Dataset]: + # check which group the dataset belongs to + for group_index in range(0, len(cfg.x.btag_wp_eff_groups)): + group_tag = f"btag_wp_eff_group_{group_index}" + if dataset_inst.has_tag(group_tag): + return [ + _dataset_inst + for _dataset_inst in cfg.datasets + if _dataset_inst.has_tag(group_tag) + ] + raise NotImplementedError(f"btag WP efficiency group not implemented for dataset {dataset_inst.name}") + + cfg.x.btag_wp_sf_config = BTagWPSFConfig( + jet_name="Jet", + btag_column=discr, + correction_set="UParTAK4_merged", + btag_wps=cfg.x.btag_working_points.btagUParTAK4B.fixed_wp, + dataset_groups=dataset_groups, + systs=btag_uncs, + # further merge eta bins for sufficient statistics in each bin + abs_eta_edges=(0.0, 1.5, 5.0), + wp_merging={ + # remove xxtight for better stats + "loose": ["loose"], + "medium": ["medium"], + "tight": ["tight"], + "xtight": ["xtight"], + # "xxtight": ["xxtight"], + }, + pt_edges=(0, 20, 30, 50, 70, 100, 140, 200, 300, 600, 10_000) if not limit_dataset_files == 2 else (0, 10_000), # no pt binning for testing with limited files # noqa + ) + else: + cfg.add_tag("skip_btag_wp_weights") # skip fixed WP based btag weights for 2022/2023, apply shape based SF + logger.debug("Setting up shape based btag SFs for 2022/2023.") + logger.warning_once("Evaluate used processes for normalized btag SFs for 2022/2023, set to 'tt' + 'st' for now.") + cfg.x.btag_sf = BTagSFConfig( + correction_set="deepJet_shape", + jec_sources=cfg.x.btag_sf_jec_sources, + discriminator=discr, + ) + # implementation from hbt analysis: + # https://github.com/uhh-cms/hh2bbtautau/blob/4b2f1bc57a9c2ada18776e5ac6f0372269e1e26c/hbt/config/configs_hbt.py#L1410 # noqa + cfg.x.btag_wp_count_config = BTagWPCountConfig( + jet_name="Dummy", + ) + cfg.x.btag_wp_sf_config = BTagWPSFConfig( + jet_name="Dummy", + ) + + # top pt reweighting parameters + # https://twiki.cern.ch/twiki/bin/viewauth/CMS/TopPtReweighting#TOP_PAG_corrections_based_on_dat?rev=31 + cfg.x.top_pt_reweighting_params = { + "a": 0.0615, + "b": -0.0005, + } + + # V+jets reweighting + # FIXME update to Run 3 k-factors + cfg.x.vjets_reweighting = DotDict.wrap({ + "w": { + "value": "wjets_kfactor_value", + "error": "wjets_kfactor_error", + }, + "z": { + "value": "zjets_kfactor_value", + "error": "zjets_kfactor_error", + }, + }) + + # + # systematic shifts + # + + # read in JEC sources from file + with open(os.path.join(thisdir, "jec_sources.yaml"), "r") as f: + all_jec_sources = yaml.load(f, yaml.Loader)["names"] + btag_uncs_bc = [ + "correlated", + "uncorrelated", + "bfragmentation", + "fsrdef", + "hdamp", + "isrdef", + "jer", + "jes", + "muf", + "mur", + "pdfas", + "pileup", + "statistic", + "topmass", + "type3p", + ] + btag_uncs_bc_full = [f"{unc}_bc" for unc in btag_uncs_bc] + ["bc"] + btag_uncs_light = [ + "", + "correlated", "uncorrelated", + ] + btag_uncs_light_full = [f"{unc}_light" for unc in btag_uncs_light] + ["light"] + + # declare the shifts + def add_shifts(cfg): + # nominal shift + cfg.add_shift(name="nominal", id=0) + + # tune shifts are covered by dedicated, varied datasets, so tag the shift as "disjoint_from_nominal" + # (this is currently used to decide whether ML evaluations are done on the full shifted dataset) + cfg.add_shift(name="tune_up", id=1, type="shape", tags={"disjoint_from_nominal"}) + cfg.add_shift(name="tune_down", id=2, type="shape", tags={"disjoint_from_nominal"}) + + cfg.add_shift(name="hdamp_up", id=3, type="shape", tags={"disjoint_from_nominal"}) + cfg.add_shift(name="hdamp_down", id=4, type="shape", tags={"disjoint_from_nominal"}) + + # pileup / minimum bias cross section variations + cfg.add_shift(name="minbias_xs_up", id=7, type="shape") + cfg.add_shift(name="minbias_xs_down", id=8, type="shape") + add_shift_aliases( + cfg, + "minbias_xs", + { + "normalized_pu_weight": "normalized_pu_weight_{name}", + "pu_weight": "pu_weight_{name}", + }, + ) + + # top pt reweighting + cfg.add_shift(name="top_pt_up", id=9, type="shape") + cfg.add_shift(name="top_pt_down", id=10, type="shape") + add_shift_aliases(cfg, "top_pt", {"top_pt_weight": "top_pt_weight_{direction}"}) + + # renormalization scale + cfg.add_shift(name="mur_up", id=901, type="shape") + cfg.add_shift(name="mur_down", id=902, type="shape") + + # factorization scale + cfg.add_shift(name="muf_up", id=903, type="shape") + cfg.add_shift(name="muf_down", id=904, type="shape") + + # combined renormalization and factorization scale variation + cfg.add_shift(name="murmuf_up", id=907, type="shape") + cfg.add_shift(name="murmuf_down", id=908, type="shape") + cfg.add_shift(name="murmuf_envelope_up", id=909, type="shape") + cfg.add_shift(name="murmuf_envelope_down", id=910, type="shape") + add_shift_aliases(cfg, "murmuf", {"murmuf_weight": "murmuf_weight_{direction}"}) + add_shift_aliases(cfg, "murmuf", {"murmuf_envelope_weight": "murmuf_envelope_weight_{direction}"}) + + # scale variation (?) + cfg.add_shift(name="scale_up", id=905, type="shape") + cfg.add_shift(name="scale_down", id=906, type="shape") + + # pdf variations + cfg.add_shift(name="pdf_up", id=951, type="shape") + cfg.add_shift(name="pdf_down", id=952, type="shape") + + # alpha_s variation + cfg.add_shift(name="alpha_up", id=961, type="shape") + cfg.add_shift(name="alpha_down", id=962, type="shape") + + # PSWeight variations + cfg.add_shift(name="isr_up", id=7001, type="shape") # PS weight [0] ISR=2 FSR=1 + cfg.add_shift(name="isr_down", id=7002, type="shape") # PS weight [2] ISR=0.5 FSR=1 + add_shift_aliases(cfg, "isr", {"isr": "isr_{direction}"}) + cfg.add_shift(name="fsr_up", id=7003, type="shape") # PS weight [1] ISR=1 FSR=2 + cfg.add_shift(name="fsr_down", id=7004, type="shape") # PS weight [3] ISR=1 FSR=0.5 + add_shift_aliases(cfg, "fsr", {"fsr": "fsr_{direction}"}) + + for unc in ["mur", "muf", "murmuf_envelope", "pdf", "isr", "fsr"]: + col = unc + add_shift_aliases( + cfg, + unc, + { + f"normalized_{col}_weight": f"normalized_{col}_weight_" + "{direction}", + f"{col}_weight": f"{col}_weight_" + "{direction}", + }, + ) + + # event weights due to muon scale factors + if not cfg.has_tag("skip_muon_weights"): + # cfg.add_shift(name="muon_up", id=111, type="shape") + # cfg.add_shift(name="muon_down", id=112, type="shape") + # add_shift_aliases(cfg, "muon", {"muon_weight": "muon_weight_{direction}"}) + cfg.add_shift(name="muon_reco_up", id=113, type="shape") + cfg.add_shift(name="muon_reco_down", id=114, type="shape") + add_shift_aliases(cfg, "muon_reco", {"muon_reco_weight": "muon_reco_weight_{direction}"}) + cfg.add_shift(name="muon_id_up", id=115, type="shape") + cfg.add_shift(name="muon_id_down", id=116, type="shape") + add_shift_aliases(cfg, "muon_id", {"muon_id_weight": "muon_id_weight_{direction}"}) + cfg.add_shift(name="muon_iso_up", id=117, type="shape") + cfg.add_shift(name="muon_iso_down", id=118, type="shape") + add_shift_aliases(cfg, "muon_iso", {"muon_iso_weight": "muon_iso_weight_{direction}"}) + cfg.add_shift(name="muon_trigger_up", id=119, type="shape") + cfg.add_shift(name="muon_trigger_down", id=120, type="shape") + add_shift_aliases(cfg, "muon_trigger", {"muon_trigger_weight": "muon_trigger_weight_{direction}"}) + + # event weights due to electron scale factors + if not cfg.has_tag("skip_electron_weights"): + # cfg.add_shift(name="electron_up", id=121, type="shape") + # cfg.add_shift(name="electron_down", id=122, type="shape") + # add_shift_aliases(cfg, "electron", {"electron_weight": "electron_weight_{direction}"}) + cfg.add_shift(name="electron_reco_up", id=123, type="shape") + cfg.add_shift(name="electron_reco_down", id=124, type="shape") + add_shift_aliases(cfg, "electron_reco", {"electron_reco_weight": "electron_reco_weight_{direction}"}) + cfg.add_shift(name="electron_id_iso_up", id=125, type="shape") + cfg.add_shift(name="electron_id_iso_down", id=126, type="shape") + add_shift_aliases(cfg, "electron_id_iso", {"electron_id_iso_weight": "electron_id_iso_weight_{direction}"}) + cfg.add_shift(name="electron_trigger_up", id=127, type="shape") + cfg.add_shift(name="electron_trigger_down", id=128, type="shape") + add_shift_aliases(cfg, "electron_trigger", {"electron_trigger_weight": "electron_trigger_weight_{direction}"}) + + # V+jets reweighting + cfg.add_shift(name="vjets_up", id=201, type="shape") + cfg.add_shift(name="vjets_down", id=202, type="shape") + add_shift_aliases(cfg, "vjets", {"vjets_weight": "vjets_weight_{direction}"}) + + # b-tagging shifts + if year != 2024: + logger.debug("adding shape based btag SF shifts for 2022/2023") + btag_uncs = [ + "hf", "lf", + "hfstats1", "hfstats2", + "lfstats1", "lfstats2", + "cferr1", "cferr2", + ] + for i, unc in enumerate(btag_uncs): + logger.debug( + f"adding btag SF shift for unc. source '{unc}' with id {500 + 2 * i} (up) and {501 + 2 * i} (down)" + ) + cfg.add_shift(name=f"btag_{unc}_up", id=500 + 2 * i, type="shape") + cfg.add_shift(name=f"btag_{unc}_down", id=501 + 2 * i, type="shape") + add_shift_aliases( + cfg, + f"btag_{unc}", + { + btag_weight: f"{btag_weight}_{unc}_" + "{direction}" + for btag_weight in ( + "btag_weight", + # "normalized_btag_weight", + # "normalized_njet_btag_weight", + # "normalized_ht_njet_btag_weight", + "normalized_ht_njet_nhf_btag_weight", + # "normalized_ht_btag_weight", + ) + }, + ) + else: + # https://cms-analysis-corrections.docs.cern.ch/corrections_era/Run3-24CDEReprocessingFGHIPrompt-Summer24-NanoAODv15/BTV/2025-08-19/#btagging_preliminaryjsongz # noqa + btag_uncs_bc = [ + "fsrdef", "isrdef", + "hdamp", "jer", "jes", + "mass", "statistic", + "tune", + ] + btag_uncs_light = [ + "correlated", "uncorrelated", + ] + for i, unc in enumerate(btag_uncs_bc): + cfg.add_shift(name=f"btag_{unc}_bc_up", id=501 + 4 * i, type="shape") + cfg.add_shift(name=f"btag_{unc}_bc_down", id=502 + 4 * i, type="shape") + add_shift_aliases( + cfg, + f"btag_{unc}_bc", + { + f"btag_weight": f"btag_weight_{unc}_bc_" + "{direction}", + }, + ) + for i, unc in enumerate(btag_uncs_light): + cfg.add_shift(name=f"btag_{unc}_light_up", id=503 + 4 * i, type="shape") + cfg.add_shift(name=f"btag_{unc}_light_down", id=504 + 4 * i, type="shape") + add_shift_aliases( + cfg, + f"btag_{unc}_light", + { + f"btag_weight": f"btag_weight_{unc}_light_" + "{direction}", + }, + ) + + cfg.add_shift(name="btag_bc_up", id=501 + 4 * len(btag_uncs_bc), type="shape") + cfg.add_shift(name="btag_bc_down", id=502 + 4 * len(btag_uncs_bc), type="shape") + cfg.add_shift(name="btag_light_up", id=503 + 4 * len(btag_uncs_light), type="shape") + cfg.add_shift(name="btag_light_down", id=504 + 4 * len(btag_uncs_light), type="shape") + add_shift_aliases( + cfg, + "btag_bc", + { + "btag_weight": "btag_weight_bc_" + "{direction}", + }, + ) + add_shift_aliases( + cfg, + "btag_light", + { + "btag_weight": "btag_weight_light_" + "{direction}", + }, + ) + + # jet energy scale (JEC) uncertainty variations + for jec_source in cfg.x.jec.Jet.uncertainty_sources: + idx = all_jec_sources.index(jec_source) + cfg.add_shift( + name=f"jec_{jec_source}_up", + id=5000 + 2 * idx, + type="shape", + tags={"jec"}, + aux={ + "jec_source": jec_source, + "version": 1, + }, + ) + cfg.add_shift( + name=f"jec_{jec_source}_down", + id=5001 + 2 * idx, + type="shape", + tags={"jec"}, + aux={ + "jec_source": jec_source, + "version": 1, + }, + ) + add_shift_aliases( + cfg, + f"jec_{jec_source}", + { + "Jet.pt": "Jet.pt_{name}", + "Jet.mass": "Jet.mass_{name}", + "PuppiMET.pt": "PuppiMET.pt_{name}", + "PuppiMET.phi": "PuppiMET.phi_{name}", + "FatJet.pt": "FatJet.pt_{name}", + "FatJet.mass": "FatJet.mass_{name}", + }, + ) + + if jec_source in ["Total", *cfg.x.btag_sf_jec_sources]: + # when jec_source is a known btag SF source, add aliases for btag weight column + add_shift_aliases( + cfg, + f"jec_{jec_source}", + { + btag_weight: f"{btag_weight}_jec_{jec_source}_" + "{direction}" + for btag_weight in ( + "btag_weight", + # "normalized_btag_weight", + # "normalized_njet_btag_weight", + # "normalized_ht_njet_btag_weight", + "normalized_ht_njet_nhf_btag_weight", + # "normalized_ht_btag_weight", + ) + }, + ) + + # jet energy resolution (JER) scale factor variations + cfg.add_shift(name="jer_up", id=6000, type="shape") + cfg.add_shift(name="jer_down", id=6001, type="shape") + add_shift_aliases( + cfg, + "jer", + { + "Jet.pt": "Jet.pt_{name}", + "Jet.mass": "Jet.mass_{name}", + "PuppiMET.pt": "PuppiMET.pt_{name}", + "PuppiMET.phi": "PuppiMET.phi_{name}", + "FatJet.pt": "FatJet.pt_{name}", + "FatJet.mass": "FatJet.mass_{name}", + }, + ) + + # add the shifts + add_shifts(cfg) + + # + # external files + # setup taken from https://github.com/uhh-cms/hh2bbtautau/blob/ed8f363ac239b0257fc7f470b96f5c09a0572c34/hbt/config/configs_hbt.py#L1574 # noqa: E501 + # https://cms-analysis-corrections.docs.cern.ch + # + + cfg.x.external_files = DotDict() + + # helper + def add_external(name, value): + if isinstance(value, dict): + value = DotDict.wrap(value) + cfg.x.external_files[name] = value + + # prepare run/era/nano meta data info to determine files in the CAT metadata structure + # see https://cms-analysis-corrections.docs.cern.ch + cat_info = { + (2022, "", 12): CATInfo( + run=3, + vnano=12, + era="22CDSep23-Summer22", + pog_directories={"dc": "Collisions22"}, + snapshot=CATSnapshot(btv="2025-08-20", dc="2025-07-25", egm="2025-12-15", jme="2026-04-13", lum="2024-01-31", muo="2026-04-28", tau="2025-12-25"), # noqa: E501 + ), + (2022, "EE", 12): CATInfo( + run=3, + vnano=12, + era="22EFGSep23-Summer22EE", + pog_directories={"dc": "Collisions22"}, + snapshot=CATSnapshot(btv="2025-08-20", dc="2025-07-25", egm="2025-12-15", jme="2026-04-13", lum="2024-01-31", muo="2026-04-28", tau="2025-12-25"), # noqa: E501 + ), + (2023, "", 12): CATInfo( + run=3, + vnano=12, + era="23CSep23-Summer23", + pog_directories={"dc": "Collisions23"}, + snapshot=CATSnapshot(btv="2025-08-20", dc="2025-07-25", egm="2025-12-15", jme="2026-04-13", lum="2024-01-31", muo="2026-04-28", tau="2025-12-25"), # noqa: E501 + ), + (2023, "BPix", 12): CATInfo( + run=3, + vnano=12, + era="23DSep23-Summer23BPix", + pog_directories={"dc": "Collisions23"}, + snapshot=CATSnapshot(btv="2025-08-20", dc="2025-07-25", egm="2025-12-15", jme="2026-04-13", lum="2024-01-31", muo="2026-04-28", tau="2025-12-25"), # noqa: E501 + ), + (2024, "", 15): CATInfo( + run=3, + vnano=15, + era="24CDEReprocessingFGHIPrompt-Summer24", + pog_directories={"dc": "Collisions24"}, + snapshot=CATSnapshot(btv="2026-03-10", dc="2025-07-25", egm="2025-12-15", jme="2025-12-02", muo="2026-04-28", lum="2026-04-15"), # noqa: E501 + ), + }[(year, campaign.x.postfix, vnano)] + cfg.x.cat_info = cat_info + + # common files + # (versions in the end are for hashing in cases where file contents changed but paths did not) + add_external("lumi", { + "golden": { + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=161#Year_2022 + 2022: (cat_info.get_file("dc", "Cert_Collisions2022_355100_362760_Golden.json"), "v1"), + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=161#Year_2023 + 2023: (cat_info.get_file("dc", "Cert_Collisions2023_366442_370790_Golden.json"), "v1"), + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=180#Year_2024 + # not yet available at CAT space + # 2024: (cat_info.get_file("dc", "Cert_Collisions2024_378981_386951_Golden.json"), "v1"), + 2024: ("https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions24/Cert_Collisions2024_378981_386951_Golden.json", "v1"), # noqa: E501 + }[year], + "normtag": { + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=161#Year_2022 + 2022: ("/cvmfs/cms-bril.cern.ch/cms-lumi-pog/Normtags/normtag_BRIL.json", "v1"), + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=161#Year_2023 + 2023: ("/cvmfs/cms-bril.cern.ch/cms-lumi-pog/Normtags/normtag_BRIL.json", "v1"), + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=180#Year_2024 + 2024: ("/cvmfs/cms-bril.cern.ch/cms-lumi-pog/Normtags/normtag_BRIL.json", "v1"), # TODO: correct? + }[year], + }) + + # pileup weight corrections + if year != 2024: # TODO: not yet available, see https://cms-analysis-corrections.docs.cern.ch + add_external("pu_sf", (cat_info.get_file("lum", "puWeights.json.gz"), "v1")) + elif year == 2024: + add_external("pu_sf", (cat_info.get_file("lum", "puWeights_CDEFGHI.json.gz"), "v1")) + + # jet energy correction + add_external("jet_jerc", (cat_info.get_file("jme", "jet_jerc.json.gz"), "v1")) + + # fat jet energy correction + add_external("fat_jet_jerc", (cat_info.get_file("jme", "fatJet_jerc.json.gz"), "v1")) # noqa: E501 + + # jet veto map + add_external("jet_veto_map", (cat_info.get_file("jme", "jetvetomaps.json.gz"), "v1")) + + # btag scale factor + if year != 2024: + add_external("btag_sf_corr", (cat_info.get_file("btv", "btagging.json.gz"), "v1")) + else: + # keep this in case we want to switch back to the fixed wp for 2024 + add_external("btag_sf_corr", (cat_info.get_file("btv", "btagging.json.gz"), "v1")) # noqa: E501 + # use custom file with merged SF for both b/c and light jets + add_external("btag_wp_sf_corr", ("/data/dust/user/matthiej/topsf/topsf/config/run3/btagging_merged.json.gz", "v1")) # noqa: E501 + + # updated jet id + add_external("jet_id", (cat_info.get_file("jme", "jetid.json.gz"), "v1")) + + # muon scale factors + add_external("muon_sf", (cat_info.get_file("muo", "muon_HighPt.json.gz"), "v1")) + + # met phi correction + if year != 2024: # TODO: not yet available for 2024 + add_external("met_phi_corr", (cat_info.get_file("jme", f"met_xyCorrections_{year}_{year}{campaign.x.postfix}.json.gz"), "v1")) # noqa: E501 + + # electron scale factors + add_external("electron_sf", (cat_info.get_file("egm", "electron.json.gz"), "v1")) + add_external("electron_trigger_sf", (cat_info.get_file("egm", "electronHlt.json.gz"), "v1")) + # electron energy correction and smearing + add_external("electron_ss", (cat_info.get_file("egm", "electronSS_EtDependent.json.gz"), "v1")) # FIXME correct for us? # noqa: E501 + + # # top-tagging scale factors (TODO) + # "toptag_sf": (f"{sources['jet']}/JMAR/???/???.json", "v1"), # noqa + + # # V+jets reweighting + # "vjets_reweighting": f"{sources['local_repo']}/data/json/vjets_reweighting.json", + + # + # event reduction configuration + # + + # target file size after MergeReducedEvents in MB + cfg.x.reduced_file_size = 512.0 + + # columns to keep after certain steps + cfg.x.keep_columns = DotDict.wrap({ + "cf.ReduceEvents": { + # + # NanoAOD columns + # + + # general event info + "run", "luminosityBlock", "event", + + # weights + "genWeight", + "LHEWeight.*", + "LHEPdfWeight", "LHEScaleWeight", + "PSWeight", + + # muons + "Muon.pt", "Muon.eta", "Muon.phi", "Muon.mass", "Muon.tunepRelPt", "Muon.rawPt", + "Muon.pdgId", + "Muon.jetIdx", + "Muon.nStations", + "Muon.pfRelIso03_all", "Muon.pfRelIso04_all", "Muon.tkRelIso", + + # electrons + "Electron.pt", "Electron.eta", "Electron.phi", "Electron.mass", + "Electron.pdgId", + "Electron.jetIdx", + "Electron.deltaEtaSC", + "Electron.pfRelIso03_all", + + # photons (for L1 prefiring) + "Photon.pt", "Photon.eta", "Photon.phi", "Photon.mass", + "Photon.jetIdx", + + # columns for btag reweighting crosschecks + "njet", "ht", "nhf", + + # AK4 jets + "Jet.pt", "Jet.eta", "Jet.phi", "Jet.mass", + "Jet.rawFactor", + "Jet.btagDeepFlavB", "Jet.hadronFlavour", "Jet.btagUParTAK4B", + # # optional, enable if needed + # "Jet.area", + # "Jet.hadronFlavour", "Jet.partonFlavour", + # "Jet.jetId", "Jet.puId", "Jet.puIdDisc", + # # cleaning + # "Jet.cleanmask", + # "Jet.muonSubtrFactor", + # # indices to other collections + # "Jet.electronIdx*", + # "Jet.muonIdx*", + # "Jet.genJetIdx*", + # # number of jet constituents + # "Jet.nConstituents", + # "Jet.nElectrons", + # "Jet.nMuons", + # # PF energy fractions + # "Jet.chEmEF", + # "Jet.chHEF", + # "Jet.neEmEF", + # "Jet.neHEF", + # "Jet.muEF", + # # taggers + # "Jet.qgl", + # "Jet.btag*", + + # AK8 jets + "FatJet.pt", "FatJet.eta", "FatJet.phi", "FatJet.mass", "FatJet.msoftdrop", + "FatJet.rawFactor", + "FatJet.tau1", "FatJet.tau2", "FatJet.tau3", "FatJet.tau4", + "FatJet.subJetIdx1", "FatJet.subJetIdx2", + # # optional, enable if needed + # "FatJet.area", "FatJet.jetId", "FatJet.hadronFlavour", + # "FatJet.genJetAK8Idx", + # "FatJet.muonIdx3SJ", "FatJet.electronIdx3SJ", + # "FatJet.nBHadrons", "FatJet.nCHadrons", + # # taggers + # "FatJet.btag*", "FatJet.deepTag*", "FatJet.particleNet*", + + # subjets + "SubJet.btagDeepB", "SubJet.btagUParTAK4B" + + # generator quantities + "Generator.*", + + # # generator particles + # "GenPart.pt", "GenPart.eta", "GenPart.phi", "GenPart.mass", + # "GenPart.pdgId", + # "GenPart.*", + + # missing transverse momentum + "MET.pt", "MET.phi", "MET.significance", "MET.covXX", "MET.covXY", "MET.covYY", + "PuppiMET.phi", "PuppiMET.pt", + + # number of primary vertices + "PV.npvs", + "PV.npvsGood", + + # average number of pileup interactions + "Pileup.nTrueInt", + + # + # columns added during selection + # + + # generator particle info + "GenTopDecay.*", + "GenTopAssociatedDecay.*", + "GenPartonTop.*", + "GenVBoson.*", + + # generic leptons (merger of Muon/Electron) + "Lepton.*", + + # probe jet + "ProbeJet.*", + + # columns for PlotCutflowVariables + "cutflow.*", + + # other columns, required by various tasks + "channel_id", "category_ids", "process_id", + "deterministic_seed", + "mc_weight", + "pu_weight*", + "pdf_weight*", "fsr_weight*", "isr_weight*", + "muf_weight*", "mur_weight*", "murmuf_weight*", "murmuf_envelope*", + }, + "cf.MergeSelectionMasks": { + "channel_id", "process_id", "category_ids", + "normalization_weight", + "cutflow.*", + "mc_weight", + }, + "cf.UniteColumns": { + "*", + }, + }) + + # + # event weights + # + + # event weight columns as keys in an OrderedDict, mapped to shift instances they depend on + get_shifts = functools.partial(get_shifts_from_sources, cfg) + # add b tagging weights + btag_shifts = ["hf", "lf", "hfstats1", "hfstats2", "lfstats1", "lfstats2", "cferr1", "cferr2"] + full_btag_uncs = btag_uncs_bc_full + btag_uncs_light_full + cfg.x.event_weights = DotDict({ + "normalization_weight": [], + "normalized_pu_weight": get_shifts("minbias_xs"), + "muon_reco_weight": get_shifts("muon_reco"), + "muon_id_weight": get_shifts("muon_id"), + "muon_iso_weight": get_shifts("muon_iso"), + "muon_trigger_weight": get_shifts("muon_trigger"), + "electron_id_iso_weight": get_shifts("electron_id_iso"), + "electron_reco_weight": get_shifts("electron_reco"), + }) + if cfg.has_tag("use_non_normalized_weights"): + logger.debug("Using non-normalized event weights.") + # store non normalized for future checks + cfg.x.event_weights["btag_weight"] = get_shifts("btag") + cfg.x.event_weights["fsr_weight"] = get_shifts("fsr") + cfg.x.event_weights["isr_weight"] = get_shifts("isr") + else: + logger.debug("Using normalized event weights.") + if not cfg.x.year == 2024: + logger.debug("Use normalized btag weights for 2022/2023, with shape-based SF and uncertainties.") + cfg.x.event_weights["normalized_ht_njet_nhf_btag_weight"] = get_shifts( + *(f"btag_{unc}" for unc in btag_shifts) + ) + # cfg.x.event_weights["normalized_njet_btag_weight"] = get_shifts("btag") + # cfg.x.event_weights["normalized_ht_btag_weight"] = get_shifts("btag") + + if cfg.x.year == 2024: + logger.debug("Using fixed wp btag weights for 2024, with separate uncertainties for b/c and light jets.") + cfg.x.event_weights["btag_weight"] = get_shifts(*(f"btag_{unc}" for unc in full_btag_uncs)) + + for dataset in cfg.datasets: + dataset.x.event_weights = DotDict() + if dataset.has_tag("is_ttbar"): + # top pt reweighting + dataset.x.event_weights["top_pt_weight"] = get_shifts("top_pt") + if not has_tag("skip_kfactor_weights", cfg, dataset, operator=any) and dataset.has_tag("is_v_jets"): + # V+jets QCD NLO reweighting + dataset.x.event_weights["vjets_weight"] = get_shifts("vjets") + # add PSWeight variations for all datasets but qcd + if not cfg.has_tag("use_non_normalized_weights"): + if not dataset.has_tag("is_qcd") and dataset.is_mc: + logger.debug_once("Use normalized ps weights.") + dataset.x.event_weights["normalized_isr_weight"] = get_shifts("isr") + dataset.x.event_weights["normalized_fsr_weight"] = get_shifts("fsr") + if dataset.has_tag("has_top"): + logger.debug_once("Use normalized scale variation weights.") + dataset.x.event_weights["normalized_mur_weight"] = get_shifts("mur") + dataset.x.event_weights["normalized_muf_weight"] = get_shifts("muf") + # switch to combined if needed + # dataset.x.event_weights["normalized_murmuf_envelope_weight"] = get_shifts("murmuf_envelope") + # dataset.x.event_weights["normalized_murmuf_weight"] = get_shifts("murmuf") + if not has_tag("skip_pdf", cfg, dataset): + logger.debug_once("Use normalized pdf weights.") + dataset.x.event_weights["normalized_pdf_weight"] = get_shifts("pdf") + + # group datasets together for btag WP efficiency calculation in 2024 + if year == 2024: + # TODO figure out which datasets should be grouped together; + # for now, group all datasets together + cfg.x.btag_wp_eff_groups = [ + ["tt_*", "st_*", "qcd_*", "ww_*", "dy_*", "w_lnu_*", "wz_*", "zz_*"], + # ["tt_*"], + # ["st_*"], + # ["dy_*"], + # ["w_lnu_*"], + # ["ww_*", "wz_*", "zz_*"], + # ["qcd_*"], + # ["dy_*", "w_lnu_*", "wz_*", "zz_*", "ww_*", "qcd_*"], + # ["tt_*", "st_*"], + ] + group_matched = False + for i, dataset_pattern in enumerate(cfg.x.btag_wp_eff_groups): + if law.util.multi_match(dataset.name, dataset_pattern): + if group_matched: + raise ValueError( + f"dataset '{dataset.name}' already has a btag WP group assigned! Cannot assign it to more " + "than one group", + ) + group_matched = True + dataset.add_tag(f"btag_wp_eff_group_{i}") + if not group_matched and dataset.is_mc: + raise ValueError(f"no btag_wp_eff_group_* assigned to dataset '{dataset.name}'") + if group_matched and dataset.is_data: + raise ValueError(f"must not assign btag_wp_eff_group_* to dataset '{dataset.name}'") + + # # + # # versions + # # + # cfg.x.versions = { + # "tt_*": "test_v7", + # "st_*": "test_v7", + # "dy_*": "test_v7", + # "w_*": "test_v7", + # "ww_*": "test_v7", + # "wz_*": "test_v7", + # "zz_*": "test_v7", + # "data_*": "test_v7", + # "topsf.CreateDatacards": "test_v8", + # } + + # # named references to actual versions to use for certain sets of tasks + # main_ver = "test_v4" + # cfg.x.named_versions = DotDict.wrap({ + # "default": f"{main_ver}", + # "calibrate": "test_v4", + # "select": "test_v4", + # "reduce": f"{main_ver}", + # "merge": f"{main_ver}", + # "produce": f"{main_ver}", + # "hist": f"{main_ver}", + # "plot": f"{main_ver}", + # "datacards": f"{main_ver}", + # }) + + # # versions per task family and optionally also dataset and shift + # # None can be used as a key to define a default value + # cfg.x.versions = { + # None: cfg.x.named_versions["default"], + # # CSR tasks + # "cf.CalibrateEvents": cfg.x.named_versions["calibrate"], + # "cf.SelectEvents": cfg.x.named_versions["select"], + # "cf.ReduceEvents": cfg.x.named_versions["reduce"], + # # merging tasks + # "cf.MergeSelectionStats": cfg.x.named_versions["merge"], + # "cf.MergeSelectionMasks": cfg.x.named_versions["merge"], + # "cf.MergeReducedEvents": cfg.x.named_versions["merge"], + # "cf.MergeReductionStats": cfg.x.named_versions["merge"], + # # column production + # "cf.ProduceColumns": cfg.x.named_versions["produce"], + # # histogramming + # "cf.CreateCutflowHistograms": cfg.x.named_versions["hist"], + # "cf.CreateHistograms": cfg.x.named_versions["hist"], + # "cf.MergeHistograms": cfg.x.named_versions["hist"], + # "cf.MergeShiftedHistograms": cfg.x.named_versions["hist"], + # # plotting + # "cf.PlotVariables1D": cfg.x.named_versions["plot"], + # "cf.PlotVariables2D": cfg.x.named_versions["plot"], + # "cf.PlotVariablesPerProcess2D": cfg.x.named_versions["plot"], + # "cf.PlotShiftedVariables1D": cfg.x.named_versions["plot"], + # "cf.PlotShiftedVariablesPerProcess1D": cfg.x.named_versions["plot"], + # # + # "cf.PlotCutflow": cfg.x.named_versions["plot"], + # "cf.PlotCutflowVariables1D": cfg.x.named_versions["plot"], + # "cf.PlotCutflowVariables2D": cfg.x.named_versions["plot"], + # "cf.PlotCutflowVariablesPerProcess2D": cfg.x.named_versions["plot"], + # # datacards + # "cf.CreateDatacards": cfg.x.named_versions["datacards"], + # } + + # + # finalization + # + + # add categories + add_categories(cfg) + + # add variables + add_variables(cfg) + + # add channels + cfg.add_channel("e", id=1) + cfg.add_channel("mu", id=2) + + return cfg diff --git a/topsf/config/run3/config_wp.py b/topsf/config/run3/config_wp.py index 24d4feb..5091671 100644 --- a/topsf/config/run3/config_wp.py +++ b/topsf/config/run3/config_wp.py @@ -16,12 +16,16 @@ from scinum import Number from columnflow.util import DotDict +from columnflow.cms_util import CATInfo, CATSnapshot from columnflow.config_util import ( add_shift_aliases, get_root_processes_from_campaign, get_shifts_from_sources, verify_config_processes, ) +from columnflow.production.cms.electron import ElectronSFConfig +from columnflow.production.cms.jet import JetIdConfig +from columnflow.calibration.cms.met import METPhiConfig from topsf.config.variables import add_variables from topsf.config.categories_wp import add_categories @@ -41,6 +45,7 @@ def add_config( Configurable function for creating a config for a run2 analysis given a base *analysis* object and a *campaign* (i.e. set of datasets). """ + raise NotImplementedError("This function is old and kept as a reference for now and not to be used.") # validation assert campaign.x.year in [2022, 2023] if campaign.x.year == 2022: @@ -57,7 +62,7 @@ def add_config( elif year == 2023: corr_postfix = f"{campaign.x.BPix}BPix" - implemented_years = [2022] + implemented_years = [2022, 2023] if year not in implemented_years: raise NotImplementedError("For now, only 2022 campaign is fully implemented") @@ -185,10 +190,20 @@ def add_config( "qcd_em_pt170to300_pythia", "qcd_em_pt300toinf_pythia", ] - # if campaign.x.EE == "post": - # dataset_names += [ - # "qcd_mu_pt20to30_pythia", - # ] + if year == 2022: + if campaign.x.EE == "pre": + dataset_names += [ + ] + elif campaign.x.EE == "post": + dataset_names += [ + ] + elif year == 2023: + if campaign.has_tag("preBPix"): + dataset_names += [ + ] + elif campaign.has_tag("postBPix"): + dataset_names += [ + ] for dataset_name in dataset_names: # add the dataset dataset = cfg.add_dataset(campaign.get_dataset(dataset_name)) @@ -312,6 +327,17 @@ def add_config( "lumi_13TeV_2022": 0.01j, "lumi_13TeV_correlated": 0.006j, }) + elif year == 2023: + if campaign.has_tag("preBPix"): + cfg.x.luminosity = Number(17794, { + "lumi_13TeV_2023": 0.01j, + "lumi_13TeV_correlated": 0.006j, + }) + elif campaign.has_tag("postBPix"): + cfg.x.luminosity = Number(9451, { + "lumi_13TeV_2023": 0.01j, + "lumi_13TeV_correlated": 0.006j, + }) else: raise NotImplementedError(f"Luminosity for year {year} is not defined.") @@ -350,19 +376,27 @@ def add_config( # jec configuration taken from HBW # https://github.com/uhh-cms/hh2bbww/blob/master/hbw/config/config_run2.py#L138C5-L269C1 # https://twiki.cern.ch/twiki/bin/view/CMS/JECDataMC?rev=201 - jerc_postfix = "" - if year == 2022 and campaign.x.EE == "post": - jerc_postfix = "EE" + jerc_postfix = campaign.x.postfix + if jerc_postfix not in ("", "EE", "BPix"): + raise ValueError(f"Invalid JERC postfix '{jerc_postfix}' for campaign {campaign.name}.") + if year == 2022: + jer_campaign = jec_campaign = f"Summer{year2}{jerc_postfix}_22Sep2023" + elif year == 2023: + era = "Cv1234" if campaign.has_tag("preBPix") else "D" + jer_campaign = f"Summer{year2}{jerc_postfix}Prompt{year2}_Run{era}" + jec_campaign = f"Summer{year2}{jerc_postfix}Prompt{year2}" - if cfg.x.run == 3: - jerc_campaign = f"Summer{year2}{jerc_postfix}_22Sep2023" - jet_type = "AK4PFPuppi" - fatjet_type = "AK8PFPuppi" + jet_type = "AK4PFPuppi" + fatjet_type = "AK8PFPuppi" + jec_ak4_version = jec_ak8_version = { + 2022: "V2", + 2023: "V2" if jerc_postfix == "" else "V3", + }[year] cfg.x.jec = DotDict.wrap({ "Jet": { - "campaign": jerc_campaign, - "version": {2016: "V7", 2017: "V5", 2018: "V5", 2022: "V2"}[year], + "campaign": jec_campaign, + "version": jec_ak4_version, "jet_type": jet_type, "levels": ["L1FastJet", "L2Relative", "L2L3Residual", "L3Absolute"], "levels_for_type1_met": ["L1FastJet"], @@ -424,10 +458,11 @@ def add_config( "CorrelationGroupFlavor", "CorrelationGroupUncorrelated", ], + "data_per_era": False if year == 2023 else True, }, "FatJet": { - "campaign": jerc_campaign, - "version": {2016: "V7", 2017: "V5", 2018: "V5", 2022: "V2"}[year], + "campaign": jec_campaign, + "version": jec_ak8_version, "jet_type": fatjet_type, "levels": ["L1FastJet", "L2Relative", "L2L3Residual", "L3Absolute"], "levels_for_type1_met": ["L1FastJet"], @@ -489,10 +524,11 @@ def add_config( "CorrelationGroupFlavor", "CorrelationGroupUncorrelated", ], + "data_per_era": False if year == 2023 else True, }, "SubJet": { - "campaign": jerc_campaign, - "version": {2016: "V7", 2017: "V5", 2018: "V5", 2022: "V2"}[year], + "campaign": jec_campaign, + "version": jec_ak4_version, "jet_type": jet_type, "levels": ["L1FastJet", "L2Relative", "L2L3Residual", "L3Absolute"], "levels_for_type1_met": ["L1FastJet"], @@ -554,6 +590,7 @@ def add_config( # "CorrelationGroupFlavor", # "CorrelationGroupUncorrelated", ], + "data_per_era": False if year == 2023 else True, }, }) @@ -562,21 +599,27 @@ def add_config( # TODO: get jerc working for Run3 cfg.x.jer = DotDict.wrap({ "Jet": { - "campaign": jerc_campaign, - "version": {2022: "JRV1"}[year], + "campaign": jer_campaign, + "version": {2022: "JRV1", 2023: "JRV1"}[year], "jet_type": jet_type, }, "FatJet": { - "campaign": jerc_campaign, - "version": {2022: "JRV1"}[year], + "campaign": jer_campaign, + "version": {2022: "JRV1", 2023: "JRV1"}[year], "jet_type": fatjet_type, }, "SubJet": { - "campaign": jerc_campaign, - "version": {2022: "JRV1"}[year], + "campaign": jer_campaign, + "version": {2022: "JRV1", 2023: "JRV1"}[year], "jet_type": jet_type, }, }) + cfg.x.jet_id = JetIdConfig( + corrections={ + "AK4PUPPI_Tight": 2, + "AK4PUPPI_TightLeptonVeto": 6, + }, + ) # JEC uncertainty sources propagated to btag scale factors # (names derived from contents in BTV correctionlib file) @@ -620,6 +663,26 @@ def add_config( "TimePtEta", ] + if cfg.x.run == 2: + cfg.x.met_phi_correction_set = "{variable}_metphicorr_pfmet_{data_source}" + else: + cfg.x.met_phi_correction = METPhiConfig( + met_name="PuppiMET", + met_type="PuppiMET", + correction_set="met_xy_corrections", + keep_uncorrected=True, # TODO do we need this? + pt_phi_variations={ + "stat_xdn": "metphi_statx_down", + "stat_xup": "metphi_statx_up", + "stat_ydn": "metphi_staty_down", + "stat_yup": "metphi_staty_up", + }, + variations={ + "pu_dn": "minbias_xs_down", + "pu_up": "minbias_xs_up", + }, + ) + # # tagger working points # @@ -628,28 +691,29 @@ def add_config( # https://btv-wiki.docs.cern.ch/ScaleFactors/Run3Summer22/ # https://btv-wiki.docs.cern.ch/ScaleFactors/Run3Summer22EE/ # TODO: add correct 2022 + 2022preEE WP for deepcsv if needed - btag_key = f"2022{campaign.x.EE}EE" if year == 2022 else year + # TODO: use PNet? + btag_key = cfg.x.cpn_tag cfg.x.btag_working_points = DotDict.wrap({ "deepjet": { "loose": { - "2022preEE": 0.0583, "2022postEE": 0.0614, + "2022preEE": 0.0583, "2022postEE": 0.0614, "2023preBPix": 0.0479, "2023postBPix": 0.048, }[btag_key], "medium": { - "2022preEE": 0.3086, "2022postEE": 0.3196, + "2022preEE": 0.3086, "2022postEE": 0.3196, "2023preBPix": 0.2431, "2023postBPix": 0.2435, }[btag_key], "tight": { - "2022preEE": 0.7183, "2022postEE": 0.7300, + "2022preEE": 0.7183, "2022postEE": 0.7300, "2023preBPix": 0.6553, "2023postBPix": 0.6563, }[btag_key], }, "deepcsv": { "loose": { - "2022preEE": 0.1208, "2022postEE": 0.1208, + "2022preEE": 0.1208, "2022postEE": 0.1208, "2023preBPix": 0.1208, "2023postBPix": 0.1208, }[btag_key], "medium": { - "2022preEE": 0.4168, "2022postEE": 0.4168, + "2022preEE": 0.4168, "2022postEE": 0.4168, "2023preBPix": 0.4168, "2023postBPix": 0.4168, }[btag_key], "tight": { - "2022preEE": 0.7665, "2022postEE": 0.7665, + "2022preEE": 0.7665, "2022postEE": 0.7665, "2023preBPix": 0.7665, "2023postBPix": 0.7665, }[btag_key], }, }) @@ -730,19 +794,23 @@ def add_config( "min_pt": 300, "max_abseta": 2.4, # note: SF analysis has 2.5 "msoftdrop_range": (105, 210), + # https://twiki.cern.ch/twiki/bin/view/CMS/JetID13p6TeV + "jetId": 2, # bit2 (2): pass tight ID, fail tightLepVeto, bit3 (6): pass tight and tightLepVeto ID # probe jet pt bins (used by category builder) "pt_bins": [300, 400, 480, 600, None], # parameters for b-tagged subjets "subjet_column": "SubJet", "subjet_btag": "btagDeepB", - "subjet_btag_wp": cfg.x.btag_working_points.deepcsv.loose, + "subjet_btag_wp": cfg.x.btag_working_points.deepcsv.loose, # FIXME: use DeepJet or PNet? }, }) # MET selection parameters + # FIXME: use PuppiMET for Run 3? What's the difference? It's better? + # https://twiki.cern.ch/twiki/bin/viewauth/CMS/MissingETRun2Corrections?rev=79#xy_Shift_Correction_MET_phi_modu cfg.x.met_selection = DotDict.wrap({ "default": { - "column": "MET", + "column": "PuppiMET" if cfg.x.run == 3 else "MET", "min_pt": 50, }, }) @@ -751,28 +819,45 @@ def add_config( # producer configurations # - if cfg.x.run == 3: - # TODO: check that everyting is setup as intended - - # btag weight configuration - cfg.x.btag_sf = ("deepJet_shape", cfg.x.btag_sf_jec_sources) - - # lepton sf taken from - # https://github.com/uhh-cms/hh2bbww/blob/master/hbw/config/config_run2.py#L338C1-L352C85 - # names of electron correction sets and working points - # (used in the electron_sf producer) - if cfg.x.cpn_tag == "2022postEE": - # TODO: we need to use different SFs for control regions - cfg.x.electron_sf_names = ("Electron-ID-SF", "2022Re-recoE+PromptFG", "Tight") - elif cfg.x.cpn_tag == "2022preEE": - cfg.x.electron_sf_names = ("Electron-ID-SF", "2022Re-recoBCD", "Tight") + # btag weight configuration + cfg.x.btag_sf = ("deepJet_shape", cfg.x.btag_sf_jec_sources) - # names of muon correction sets and working points - # (used in the muon producer) + # lepton sf taken from + # https://github.com/uhh-cms/hh2bbww/blob/master/hbw/config/config_run2.py#L338C1-L352C85 + # names of electron correction sets and working points + # (used in the electron_sf producer) + if cfg.x.cpn_tag == "2022postEE": # TODO: we need to use different SFs for control regions - cfg.x.muon_sf_names = ("NUM_TightPFIso_DEN_TightID", f"{cfg.x.cpn_tag}") - cfg.x.muon_id_sf_names = ("NUM_TightID_DEN_TrackerMuons", f"{cfg.x.cpn_tag}") - cfg.x.muon_iso_sf_names = ("NUM_TightPFIso_DEN_TightID", f"{cfg.x.cpn_tag}") + cfg.x.electron_sf_names = ElectronSFConfig( + correction="Electron-ID-SF", + campaign="2022Re-recoE+PromptFG", + working_point="Tight", + ) + elif cfg.x.cpn_tag == "2022preEE": + cfg.x.electron_sf_names = ElectronSFConfig( + correction="Electron-ID-SF", + campaign="2022Re-recoBCD", + working_point="Tight", + ) + elif cfg.x.cpn_tag == "2023preBPix": + cfg.x.electron_sf_names = ElectronSFConfig( + correction="Electron-ID-SF", + campaign="2023PromptC", + working_point="Tight", + ) + elif cfg.x.cpn_tag == "2023postBPix": + cfg.x.electron_sf_names = ElectronSFConfig( + correction="Electron-ID-SF", + campaign="2023PromptD", + working_point="Tight", + ) + + # names of muon correction sets and working points + # (used in the muon producer) + # TODO: we need to use different SFs for control regions + cfg.x.muon_sf_names = ("NUM_TightPFIso_DEN_TightID", f"{cfg.x.cpn_tag}") + cfg.x.muon_id_sf_names = ("NUM_TightID_DEN_TrackerMuons", f"{cfg.x.cpn_tag}") + cfg.x.muon_iso_sf_names = ("NUM_TightPFIso_DEN_TightID", f"{cfg.x.cpn_tag}") # top pt reweighting parameters # https://twiki.cern.ch/twiki/bin/viewauth/CMS/TopPtReweighting#TOP_PAG_corrections_based_on_dat?rev=31 @@ -943,101 +1028,214 @@ def add_shifts(cfg): # add the shifts add_shifts(cfg) - # - # external files - # taken from - # https://github.com/uhh-cms/hh2bbww/blob/master/hbw/config/config_run2.py#L535C7-L579C84 - # + # # + # # external files + # # taken from + # # https://github.com/uhh-cms/hh2bbww/blob/master/hbw/config/config_run2.py#L535C7-L579C84 + # # - # external files - json_mirror = "/afs/cern.ch/user/j/jmatthie/public/mirrors/jsonpog-integration-b7a48c75" - local_repo = "/data/dust/user/matthiej/topsf" # TODO: avoid hardcoding path + # # external files + # json_mirror = "/afs/cern.ch/user/j/jmatthie/public/mirrors/jsonpog-integration-406118ec" # updated 31.07.25 + # local_repo = "/data/dust/user/matthiej/topsf" # TODO: avoid hardcoding path - if cfg.x.run == 3: - corr_tag = f"{year}_Summer22{jerc_postfix}" + # if cfg.x.cpn_tag == "2022preEE" or cfg.x.cpn_tag == "2022postEE": + # corr_tag = f"{year}_Summer22{jerc_postfix}" + # elif cfg.x.cpn_tag == "2023preBPix" or cfg.x.cpn_tag == "2023postBPix": + # corr_tag = f"{year}_Summer23{jerc_postfix}" - cfg.x.external_files = DotDict.wrap({ - # pileup weight corrections - "pu_sf": (f"{json_mirror}/POG/LUM/{corr_tag}/puWeights.json.gz", "v1"), + # cfg.x.external_files = DotDict.wrap({ + # # pileup weight corrections + # "pu_sf": (f"{json_mirror}/POG/LUM/{corr_tag}/puWeights.json.gz", "v1"), - # jet energy correction - "jet_jerc": (f"{json_mirror}/POG/JME/{corr_tag}/jet_jerc.json.gz", "v1"), + # # jet energy correction + # "jet_jerc": (f"{json_mirror}/POG/JME/{corr_tag}/jet_jerc.json.gz", "v1"), - # jet energy correction ak8 - "fat_jet_jerc": (f"{json_mirror}/POG/JME/{corr_tag}/fatJet_jerc.json.gz", "v1"), + # # jet energy correction ak8 + # "fat_jet_jerc": (f"{json_mirror}/POG/JME/{corr_tag}/fatJet_jerc.json.gz", "v1"), - # jet veto map - "jet_veto_map": (f"{json_mirror}/POG/JME/{corr_tag}/jetvetomaps.json.gz", "v1"), + # # jet veto map + # "jet_veto_map": (f"{json_mirror}/POG/JME/{corr_tag}/jetvetomaps.json.gz", "v1"), - # electron scale factors - "electron_sf": (f"{json_mirror}/POG/EGM/{corr_tag}/electron.json.gz", "v1"), + # # electron scale factors + # "electron_sf": (f"{json_mirror}/POG/EGM/{corr_tag}/electron.json.gz", "v1"), - # muon scale factors - "muon_sf": (f"{json_mirror}/POG/MUO/{corr_tag}/muon_Z.json.gz", "v1"), + # # muon scale factors + # "muon_sf": (f"{json_mirror}/POG/MUO/{corr_tag}/muon_Z.json.gz", "v1"), - # btag scale factor - "btag_sf_corr": (f"{json_mirror}/POG/BTV/{corr_tag}/btagging.json.gz", "v1"), + # # btag scale factor + # "btag_sf_corr": (f"{json_mirror}/POG/BTV/{corr_tag}/btagging.json.gz", "v1"), - # met phi corrector - "met_phi_corr": (f"{json_mirror}/POG/JME/{corr_tag}/met.json.gz", "v1"), + # # V+jets reweighting + # "vjets_reweighting": f"{local_repo}/data/json/vjets_reweighting.json.gz", - # V+jets reweighting - "vjets_reweighting": f"{local_repo}/data/json/vjets_reweighting.json.gz", + # # jet id + # "jet_id": f"{json_mirror}/POG/JME/{corr_tag}/jetid.json.gz", + # }) + + # if cfg.x.run == 2: + # cfg.x.external_files.update(DotDict.wrap({ + # "met_phi_corr": (f"{json_mirror}/POG/JME/{corr_tag}/met.json.gz", "v1"), + # })) + # elif cfg.x.run == 3: + # met_corr_tag = f"{year}_{year}{jerc_postfix}" + # cfg.x.external_files.update(DotDict.wrap({ + # # met phi corrector + # "met_phi_corr": (f"{json_mirror}/POG/JME/{corr_tag}/met_xyCorrections_{met_corr_tag}.json.gz", "v1"), + # })) + + # if cfg.x.cpn_tag == "2022preEE": + # cfg.x.external_files.update(DotDict.wrap({ + # # files from https://twiki.cern.ch/twiki/bin/view/CMSPublic/SWGuideGoodLumiSectionsJSONFile + # "lumi": { + # "golden": ("https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/Cert_Collisions2022_355100_362760_Golden.json", "v1"), # noqa + # "normtag": ("/afs/cern.ch/user/l/lumipro/public/Normtags/normtag_PHYSICS.json", "v1"), + # }, + # })) + # elif cfg.x.cpn_tag == "2022postEE": + # cfg.x.external_files.update(DotDict.wrap({ + # # files from https://twiki.cern.ch/twiki/bin/view/CMSPublic/SWGuideGoodLumiSectionsJSONFile + # "lumi": { + # "golden": ("https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/Cert_Collisions2022_355100_362760_Golden.json", "v1"), # noqa + # "normtag": ("/afs/cern.ch/user/l/lumipro/public/Normtags/normtag_PHYSICS.json", "v1"), + # }, + # })) + # elif cfg.x.cpn_tag == "2023preBPix": + # cfg.x.external_files.update(DotDict.wrap({ + # # files from https://twiki.cern.ch/twiki/bin/view/CMSPublic/SWGuideGoodLumiSectionsJSONFile + # "lumi": { + # "golden": ("https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions23/Cert_Collisions2023_366442_370790_Golden.json", "v1"), # noqa + # "normtag": ("/afs/cern.ch/user/l/lumipro/public/Normtags/normtag_PHYSICS.json", "v1"), + # }, + # })) + # elif cfg.x.cpn_tag == "2023postBPix": + # cfg.x.external_files.update(DotDict.wrap({ + # # files from https://twiki.cern.ch/twiki/bin/view/CMSPublic/SWGuideGoodLumiSectionsJSONFile + # "lumi": { + # "golden": ("https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions23/Cert_Collisions2023_366442_370790_Golden.json", "v1"), # noqa + # "normtag": ("/afs/cern.ch/user/l/lumipro/public/Normtags/normtag_PHYSICS.json", "v1"), + # }, + # })) + # else: + # raise NotImplementedError(f"No lumi and pu files provided for year {year}") + + + # + # external files + # setup taken from https://github.com/uhh-cms/hh2bbtautau/blob/ed8f363ac239b0257fc7f470b96f5c09a0572c34/hbt/config/configs_hbt.py#L1574 # noqa: E501 + # https://cms-analysis-corrections.docs.cern.ch + # + + cfg.x.external_files = DotDict() + + # helper + def add_external(name, value): + if isinstance(value, dict): + value = DotDict.wrap(value) + cfg.x.external_files[name] = value + + # prepare run/era/nano meta data info to determine files in the CAT metadata structure + # see https://cms-analysis-corrections.docs.cern.ch + cat_info = { + (2022, "", 12): CATInfo( + run=3, + vnano=12, + era="22CDSep23-Summer22", + pog_directories={"dc": "Collisions22"}, + snapshot=CATSnapshot(btv="2025-08-20", dc="2025-07-25", egm="2025-04-15", jme="2025-09-23", lum="2024-01-31", muo="2025-08-14", tau="2025-10-01"), # noqa: E501 + ), + (2022, "EE", 12): CATInfo( + run=3, + vnano=12, + era="22EFGSep23-Summer22EE", + pog_directories={"dc": "Collisions22"}, + snapshot=CATSnapshot(btv="2025-08-20", dc="2025-07-25", egm="2025-04-15", jme="2025-10-07", lum="2024-01-31", muo="2025-08-14", tau="2025-10-01"), # noqa: E501 + ), + (2023, "", 12): CATInfo( + run=3, + vnano=12, + era="23CSep23-Summer23", + # pog_eras={"tau": "23CSep23-Summer22"}, # TODO: remove once typo in CAT repo is fixed + pog_directories={"dc": "Collisions23"}, + snapshot=CATSnapshot(btv="2025-08-20", dc="2025-07-25", egm="2025-04-15", jme="2025-10-07", lum="2024-01-31", muo="2025-08-14", tau="2025-10-01"), # noqa: E501 + ), + (2023, "BPix", 12): CATInfo( + run=3, + vnano=12, + era="23DSep23-Summer23BPix", + pog_directories={"dc": "Collisions23"}, + snapshot=CATSnapshot(btv="2025-08-20", dc="2025-07-25", egm="2025-04-15", jme="2025-10-07", lum="2024-01-31", muo="2025-08-14", tau="2025-10-01"), # noqa: E501 + ), + (2024, "", 15): CATInfo( + run=3, + vnano=15, + era="24CDEReprocessingFGHIPrompt-Summer24", + pog_directories={"dc": "Collisions24"}, + snapshot=CATSnapshot(btv="2026-01-30", dc="2025-07-25", egm="2025-12-15", jme="2025-12-02", muo="2025-11-27", lum="2025-12-02"), # noqa: E501 + ), + }[(year, campaign.x.postfix, vnano)] + cfg.x.cat_info = cat_info + + # common files + # (versions in the end are for hashing in cases where file contents changed but paths did not) + add_external("lumi", { + "golden": { + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=161#Year_2022 + 2022: (cat_info.get_file("dc", "Cert_Collisions2022_355100_362760_Golden.json"), "v1"), + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=161#Year_2023 + 2023: (cat_info.get_file("dc", "Cert_Collisions2023_366442_370790_Golden.json"), "v1"), + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=180#Year_2024 + # not yet available at CAT space + # 2024: (cat_info.get_file("dc", "Cert_Collisions2024_378981_386951_Golden.json"), "v1"), + 2024: ("https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions24/Cert_Collisions2024_378981_386951_Golden.json", "v1"), # noqa: E501 + }[year], + "normtag": { + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=161#Year_2022 + 2022: ("/cvmfs/cms-bril.cern.ch/cms-lumi-pog/Normtags/normtag_BRIL.json", "v1"), + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=161#Year_2023 + 2023: ("/cvmfs/cms-bril.cern.ch/cms-lumi-pog/Normtags/normtag_BRIL.json", "v1"), + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=180#Year_2024 + 2024: ("/cvmfs/cms-bril.cern.ch/cms-lumi-pog/Normtags/normtag_BRIL.json", "v1"), # TODO: correct? + }[year], }) - # temporary fix due to missing corrections in run 3 - # electron and met still missing - if cfg.x.run == 3: - # cfg.add_tag("skip_electron_weights") - # cfg.add_tag("skip_muon_weights") + # pileup weight corrections + if year != 2024: # TODO: not yet available, see https://cms-analysis-corrections.docs.cern.ch + add_external("pu_sf", (cat_info.get_file("lum", "puWeights.json.gz"), "v1")) + elif year == 2024: + add_external("pu_sf", (cat_info.get_file("lum", "puWeights_BCDEFGHI.json.gz"), "v1")) - cfg.x.external_files.pop("met_phi_corr") + # jet energy correction + add_external("jet_jerc", (cat_info.get_file("jme", "jet_jerc.json.gz"), "v1")) - if year == 2022 and campaign.x.EE == "pre": - cfg.x.external_files.update(DotDict.wrap({ - # files from https://twiki.cern.ch/twiki/bin/view/CMSPublic/SWGuideGoodLumiSectionsJSONFile - "lumi": { - "golden": ("https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/Cert_Collisions2022_355100_362760_Golden.json", "v1"), # noqa - "normtag": ("/afs/cern.ch/user/l/lumipro/public/Normtags/normtag_PHYSICS.json", "v1"), - }, - "pu": { - # "json": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/BCD/pileup_JSON.txt", "v1"), # noqa - "json": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/BCDEFG/pileup_JSON.txt", "v1"), # noqa - "mc_profile": ("https://raw.githubusercontent.com/cms-sw/cmssw/bb525104a7ddb93685f8ced6fed1ab793b2d2103/SimGeneral/MixingModule/python/Run3_2022_LHC_Simulation_10h_2h_cfi.py", "v1"), # noqa - "data_profile": { - # "nominal": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/BCD/pileupHistogram-Cert_Collisions2022_355100_357900_eraBCD_GoldenJson-13p6TeV-69200ub-99bins.root", "v1"), # noqa - # "minbias_xs_up": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/BCD/pileupHistogram-Cert_Collisions2022_355100_357900_eraBCD_GoldenJson-13p6TeV-72400ub-99bins.root", "v1"), # noqa - # "minbias_xs_down": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/BCD/pileupHistogram-Cert_Collisions2022_355100_357900_eraBCD_GoldenJson-13p6TeV-66000ub-99bins.root", "v1"), # noqa - "nominal": (f"/afs/cern.ch/user/a/anhaddad/public/Collisions22/pileupHistogram-Cert_Collisions2022_355100_362760_GoldenJson-13p6TeV-69200ub-100bins.root", "v1"), # noqa - "minbias_xs_up": (f"/afs/cern.ch/user/a/anhaddad/public/Collisions22/pileupHistogram-Cert_Collisions2022_355100_362760_GoldenJson-13p6TeV-72400ub-100bins.root", "v1"), # noqa - "minbias_xs_down": (f"/afs/cern.ch/user/a/anhaddad/public/Collisions22/pileupHistogram-Cert_Collisions2022_355100_362760_GoldenJson-13p6TeV-66000ub-100bins.root", "v1"), # noqa - }, - }, - })) - elif year == 2022 and campaign.x.EE == "post": - cfg.x.external_files.update(DotDict.wrap({ - # files from https://twiki.cern.ch/twiki/bin/view/CMSPublic/SWGuideGoodLumiSectionsJSONFile - "lumi": { - "golden": ("https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/Cert_Collisions2022_355100_362760_Golden.json", "v1"), # noqa - "normtag": ("/afs/cern.ch/user/l/lumipro/public/Normtags/normtag_PHYSICS.json", "v1"), - }, - "pu": { - # "json": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/EFG/pileup_JSON.txt", "v1"), # noqa - "json": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/BCDEFG/pileup_JSON.txt", "v1"), # noqa - "mc_profile": ("https://raw.githubusercontent.com/cms-sw/cmssw/bb525104a7ddb93685f8ced6fed1ab793b2d2103/SimGeneral/MixingModule/python/Run3_2022_LHC_Simulation_10h_2h_cfi.py", "v1"), # noqa - "data_profile": { - # data profiles were produced with 99 bins instead of 100 --> use custom produced data profiles instead # noqa - # "nominal": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/EFG/pileupHistogram-Cert_Collisions2022_359022_362760_eraEFG_GoldenJson-13p6TeV-69200ub-99bins.root", "v1"), # noqa - # "minbias_xs_up": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/EFG/pileupHistogram-Cert_Collisions2022_359022_362760_eraEFG_GoldenJson-13p6TeV-72400ub-99bins.root", "v1"), # noqa - # "minbias_xs_down": (f"https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions22/PileUp/EFG/pileupHistogram-Cert_Collisions2022_359022_362760_eraEFG_GoldenJson-13p6TeV-66000ub-99bins.root", "v1"), # noqa - "nominal": (f"/afs/cern.ch/user/a/anhaddad/public/Collisions22/pileupHistogram-Cert_Collisions2022_355100_362760_GoldenJson-13p6TeV-69200ub-100bins.root", "v1"), # noqa - "minbias_xs_up": (f"/afs/cern.ch/user/a/anhaddad/public/Collisions22/pileupHistogram-Cert_Collisions2022_355100_362760_GoldenJson-13p6TeV-72400ub-100bins.root", "v1"), # noqa - "minbias_xs_down": (f"/afs/cern.ch/user/a/anhaddad/public/Collisions22/pileupHistogram-Cert_Collisions2022_355100_362760_GoldenJson-13p6TeV-66000ub-100bins.root", "v1"), # noqa - }, - }, - })) + # fat jet energy correction + add_external("fat_jet_jerc", (cat_info.get_file("jme", "fat_jet_jerc.json.gz" if year != 2024 else "fatJet_jerc.json.gz"), "v1")) # noqa: E501 + + # jet veto map + add_external("jet_veto_map", (cat_info.get_file("jme", "jetvetomaps.json.gz"), "v1")) + + # btag scale factor + if year != 2024: + add_external("btag_sf_corr", (cat_info.get_file("btv", "btagging.json.gz"), "v1")) else: - raise NotImplementedError(f"No lumi and pu files provided for year {year}") + # SF stored in preliminary file for 2024 for now + # add_external("btag_sf_corr", (cat_info.get_file("btv", "btagging_preliminary.json.gz"), "v1")) # noqa: E501 + # use custom file with merged SF for both b/c and light jets + add_external("btag_wp_sf_corr", ("/data/dust/user/matthiej/mttbar/mtt/config/run3/btagging_preliminary_merged.json.gz", "v1")) # noqa: E501 + + # updated jet id + add_external("jet_id", (cat_info.get_file("jme", "jetid.json.gz"), "v1")) + + # muon scale factors + add_external("muon_sf", (cat_info.get_file("muo", "muon_Z.json.gz"), "v1")) + + # met phi correction + if year != 2024: # TODO: not yet available for 2024 + add_external("met_phi_corr", (cat_info.get_file("jme", f"met_xyCorrections_{year}_{year}{campaign.x.postfix}.json.gz"), "v1")) # noqa: E501 + + # electron scale factors + add_external("electron_sf", (cat_info.get_file("egm", "electron.json.gz"), "v1")) + # electron energy correction and smearing + add_external("electron_ss", (cat_info.get_file("egm", "electronSS_EtDependent.json.gz"), "v1")) # FIXME correct for us? # noqa: E501 # # event reduction configuration diff --git a/topsf/config/run3/config_wp_new.py b/topsf/config/run3/config_wp_new.py new file mode 100644 index 0000000..04c48cd --- /dev/null +++ b/topsf/config/run3/config_wp_new.py @@ -0,0 +1,1204 @@ +# coding: utf-8 + +""" +Configuration creation for top-tagging working point +measurement using Run3 samples. +""" + +from __future__ import annotations + +import functools +import os +import law + +import order as od +import yaml + +from scinum import Number + +from columnflow.util import DotDict +from columnflow.cms_util import CATInfo, CATSnapshot +from columnflow.config_util import ( + add_shift_aliases, + get_root_processes_from_campaign, + get_shifts_from_sources, + verify_config_processes, +) +from columnflow.production.cms.electron import ElectronSFConfig +from columnflow.production.cms.muon import MuonSFConfig +from columnflow.production.cms.jet import JetIdConfig + +from topsf.config.variables import add_variables +from topsf.config.categories_wp import add_categories +from topsf.util import has_tag +from topsf.config.datasets import add_datasets_from_yaml +from topsf.config.taggers import btag_wps, toptag_wps + +thisdir = os.path.dirname(os.path.abspath(__file__)) +logger = law.logger.get_logger(__name__) + + +def add_new_config( + analysis: od.Analysis, + campaign: od.Campaign, + config_name: str | None = None, + config_id: int | None = None, + limit_dataset_files: int | None = None, +) -> od.Config: + """ + Configurable function for creating a config for a run2 analysis given + a base *analysis* object and a *campaign* (i.e. set of datasets). + """ + # validation + assert campaign.x.year in [2022, 2023, 2024] + if campaign.x.year == 2022: + assert campaign.x.EE in ["pre", "post"] + elif campaign.x.year == 2023: + assert campaign.x.BPix in ["pre", "post"] + + # gather campaign data + year = campaign.x.year + year2 = year % 100 + corr_postfix = "" + if year == 2022: + corr_postfix = f"{campaign.x.EE}EE" + elif year == 2023: + corr_postfix = f"{campaign.x.BPix}BPix" + + implemented_years = [2022, 2023, 2024] + + if year not in implemented_years: + raise NotImplementedError("For now, only 2022, 2023, and 2024 campaigns are fully implemented") + + # create a config by passing the campaign + # (if id and name are not set they will be taken from the campaign) + cfg = analysis.add_config(campaign, name=config_name, id=config_id) + + # add some important tags to the config + cfg.x.run = 3 + cfg.x.cpn_tag = f"{year}{corr_postfix}" + cfg.x.year = year + vnano = campaign.x.version + logger.info(f"Creating config '{cfg.name}' for campaign '{campaign.name}' with year {year} and version {vnano}") + cfg.add_tag("skip_btag_weights") + cfg.add_tag("skip_btag_wp_weights") + cfg.add_tag("skip_muon_weights") + cfg.add_tag("skip_electron_weights") + cfg.add_tag("skip_scale") + cfg.add_tag("skip_pdf") + cfg.add_tag("no_ps_weights") + + # + # configure processes + # + + # get all root processes + procs = get_root_processes_from_campaign(campaign) + + # set color of some processes + colors = { + "tt": "#E04F21", # red + "qcd": "#5E8FFC", # blue + } + + # add processes we are interested in + process_names = [ + "tt", + "qcd", + # # split qcd into bins for studies + # "qcd_mu_pt15to20", + # "qcd_mu_pt20to30", + # "qcd_mu_pt30to50", + # "qcd_mu_pt50to80", + # "qcd_mu_pt80to120", + # "qcd_mu_pt120to170", + # "qcd_mu_pt170to300", + # "qcd_mu_pt300to470", + # "qcd_mu_pt470to600", + # "qcd_mu_pt600to800", + # "qcd_mu_pt800to1000", + # "qcd_mu_pt1000toinf", + # "qcd_em_pt10to30", + # "qcd_em_pt30to50", + # "qcd_em_pt50to80", + # "qcd_em_pt80to120", + # "qcd_em_pt120to170", + # "qcd_em_pt170to300", + # "qcd_em_pt300toinf", + ] + + blue_shades_mu = [ + "#ADD8E6", "#87CEEB", "#5CB3FF", "#4682B4", + "#4169E1", "#0000FF", "#0000CD", "#00008B", + "#191970", "#0F52BA", "#082567", "#001F3F", + ] + green_shades_em = [ + "#98FB98", "#7CFC00", "#32CD32", + "#228B22", "#006400", "#013220", "#002B00", + ] + + for process_name in process_names: + # add the process + proc = cfg.add_process(procs.get(process_name)) + + # mark the presence of a top quark + if proc.name.startswith("tt"): + proc.add_tag({"has_top", "is_ttbar"}) + + # assign colors of qcd_mu processes using the blue shades + if proc.name.startswith("qcd_mu"): + proc.color = blue_shades_mu.pop(0) + + # assign colors of qcd_em processes using the green shades + if proc.name.startswith("qcd_em"): + proc.color = green_shades_em.pop(0) + + # assign colors of tt processes + if proc.name == "tt": + proc.color = colors["tt"] + + # assign colors of qcd processes + if proc.name == "qcd": + proc.color = colors["qcd"] + + # + # datasets + # + + # add datasets we need to study + dataset_names = add_datasets_from_yaml(cfg, limit_dataset_files=limit_dataset_files, dataset_types=["tt_fh", "qcd_pt"]) + + # verify that the root processes of each dataset (or one of their + # ancestor processes) are registered in the config + verify_config_processes(cfg, warn=True) + logger.info(f"Added {len(cfg.processes)} processes and {len(cfg.datasets)} datasets to config '{cfg.name}'") + + # + # defaults + # + + # default objects, such as calibrator, selector, producer, + # ml model, inference model, etc + cfg.x.default_calibrator = "default" + cfg.x.default_selector = "wp" + cfg.x.default_reducer = "cf_default" + cfg.x.default_producer = "default" + cfg.x.default_hist_producer = "all_weights" # NOTE: no need to further differentiate for now, use more fine grained hist producers if needed in the future + cfg.x.default_ml_model = None + cfg.x.default_inference_model = None + cfg.x.default_categories = ("incl",) + cfg.x.default_variables = ( + "fatjet_pt", + "fatjet_tau32", + ) + + # + # parameter groups + # + + # process groups for conveniently looping over certain processs + # (used in wrapper_factory and during plotting) + cfg.x.process_groups = { + "all": process_names, + "qcd": ["qcd*"], + "tt": ["tt*"], + } + + # dataset groups for conveniently looping over certain datasets + # (used in wrapper_factory and during plotting) + cfg.x.dataset_groups = { + "all": dataset_names, + "qcd": ["qcd*"], + "tt": ["tt*"], + } + + # category groups for conveniently looping over certain categories + # (used during plotting) + cfg.x.category_groups = {} + + # variable groups for conveniently looping over certain variables + # (used during plotting) + cfg.x.variable_groups = {} + + # shift groups for conveniently looping over certain shifts + # (used during plotting) + cfg.x.shift_groups = {} + + # selector step groups for conveniently looping over certain steps + # (used in cutflow tasks) + cfg.x.selector_step_groups = { + "default": [ + # "FatJet", "METFilters", + "cleanup", "FatJet" + ], + } + + # Exception: no weight producer configured for task. cf.MergeShiftedHistograms. + # As of 02.05.2024, it is required to pass a weight_producer for tasks creating histograms. + # You can add a 'default_weight_producer' to your config or directly add the weight_producer + # on command line via the '--weight_producer' parameter. To reproduce results from before this date, + # you can use the 'all_weights' weight_producer defined in columnflow.weight.all_weights: + # With cf 0.3.x, the 'weight_producer' has been renamed to 'hist_producer'. + + # custom labels for selector steps + cfg.x.selector_step_labels = {} + + # plotting settings groups + cfg.x.general_settings_groups = {} + cfg.x.process_settings_groups = {} + cfg.x.variable_settings_groups = {} + + # + # dataset customization + # + + # custom method and sandbox for determining dataset lfns + cfg.x.get_dataset_lfns = None + cfg.x.get_dataset_lfns_sandbox = None + + # whether to validate the number of obtained LFNs in GetDatasetLFNs + cfg.x.validate_dataset_lfns = limit_dataset_files is None + + # + # tagger working points + # + + # full b-tag working points dict + cfg.x.btag_working_points = btag_wps(cfg, full=True) + # store only upart wps for fixed wp sf producer + cfg.x.btag_working_points.btagUParTAK4B.fixed_wp = btag_wps(cfg, full=False) + + # jet selection parameters + subjet_wp_key = "btagUParTAK4B" if year == 2024 else "deepcsv" + subjet_btag_wp = getattr(cfg.x.btag_working_points, subjet_wp_key).loose + if subjet_btag_wp < 0: + raise ValueError(f"Invalid subjet b-tag working point for year {year}. Please check the configuration.") + + cfg.x.jet_selection = DotDict.wrap({ + "ak8": { + "column": "FatJet", + "min_pt": 300, + "max_abseta": 2.4, # note: SF analysis has 2.5 + "msoftdrop_range": (105, 210), + # https://twiki.cern.ch/twiki/bin/view/CMS/JetID13p6TeV + "jetId": 2, # bit2 (2): pass tight ID, fail tightLepVeto, bit3 (6): pass tight and tightLepVeto ID + # probe jet pt bins (used by category builder) + "pt_bins": [300, 400, 480, 600, None], + # parameters for b-tagged subjets + "subjet_column": "SubJet", + "subjet_btag": "btagDeepB" if not year == 2024 else "btagUParTAK4B", + "subjet_btag_wp": subjet_btag_wp, + }, + }) + + # MET selection parameters + # FIXME: use PuppiMET for Run 3? What's the difference? It's better? + # https://twiki.cern.ch/twiki/bin/viewauth/CMS/MissingETRun2Corrections?rev=79#xy_Shift_Correction_MET_phi_modu + cfg.x.met_selection = DotDict.wrap({ + "default": { + "column": "PuppiMET" if cfg.x.run == 3 else "MET", + "min_pt": 50, + }, + }) + + # + # luminosity + # + + # lumi values in inverse pb + # https://twiki.cern.ch/twiki/bin/viewauth/CMS/PdmVRun3Analysis + if year == 2022: + if campaign.x.EE == "pre": + cfg.x.luminosity = Number(7971, { + "lumi_13TeV_2022": 0.01j, + "lumi_13TeV_correlated": 0.006j, + }) + elif campaign.x.EE == "post": + cfg.x.luminosity = Number(26337, { + "lumi_13TeV_2022": 0.01j, + "lumi_13TeV_correlated": 0.006j, + }) + elif year == 2023: + if campaign.has_tag("preBPix"): + cfg.x.luminosity = Number(17794, { + "lumi_13TeV_2023": 0.01j, + "lumi_13TeV_correlated": 0.006j, + }) + elif campaign.has_tag("postBPix"): + cfg.x.luminosity = Number(9451, { + "lumi_13TeV_2023": 0.01j, + "lumi_13TeV_correlated": 0.006j, + }) + elif year == 2024: + # Total - EraB + cfg.x.luminosity = Number(109_950.0 - 130.0, { + "lumi_13p6TeV_2024": 0.016j, # CERN-CMS-DP-2026-003 + }) + else: + raise NotImplementedError(f"Luminosity for year {year} is not defined.") + + # + # MET filters + # + + # https://twiki.cern.ch/twiki/bin/view/CMS/MissingETOptionalFiltersRun2 + cfg.x.met_filters = { + "Flag.goodVertices", + "Flag.globalSuperTightHalo2016Filter", + "Flag.EcalDeadCellTriggerPrimitiveFilter", + "Flag.BadPFMuonFilter", + "Flag.BadPFMuonDzFilter", + "Flag.eeBadScFilter", + "Flag.ecalBadCalibFilter", + } + if year == 2024: + cfg.x.met_filters.add("Flag.hfNoisyHitsFilter") + + # + # JEC & JER # FIXME: Taken from HBW + # https://github.com/uhh-cms/hh2bbww/blob/master/hbw/config/config_run2.py#L138C5-L269C1 + # + + # jec configuration + # https://twiki.cern.ch/twiki/bin/view/CMS/JECDataMC?rev=2017#Jet_Energy_Corrections_in_Run2 + + # jec configuration taken from HBW + # https://github.com/uhh-cms/hh2bbww/blob/master/hbw/config/config_run2.py#L138C5-L269C1 + # https://twiki.cern.ch/twiki/bin/view/CMS/JECDataMC?rev=201 + jerc_postfix = campaign.x.postfix + if jerc_postfix not in ("", "EE", "BPix"): + raise ValueError(f"Invalid JERC postfix '{jerc_postfix}' for campaign {campaign.name}.") + if year == 2022: + jer_campaign = jec_campaign = f"Summer{year2}{jerc_postfix}_22Sep2023" + elif year == 2023: + era = "Cv1234" if campaign.has_tag("preBPix") else "D" + jer_campaign = f"Summer{year2}{jerc_postfix}Prompt{year2}_Run{era}" + jec_campaign = f"Summer{year2}{jerc_postfix}Prompt{year2}" + elif year == 2024: + jec_campaign = "Summer24Prompt24" + jer_campaign = "Summer23BPixPrompt23_RunD" # no 2024 JER yet, use 2023 BPix: https://cms-jerc.web.cern.ch/Recommendations/#2024_1 # noqa + else: + raise NotImplementedError(f"JEC/JER configuration for year {year} is not defined.") + + jet_type = "AK4PFPuppi" + fatjet_type = "AK8PFPuppi" + jec_ak4_version = jec_ak8_version = { + 2022: "V3", + 2023: "V2" if not year == 2023 else "V3", + 2024: "V2", + }[year] + + cfg.x.jec = DotDict.wrap({ + "Jet": { + "campaign": jec_campaign, + "version": jec_ak4_version, + "jet_type": jet_type, + "levels": ["L1FastJet", "L2Relative", "L2L3Residual", "L3Absolute"], + "levels_for_type1_met": ["L1FastJet"], + "uncertainty_sources": [ + # "AbsoluteStat", + # "AbsoluteScale", + # "AbsoluteSample", + # "AbsoluteFlavMap", + # "AbsoluteMPFBias", + # "Fragmentation", + # "SinglePionECAL", + # "SinglePionHCAL", + # "FlavorQCD", + # "TimePtEta", + # "RelativeJEREC1", + # "RelativeJEREC2", + # "RelativeJERHF", + # "RelativePtBB", + # "RelativePtEC1", + # "RelativePtEC2", + # "RelativePtHF", + # "RelativeBal", + # "RelativeSample", + # "RelativeFSR", + # "RelativeStatFSR", + # "RelativeStatEC", + # "RelativeStatHF", + # "PileUpDataMC", + # "PileUpPtRef", + # "PileUpPtBB", + # "PileUpPtEC1", + # "PileUpPtEC2", + # "PileUpPtHF", + # "PileUpMuZero", + # "PileUpEnvelope", + # "SubTotalPileUp", + # "SubTotalRelative", + # "SubTotalPt", + # "SubTotalScale", + # "SubTotalAbsolute", + # "SubTotalMC", + "Total", + # "TotalNoFlavor", + # "TotalNoTime", + # "TotalNoFlavorNoTime", + # "FlavorZJet", + # "FlavorPhotonJet", + # "FlavorPureGluon", + # "FlavorPureQuark", + # "FlavorPureCharm", + # "FlavorPureBottom", + # "TimeRunA", + # "TimeRunB", + # "TimeRunC", + # "TimeRunD", + "CorrelationGroupMPFInSitu", + "CorrelationGroupIntercalibration", + "CorrelationGroupbJES", + "CorrelationGroupFlavor", + "CorrelationGroupUncorrelated", + ], + "data_per_era": True if year == 2022 else False, + }, + "FatJet": { + "campaign": jec_campaign, + "version": jec_ak8_version, + "jet_type": fatjet_type, + "levels": ["L1FastJet", "L2Relative", "L2L3Residual", "L3Absolute"], + "levels_for_type1_met": ["L1FastJet"], + "uncertainty_sources": [ + # "AbsoluteStat", + # "AbsoluteScale", + # "AbsoluteSample", + # "AbsoluteFlavMap", + # "AbsoluteMPFBias", + # "Fragmentation", + # "SinglePionECAL", + # "SinglePionHCAL", + # "FlavorQCD", + # "TimePtEta", + # "RelativeJEREC1", + # "RelativeJEREC2", + # "RelativeJERHF", + # "RelativePtBB", + # "RelativePtEC1", + # "RelativePtEC2", + # "RelativePtHF", + # "RelativeBal", + # "RelativeSample", + # "RelativeFSR", + # "RelativeStatFSR", + # "RelativeStatEC", + # "RelativeStatHF", + # "PileUpDataMC", + # "PileUpPtRef", + # "PileUpPtBB", + # "PileUpPtEC1", + # "PileUpPtEC2", + # "PileUpPtHF", + # "PileUpMuZero", + # "PileUpEnvelope", + # "SubTotalPileUp", + # "SubTotalRelative", + # "SubTotalPt", + # "SubTotalScale", + # "SubTotalAbsolute", + # "SubTotalMC", + "Total", + # "TotalNoFlavor", + # "TotalNoTime", + # "TotalNoFlavorNoTime", + # "FlavorZJet", + # "FlavorPhotonJet", + # "FlavorPureGluon", + # "FlavorPureQuark", + # "FlavorPureCharm", + # "FlavorPureBottom", + # "TimeRunA", + # "TimeRunB", + # "TimeRunC", + # "TimeRunD", + "CorrelationGroupMPFInSitu", + "CorrelationGroupIntercalibration", + "CorrelationGroupbJES", + "CorrelationGroupFlavor", + "CorrelationGroupUncorrelated", + ], + "data_per_era": True if year == 2022 else False, + }, + "SubJet": { + "campaign": jec_campaign, + "version": jec_ak4_version, + "jet_type": jet_type, + "levels": ["L1FastJet", "L2Relative", "L2L3Residual", "L3Absolute"], + "levels_for_type1_met": ["L1FastJet"], + "uncertainty_sources": [ + # "AbsoluteStat", + # "AbsoluteScale", + # "AbsoluteSample", + # "AbsoluteFlavMap", + # "AbsoluteMPFBias", + # "Fragmentation", + # "SinglePionECAL", + # "SinglePionHCAL", + # "FlavorQCD", + # "TimePtEta", + # "RelativeJEREC1", + # "RelativeJEREC2", + # "RelativeJERHF", + # "RelativePtBB", + # "RelativePtEC1", + # "RelativePtEC2", + # "RelativePtHF", + # "RelativeBal", + # "RelativeSample", + # "RelativeFSR", + # "RelativeStatFSR", + # "RelativeStatEC", + # "RelativeStatHF", + # "PileUpDataMC", + # "PileUpPtRef", + # "PileUpPtBB", + # "PileUpPtEC1", + # "PileUpPtEC2", + # "PileUpPtHF", + # "PileUpMuZero", + # "PileUpEnvelope", + # "SubTotalPileUp", + # "SubTotalRelative", + # "SubTotalPt", + # "SubTotalScale", + # "SubTotalAbsolute", + # "SubTotalMC", + "Total", + # "TotalNoFlavor", + # "TotalNoTime", + # "TotalNoFlavorNoTime", + # "FlavorZJet", + # "FlavorPhotonJet", + # "FlavorPureGluon", + # "FlavorPureQuark", + # "FlavorPureCharm", + # "FlavorPureBottom", + # "TimeRunA", + # "TimeRunB", + # "TimeRunC", + # "TimeRunD", + # "CorrelationGroupMPFInSitu", + # "CorrelationGroupIntercalibration", + # "CorrelationGroupbJES", + # "CorrelationGroupFlavor", + # "CorrelationGroupUncorrelated", + ], + "data_per_era": True if year == 2022 else False, + }, + }) + + # JER + # https://twiki.cern.ch/twiki/bin/view/CMS/JetResolution?rev=107 + cfg.x.jer = DotDict.wrap({ + "Jet": { + "campaign": jer_campaign, + "version": {2022: "JRV1", 2023: "JRV1", 2024: "JRV1"}[year], + "jet_type": jet_type, + }, + "FatJet": { + "campaign": jer_campaign, + "version": {2022: "JRV1", 2023: "JRV1", 2024: "JRV1"}[year], + "jet_type": fatjet_type, + }, + "SubJet": { + "campaign": jer_campaign, + "version": {2022: "JRV1", 2023: "JRV1", 2024: "JRV1"}[year], + "jet_type": jet_type, + }, + }) + cfg.x.jet_id = JetIdConfig( + corrections={ + "AK4PUPPI_Tight": 2, + "AK4PUPPI_TightLeptonVeto": 6, + }, + ) + + # JEC uncertainty sources propagated to btag scale factors + # (names derived from contents in BTV correctionlib file) + cfg.x.btag_sf_jec_sources = [ + "", # total + "Absolute", + "AbsoluteMPFBias", + "AbsoluteScale", + "AbsoluteStat", + f"Absolute_{year}", + "BBEC1", + f"BBEC1_{year}", + "EC2", + f"EC2_{year}", + "FlavorQCD", + "Fragmentation", + "HF", + f"HF_{year}", + "PileUpDataMC", + "PileUpPtBB", + "PileUpPtEC1", + "PileUpPtEC2", + "PileUpPtHF", + "PileUpPtRef", + "RelativeBal", + "RelativeFSR", + "RelativeJEREC1", + "RelativeJEREC2", + "RelativeJERHF", + "RelativePtBB", + "RelativePtEC1", + "RelativePtEC2", + "RelativePtHF", + "RelativeSample", + f"RelativeSample_{year}", + "RelativeStatEC", + "RelativeStatFSR", + "RelativeStatHF", + "SinglePionECAL", + "SinglePionHCAL", + "TimePtEta", + ] + + from columnflow.calibration.cms.met import METPhiConfig + met_column = cfg.x.met_selection.default.column + cfg.x.met_phi_correction = METPhiConfig( + met_name=met_column, + met_type=met_column, + correction_set="met_xy_corrections", + keep_uncorrected=True, # TODO do we need this? + pt_phi_variations={ + "stat_xdn": "metphi_statx_down", + "stat_xup": "metphi_statx_up", + "stat_ydn": "metphi_staty_down", + "stat_yup": "metphi_staty_up", + }, + variations={ + "pu_dn": "minbias_xs_down", + "pu_up": "minbias_xs_up", + }, + ) + + # + # producer configurations + # + # lepton scale factor producers not needed + + if year == 2024: + cfg.x.jet_id = JetIdConfig( + corrections={"AK4PUPPI_Tight": 2, "AK4PUPPI_TightLeptonVeto": 3}, + ) + cfg.x.fatjet_id = JetIdConfig( + corrections={"AK8PUPPI_Tight": 2, "AK8PUPPI_TightLeptonVeto": 3}, + ) + else: + logger.warning(f"(Fat)Jet ID recalculation not configured for {cfg.x.cpn_tag} campaign. Will be skipped.") + cfg.add_tag("skip_jet_ids") + + # top pt reweighting parameters + # https://twiki.cern.ch/twiki/bin/viewauth/CMS/TopPtReweighting#TOP_PAG_corrections_based_on_dat?rev=31 + cfg.x.top_pt_reweighting_params = { + "a": 0.0615, + "a_up": 0.0615 * 1.5, + "a_down": 0.0615 * 0.5, + "b": -0.0005, + "b_up": -0.0005 * 1.5, + "b_down": -0.0005 * 0.5, + } + + # V+jets reweighting + cfg.x.vjets_reweighting = DotDict.wrap({ + "w": { + "value": "wjets_kfactor_value", + "error": "wjets_kfactor_error", + }, + "z": { + "value": "zjets_kfactor_value", + "error": "zjets_kfactor_error", + }, + }) + + # + # systematic shifts + # + + # read in JEC sources from file + # FIXME same as Run2? + with open(os.path.join(thisdir, "jec_sources.yaml"), "r") as f: + all_jec_sources = yaml.load(f, yaml.Loader)["names"] + + # declare the shifts + def add_shifts(cfg): + logger.warn_once("The shift definitions in the config are currently not kept up as they are not needed for now. Make sure to adapt if needed.") # noqa + # nominal shift + cfg.add_shift(name="nominal", id=0) + + # tune shifts are covered by dedicated, varied datasets, so tag the shift as "disjoint_from_nominal" + # (this is currently used to decide whether ML evaluations are done on the full shifted dataset) + cfg.add_shift(name="tune_up", id=1, type="shape", tags={"disjoint_from_nominal"}) + cfg.add_shift(name="tune_down", id=2, type="shape", tags={"disjoint_from_nominal"}) + + cfg.add_shift(name="hdamp_up", id=3, type="shape", tags={"disjoint_from_nominal"}) + cfg.add_shift(name="hdamp_down", id=4, type="shape", tags={"disjoint_from_nominal"}) + + # pileup / minimum bias cross section variations + cfg.add_shift(name="minbias_xs_up", id=7, type="shape") + cfg.add_shift(name="minbias_xs_down", id=8, type="shape") + add_shift_aliases(cfg, "minbias_xs", {"pu_weight": "pu_weight_{name}"}) + + # top pt reweighting + cfg.add_shift(name="top_pt_up", id=9, type="shape") + cfg.add_shift(name="top_pt_down", id=10, type="shape") + add_shift_aliases(cfg, "top_pt", {"top_pt_weight": "top_pt_weight_{direction}"}) + + # renormalization scale + cfg.add_shift(name="mur_up", id=901, type="shape") + cfg.add_shift(name="mur_down", id=902, type="shape") + + # factorization scale + cfg.add_shift(name="muf_up", id=903, type="shape") + cfg.add_shift(name="muf_down", id=904, type="shape") + + # scale variation (?) + cfg.add_shift(name="scale_up", id=905, type="shape") + cfg.add_shift(name="scale_down", id=906, type="shape") + + # pdf variations + cfg.add_shift(name="pdf_up", id=951, type="shape") + cfg.add_shift(name="pdf_down", id=952, type="shape") + + # alpha_s variation + cfg.add_shift(name="alpha_up", id=961, type="shape") + cfg.add_shift(name="alpha_down", id=962, type="shape") + + # TODO: murf_envelope? + for unc in ["mur", "muf", "scale", "pdf", "alpha"]: + add_shift_aliases(cfg, unc, { + # TODO: normalized? + f"{unc}_weight": f"{unc}_weight_{{direction}}", + }) + + # event weights due to muon scale factors + if not has_tag("skip_muon_weights", cfg, operator=any): + cfg.add_shift(name="muon_up", id=111, type="shape") + cfg.add_shift(name="muon_down", id=112, type="shape") + add_shift_aliases(cfg, "muon", {"muon_weight": "muon_weight_{direction}"}) + + # event weights due to electron scale factors + cfg.add_shift(name="electron_up", id=121, type="shape") + cfg.add_shift(name="electron_down", id=122, type="shape") + add_shift_aliases(cfg, "electron", {"electron_weight": "electron_weight_{direction}"}) + + # V+jets reweighting + cfg.add_shift(name="vjets_up", id=201, type="shape") + cfg.add_shift(name="vjets_down", id=202, type="shape") + add_shift_aliases(cfg, "vjets", {"vjets_weight": "vjets_weight_{direction}"}) + + # prefiring weights + cfg.add_shift(name="l1_ecal_prefiring_up", id=301, type="shape") + cfg.add_shift(name="l1_ecal_prefiring_down", id=302, type="shape") + add_shift_aliases( + cfg, + "l1_ecal_prefiring", + { + "l1_ecal_prefiring_weight": "l1_ecal_prefiring_weight_{direction}", + }, + ) + + # b-tagging shifts + btag_uncs = [ + "hf", "lf", + f"hfstats1_{year}", f"hfstats2_{year}", + f"lfstats1_{year}", f"lfstats2_{year}", + "cferr1", "cferr2", + ] + for i, unc in enumerate(btag_uncs): + cfg.add_shift(name=f"btag_{unc}_up", id=501 + 2 * i, type="shape") + cfg.add_shift(name=f"btag_{unc}_down", id=502 + 2 * i, type="shape") + add_shift_aliases( + cfg, + f"btag_{unc}", + { + # TODO: normalized? + #"normalized_btag_weight": f"normalized_btag_weight_{unc}_{{direction}}", # noqa + #"normalized_njet_btag_weight": f"normalized_njet_btag_weight_{unc}_{{direction}}", # noqa + "btag_weight": f"btag_weight_{unc}_{{direction}}", + }, + ) + + # jet energy scale (JEC) uncertainty variations + for jec_source in cfg.x.jec.Jet.uncertainty_sources: + idx = all_jec_sources.index(jec_source) + cfg.add_shift(name=f"jec_{jec_source}_up", id=5000 + 2 * idx, type="shape") + cfg.add_shift(name=f"jec_{jec_source}_down", id=5001 + 2 * idx, type="shape") + add_shift_aliases( + cfg, + f"jec_{jec_source}", + { + "Jet.pt": "Jet.pt_{name}", + "Jet.mass": "Jet.mass_{name}", + "MET.pt": "MET.pt_{name}", + }, + ) + + # jet energy resolution (JER) scale factor variations + cfg.add_shift(name="jer_up", id=6000, type="shape") + cfg.add_shift(name="jer_down", id=6001, type="shape") + add_shift_aliases( + cfg, + "jer", + { + "Jet.pt": "Jet.pt_{name}", + "Jet.mass": "Jet.mass_{name}", + "MET.pt": "MET.pt_{name}", + }, + ) + + # PSWeight variations + cfg.add_shift(name="ISR_up", id=7001, type="shape") # PS weight [0] ISR=2 FSR=1 + cfg.add_shift(name="ISR_down", id=7002, type="shape") # PS weight [2] ISR=0.5 FSR=1 + add_shift_aliases(cfg, "ISR", {"ISR": "ISR_{direction}"}) + cfg.add_shift(name="FSR_up", id=7003, type="shape") # PS weight [1] ISR=1 FSR=2 + cfg.add_shift(name="FSR_down", id=7004, type="shape") # PS weight [3] ISR=1 FSR=0.5 + add_shift_aliases(cfg, "FSR", {"FSR": "FSR_{direction}"}) + + # add the shifts + # add_shifts(cfg) # NOTE: currently not needed for WP analysis, as no shifts are currently considered. Make sure to adapt if needed. + cfg.add_shift(name="nominal", id=0) + + # + # external files + # setup taken from https://github.com/uhh-cms/hh2bbtautau/blob/ed8f363ac239b0257fc7f470b96f5c09a0572c34/hbt/config/configs_hbt.py#L1574 # noqa: E501 + # https://cms-analysis-corrections.docs.cern.ch + # + + cfg.x.external_files = DotDict() + + # helper + def add_external(name, value): + if isinstance(value, dict): + value = DotDict.wrap(value) + cfg.x.external_files[name] = value + + # prepare run/era/nano meta data info to determine files in the CAT metadata structure + # see https://cms-analysis-corrections.docs.cern.ch + cat_info = { + (2022, "", 12): CATInfo( + run=3, + vnano=12, + era="22CDSep23-Summer22", + pog_directories={"dc": "Collisions22"}, + snapshot=CATSnapshot(btv="2025-08-20", dc="2025-07-25", egm="2025-12-15", jme="2026-04-13", lum="2024-01-31", muo="2026-04-28", tau="2025-12-25"), # noqa: E501 + ), + (2022, "EE", 12): CATInfo( + run=3, + vnano=12, + era="22EFGSep23-Summer22EE", + pog_directories={"dc": "Collisions22"}, + snapshot=CATSnapshot(btv="2025-08-20", dc="2025-07-25", egm="2025-12-15", jme="2026-04-13", lum="2024-01-31", muo="2026-04-28", tau="2025-12-25"), # noqa: E501 + ), + (2023, "", 12): CATInfo( + run=3, + vnano=12, + era="23CSep23-Summer23", + pog_directories={"dc": "Collisions23"}, + snapshot=CATSnapshot(btv="2025-08-20", dc="2025-07-25", egm="2025-12-15", jme="2026-04-13", lum="2024-01-31", muo="2026-04-28", tau="2025-12-25"), # noqa: E501 + ), + (2023, "BPix", 12): CATInfo( + run=3, + vnano=12, + era="23DSep23-Summer23BPix", + pog_directories={"dc": "Collisions23"}, + snapshot=CATSnapshot(btv="2025-08-20", dc="2025-07-25", egm="2025-12-15", jme="2026-04-13", lum="2024-01-31", muo="2026-04-28", tau="2025-12-25"), # noqa: E501 + ), + (2024, "", 15): CATInfo( + run=3, + vnano=15, + era="24CDEReprocessingFGHIPrompt-Summer24", + pog_directories={"dc": "Collisions24"}, + snapshot=CATSnapshot(btv="2026-03-10", dc="2025-07-25", egm="2025-12-15", jme="2025-12-02", muo="2026-04-28", lum="2026-04-15"), # noqa: E501 + ), + }[(year, campaign.x.postfix, vnano)] + cfg.x.cat_info = cat_info + + # common files + # (versions in the end are for hashing in cases where file contents changed but paths did not) + add_external("lumi", { + "golden": { + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=161#Year_2022 + 2022: (cat_info.get_file("dc", "Cert_Collisions2022_355100_362760_Golden.json"), "v1"), + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=161#Year_2023 + 2023: (cat_info.get_file("dc", "Cert_Collisions2023_366442_370790_Golden.json"), "v1"), + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=180#Year_2024 + # not yet available at CAT space + # 2024: (cat_info.get_file("dc", "Cert_Collisions2024_378981_386951_Golden.json"), "v1"), + 2024: ("https://cms-service-dqmdc.web.cern.ch/CAF/certification/Collisions24/Cert_Collisions2024_378981_386951_Golden.json", "v1"), # noqa: E501 + }[year], + "normtag": { + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=161#Year_2022 + 2022: ("/cvmfs/cms-bril.cern.ch/cms-lumi-pog/Normtags/normtag_BRIL.json", "v1"), + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=161#Year_2023 + 2023: ("/cvmfs/cms-bril.cern.ch/cms-lumi-pog/Normtags/normtag_BRIL.json", "v1"), + # https://twiki.cern.ch/twiki/bin/view/CMS/PdmVRun3Analysis?rev=180#Year_2024 + 2024: ("/cvmfs/cms-bril.cern.ch/cms-lumi-pog/Normtags/normtag_BRIL.json", "v1"), # TODO: correct? + }[year], + }) + + # pileup weight corrections + if year != 2024: # TODO: not yet available, see https://cms-analysis-corrections.docs.cern.ch + add_external("pu_sf", (cat_info.get_file("lum", "puWeights.json.gz"), "v1")) + elif year == 2024: + add_external("pu_sf", (cat_info.get_file("lum", "puWeights_BCDEFGHI.json.gz"), "v1")) + + # jet energy correction + add_external("jet_jerc", (cat_info.get_file("jme", "jet_jerc.json.gz"), "v1")) + + # fat jet energy correction + add_external("fat_jet_jerc", (cat_info.get_file("jme", "fatJet_jerc.json.gz"), "v1")) # noqa: E501 + + # jet veto map + add_external("jet_veto_map", (cat_info.get_file("jme", "jetvetomaps.json.gz"), "v1")) + + # btag scale factor + if year != 2024: + add_external("btag_sf_corr", (cat_info.get_file("btv", "btagging.json.gz"), "v1")) + else: + # SF stored in preliminary file for 2024 for now + # add_external("btag_sf_corr", (cat_info.get_file("btv", "btagging_preliminary.json.gz"), "v1")) # noqa: E501 + # use custom file with merged SF for both b/c and light jets + add_external("btag_wp_sf_corr", ("/data/dust/user/matthiej/mttbar/mtt/config/run3/btagging_preliminary_merged.json.gz", "v1")) # noqa: E501 + + # updated jet id + add_external("jet_id", (cat_info.get_file("jme", "jetid.json.gz"), "v1")) + + # muon scale factors + add_external("muon_sf", (cat_info.get_file("muo", "muon_HighPt.json.gz"), "v1")) + + # met phi correction + if year != 2024: # TODO: not yet available for 2024 + add_external("met_phi_corr", (cat_info.get_file("jme", f"met_xyCorrections_{year}_{year}{campaign.x.postfix}.json.gz"), "v1")) # noqa: E501 + + # electron scale factors + add_external("electron_sf", (cat_info.get_file("egm", "electron.json.gz"), "v1")) + # electron energy correction and smearing + add_external("electron_ss", (cat_info.get_file("egm", "electronSS_EtDependent.json.gz"), "v1")) # FIXME correct for us? # noqa: E501 + + # + # event reduction configuration + # + + # target file size after MergeReducedEvents in MB + cfg.x.reduced_file_size = 512.0 + + # columns to keep after certain steps + cfg.x.keep_columns = DotDict.wrap({ + "cf.ReduceEvents": { + # + # NanoAOD columns + # + + # general event info + "run", "luminosityBlock", "event", + + # weights + "genWeight", + "LHEWeight.*", + "LHEPdfWeight", "LHEScaleWeight", + "PSWeight", + + # muons + "Muon.pt", "Muon.eta", "Muon.phi", "Muon.mass", "Muon.tunepRelPt", + "Muon.pdgId", + "Muon.jetIdx", + "Muon.nStations", + "Muon.pfRelIso03_all", "Muon.pfRelIso04_all", "Muon.tkRelIso", + + # electrons + "Electron.pt", "Electron.eta", "Electron.phi", "Electron.mass", + "Electron.pdgId", + "Electron.jetIdx", + "Electron.deltaEtaSC", + "Electron.pfRelIso03_all", + + # photons (for L1 prefiring) + "Photon.pt", "Photon.eta", "Photon.phi", "Photon.mass", + "Photon.jetIdx", + + # AK4 jets + "Jet.pt", "Jet.eta", "Jet.phi", "Jet.mass", + "Jet.rawFactor", + "Jet.btagDeepFlavB", "Jet.hadronFlavour", + # # optional, enable if needed + # "Jet.area", + # "Jet.hadronFlavour", "Jet.partonFlavour", + # "Jet.jetId", "Jet.puId", "Jet.puIdDisc", + # # cleaning + # "Jet.cleanmask", + # "Jet.muonSubtrFactor", + # # indices to other collections + # "Jet.electronIdx*", + # "Jet.muonIdx*", + # "Jet.genJetIdx*", + # # number of jet constituents + # "Jet.nConstituents", + # "Jet.nElectrons", + # "Jet.nMuons", + # # PF energy fractions + # "Jet.chEmEF", + # "Jet.chHEF", + # "Jet.neEmEF", + # "Jet.neHEF", + # "Jet.muEF", + # # taggers + # "Jet.qgl", + # "Jet.btag*", + + # AK8 jets + "FatJet.pt", "FatJet.eta", "FatJet.phi", "FatJet.mass", "FatJet.msoftdrop", + "FatJet.rawFactor", + "FatJet.tau1", "FatJet.tau2", "FatJet.tau3", "FatJet.tau4", + "FatJet.subJetIdx1", "FatJet.subJetIdx2", + "FatJet.globalParT3_withMassTopvsQCD", "FatJet.globalParT3_withMassWvsQCD", + # # optional, enable if needed + # "FatJet.area", "FatJet.jetId", "FatJet.hadronFlavour", + # "FatJet.genJetAK8Idx", + # "FatJet.muonIdx3SJ", "FatJet.electronIdx3SJ", + # "FatJet.nBHadrons", "FatJet.nCHadrons", + # # taggers + # "FatJet.btag*", "FatJet.deepTag*", "FatJet.particleNet*", + + # subjets + "SubJet.btagDeepB", "SubJet.btagUParTAK4B" + + # generator quantities + "Generator.*", + + # generator particles + # "GenPart.pt", "GenPart.eta", "GenPart.phi", "GenPart.mass", + # "GenPart.pdgId", + # "GenPart.*", + + # number of primary vertices + "PV.npvs", + + # average number of pileup interactions + "Pileup.nTrueInt", + + # + # columns added during selection + # + + # generator particle info + "GenPartonTop.*", + + # columns for PlotCutflowVariables + "cutflow.*", + + # other columns, required by various tasks + "channel_id", "category_ids", "process_id", + "deterministic_seed", + "mc_weight", + "pu_weight*", + "FSR*", "ISR*", "pdf_weight*", + "muf_weight*", "mur_weight*", "murmuf_weight*", "murmuf_envelope*", + }, + "cf.MergeSelectionMasks": { + "channel_id", "process_id", "category_ids", + "normalization_weight", + "cutflow.*", + "mc_weight", + }, + "cf.UniteColumns": { + "*", + }, + }) + + # + # event weights + # + + # event weight columns as keys in an OrderedDict, mapped to shift instances they depend on + get_shifts = functools.partial(get_shifts_from_sources, cfg) + cfg.x.event_weights = DotDict({ + "normalization_weight": [], + "normalized_pu_weight": get_shifts("minbias_xs"), + }) + + # event weights only present in certain datasets or configs + for dataset in cfg.datasets: + dataset.x.event_weights = DotDict() + if dataset.has_tag("is_ttbar"): + # top pt reweighting + dataset.x.event_weights["top_pt_weight"] = get_shifts("top_pt") + if dataset.has_tag("is_v_jets"): + # V+jets QCD NLO reweighting + dataset.x.event_weights["vjets_weight"] = get_shifts("vjets") + # add PSWeight variations for all datasets but qcd + # if not dataset.has_tag("is_qcd"): + # dataset.x.event_weights["ISR"] = get_shifts("ISR") + # dataset.x.event_weights["FSR"] = get_shifts("FSR") + + # # + # # versions + # # + # cfg.x.versions = { + # "tt_*": "test_v7", + # } + + # # named references to actual versions to use for certain sets of tasks + # main_ver = "v1" + # cfg.x.named_versions = DotDict.wrap({ + # "default": f"{main_ver}", + # "calibrate": f"{main_ver}", + # "select": f"{main_ver}", + # "reduce": f"{main_ver}", + # "merge": f"{main_ver}", + # "produce": f"{main_ver}", + # "hist": f"{main_ver}", + # "plot": f"{main_ver}", + # "datacards": f"{main_ver}", + # }) + + # # versions per task family and optionally also dataset and shift + # # None can be used as a key to define a default value + # cfg.x.versions = { + # None: cfg.x.named_versions["default"], + # # CSR tasks + # "cf.CalibrateEvents": cfg.x.named_versions["calibrate"], + # "cf.SelectEvents": cfg.x.named_versions["select"], + # "cf.ReduceEvents": cfg.x.named_versions["reduce"], + # # merging tasks + # "cf.MergeSelectionStats": cfg.x.named_versions["merge"], + # "cf.MergeSelectionMasks": cfg.x.named_versions["merge"], + # "cf.MergeReducedEvents": cfg.x.named_versions["merge"], + # "cf.MergeReductionStats": cfg.x.named_versions["merge"], + # # column production + # "cf.ProduceColumns": cfg.x.named_versions["produce"], + # # histogramming + # "cf.CreateCutflowHistograms": cfg.x.named_versions["hist"], + # "cf.CreateHistograms": cfg.x.named_versions["hist"], + # "cf.MergeHistograms": cfg.x.named_versions["hist"], + # "cf.MergeShiftedHistograms": cfg.x.named_versions["hist"], + # # plotting + # "cf.PlotVariables1D": cfg.x.named_versions["plot"], + # "cf.PlotVariables2D": cfg.x.named_versions["plot"], + # "cf.PlotVariablesPerProcess2D": cfg.x.named_versions["plot"], + # "cf.PlotShiftedVariables1D": cfg.x.named_versions["plot"], + # "cf.PlotShiftedVariablesPerProcess1D": cfg.x.named_versions["plot"], + # # + # "cf.PlotCutflow": cfg.x.named_versions["plot"], + # "cf.PlotCutflowVariables1D": cfg.x.named_versions["plot"], + # "cf.PlotCutflowVariables2D": cfg.x.named_versions["plot"], + # "cf.PlotCutflowVariablesPerProcess2D": cfg.x.named_versions["plot"], + # # datacards + # "cf.CreateDatacards": cfg.x.named_versions["datacards"], + # } + + # + # finalization + # + + # add categories + add_categories(cfg) + + # add variables + add_variables(cfg) + + return cfg diff --git a/topsf/config/taggers.py b/topsf/config/taggers.py new file mode 100644 index 0000000..da18c64 --- /dev/null +++ b/topsf/config/taggers.py @@ -0,0 +1,182 @@ +# coding: utf-8 + +""" +Stores the taggers information for the top-tagging scale factor analysis. +""" +from __future__ import annotations +import law + +from columnflow.util import DotDict + +logger = law.logger.get_logger(__name__) + + +def btag_wps( + config, + full=True, +) -> DotDict: + # b-tag working points + # https://btv-wiki.docs.cern.ch/ScaleFactors/Run3Summer22/ + # https://btv-wiki.docs.cern.ch/ScaleFactors/Run3Summer22EE/ + # TODO: use PNet? -> not available for SubJet tagging, only DeepCSV for v12, and UnifiedParT for v15 + btag_key = config.x.cpn_tag + btag_working_points = DotDict.wrap({ + "deepjet": { + "loose": { + "2022preEE": 0.0583, "2022postEE": 0.0614, "2023preBPix": 0.0479, "2023postBPix": 0.048, "2024": -10.0, + }[btag_key], + "medium": { + "2022preEE": 0.3086, "2022postEE": 0.3196, "2023preBPix": 0.2431, "2023postBPix": 0.2435, "2024": -10.0, + }[btag_key], + "tight": { + "2022preEE": 0.7183, "2022postEE": 0.7300, "2023preBPix": 0.6553, "2023postBPix": 0.6563, "2024": -10.0, + }[btag_key], + }, + "deepcsv": { + "loose": { + "2022preEE": 0.1208, "2022postEE": 0.1208, "2023preBPix": 0.1208, "2023postBPix": 0.1208, "2024": -10.0, + }[btag_key], + "medium": { + "2022preEE": 0.4168, "2022postEE": 0.4168, "2023preBPix": 0.4168, "2023postBPix": 0.4168, "2024": -10.0, + }[btag_key], + "tight": { + "2022preEE": 0.7665, "2022postEE": 0.7665, "2023preBPix": 0.7665, "2023postBPix": 0.7665, "2024": -10.0, + }[btag_key], + }, + "btagUParTAK4B": { + "loose": { + "2022preEE": -10.0, "2022postEE": -10.0, "2023preBPix": -10.0, "2023postBPix": -10.0, "2024": 0.0246 + }[btag_key], + "medium": { + "2022preEE": -10.0, "2022postEE": -10.0, "2023preBPix": -10.0, "2023postBPix": -10.0, "2024": 0.1272 + }[btag_key], + "tight": { + "2022preEE": -10.0, "2022postEE": -10.0, "2023preBPix": -10.0, "2023postBPix": -10.0, "2024": 0.4648 + }[btag_key], + "xtight": { + "2022preEE": -10.0, "2022postEE": -10.0, "2023preBPix": -10.0, "2023postBPix": -10.0, "2024": 0.6298 + }[btag_key], + "xxtight": { + "2022preEE": -10.0, "2022postEE": -10.0, "2023preBPix": -10.0, "2023postBPix": -10.0, "2024": 0.9739 + }[btag_key], + }, + }) + # store upart wp different for fixed wp sf producer + btagUParTAK4B__fixed_wp = DotDict.wrap({ + "loose": 0.0246, + "medium": 0.1272, + "tight": 0.4648, + "xtight": 0.6298, + # "xxtight": 0.9739, + }) + result = btag_working_points if full else btagUParTAK4B__fixed_wp + + return result + + +def toptag_wps(era) -> DotDict: + # top-tag working points + toptag_working_points = DotDict.wrap({ + "tau32_run2": { + # stored here for reference + # https://twiki.cern.ch/twiki/bin/view/CMS/JetTopTagging?rev=41 + "very_loose": 0.69, + "loose": 0.61, + "medium": 0.52, + "tight": 0.47, + "very_tight": 0.38, + }, + "tau32_v7": { + # v7, 2223 values + # with mass constraint + "very_loose": 0.761, + "loose": 0.680, + "medium": 0.579, + "tight": 0.514, + "very_tight": 0.395, + }, + "tau32_v8_run3": { + # with mass constraint + "very_loose": 0.73, + "loose": 0.63, + "medium": 0.53, + "tight": 0.47, + "very_tight": 0.36, + }, + "tau32_v11_run3": { + # with mass constraint + # [0.36944236, 0.47045506, 0.52725393, 0.62049409, 0.71313679] + "very_loose": 0.71, + "loose": 0.62, + "medium": 0.53, + "tight": 0.47, + "very_tight": 0.37, + } + }) + # era-specific working points (v8) + toptag_working_points_eras = DotDict.wrap({ + "2022preEE": { + "tau32": { + # 0.36660745, 0.47563951, 0.53662783, 0.63603016, 0.73238039 + "very_loose": 0.73, + "loose": 0.64, + "medium": 0.54, + "tight": 0.48, + "very_tight": 0.37, + }, + }, + "2022postEE": { + "tau32": { + # 0.36733566, 0.47297406, 0.53398852, 0.63312435, 0.72942649 + "very_loose": 0.73, + "loose": 0.63, + "medium": 0.53, + "tight": 0.47, + "very_tight": 0.37, + }, + }, + "2023preBPix": { + "tau32": { + # 0.36037594, 0.46878631, 0.5314992, 0.63098395, 0.72912963 + "very_loose": 0.73, + "loose": 0.63, + "medium": 0.53, + "tight": 0.47, + "very_tight": 0.36, + }, + }, + "2023postBPix": { + "tau32": { + # 0.35929614, 0.47072125, 0.53208042, 0.63179965, 0.72908621 + "very_loose": 0.73, + "loose": 0.63, + "medium": 0.53, + "tight": 0.47, + "very_tight": 0.36, + }, + }, + "2024": { + "tau32": { + # 0.37950085, 0.49627566, 0.55961938, 0.66063325, 0.75729332 + "very_loose": 0.76, + "loose": 0.66, + "medium": 0.56, + "tight": 0.50, + "very_tight": 0.38, + }, + }, + "222324": { + "tau32": { + # 0.36372085, 0.47182575, 0.53341095, 0.63278459, 0.72981009 + "very_loose": 0.73, + "loose": 0.63, + "medium": 0.53, + "tight": 0.47, + "very_tight": 0.36, + }, + }, + }) + # logger.warning_once("Reminder: As of v8, the topwp values deviate in between the eras. Consider using era-specific WPs if we want to be precise. -> update numbers when rehistogramming!") + # return toptag_working_points_eras[era] + logger.info_once("Reminder: As of v11, the topwp values are the same between the eras. Using the same WPs again for all Eras.") + return toptag_working_points["tau32_v11_run3"] diff --git a/topsf/config/variables.py b/topsf/config/variables.py index e2f7f7c..c2eb348 100644 --- a/topsf/config/variables.py +++ b/topsf/config/variables.py @@ -46,6 +46,24 @@ def add_variables(config: od.Config) -> None: binning=(200, -10, 10), x_title="MC weight", ) + config.add_variable( + name="weight_unweighted", # used in histProducer specificially accessing this name to set weight=1 + expression="weight", + binning=(100, -10, 10), + x_title="Event weight per MC event", + ) + config.add_variable( + name="weight_unweighted1", # used in histProducer specificially accessing this name to set weight=1 + expression="weight", + binning=(100, -1, 1), + x_title="Event weight per MC event", + ) + config.add_variable( + name="weight_unweighted2", # used in histProducer specificially accessing this name to set weight=1 + expression="weight", + binning=(100, -0.1, 0.1), + x_title="Event weight per MC event", + ) # Event properties config.add_variable( @@ -103,11 +121,12 @@ def add_variables(config: od.Config) -> None: aux={ "inputs": {"FatJet.tau3", "FatJet.tau2"}, "short_label": "$\tau_{3}/\tau_{2}$", + "signal_side": "left", }, ) config.add_variable( name="fatjet_tau32_fine", - expression=lambda events: events.FatJet.tau3 / events.FatJet.tau2, + expression=lambda events: events["FatJet"]["tau3"] / events["FatJet"]["tau2"], null_value=EMPTY_FLOAT, binning=(500, 0, 1), x_title=r"$\tau_{3}/\tau_{2}$ upper limit", @@ -127,6 +146,17 @@ def add_variables(config: od.Config) -> None: "short_label": "$m_{SD}$", }, ) + config.add_variable( + name="fatjet_msoftdrop", + expression="FatJet.msoftdrop", + null_value=EMPTY_FLOAT, + binning=(100, 0, 500), + x_title=r"AK8 jet $m_{SD}$", + unit="GeV", + aux={ + "short_label": "$m_{SD}$", + }, + ) config.add_variable( name="fatjet_is_unique_top_matched", expression="FatJet.is_unique_top_matched", @@ -134,6 +164,16 @@ def add_variables(config: od.Config) -> None: binning=[-0.5, 0.5, 1.5], x_title=r"AK8 jet has unique top quark match", ) + config.add_variable( + name="fatjet_globalParT3_withMassTopvsQCD", + expression="FatJet.globalParT3_withMassTopvsQCD", + null_value=EMPTY_FLOAT, + binning=(50, 0, 1), + x_title=r"AK8 jet $GlobalParT3$ with Mass Top vs QCD", + aux={ + "signal_side": "right", + } + ) # pt-leading jets for i in range(4): @@ -216,6 +256,21 @@ def add_variables(config: od.Config) -> None: binning=(40, -3.2, 3.2), x_title=r"MET $\phi$", ) + config.add_variable( + name="puppimet_pt", + expression="PuppiMET.pt[:,0]", + null_value=EMPTY_FLOAT, + binning=(40, 0., 400.), + unit="GeV", + x_title=r"PuppiMET $p_{T}$", + ) + config.add_variable( + name="puppimet_phi", + expression="PuppiMET.phi[:,0]", + null_value=EMPTY_FLOAT, + binning=(40, -3.2, 3.2), + x_title=r"PuppiMET $\phi$", + ) # probe jet properties config.add_variable( @@ -324,6 +379,35 @@ def add_variables(config: od.Config) -> None: x_title=r"Probe jet $m_{SD}$", unit="GeV", ) + config.add_variable( + name="probejet_msoftdrop_inf_rebin_fix", + expression="ProbeJet.msoftdrop", + null_value=EMPTY_FLOAT, + binning=[ + 50, 70, 85, + 105, 120, 140, + 155, 170, 185, + 200, 210, 220, + 230, 250, 500, + ], + x_title=r"Probe jet $m_{SD}$", + unit="GeV", + ) + config.add_variable( + name="probejet_msoftdrop_inf_rebin_highmass", + expression="ProbeJet.msoftdrop", + null_value=EMPTY_FLOAT, + binning=[ + # 50, 70, 85, + # 105, 120, 140, + 100, 120, 140, + 155, 170, 185, + 200, 210, 220, + 230, 250, 500, + ], + x_title=r"Probe jet $m_{SD}$", + unit="GeV", + ) config.add_variable( name="probejet_tau3", expression="ProbeJet.tau3", @@ -340,7 +424,7 @@ def add_variables(config: od.Config) -> None: ) config.add_variable( name="probejet_tau32", - expression=lambda events: events.ProbeJet.tau3 / events.ProbeJet.tau2, + expression=lambda events: events.ProbeJet["tau3"] / events.ProbeJet["tau2"], null_value=EMPTY_FLOAT, binning=(50, 0, 1), x_title=r"Probe jet $\tau_{3}/\tau_{2}$", @@ -400,3 +484,20 @@ def add_variables(config: od.Config) -> None: "inputs": {"Jet.pt"}, }, ) + + # Jet MET features + for i in range(3): + config.add_variable( + name=f"jet{i+1}_met_delta_phi", + expression=f"Jet_{i}_MET_delta_phi", + null_value=EMPTY_FLOAT, + binning=(40, 0, 3.2), + x_title=rf"$\Delta \phi$(jet {i+1}, MET)", + ) + config.add_variable( + name=f"fatjet{i+1}_met_delta_phi", + expression=f"FatJet_{i}_MET_delta_phi", + null_value=EMPTY_FLOAT, + binning=(40, 0, 3.2), + x_title=rf"$\Delta \phi$(AK8 jet {i+1}, MET)", + ) diff --git a/topsf/inference/uhh2.py b/topsf/inference/uhh2.py index f1acccf..e8deb02 100644 --- a/topsf/inference/uhh2.py +++ b/topsf/inference/uhh2.py @@ -108,7 +108,7 @@ def uhh2(self): for config_inst in self.config_insts }, mc_stats="0 1 1", # FIXME: make configurable - flow_strategy="move", + flow_strategy="ignore", empty_bin_value=0.0, # NOTE: remove this when removing custom rebin task ) @@ -230,7 +230,7 @@ def uhh2(self): for proc in processes: for unc in uncertainty_shifts: - if proc == "mj" and unc in ["FSR", "ISR"]: + if proc == "mj" and unc in ["fsr", "isr"]: continue if unc in ["mur", "muf"] and not (proc.startswith("st") or proc.startswith("tt")): continue diff --git a/topsf/plotting/plot_all.py b/topsf/plotting/plot_all.py index 5583a83..4fb6d15 100644 --- a/topsf/plotting/plot_all.py +++ b/topsf/plotting/plot_all.py @@ -6,6 +6,8 @@ from __future__ import annotations +import law + from columnflow.util import DotDict, maybe_import from columnflow.plotting.plot_util import get_position from columnflow.plotting.plot_all import ( @@ -23,6 +25,8 @@ mplhep = maybe_import("mplhep") od = maybe_import("order") +logger = law.logger.get_logger(__name__) + def draw_efficiency( ax: plt.Axes, @@ -31,6 +35,7 @@ def draw_efficiency( norm: float = 1.0, data: dict | None = None, plot_mode: str | None = "roc", + signal_side: str | None = None, **kwargs, ) -> None: @@ -46,13 +51,13 @@ def draw_efficiency( for key, h in hists.items(): totals_[key] = totals.get(key, h.sum(flow=True).value) - passes[key] = np.array([ - h[:hist.loc(v)].sum().value - for v in values - ]) - pass_fractions[key] = ( - passes[key] / totals_[key] - ) + if signal_side == "left": + passes[key] = np.array([h[:hist.loc(v)].sum().value for v in values]) + elif signal_side == "right": + passes[key] = np.array([h[hist.loc(v):].sum().value for v in values]) + else: + raise ValueError(f"invalid signal_side: {signal_side}") + pass_fractions[key] = passes[key] / totals_[key] plot_kwargs = { # "linewidth": 1, @@ -83,11 +88,36 @@ def draw_efficiency( # mask values that are not strictly monotonically increasing # (strict monotonicity required by interpolation) - mask = (pass_fraction_background[1:] / pass_fraction_background[:-1]) > 1.001 + # mask = (pass_fraction_background[1:] / pass_fraction_background[:-1]) > 1.001 + if signal_side == "right": + print("signal side is right, masking non-monotonic points by looking at ratios of neighboring pass fractions < 1/1.001") + if plot_mode == "background": + mask = (pass_fraction_background[1:] / pass_fraction_background[:-1]) < (1 / 1.001) + elif plot_mode == "signal": + mask = (pass_fraction_signal[1:] / pass_fraction_signal[:-1]) < (1 / 1.001) + elif plot_mode == "roc": + mask = (pass_fraction_background[1:] / pass_fraction_background[:-1]) < (1 / 1.001) + else: + if plot_mode == "background": + mask = (pass_fraction_background[1:] / pass_fraction_background[:-1]) > 1.001 + elif plot_mode == "signal": + mask = (pass_fraction_signal[1:] / pass_fraction_signal[:-1]) > 1.001 + elif plot_mode == "roc": + mask = (pass_fraction_background[1:] / pass_fraction_background[:-1]) > 1.001 mask = np.array([True] + list(mask)) pass_fraction_background = pass_fraction_background[mask] pass_fraction_signal = pass_fraction_signal[mask] discriminator_values = np.array(data["discriminator_values"])[mask] + if plot_mode == "background": + idx = np.argsort(pass_fraction_background) + elif plot_mode == "signal": + idx = np.argsort(pass_fraction_signal) + elif plot_mode == "roc": + idx = np.argsort(pass_fraction_signal) + + pass_fraction_background = pass_fraction_background[idx] + pass_fraction_signal = pass_fraction_signal[idx] + discriminator_values = discriminator_values[idx] # efficiency values for which to derive cut on discriminating variable pass_fraction_background_wp = np.array( @@ -172,6 +202,9 @@ def draw_efficiency( color="r", fontsize=16, ) + logger.info(f"WP discriminator values: {np.array(discriminator_values_wp)}") + logger.info(f"WP signal efficiencies: {np.array(pass_fraction_signal_wp)}") + logger.info(f"WP background efficiencies: {np.array(pass_fraction_background_wp)}") return artists diff --git a/topsf/plotting/plot_roc_curve.py b/topsf/plotting/plot_roc_curve.py index 0afe152..71feb63 100644 --- a/topsf/plotting/plot_roc_curve.py +++ b/topsf/plotting/plot_roc_curve.py @@ -23,9 +23,12 @@ mplhep = maybe_import("mplhep") od = maybe_import("order") +logger = law.logger.get_logger(__name__) + def plot_roc_curve( hists: OrderedDict, + totals: dict, config_inst: od.Config, category_inst: od.Category, variable_inst: od.Variable, @@ -33,6 +36,7 @@ def plot_roc_curve( # hide_errors: bool | None = None, # variable_settings: dict | None = None, binning_variable_labels: list | None = None, + signal_side: str | None = None, **kwargs, ) -> plt.Figure: """ @@ -48,8 +52,8 @@ def plot_roc_curve( if "signal" not in hists or "background" not in hists: hists_keys_str = ", ".join(hists) - print( - f"WARNING: `hists` should contain the keys 'signal' and 'background', got: {hists_keys_str}", + logger.warning( + f"`hists` should contain the keys 'signal' and 'background', got: {hists_keys_str}", ) # plot config with a single entry for drawing the ROC curve @@ -59,6 +63,8 @@ def plot_roc_curve( "hist": hists, "kwargs": { "plot_mode": "roc", + "totals": totals, + "signal_side": signal_side, }, }, } @@ -114,6 +120,7 @@ def plot_efficiency( # hide_errors: bool | None = None, # variable_settings: dict | None = None, binning_variable_labels: list | None = None, + signal_side: str | None = None, **kwargs, ) -> plt.Figure: """ @@ -133,13 +140,13 @@ def plot_efficiency( "background" not in hists ): hists_keys_str = ", ".join(hists) - print( - f"WARNING: `hists` should contain the keys 'signal' and 'background', got: {hists_keys_str}", + logger.warning( + f"`hists` should contain the keys 'signal' and 'background', got: {hists_keys_str}", ) elif plot_mode not in hists: hists_keys_str = ", ".join(hists) - print( - f"WARNING: `hists` should contain the key '{plot_mode}', got: {hists_keys_str}", + logger.warning( + f"`hists` should contain the key '{plot_mode}', got: {hists_keys_str}", ) # plot config with a single entry for drawing the ROC curve @@ -150,6 +157,7 @@ def plot_efficiency( "kwargs": { "totals": totals, "plot_mode": plot_mode, + "signal_side": signal_side, }, }, } diff --git a/topsf/production/features.py b/topsf/production/features.py index bd641a5..3d2c1d7 100644 --- a/topsf/production/features.py +++ b/topsf/production/features.py @@ -8,8 +8,10 @@ from columnflow.util import maybe_import from columnflow.columnar_util import set_ak_column from columnflow.production.util import attach_coffea_behavior +from topsf.util import has_tag ak = maybe_import("awkward") +np = maybe_import("numpy") coffea = maybe_import("coffea") maybe_import("coffea.nanoevents.methods.nanoaod") @@ -38,16 +40,19 @@ def jet_energy_shifts_init(self: Producer) -> None: uses={ attach_coffea_behavior, "event", - "Jet.pt", - "FatJet.pt", + "Jet.pt", "Jet.eta", "Jet.phi", "Jet.mass", + # "BJet.pt", "BJet.eta", "BJet.phi", "BJet.mass", + "FatJet.pt", "FatJet.eta", "FatJet.phi", "FatJet.mass", "Muon.pt", "Electron.pt", + "MET.phi", "MET.pt", }, produces={ attach_coffea_behavior, "dummy", "n_jet", "n_fatjet", + # "n_bjet", "n_muon", "n_electron", }, @@ -62,13 +67,121 @@ def features(self: Producer, events: ak.Array, **kwargs) -> ak.Array: events = set_ak_column(events, "dummy", ak.ones_like(events.event)) # count jets and fatjets - jet = ak.without_parameters(events["Jet"]) + jet = ak.with_name(events.Jet, "Jet") fatjet = ak.without_parameters(events["FatJet"]) + # bjet = ak.without_parameters(events["BJet"]) muon = ak.without_parameters(events["Muon"]) electron = ak.without_parameters(events["Electron"]) events = set_ak_column(events, "n_jet", ak.num(jet.pt, axis=-1)) events = set_ak_column(events, "n_fatjet", ak.num(fatjet.pt, axis=-1)) + # events = set_ak_column(events, "n_bjet", ak.num(bjet.pt, axis=-1)) events = set_ak_column(events, "n_muon", ak.num(muon.pt, axis=-1)) events = set_ak_column(events, "n_electron", ak.num(electron.pt, axis=-1)) + jet = events.Jet[ak.argsort(events.Jet.pt, axis=1, ascending=False)] + # bjet = events.BJet[ak.argsort(events.BJet.pt, axis=1, ascending=False)] + fatjet = events.FatJet[ak.argsort(events.FatJet.pt, axis=1, ascending=False)] + jet_phi_padded = ak.pad_none(jet.phi, 3, axis=1, clip=True) + fatjet_phi_padded = ak.pad_none(fatjet.phi, 3, axis=1, clip=True) + # bjet_phi_padded = ak.pad_none(bjet.phi, 3, axis=1, clip=True) + + for i in range(3): + dphi_jet_met = np.abs(jet_phi_padded[:, i] - events.MET.phi) + dphi_jet_met = ak.where(dphi_jet_met > np.pi, 2 * np.pi - dphi_jet_met, dphi_jet_met) + events = set_ak_column(events, f"Jet_{i}_MET_delta_phi", dphi_jet_met) + + for i in range(3): + dphi_fatjet_met = np.abs(fatjet_phi_padded[:, i] - events.MET.phi) + dphi_fatjet_met = ak.where(dphi_fatjet_met > np.pi, 2 * np.pi - dphi_fatjet_met, dphi_fatjet_met) + events = set_ak_column(events, f"FatJet_{i}_MET_delta_phi", dphi_fatjet_met) + + # for i in range(3): + # dphi_bjet_met = np.abs(bjet_phi_padded[:, i] - events.MET.phi) + # dphi_bjet_met = ak.where(dphi_bjet_met > np.pi, 2 * np.pi - dphi_bjet_met, dphi_bjet_met) + # events = set_ak_column(events, f"BJet_{i}_MET_delta_phi", dphi_bjet_met) + + # btag score for AK4 jets + # get btagging working points for the given column from config + if has_tag("skip_btag_weights", self.config_inst): + print("Skipping btag weight features as 'skip_btag_weights' tag is set in config.") + btag_col = self.config_inst.x.jet_selection.ak4.btag_column + wp_dict = self.config_inst.x.btag_working_points[btag_col].fixed_wp + edges = sorted(wp_dict.values()) + + scores = jet[btag_col] + + # count how many WPs are passed + buckets = sum(scores >= edge for edge in edges) + + jet = ak.with_field(jet, buckets, f"{btag_col}_buckets") + events = set_ak_column(events, f"Jet.{btag_col}_buckets", buckets) + + # store btag pass/fail decision as boolean for each WP as well + # 1: pass, 0: fail, -1: undefined (e.g. no jet or no score) + for wp, edge in wp_dict.items(): + pass_fail = ak.where(scores >= edge, 1, 0) + pass_fail = ak.where(ak.is_none(scores), -1, pass_fail) + jet = ak.with_field(jet, pass_fail, f"{btag_col}_pass_{wp}") + + # set some default value for undefined btag scores, and the number of jets to store the features + DEFAULT_VAL = -10.0 + N_JETS = 5 + + # pad per-jet arrays to N_JETS so indexing per jet is safe even if events have fewer jets + padded_scores = ak.pad_none(ak.nan_to_none(scores), N_JETS, axis=1) + padded_buckets = ak.pad_none(buckets, N_JETS, axis=1) + + # precompute pass/fail arrays for each WP and pad them + padded_pass_fail = {} + for wp, edge in wp_dict.items(): + print(f"Computing pass/fail for WP {wp} with edge {edge}") + pf_int = ak.where(scores >= edge, 1, 0) + padded_pass_fail[wp] = ak.pad_none(pf_int, N_JETS, axis=1) + print(f" Pass/fail for WP {wp}: {pf_int}") + + for i in range(0, N_JETS): + # take the i-th jet across events (padded with None where missing) and fill defaults + score_i = padded_scores[:, i] + score_i = ak.fill_none(score_i, DEFAULT_VAL) + events = set_ak_column(events, f"Jet_{i}_{btag_col}", score_i) + + buckets_i = padded_buckets[:, i] + buckets_i = ak.fill_none(buckets_i, 0) + events = set_ak_column(events, f"Jet_{i}_{btag_col}_buckets", buckets_i) + + for wp in wp_dict.keys(): + pf_i = padded_pass_fail[wp][:, i] + # convert missing -> -1 (undefined), keep 0/1 otherwise + pf_i = ak.fill_none(pf_i, -1) + events = set_ak_column(events, f"Jet_{i}_{btag_col}_pass_{wp}", pf_i) + return events + + +@features.init +def features_init(self: Producer) -> None: + self.produces |= { + f"Jet_{i}_MET_delta_phi" for i in range(3) + } | { + f"FatJet_{i}_MET_delta_phi" for i in range(3) + } + # } | { + # f"BJet_{i}_MET_delta_phi" for i in range(3) + # } + if has_tag("skip_btag_weights", self.config_inst): + btag_col = self.config_inst.x.jet_selection.ak4.btag_column + fixed_wps = self.config_inst.x.btag_working_points[btag_col].fixed_wp.keys() + self.produces |= { + f"Jet_{i}_{btag_col}" for i in range(5) + } | { + f"Jet_{i}_{btag_col}_buckets" for i in range(5) + } | { + f"Jet_{i}_{btag_col}_pass_{wp}" for i in range(5) for wp in fixed_wps + } | { + f"Jet.{btag_col}_buckets" + } | { + f"Jet.{btag_col}" + } + self.uses |= { + f"Jet.{btag_col}", + } diff --git a/topsf/production/filter.py b/topsf/production/filter.py new file mode 100644 index 0000000..d294a6a --- /dev/null +++ b/topsf/production/filter.py @@ -0,0 +1,82 @@ +# coding: utf-8 + +""" +Production modules related to electrons. +""" + +from __future__ import annotations + +from functools import partial + + +import law +from columnflow.util import maybe_import + +from columnflow.production import Producer, producer +from columnflow.columnar_util import set_ak_column + +np = maybe_import("numpy") +ak = maybe_import("awkward") + + +set_ak_bool = partial(set_ak_column, value_type=np.bool_) + + +logger = law.logger.get_logger(__name__) + + +@producer( + uses={"run", "PuppiMET.{pt,phi}", "Jet.{pt,eta,phi,mass,neEmEF,chEmEF}"}, + produces={"patchedEcalBadCalibFilter"}, + data_only=True, +) +def ECALBadCalibrationFilter( + self: Producer, events: ak.Array, **kwargs, +) -> ak.Array: + """ + Producer that applies the ECAL bad calibration filter. + + At the time of writing this, the filter only needs to be applied to 2022 Era F+G + since these datasets are still Prompt data. + + Resources: + - https://twiki.cern.ch/twiki/bin/viewauth/CMS/MissingETOptionalFiltersRun2?rev=167#ECal_BadCalibration_Filter_Flag + - https://cms-talk.web.cern.ch/t/noise-met-filters-in-run-3/63346/5 + + """ + set(events.run) + logger.debug(f"{events.run}") + jet_mask = ( + (events.Jet.pt > 50) & + (events.Jet.eta > -0.5) & + (events.Jet.eta < 0.1) & + (events.Jet.phi > -2.1) & + (events.Jet.phi < -1.8) & + (events.Jet.neEmEF > 0.9) & + (events.Jet.chEmEF > 0.9) & + (events.Jet.delta_phi(events.PuppiMET) > 2.9) + ) + reject = ( + (events.run >= 362433) & + (events.run <= 367144) & + (events.PuppiMET.pt > 100) & + (ak.any(jet_mask, axis=1)) + ) + ecal_bad_calibration_mask = ~reject + logger.info(f"{ak.sum(reject)} / {len(events)} events affected by ECAL bad calibration filter") + + events = set_ak_bool(events, "patchedEcalBadCalibFilter", ecal_bad_calibration_mask) + + return events + + +@ECALBadCalibrationFilter.skip +def ECALBadCalibrationFilter_skip(self: Producer) -> None: + if ( + self.config_inst.campaign.x.year == 2022 and + self.dataset_inst.is_data and + self.dataset_inst.x.era in "FG" + ): + # only run on prompt data + return False + return True \ No newline at end of file diff --git a/topsf/production/gen_v.py b/topsf/production/gen_v.py index b6210d9..c67b08b 100644 --- a/topsf/production/gen_v.py +++ b/topsf/production/gen_v.py @@ -193,11 +193,11 @@ def vjets_weight_skip(self: Producer) -> bool: ) -@vjets_weight.init -def vjets_weight_init(self: Producer) -> None: - shift_inst = getattr(self, "local_shift_inst", None) - if not shift_inst: - return +# @vjets_weight.init +# def vjets_weight_init(self: Producer) -> None: +# shift_inst = getattr(self, "local_shift_inst", None) +# if not shift_inst: +# return @vjets_weight.requires diff --git a/topsf/production/l1_prefiring.py b/topsf/production/l1_prefiring.py index 65292b9..335c87c 100644 --- a/topsf/production/l1_prefiring.py +++ b/topsf/production/l1_prefiring.py @@ -150,11 +150,11 @@ def get_eff(obj_name, key, obj): return events -@l1_prefiring_weights.init -def l1_prefiring_weights_init(self: Producer) -> None: - shift_inst = getattr(self, "local_shift_inst", None) - if not shift_inst: - return +# @l1_prefiring_weights.init +# def l1_prefiring_weights_init(self: Producer) -> None: +# shift_inst = getattr(self, "local_shift_inst", None) +# if not shift_inst: +# return @l1_prefiring_weights.requires diff --git a/topsf/production/normalized_weights.py b/topsf/production/normalized_weights.py new file mode 100644 index 0000000..1df0543 --- /dev/null +++ b/topsf/production/normalized_weights.py @@ -0,0 +1,226 @@ +# coding: utf-8 + +""" +Column production methods related to generic event weights. +""" + +from typing import Iterable, Callable + +import law + +from columnflow.production import Producer, producer +from columnflow.util import maybe_import, safe_div +from columnflow.columnar_util import set_ak_column # , EMPTY_FLOAT +from columnflow.production.cms.btag import btag_weights + +ak = maybe_import("awkward") +np = maybe_import("numpy") + + +logger = law.logger.get_logger(__name__) + + +def normalized_weight_factory( + producer_name: str, + weight_producers: Iterable[Producer], + **kwargs, +) -> Callable: + + @producer( + # TODO: w.produces does not work as intended anymore, so we have to initialize the Producers here + uses=set(weight_producers) | set().union(*[w().produced_columns for w in weight_producers]) | {"process_id"}, + cls_name=producer_name, + mc_only=True, + # skip the checking existence of used/produced columns because not all columns are there + check_used_columns=False, + check_produced_columns=False, + # remaining produced columns are defined in the init function below + ) + def normalized_weight(self: Producer, events: ak.Array, **kwargs) -> ak.Array: + # check existence of requested weights to normalize and run producer if missing + missing_weights = self.weight_names.difference(events.fields) + + if missing_weights: + logger.warning(f"Missing weight columns: {missing_weights}") + # try to produce missing weights + for prod in self.weight_producers: + if ( + self[prod].produced_columns.difference(events.fields) and + self[prod].used_columns.intersection(events.fields) + ): + logger.info(f"Rerun producer {self[prod].cls_name}") + events = self[prod](events, **kwargs) + + # Create normalized weight columns if possible + if not_reproduced := missing_weights.difference(events.fields): + logger.warning(f"Weight columns {not_reproduced} could not be reproduced") + + for weight_name in self.weight_names.intersection(events.fields): + logger.debug(f"Creating normalized weight column for {weight_name}") + # create a weight vector starting with ones + norm_weight_per_pid = np.ones(len(events), dtype=np.float32) + + # fill weights with a new mask per unique process id (mostly just one) + for pid in self.unique_process_ids: + pid_mask = events.process_id == pid + norm_weight_per_pid[pid_mask] = self.ratio_per_pid[weight_name][pid] + + # multiply with actual weight + norm_weight_per_pid = norm_weight_per_pid * events[weight_name] + + # store it + norm_weight_per_pid = ak.values_astype(norm_weight_per_pid, np.float32) + events = set_ak_column(events, f"normalized_{weight_name}", norm_weight_per_pid) + + return events + + @normalized_weight.post_init + def normalized_weight_post_init(self: Producer, task: law.Task) -> None: + self.weight_producers = weight_producers + + # resolve weight names + self.weight_names = set() + for col in self.used_columns: + col = col.string_nano_column + if task.shift != "nominal" and (col.endswith("_up") or col.endswith("_down")): + # skip the up/down variations + continue + if "weight" in col and "normalized" not in col and "btag" not in col: + self.weight_names.add(col) + + self.produces |= set(f"normalized_{weight_name}" for weight_name in self.weight_names) + + @normalized_weight.requires + def normalized_weight_requires(self: Producer, task: law.Task, reqs: dict) -> None: + from columnflow.tasks.selection import MergeSelectionStats + reqs["selection_stats"] = MergeSelectionStats.req( + task, + branch=-1, + ) + + @normalized_weight.setup + def normalized_weight_setup( + self: Producer, task: law.Task, reqs: dict, inputs: dict, reader_targets: law.util.InsertableDict, + ) -> None: + # load the selection stats + stats = inputs["selection_stats"]["collection"][0]["stats"].load(formatter="json") + + # get the unique process ids in that dataset + key = "sum_mc_weight_per_process" + self.unique_process_ids = list(map(int, stats[key].keys())) + + # helper to get numerators and denominators + def numerator_per_pid(pid): + key = "sum_mc_weight_per_process" + return stats[key].get(str(pid), 0.0) + + def denominator_per_pid(weight_name, pid): + key = f"sum_mc_weight_{weight_name}_per_process" + return stats[key].get(str(pid), 0.0) + + # extract the ratio per weight and pid + self.ratio_per_pid = { + weight_name: { + pid: safe_div(numerator_per_pid(pid), denominator_per_pid(weight_name, pid)) + for pid in self.unique_process_ids + } + for weight_name in self.weight_names + } + + return normalized_weight + + +@producer( + uses={ + btag_weights.PRODUCES, "process_id", "Jet.{pt,eta,phi}", "njet", "ht", "nhf", + }, + # produced columns are defined in the init function below + mc_only=True, + modes=["ht_njet_nhf"], + # modes=["ht_njet_nhf", "ht_njet", "njet", "ht"], + from_file=False, +) +def normalized_btag_weights(self: Producer, events: ak.Array, **kwargs) -> ak.Array: + variable_map = { + # NOTE: might be cleaner to use the ht and njet reconstructed during the selection (and also compare?) + "ht": ak.sum(events.Jet.pt, axis=1), + "njet": ak.num(events.Jet.pt, axis=1), + "nhf": events.nhf, + } + + # sanity check + for var in ("ht", "njet"): + consistency_check = np.isclose(events[var], variable_map[var], rtol=0.0001) + if not ak.all(consistency_check): + logger.warning(f"Variable {var} is not consistent between before and after event selection. Please check the consistency of {var} and the selection steps.") + # raise ValueError(f"Variable {var} is not consistent between before and after event selection") + + for mode in self.modes: + if mode not in ("ht_njet_nhf", "ht_njet", "njet", "ht"): + raise NotImplementedError( + f"Normalization mode {mode} not implemented (see topsf.tasks.corrections.GetBtagNormalizationSF)", + ) + for weight_route in self[btag_weights].produced_columns: + weight_name = weight_route.string_column + if not weight_name.startswith("btag_weight"): + continue + + correction_key = f"{mode}_{weight_name}" + if correction_key not in set(self.correction_set.keys()): + raise KeyError(f"Missing scale factor for {correction_key}") + + sf = self.correction_set[correction_key] + inputs = [variable_map[inp.name] for inp in sf.inputs] + + norm_weight = sf.evaluate(*inputs) + norm_weight = norm_weight * events[weight_name] + events = set_ak_column(events, f"normalized_{mode}_{weight_name}", norm_weight, value_type=np.float32) + + return events + + +@normalized_btag_weights.post_init +def normalized_btag_weights_post_init(self: Producer, task: law.Task) -> None: + # NOTE: self[btag_weights].produced_columns is empty during the `init`, therefore changed to `post_init` + # this means that running this Producer directly on command line would not be triggered due to empty produces + # during task initialization + for weight_route in self[btag_weights].produced_columns: + weight_name = weight_route.string_column + if not weight_name.startswith("btag_weight"): + continue + for mode in self.modes: + self.produces.add(f"normalized_{mode}_{weight_name}") + + +@normalized_btag_weights.requires +def normalized_btag_weights_requires(self: Producer, task: law.Task, reqs: dict) -> None: + from topsf.tasks.corrections import GetBtagNormalizationSF + reqs["btag_renormalization_sf"] = GetBtagNormalizationSF.req(task) + + +normalized_btag_weights_full = normalized_btag_weights.derive("normalized_btag_weights_full", cls_dict=dict( + modes=["ht_njet_nhf", "ht_njet", "njet", "ht"], +)) + + +@normalized_btag_weights.setup +def normalized_btag_weights_setup( + self: Producer, + task: law.Task, + reqs: dict, + inputs: dict, + reader_targets: law.util.InsertableDict, +) -> None: + # create the corrector + import correctionlib + correctionlib.highlevel.Correction.__call__ = correctionlib.highlevel.Correction.evaluate + if self.from_file: + # used when the correction is stored as a JSON dict + self.correction_set = correctionlib.CorrectionSet.from_file( + inputs["btag_renormalization_sf"]["btag_renormalization_sf"].fn, + ) + else: + # used when correction is stored as a JSON string + self.correction_set = correctionlib.CorrectionSet.from_string( + inputs["btag_renormalization_sf"]["btag_renormalization_sf"].load(formatter="json"), + ) diff --git a/topsf/production/util.py b/topsf/production/util.py index dbe9419..36c9316 100644 --- a/topsf/production/util.py +++ b/topsf/production/util.py @@ -6,6 +6,7 @@ from functools import partial from columnflow.util import maybe_import +from columnflow.production.util import _lv_base ak = maybe_import("awkward") np = maybe_import("numpy") @@ -38,7 +39,7 @@ def ak_extract_fields(arr, fields, **kwargs): # functions for operating on lorentz vectors # -_lv_base = partial(ak_extract_fields, behavior=coffea.nanoevents.methods.nanoaod.behavior) +# _lv_base = partial(ak_extract_fields, behavior=coffea.nanoevents.methods.nanoaod.behavior) # broken in coffea :( lv_xyzt = partial(_lv_base, fields=["x", "y", "z", "t"], with_name="LorentzVector") lv_xyzt.__doc__ = """Construct a `LorentzVectorArray` from an input array.""" diff --git a/topsf/production/weights.py b/topsf/production/weights.py index c09af7b..f0da8ff 100644 --- a/topsf/production/weights.py +++ b/topsf/production/weights.py @@ -4,23 +4,151 @@ """ Producers related to event weights. """ +import law from columnflow.production import Producer, producer -from columnflow.production.cms.electron import electron_weights -from columnflow.production.cms.mc_weight import mc_weight -from columnflow.production.cms.muon import muon_weights +from columnflow.production.cms.electron import electron_weights, ElectronSFConfig +# from columnflow.production.cms.mc_weight import mc_weight +from columnflow.production.cms.muon import muon_weights, MuonSFConfig from columnflow.production.cms.pileup import pu_weight from columnflow.production.cms.scale import murmuf_weights, murmuf_envelope_weights +from columnflow.production.cms.btag import btag_weights, btag_wp_weights +from columnflow.production.cms.pdf import pdf_weights from columnflow.util import maybe_import +from columnflow.selection import SelectionResult +from columnflow.columnar_util import fill_at, set_ak_column from topsf.production.normalization import normalization_weights from topsf.production.gen_top import top_pt_weight from topsf.production.gen_v import vjets_weight -from topsf.production.ps_weights import ps_weights -from topsf.production.l1_prefiring import l1_prefiring_weights -from topsf.util import has_tag +# from topsf.production.ps_weights import ps_weights +from columnflow.production.cms.parton_shower import ps_weights +from topsf.production.normalized_weights import normalized_weight_factory, normalized_btag_weights +from topsf.util import has_tag, record_calls ak = maybe_import("awkward") +np = maybe_import("numpy") + +logger = law.logger.get_logger(__name__) + + +def high_pt_muon_reco(producer, corrector, variable_map): + # for muon reco SFs, use pT and eta as variables and apply a mask to only compute weights for high pT muons + momentum = variable_map["pt"] * np.cosh(variable_map["eta"]) + variable_map["p"] = momentum + return variable_map + + +muon_reco_weights = muon_weights.derive( + "muon_reco_weights", + cls_dict={ + "weight_name": "muon_reco_weight", + "get_muon_config": (lambda self: MuonSFConfig.new(self.config_inst.x.muon_reco_sf_config)), + "update_corrector_variables": (lambda self, corrector, variables: high_pt_muon_reco(self, corrector, variables)), + } +) +muon_id_weights = muon_weights.derive( + "muon_id_weights", + cls_dict={ + "weight_name": "muon_id_weight", + "get_muon_config": (lambda self: MuonSFConfig.new(self.config_inst.x.muon_iso_sf_config)), + } +) +muon_iso_weights = muon_weights.derive( + "muon_iso_weights", + cls_dict={ + "weight_name": "muon_iso_weight", + "get_muon_config": (lambda self: MuonSFConfig.new(self.config_inst.x.muon_id_sf_config)), + } +) +muon_trigger_weights = muon_weights.derive( + "muon_trigger_weights", + cls_dict={ + "weight_name": "muon_trigger_weight", + "get_muon_config": (lambda self: MuonSFConfig.new(self.config_inst.x.muon_trigger_sf_config)), + } +) + +electron_reco_weights = electron_weights.derive( + "electron_reco_weights", + cls_dict={ + "weight_name": "electron_reco_weight", + "get_electron_config": (lambda self: ElectronSFConfig.new(self.config_inst.x.electron_reco_sf_config)), + } +) +electron_id_iso_weights = electron_weights.derive( + "electron_id_iso_weights", + cls_dict={ + "weight_name": "electron_id_iso_weight", + "get_electron_config": (lambda self: ElectronSFConfig.new(self.config_inst.x.electron_id_iso_sf_config)), + } +) + +electron_trigger_weights = electron_weights.derive( + "electron_trigger_weights", + cls_dict={ + "weight_name": "electron_trigger_weight", + "get_electron_config": (lambda self: ElectronSFConfig.new(self.config_inst.x.electron_trigger_sf_config)), + "get_electron_file": (lambda self, external_files: external_files.electron_trigger_sf), + } +) + + +@producer( + uses={"Electron.pt"}, + produces={"electron_norm_fix_weight"}, + mc_only=True, +) +def electron_norm_fix_weights(self: Producer, events: ak.Array, **kwargs) -> ak.Array: + """ + Add a column with a flat 1.2 SF. + """ + electron_mask = ak.num(events.Electron) == 1 + electron_norm_weight = ak.where(electron_mask, 1.2, 1.0) + events = set_ak_column(events, "electron_norm_fix_weight", electron_norm_weight) + return events + + +@producer( + uses={ + muon_reco_weights, + muon_id_weights, + muon_iso_weights + }, + produces={ + muon_reco_weights, + muon_id_weights, + muon_iso_weights + }, + mc_only=True, +) +def muon_reco_id_iso_weights(self: Producer, events: ak.Array, **kwargs) -> ak.Array: + """ + Producer to compute muon ID and isolation weights separately. + """ + muon_mask = (events.Muon["pt"] >= 30) & (abs(events.Muon["eta"]) < 2.4) + events = self[muon_reco_weights](events, muon_mask=muon_mask, **kwargs) + events = self[muon_id_weights](events, muon_mask=muon_mask, **kwargs) + events = self[muon_iso_weights](events, muon_mask=muon_mask, **kwargs) + return events + + +@producer( + uses={electron_reco_weights, electron_id_iso_weights}, + produces={electron_reco_weights, electron_id_iso_weights}, + mc_only=True, +) +def electron_reco_id_iso_weights(self: Producer, events: ak.Array, **kwargs) -> ak.Array: + """ + Producer to compute electron reconstruction, ID and isolation weights separately. + """ + if self.config_inst.x.year in [2022, 2023]: + electron_mask = (events.Electron["pt"] >= 35) + elif self.config_inst.x.year == 2024: + electron_mask = ((events.Electron["pt"] >= 20.0) & (events.Electron["pt"] < 1000.0)) + events = self[electron_reco_weights](events, electron_mask=electron_mask, **kwargs) + events = self[electron_id_iso_weights](events, electron_mask=electron_mask, **kwargs) + return events @producer @@ -28,47 +156,76 @@ def weights(self: Producer, events: ak.Array, **kwargs) -> ak.Array: """ Main event weight producer (e.g. MC generator, scale factors, normalization). """ - if self.dataset_inst.is_mc: - if not has_tag("skip_electron_weights", self.config_inst, self.dataset_inst, operator=any): - electron_mask = (events.Electron["pt"] >= 35) - events = self[electron_weights](events, electron_mask=electron_mask, **kwargs) - if not has_tag("skip_muon_weights", self.config_inst, self.dataset_inst, operator=any): - muon_mask = (events.Muon["pt"] >= 30) & (abs(events.Muon["eta"]) < 2.4) - events = self[muon_weights](events, muon_mask=muon_mask, **kwargs) + run_list = [] + with record_calls(self, run_list): + if self.dataset_inst.is_mc: + # compute normalization weights + events = self[normalization_weights](events, **kwargs) - # # compute btag weights - # jet_mask = (events.Jet.pt >= 100) & (abs(events.Jet.eta) < 2.5) - # events = self[btag_weights](events, jet_mask=jet_mask, **kwargs) + # compute top pT weights + if self.dataset_inst.has_tag("is_ttbar"): + events = self[top_pt_weight](events, **kwargs) - # compute top pT weights - if self.dataset_inst.has_tag("is_ttbar"): - events = self[top_pt_weight](events, **kwargs) + # compute V+jets K factor weights + if not has_tag("skip_kfactor_weights", self.config_inst, self.dataset_inst, operator=any) and self.dataset_inst.has_tag("is_v_jets"): + events = self[vjets_weight](events, **kwargs) - # compute V+jets K factor weights - if self.dataset_inst.has_tag("is_v_jets"): - events = self[vjets_weight](events, **kwargs) + if not has_tag("skip_electron_weights", self.config_inst, self.dataset_inst, operator=any): + events = self[electron_reco_id_iso_weights](events, **kwargs) + events = self[electron_trigger_weights](events, **kwargs) + events = self[electron_norm_fix_weights](events, **kwargs) - if self.config_inst.x.run == 2: - # compute L1 prefiring weights - events = self[l1_prefiring_weights](events, **kwargs) + if not has_tag("skip_muon_weights", self.config_inst, self.dataset_inst, operator=any): + events = self[muon_reco_id_iso_weights](events, **kwargs) + events = self[muon_trigger_weights](events, **kwargs) - # compute normalization weights - events = self[normalization_weights](events, **kwargs) + # FIXME add trigger SF here - # compute MC weights - events = self[mc_weight](events, **kwargs) + # normalize event weights using stats + events = self[normalized_pu_weights](events, **kwargs) - # compute pu weights - events = self[pu_weight](events, **kwargs) + if not has_tag("no_ps_weights", self.config_inst, self.dataset_inst, operator=any): + logger.debug("Applying PS weights and normalizing them.") + events = self[normalized_ps_weights](events, **kwargs) - # compute scale weights (mur, muf, envelope) for st and tt datasets - if self.dataset_inst.has_tag("has_top"): - events = self[murmuf_weights](events, **kwargs) - events = self[murmuf_envelope_weights](events, **kwargs) + if not has_tag("skip_scale", self.config_inst, self.dataset_inst, operator=any): + logger.debug("Applying scale weights and normalizing them.") + events = self[normalized_scale_weights](events, **kwargs) - # compute PS weights - if not self.dataset_inst.has_tag("is_qcd"): - events = self[ps_weights](events, **kwargs) + if not has_tag("skip_pdf", self.config_inst, self.dataset_inst, operator=any): + logger.debug("Applying pdf weights and normalizing them.") + events = self[normalized_pdf_weights](events, **kwargs) + + # compute btag weights + if ( + has_tag("skip_btag_weights", self.config_inst, self.dataset_inst, operator=any) and + not has_tag("skip_btag_wp_weights", self.config_inst, self.dataset_inst, operator=any) + ): + logger.debug("Skipping shape based btag weights and applying fixed WP SF instead.") + # skip shape based btag weights and apply fixed WP SF instead (for 2024) + jet_mask = (events.Jet["pt"] < 10_000) & (abs(events.Jet["eta"]) < 2.5) + events = self[btag_wp_weights](events, jet_mask=jet_mask, **kwargs) + elif ( + has_tag("skip_btag_wp_weights", self.config_inst, self.dataset_inst, operator=any) and + not has_tag("skip_btag_weights", self.config_inst, self.dataset_inst, operator=any) + ): + logger.debug("Skipping fixed WP btag weights and applying shape based SF instead.") + # apply shape based btag weights (for 2022/23) + # and normalize + jet_mask = (events.Jet["pt"] >= 100) & (abs(events.Jet["eta"]) < 2.5) + events = self[btag_weights](events, jet_mask=jet_mask, **kwargs) + events = self[normalized_btag_weights](events, jet_mask=jet_mask, **kwargs) + else: + logger.warning("No btag weights applied.") + + # # compute MC weights + # # already run in selection, not needed here? + # events = self[mc_weight](events, **kwargs) + + logger.info_once( + "Finished computing event weights:\n" + + "\n".join(run_list) + ) return events @@ -78,16 +235,28 @@ def weights_init(self: Producer) -> None: if getattr(self, "dataset_inst", None) and self.dataset_inst.is_mc: # dynamically add dependencies if running on MC if not has_tag("skip_electron_weights", self.config_inst, self.dataset_inst, operator=any): - self.uses |= {electron_weights} - self.produces |= {electron_weights} + self.uses |= { + electron_reco_id_iso_weights, + electron_trigger_weights, + electron_norm_fix_weights, + "Electron.{pt,eta}" + } + self.produces |= { + electron_reco_id_iso_weights, + electron_trigger_weights, + electron_norm_fix_weights + } if not has_tag("skip_muon_weights", self.config_inst, self.dataset_inst, operator=any): - self.uses |= {muon_weights} - self.produces |= {muon_weights} - - if self.config_inst.x.run == 2: - self.uses |= {l1_prefiring_weights} - self.produces |= {l1_prefiring_weights} + self.uses |= { + muon_reco_id_iso_weights, + muon_trigger_weights, + "Muon.{pt,eta,phi}" + } + self.produces |= { + muon_reco_id_iso_weights, + muon_trigger_weights + } if not self.dataset_inst.has_tag("is_qcd"): self.uses |= {ps_weights} @@ -97,14 +266,188 @@ def weights_init(self: Producer) -> None: self.uses |= {top_pt_weight} self.produces |= {top_pt_weight} - if self.dataset_inst.has_tag("is_v_jets"): + if not has_tag("skip_kfactor_weights", self.config_inst, self.dataset_inst, operator=any) and self.dataset_inst.has_tag("is_v_jets"): self.uses |= {vjets_weight} self.produces |= {vjets_weight} - self.uses |= {normalization_weights, pu_weight, mc_weight} + self.uses |= {normalization_weights, normalized_pu_weights} + self.produces |= {normalization_weights, normalized_pu_weights} + + if not has_tag("no_ps_weights", self.config_inst, self.dataset_inst, operator=any): + self.uses |= {normalized_ps_weights} + self.produces |= {normalized_ps_weights} - if self.dataset_inst.has_tag("has_top"): - self.uses |= {murmuf_weights, murmuf_envelope_weights} - self.produces |= {murmuf_weights, murmuf_envelope_weights} + if not has_tag("skip_scale", self.config_inst, self.dataset_inst, operator=any): + self.uses |= {normalized_scale_weights} + self.produces |= {normalized_scale_weights} - self.produces |= {normalization_weights, pu_weight, mc_weight} + if not has_tag("skip_pdf", self.config_inst, self.dataset_inst, operator=any): + self.uses |= {normalized_pdf_weights} + self.produces |= {normalized_pdf_weights} + + if ( + has_tag("skip_btag_weights", self.config_inst, self.dataset_inst, operator=any) and + not has_tag("skip_btag_wp_weights", self.config_inst, self.dataset_inst, operator=any) + ): + logger.warning_once("Using fixed wp b-tagging weights for 2024.") + self.uses |= {btag_wp_weights} + self.produces |= {btag_wp_weights} + elif ( + has_tag("skip_btag_wp_weights", self.config_inst, self.dataset_inst, operator=any) and + not has_tag("skip_btag_weights", self.config_inst, self.dataset_inst, operator=any) + ): + logger.warning_once("Using shape based b-tagging weights for 2022/2023.") + self.uses |= {btag_weights, normalized_btag_weights} + self.produces |= {btag_weights, normalized_btag_weights} + else: + logger.warning_once("No btag weights producer loaded.") + + +@producer( + uses={ + pu_weight, + }, + # produces={ + # pu_weight, + # }, + mc_only=True, +) +def event_weights_to_normalize(self: Producer, events: ak.Array, results: SelectionResult, **kwargs) -> ak.Array: + """ + Wrapper of several event weight producers that are typically called as part of SelectEvents + since it is required to normalize them before applying certain event selections. + """ + + # compute pu weights + events = self[pu_weight](events, **kwargs) + # if self.has_dep(ps_weights): + if not has_tag("no_ps_weights", self.config_inst, self.dataset_inst, operator=any): + logger.debug("Compute PS weights for normalization") + events = self[ps_weights](events, **kwargs) + + if ( + has_tag("skip_btag_wp_weights", self.config_inst, self.dataset_inst, operator=any) and + not has_tag("skip_btag_weights", self.config_inst, self.dataset_inst, operator=any) + ): + # compute btag SF weights (for renormalization tasks) + logger.debug("Compute btag weights for normalization") + events = self[btag_weights]( + events, + jet_mask=results.aux["jet_mask"], + negative_b_score_action="ignore", + negative_b_score_log_mode="debug", + **kwargs, + ) + + # skip scale/pdf weights for some datasets (missing columns) + # if self.has_dep(murmuf_envelope_weights): + if not has_tag("skip_scale", self.config_inst, self.dataset_inst, operator=any) and self.has_dep(murmuf_envelope_weights): + # compute scale weights + logger.debug("Compute scale weights for normalization") + events = self[murmuf_envelope_weights](events, **kwargs) + + # if self.has_dep(murmuf_weights): + if not has_tag("skip_scale", self.config_inst, self.dataset_inst, operator=any) and self.has_dep(murmuf_weights): + # read out mur and weights + logger.debug("Compute murmuf weights for normalization") + events = self[murmuf_weights](events, **kwargs) + + # if self.has_dep(pdf_weights): + if not has_tag("skip_pdf", self.config_inst, self.dataset_inst, operator=any) and self.has_dep(pdf_weights): + # compute pdf weights + logger.debug("Compute pdf weights for normalization") + events = self[pdf_weights]( + events, + outlier_threshold=0.99, + outlier_action="remove", + outlier_log_mode="debug", + invalid_weights_action="ignore" if self.dataset_inst.has_tag("partial_lhe_weights") else "raise", + **kwargs, + ) + + return events + + +@event_weights_to_normalize.init +def event_weights_to_normalize_init(self) -> None: + # used Producers need to be set in the init or decorator + if ( + has_tag("skip_btag_wp_weights", self.config_inst, self.dataset_inst, operator=any) and + not has_tag("skip_btag_weights", self.config_inst, self.dataset_inst, operator=any) + ): + self.uses |= {btag_weights} + + if not has_tag("skip_scale", self.config_inst, self.dataset_inst, operator=any): + self.uses |= {murmuf_envelope_weights, murmuf_weights} + + if not has_tag("skip_pdf", self.config_inst, self.dataset_inst, operator=any): + self.uses |= {pdf_weights} + + if not has_tag("no_ps_weights", self.config_inst, self.dataset_inst, operator=any): + self.uses |= {ps_weights} + + +@event_weights_to_normalize.post_init +def event_weights_to_normalize_post_init(self, task: law.Task) -> None: + # produced columns can be set in post_init to choose stored columns based on the shift + for _cls in self.uses: + if _cls == btag_weights and task.shift == "nominal": + self.produces |= {btag_weights} + elif _cls == btag_weights: + self.produces |= self.deps[btag_weights].produced_columns + elif task.shift == "nominal": + self.produces |= self.deps[_cls].produced_columns + else: + self.produces |= { + route for route in self.deps[_cls].produced_columns + if not route.nano_column.endswith("_up") and not route.nano_column.endswith("_down") + } + + +# renormalized weights +normalized_scale_weights = normalized_weight_factory( + producer_name="normalized_scale_weights", + weight_producers={murmuf_envelope_weights, murmuf_weights}, +) +normalized_pdf_weights = normalized_weight_factory( + producer_name="normalized_pdf_weights", + weight_producers={pdf_weights}, +) +normalized_pu_weights = normalized_weight_factory( + producer_name="normalized_pu_weights", + weight_producers={pu_weight}, +) +normalized_ps_weights = normalized_weight_factory( + producer_name="normalized_ps_weights", + weight_producers={ps_weights}, +) + + +@producer( + uses={"mc_weight", "genWeight"}, + produces={"mc_weight", "genWeight"}, + mc_only=True, +) +def large_weights_killer(self: Producer, events: ak.Array, stats: dict, **kwargs) -> ak.Array: + """ + Simple producer that sets eventweights to 0 when too large. + """ + if self.dataset_inst.is_data: + raise Exception("large_weights_killer is only callable for MC") + + # set mc_weight to zero when genWeight is > 0.5 for powheg HH events + if self.dataset_inst.has_tag("is_hh") and self.dataset_inst.name.endswith("powheg"): + # TODO: this feels very unsafe because genWeight can also be just 1 for all events. To be revisited + weight_too_large = abs(events.genWeight) > 0.5 + logger.warning(f"found {ak.sum(weight_too_large)} HH events with genWeight > 0.5") + + events = fill_at(events, weight_too_large, "mc_weight", 0.0, value_type=np.float32) + + # check for anomalous weights and store in stats + median_weight = ak.sort(abs(events.mc_weight))[int(len(events) / 2)] + anomalous_weights_mask = abs(events.mc_weight) > 1000 * median_weight + if ak.any(anomalous_weights_mask): + logger.warning(f"found {ak.sum(anomalous_weights_mask)} events with weights > 1000 * median weight") + stats["num_events_anomalous_weights"] += ak.sum(anomalous_weights_mask) + + return events diff --git a/topsf/selection/bad_events.py b/topsf/selection/bad_events.py new file mode 100644 index 0000000..d92ecd4 --- /dev/null +++ b/topsf/selection/bad_events.py @@ -0,0 +1,104 @@ +# coding: utf-8 + +""" +Selection modules for HH(bbWW) to identify events that should be considered missing. +""" + +from __future__ import annotations + +import law +from columnflow.util import maybe_import +from columnflow.columnar_util import fill_at, full_like, has_ak_column + +from columnflow.selection import Selector, SelectionResult, selector + +np = maybe_import("numpy") +ak = maybe_import("awkward") + +logger = law.logger.get_logger(__name__) + + +@selector +def get_outlier_scale_weights( + self: Selector, + events: ak.Array, + results: SelectionResult, + **kwargs, +) -> ak.Array: + """ + Helper to identify bad events that should be considered missing altogether + """ + bad_mask = full_like(events.event, False, dtype=bool) + results.steps["no_sel_mask"] = ~bad_mask + if self.dataset_inst.is_data: + # no bad data events + return events, results + if self.dataset_inst.has_tag("no_lhe_weights"): + logger.debug("Dataset has tag 'no_lhe_weights', skipping checks for bad events") + # at the moment, we only check for bad events from LHE weights + return events, results + + if not has_ak_column(events, "LHEScaleWeight"): + logger.debug("Dataset does not have LHEScaleWeight column, skipping checks for bad events") + # when ScaleWeights are not present, we cannot check for bad events + return events, results + + # drop events for which we expect lhe infos but that lack them + # see https://cms-talk.web.cern.ch/t/lhe-weight-vector-empty-for-certain-events/97636/3 + if self.dataset_inst.has_tag("partial_lhe_weights"): + logger.debug("Dataset has tag 'partial_lhe_weights', dropping events with empty LHEScaleWeight") + n_weights = ak.num(events.LHEScaleWeight, axis=1) + bad_lhe_mask = (n_weights != 8) & (n_weights != 9) + if ak.any(bad_lhe_mask): + bad_mask = bad_mask | bad_lhe_mask + frac = ak.mean(bad_lhe_mask) + logger.warning( + f"found {ak.sum(bad_lhe_mask)} events ({frac * 100:.1f}%) with bad LHEScaleWeight", + ) + if self.dataset_inst.has_tag(["has_lhe_weights", "partial_lhe_weights"], mode=any): + logger.debug("Dataset has tag 'has_lhe_weights' or 'partial_lhe_weights', checking for bad LHEScaleWeights") + # check if the LHE weights are all finite (assuming bad LHEScaleWeights also means bad LHEPdfWeights) + bad_lhe_mask = ak.any(~np.isfinite(events.LHEScaleWeight), axis=1) + if ak.any(bad_lhe_mask): + bad_mask = bad_mask | bad_lhe_mask + frac = ak.mean(bad_lhe_mask) + logger.warning( + f"found {ak.sum(bad_lhe_mask)} events ({frac * 100:.1f}%) with non-finite LHEScaleWeights; " + "setting all LHE scale and pdf weights to 1", + ) + + # set LHEScaleWeight and LHEPdfWeight to 1 for these events + for col in ("LHEScaleWeight", "LHEPdfWeight"): + if has_ak_column(events, col): + # set weights to 1 for these events to avoid issues with nan values downstream + events = fill_at(events, bad_lhe_mask, col, ak.ones_like(events[col])) + # events = fill_at(events, bad_lhe_mask, col, full_like(events[col], 1.0)) + + # define "no selection" step as all events that are not considered bad from this function + no_sel = ~bad_mask + results.steps["no_sel_mask"] = no_sel + + return events, results + + +@selector +def extend_bad_events( + self: Selector, + events: ak.Array, + results: SelectionResult, + **kwargs, +) -> ak.Array: + """ + Helper to identify bad events that should be considered missing altogether + """ + if self.dataset_inst.is_data or self.dataset_inst.has_tag("no_lhe_weights"): + return events, results + + if has_ak_column(events, "pdf_weight"): + bad_mask = (events.pdf_weight == 0) + else: + logger.debug("Dataset does not have pdf_weight column, skipping checks for bad events based on pdf_weight") + return events, results + + results.steps["no_sel_mask"] = results.steps.no_sel_mask & (~bad_mask) + return events, results diff --git a/topsf/selection/common.py b/topsf/selection/common.py new file mode 100644 index 0000000..135c0b8 --- /dev/null +++ b/topsf/selection/common.py @@ -0,0 +1,244 @@ +# coding: utf-8 + +""" +Selection modules for HH(bbWW) that are used for both SL and DL. +""" + +from __future__ import annotations + +from collections import defaultdict + +import law +from columnflow.util import maybe_import +from columnflow.columnar_util import EMPTY_FLOAT, fill_at +from columnflow.production.util import attach_coffea_behavior + +from columnflow.selection import Selector, SelectionResult, selector +from columnflow.selection.cms.met_filters import met_filters +from columnflow.selection.cms.json_filter import json_filter +from columnflow.selection.cms.jets import jet_veto_map +from columnflow.production.cms.mc_weight import mc_weight +from columnflow.production.categories import category_ids +from columnflow.production.processes import process_ids +from columnflow.production.cms.seeds import deterministic_seeds + +from topsf.production.weights import event_weights_to_normalize, large_weights_killer +from topsf.production.filter import ECALBadCalibrationFilter +from topsf.selection.stats import topsf_selection_step_stats, topsf_increment_stats +from topsf.selection.hists import topsf_selection_hists +from topsf.selection.bad_events import extend_bad_events, get_outlier_scale_weights +from topsf.util import IF_MC, record_calls + +np = maybe_import("numpy") +ak = maybe_import("awkward") + +logger = law.logger.get_logger(__name__) + + +def masked_sorted_indices(mask: ak.Array, sort_var: ak.Array, ascending: bool = False) -> ak.Array: + """ + Helper function to obtain the correct indices of an object mask + """ + indices = ak.argsort(sort_var, axis=-1, ascending=ascending) + return indices[mask[indices]] + + +def get_met_filters(self: Selector): + """ custom function to skip met filter for our Run2 EOY signal samples """ + met_filters = self.config_inst.x.met_filters + + if getattr(self, "dataset_inst", None) and self.dataset_inst.has_tag("is_eoy"): + # remove filter for EOY sample + try: + met_filters.remove("Flag.BadPFMuonDzFilter") + except (KeyError, AttributeError): + pass + + return met_filters + + +topsf_met_filters = met_filters.derive("topsf_met_filters", cls_dict=dict(get_met_filters=get_met_filters)) + + +@selector( + uses={ + jet_veto_map, + topsf_met_filters, json_filter, "PV.npvsGood", + process_ids, attach_coffea_behavior, + mc_weight, large_weights_killer, + ECALBadCalibrationFilter, + }, + produces={ + topsf_met_filters, json_filter, + process_ids, attach_coffea_behavior, + mc_weight, large_weights_killer, + ECALBadCalibrationFilter, + }, + exposed=False, +) +def pre_selection( + self: Selector, + events: ak.Array, + stats: defaultdict, + task: law.Task, + **kwargs, +) -> tuple[ak.Array, SelectionResult]: + """ Methods that are called for both WP and SF before calling the selection modules """ + run_list = [] + + with record_calls(self, run_list): + # temporary fix for optional types from Calibration (e.g. events.Jet.pt --> ?float32) + # TODO: remove as soon as possible as it might lead to weird bugs when there are none entries in inputs + events = ak.fill_none(events, EMPTY_FLOAT) + + # prepare the selection results that are updated at every step + results = SelectionResult() + + # run deterministic seeds when no Calibrator has been requested + if not task.calibrators: + events = self[deterministic_seeds](events, **kwargs) + + # mc weight + if self.dataset_inst.is_mc: + events = self[mc_weight](events, **kwargs) + events = self[large_weights_killer](events, stats, **kwargs) + + # create process ids + events = self[process_ids](events, **kwargs) + + # ensure coffea behavior + events = self[attach_coffea_behavior](events, **kwargs) + + # apply some general quality criteria on events + results.steps["good_vertex"] = events.PV.npvsGood >= 1 + events, met_results = self[topsf_met_filters](events, **kwargs) # produces "met_filter" step + + # recompute ecalBadCalibrationFilter + if self.has_dep(ECALBadCalibrationFilter): + events = self[ECALBadCalibrationFilter](events, **kwargs) + logger.info("patching met_filter with patchedEcalBadCalibFilter") + met_results.steps["met_filter"] = met_results.steps.met_filter & events.patchedEcalBadCalibFilter + + results += met_results + + if self.dataset_inst.is_data: + events, json_results = self[json_filter](events, **kwargs) # produces "json" step + results += json_results + else: + results.steps["json"] = ak.Array(np.ones(len(events), dtype=bool)) + + # apply jet veto map + events, jet_veto_results = self[jet_veto_map](events, **kwargs) + results += jet_veto_results + + # combine quality criteria into a single step + results.steps["cleanup"] = ( + results.steps.jet_veto_map & + results.steps.good_vertex & + results.steps.met_filter & + results.steps.json + ) + + logger.info_once( + "Finished pre-selection steps:\n" + + "\n".join(run_list) + ) + + return events, results + + +@pre_selection.init +def pre_selection_init(self: Selector) -> None: + if not getattr(self, "dataset_inst", None) or self.dataset_inst.is_data: + return + + +@pre_selection.post_init +def pre_selection_post_init(self: Selector, task: law.Task) -> None: + if not task.calibrators: + self.uses.add(deterministic_seeds) + self.produces.add(deterministic_seeds) + + +@selector( + uses={ + get_outlier_scale_weights, extend_bad_events, + event_weights_to_normalize, + }, + produces={ + event_weights_to_normalize, + IF_MC("mc_weight"), + }, +) +def get_weights_and_no_sel_mask( + self: Selector, + events: ak.Array, + results: SelectionResult, + **kwargs, +) -> tuple[ak.Array, ak.Array]: + """ + Helper to get the weights and bad mask for the events. + """ + events, results = self[get_outlier_scale_weights](events, results=results, **kwargs) + + # produce event weights + if self.dataset_inst.is_mc: + events = self[event_weights_to_normalize](events, results=results, **kwargs) + events, results = self[extend_bad_events](events, results=results, **kwargs) + + # set mc_weight to 0 for events that are considered bad for simplified downstream processing + if ak.any(bad := ~results.steps.no_sel_mask): + logger.warning( + f"Found {ak.sum(bad)} events ({100 * ak.mean(bad):.1f}%) that are considered bad, setting mc_weight to 0", + ) + if self.dataset_inst.is_mc: + events = fill_at(events, bad, "mc_weight", 0.0, value_type=np.float32) + + return events, results + + +@selector( + uses={ + category_ids, topsf_increment_stats, topsf_selection_step_stats, + topsf_selection_hists, + }, + produces={ + category_ids, topsf_increment_stats, topsf_selection_step_stats, + topsf_selection_hists, + }, + exposed=False, +) +def post_selection( + self: Selector, + events: ak.Array, + results: SelectionResult, + stats: defaultdict, + hists: dict, + **kwargs, +) -> tuple[ak.Array, SelectionResult]: + """ Methods that are called for both SL and DL after calling the selection modules """ + + # build categories + events = self[category_ids](events, results=results, **kwargs) + + events = self[topsf_selection_step_stats](events, results, stats, **kwargs) + events = self[topsf_increment_stats](events, results, stats, **kwargs) + events = self[topsf_selection_hists](events, results, hists, **kwargs) + + def log_fraction(stats_key: str, msg: str | None = None): + if not stats.get(stats_key): + return + if not msg: + msg = "Fraction of {stats_key}" + logger.info(f"{msg}: {(100 * stats[stats_key] / stats['num_events']):.2f}%") + + log_fraction("num_negative_weights", "Fraction of negative weights") + log_fraction("num_pu_0", "Fraction of events with pu_weight == 0") + log_fraction("num_pu_100", "Fraction of events with pu_weight >= 100") + + # temporary fix for optional types from Calibration (e.g. events.Jet.pt --> ?float32) + # TODO: remove as soon as possible as it might lead to weird bugs when there are none entries in inputs + events = ak.fill_none(events, EMPTY_FLOAT) + + logger.info(f"Selected {ak.sum(results.event)} from {len(events)} events") + return events, results diff --git a/topsf/selection/default.py b/topsf/selection/default.py index 9aea994..8af2ba5 100644 --- a/topsf/selection/default.py +++ b/topsf/selection/default.py @@ -3,22 +3,18 @@ """ Exemplary selection methods. """ +from __future__ import annotations +import law from operator import and_ from functools import reduce -from collections import defaultdict, OrderedDict +from collections import defaultdict -from columnflow.util import maybe_import -from columnflow.columnar_util import optional_column as optional +from columnflow.util import maybe_import, DotDict from columnflow.columnar_util import remove_ak_column, EMPTY_FLOAT from columnflow.selection import Selector, SelectionResult, selector -from columnflow.selection.cms.met_filters import met_filters -from columnflow.selection.cms.json_filter import json_filter -from columnflow.selection.cms.jets import jet_veto_map - -from columnflow.production.cms.mc_weight import mc_weight -from columnflow.production.util import attach_coffea_behavior +from columnflow.selection.cms.btag import fill_btag_wp_count_hists from topsf.selection.lepton import lepton_selection from topsf.selection.jet import jet_selection, jet_lepton_2d_selection @@ -27,118 +23,66 @@ from topsf.selection.met import met_selection from topsf.selection.w_lep import w_lep_selection from topsf.selection.cutflow_features import cutflow_features +from topsf.selection.common import get_weights_and_no_sel_mask, pre_selection +from topsf.selection.stats import topsf_increment_stats, topsf_selection_step_stats +from topsf.selection.hists import topsf_selection_hists from topsf.production.processes import process_ids from topsf.production.probe_jet import probe_jet from topsf.production.gen_top import gen_parton_top from topsf.production.gen_v import gen_v_boson -# from columnflow.production.categories import category_ids -from topsf.production.categories import category_ids +from topsf.util import has_tag, record_calls + +from columnflow.production.categories import category_ids +# from topsf.production.categories import category_ids np = maybe_import("numpy") ak = maybe_import("awkward") +hist = maybe_import("hist") - -@selector( - uses={optional("mc_weight"), "process_id"}, - check_columns_present={"produces"}, # some used columns optional -) -def increment_stats( - self: Selector, - events: ak.Array, - results: SelectionResult, - stats: dict, - **kwargs, -) -> ak.Array: - """ - Unexposed selector that does not actually select objects but instead increments selection - *stats* in-place based on all input *events* and the final selection *mask*. - """ - # get the event mask - mask = results.event - - # ensure mask passed is boolean - mask = ak.values_astype(mask, bool) - - # increment plain counts - stats["num_events"] += len(events) - stats["num_events_selected"] += np.float64(ak.sum(mask, axis=0)) - - # create a map of entry names to (weight, mask) pairs that will be written to stats - weight_map = OrderedDict() - if self.dataset_inst.is_mc: - # mc weight for all events - weight_map["mc_weight"] = (events.mc_weight, Ellipsis) - - # mc weight for selected events - weight_map["mc_weight_selected"] = (events.mc_weight, mask) - - # add more entries here - # ... - - # get and store the weights - for name, (weights, mask) in weight_map.items(): - joinable_mask = True if mask is Ellipsis else mask - - # sum for all processes - stats[f"sum_{name}"] += np.float64(ak.sum(weights[mask])) - - # sums per process id - stats.setdefault(f"sum_{name}_per_process", defaultdict(float)) - processes = np.unique(events.process_id) - for p in processes: - stats[f"sum_{name}_per_process"][int(p)] += np.float64( - ak.sum(weights[(events.process_id == p) & joinable_mask]), - ) - - # sums per category - stats.setdefault(f"sum_{name}_per_category", defaultdict(float)) - categories = np.unique(ak.ravel(events.category_ids)) - for c in categories: - stats[f"sum_{name}_per_category"][int(c)] += np.float64( - ak.sum(weights[ak.any(events.category_ids == c, axis=-1) & joinable_mask]), - ) - - return events +logger = law.logger.get_logger(__name__) @selector( uses={ - attach_coffea_behavior, - mc_weight, process_ids, category_ids, + pre_selection, + process_ids, category_ids, cutflow_features, - met_filters, lepton_selection, met_selection, w_lep_selection, - jet_veto_map, jet_selection, bjet_lepton_selection, jet_lepton_2d_selection, fatjet_selection, - increment_stats, probe_jet, gen_parton_top, gen_v_boson, + get_weights_and_no_sel_mask, + topsf_selection_step_stats, + topsf_increment_stats, + topsf_selection_hists, }, produces={ - mc_weight, process_ids, category_ids, + pre_selection, + process_ids, category_ids, cutflow_features, - met_filters, lepton_selection, met_selection, w_lep_selection, - jet_veto_map, jet_selection, bjet_lepton_selection, jet_lepton_2d_selection, fatjet_selection, - increment_stats, probe_jet, gen_parton_top, gen_v_boson, + get_weights_and_no_sel_mask, + topsf_selection_step_stats, + topsf_increment_stats, + topsf_selection_hists, }, exposed=True, ) @@ -146,115 +90,170 @@ def default( self: Selector, events: ak.Array, stats: defaultdict, + hists: DotDict[str, hist.Hist], **kwargs, ) -> tuple[ak.Array, SelectionResult]: - # ensure coffea behavior - events = self[attach_coffea_behavior](events, **kwargs) - - # prepare the selection results that are updated at every step - results = SelectionResult() - - # MET filters - events, met_filters_results = self[met_filters](events, **kwargs) - results += met_filters_results - - # JSON filter (data-only) - if self.dataset_inst.is_data: - events, json_filter_results = self[json_filter](events, **kwargs) - results += json_filter_results - - # lepton selection - events, lepton_results = self[lepton_selection](events, **kwargs) - results += lepton_results - - # jet selection - events, jet_results = self[jet_selection](events, **kwargs) - results += jet_results - - # bjet-lepton selection - events, bjet_lepton_results = self[bjet_lepton_selection](events, **kwargs) - results += bjet_lepton_results - - # jet-lepton 2D selection - events, jet_lepton_2d_results = self[jet_lepton_2d_selection](events, results=results, **kwargs) - results += jet_lepton_2d_results - - # fatjet selection - events, fatjet_results = self[fatjet_selection](events, **kwargs) - results += fatjet_results + run_list = [] + with record_calls(self, run_list): + # ensure coffea behavior + events, results = self[pre_selection](events, stats, **kwargs) + + # lepton selection + events, lepton_results = self[lepton_selection](events, **kwargs) + results += lepton_results + + # jet selection + events, jet_results = self[jet_selection](events, **kwargs) + results += jet_results + + # bjet-lepton selection + events, bjet_lepton_results = self[bjet_lepton_selection](events, **kwargs) + results += bjet_lepton_results + + # jet-lepton 2D selection + events, jet_lepton_2d_results = self[jet_lepton_2d_selection](events, results=results, **kwargs) + results += jet_lepton_2d_results + + # fatjet selection + events, fatjet_results = self[fatjet_selection](events, **kwargs) + results += fatjet_results + + # met selection + events, met_results = self[met_selection](events, **kwargs) + results += met_results + + # w_lep selection + events, w_lep_results = self[w_lep_selection](events, **kwargs) + results += w_lep_results + + # derive event weights and add base mask of all events that are not considered bad to "cleanup" step + events, results = self[get_weights_and_no_sel_mask](events, results, **kwargs) + results.steps["cleanup"] = results.steps.cleanup & results.steps["no_sel_mask"] + + results.steps["all_but_trigger_and_bjet"] = ( + results.steps.cleanup & + results.steps.Lepton & + results.steps.AddleptonVeto & + results.steps.Jet & + results.steps.JetLepton2DCut & + results.steps.FatJet & + results.steps.MET & + results.steps.WLepPt + ) + + results.steps["all_but_bjet"] = ( + results.steps.cleanup & + results.steps.LeptonTrigger & + results.steps.Lepton & + results.steps.AddleptonVeto & + results.steps.Jet & + results.steps.JetLepton2DCut & + results.steps.FatJet & + results.steps.MET & + results.steps.WLepPt + ) + + results.steps["all"] = ( + results.steps.all_but_bjet & + results.steps.BJetLeptonDeltaR + ) + + # combined event selection after all steps + event_sel = reduce(and_, results.steps.values()) + results.event = event_sel + + for step, sel in results.steps.items(): + n_sel = ak.sum(sel, axis=-1) + logger.debug(f"{step}: {n_sel}") + + n_sel = ak.sum(event_sel, axis=-1) + if n_sel - ak.sum(results.steps['all']) != 0: + logger.debug(f"__all__: {n_sel}") + logger.warning_once( + f"Number of events passing combined selection does not match number of events passing all individual steps: {n_sel} vs {ak.sum(results.steps['all'])}" # noqa + ) + raise ValueError("Inconsistent event selection results") - # met selection - events, met_results = self[met_selection](events, **kwargs) - results += met_results + # produce features relevant for selection and event weights + if self.dataset_inst.has_tag("is_ttbar"): + events = self[gen_parton_top](events, **kwargs) - # w_lep selection - events, w_lep_results = self[w_lep_selection](events, **kwargs) - results += w_lep_results + if self.dataset_inst.has_tag("is_v_jets"): + events = self[gen_v_boson](events, **kwargs) - # apply jet veto map - events, jet_veto_results = self[jet_veto_map](events, **kwargs) - results += jet_veto_results + events = self[probe_jet](events, **kwargs) - # combined event selection after all steps - event_sel = reduce(and_, results.steps.values()) - results.event = event_sel + # create process ids + events = self[process_ids](events, **kwargs) - for step, sel in results.steps.items(): - n_sel = ak.sum(sel, axis=-1) - print(f"{step}: {n_sel}") + # build categories + events = self[category_ids](events, results=results, **kwargs) - n_sel = ak.sum(event_sel, axis=-1) - print(f"__all__: {n_sel}") + # add cutflow features + events = self[cutflow_features](events, object_masks=results.objects, **kwargs) - # produce features relevant for selection and event weights - if self.dataset_inst.has_tag("is_ttbar"): - events = self[gen_parton_top](events, **kwargs) + # increment stats + events = self[topsf_selection_step_stats](events, results, stats, **kwargs) + events = self[topsf_increment_stats](events, results, stats, **kwargs) + events = self[topsf_selection_hists](events, results, hists, **kwargs) - if self.dataset_inst.has_tag("is_v_jets"): - events = self[gen_v_boson](events, **kwargs) + if self.dataset_inst.is_mc and has_tag("skip_btag_weights", self.config_inst, self.dataset_inst, operator=any): + self[fill_btag_wp_count_hists](events, results.event, results.objects.Jet.Jet, hists, **kwargs) - events = self[probe_jet](events, **kwargs) + def log_fraction(stats_key: str, msg: str | None = None): + if not stats.get(stats_key): + return + if not msg: + msg = "Fraction of {stats_key}" + logger.info(f"{msg}: {(100 * stats[stats_key] / stats['num_events']):.2f}%") - # create process ids - events = self[process_ids](events, **kwargs) + log_fraction("num_negative_weights", "Fraction of negative weights") + log_fraction("num_pu_0", "Fraction of events with pu_weight == 0") + log_fraction("num_pu_100", "Fraction of events with pu_weight >= 100") - # build categories - events = self[category_ids](events, results=results, **kwargs) + # temporary fix for optional types from Calibration (e.g. events.Jet.pt --> ?float32) + # TODO: remove as soon as possible as it might lead to weird bugs when there are none entries in inputs + events = ak.fill_none(events, EMPTY_FLOAT) - # add cutflow features - events = self[cutflow_features](events, object_masks=results.objects, **kwargs) + # remove unused columns + for col in ["GenPart", "GenPartonTop"]: + for field in [ + "genPartIdxMother", + "statusFlags", + "genPartIdxMotherG", + "distinctParentIdxG", + "childrenIdxG", + "distinctChildrenIdxG", + "distinctChildrenDeepIdxG", + ]: + events = remove_ak_column(events, f"{col}.{field}", silent=True) - # increment stats - self[increment_stats](events, results, stats, **kwargs) + # avoid none values in events + events = ak.fill_none(events, EMPTY_FLOAT) - # remove unused columns - for col in ["GenPart", "GenPartonTop"]: - for field in [ - "genPartIdxMother", - "statusFlags", - "genPartIdxMotherG", - "distinctParentIdxG", - "childrenIdxG", - "distinctChildrenIdxG", - "distinctChildrenDeepIdxG", - ]: - events = remove_ak_column(events, f"{col}.{field}", silent=True) + logger.info(f"Selected {ak.sum(results.event)} from {len(events)} events") - # avoid none values in events - events = ak.fill_none(events, EMPTY_FLOAT) + logger.info_once( + "Finished default selector steps:\n" + + "\n".join(run_list) + ) return events, results @default.init def default_init(self: Selector): - dataset_inst = getattr(self, "dataset_inst", None) - if dataset_inst is not None and dataset_inst.is_data: - self.uses |= {json_filter} - # Add shift dependencies self.shifts |= { shift_inst.name for shift_inst in self.config_inst.shifts if shift_inst.has_tag(("jec", "jer")) } + + if hasattr(self, "dataset_inst") and self.dataset_inst.is_mc: + self.uses |= { + fill_btag_wp_count_hists, + } + self.produces |= { + fill_btag_wp_count_hists, + } diff --git a/topsf/selection/fatjet.py b/topsf/selection/fatjet.py index 8a2df9b..957516e 100644 --- a/topsf/selection/fatjet.py +++ b/topsf/selection/fatjet.py @@ -7,6 +7,7 @@ from columnflow.util import maybe_import from columnflow.selection import Selector, SelectionResult, selector +# from columnflow.production.cms.jet import jet_id. # FIXME recalculate jetId in Nano version > v12 from topsf.selection.util import masked_sorted_indices from topsf.selection.lepton import lepton_selection @@ -39,7 +40,8 @@ def fatjet_selection( # select jets fatjet_mask = ( (abs(fatjet.eta) < self.cfg.max_abseta) & - (fatjet.pt > self.cfg.min_pt) + (fatjet.pt > self.cfg.min_pt) & + (fatjet.jetId & self.cfg.jetId == self.cfg.jetId) # jetId bitmask ) fatjet_indices = masked_sorted_indices(fatjet_mask, fatjet.pt) @@ -94,4 +96,5 @@ def fatjet_selection_init(self: Selector) -> None: f"{column}.eta", f"{column}.phi", f"{column}.mass", + f"{column}.jetId", } diff --git a/topsf/selection/hists.py b/topsf/selection/hists.py new file mode 100644 index 0000000..70789b1 --- /dev/null +++ b/topsf/selection/hists.py @@ -0,0 +1,181 @@ +# coding: utf-8 + +""" +Stat-related methods. +""" + +import law +import order as od + +from columnflow.selection import Selector, SelectionResult, selector +from columnflow.selection.stats import increment_stats +# from columnflow.production.cms.btag import btag_weights +from columnflow.production.cms.btag import btag_weights +from topsf.production.weights import event_weights_to_normalize +from columnflow.columnar_util import set_ak_column + +from columnflow.util import maybe_import +from topsf.util import has_tag, IF_MC +from columnflow.hist_util import create_hist_from_variables + +np = maybe_import("numpy") +ak = maybe_import("awkward") + +logger = law.logger.get_logger(__name__) + + +@selector( + uses={increment_stats, event_weights_to_normalize, IF_MC("Jet.hadronFlavour")}, + produces=IF_MC({"ht", "njet", "nhf"}), +) +def topsf_selection_hists( + self: Selector, + events: ak.Array, + results: SelectionResult, + hists: dict, + **kwargs, +) -> ak.Array: + """ + Main selector to create and fill histograms for weight normalization. + """ + # collect important information from the results + no_weights = ak.values_astype(ak.Array(np.ones(len(events))), np.int64) + event_masks = { + "Initial": results.steps.no_sel_mask, + "selected": results.event, + } + if not has_tag("skip_btag_weights", self.config_inst, self.dataset_inst, operator=any): + event_masks["selected_no_bjet"] = results.steps.all_but_bjet + + njet = results.x.n_central_jets + ht = results.x.ht + + # store ht, njet, and nhf for consistency checks + events = set_ak_column(events, "ht", ht) + events = set_ak_column(events, "njet", njet) + + if self.dataset_inst.is_mc: + hadron_flavour = events.Jet[results.objects.Jet.Jet].hadronFlavour + nhf = ak.sum(hadron_flavour == 5, axis=1) + ak.sum(hadron_flavour == 4, axis=1) + events = set_ak_column(events, "nhf", nhf) + + # weight map definition + weight_map = { + # "num" operations + "num_events": no_weights, + } + + if self.dataset_inst.is_mc: + # "sum" operations + weight_map["sum_mc_weight"] = events.mc_weight # weights of all events + + weight_columns = set(self[event_weights_to_normalize].produced_columns) + if not has_tag("skip_btag_weights", self.config_inst, self.dataset_inst, operator=any): + # btag_weights are not produced and therefore need some manual care + weight_columns |= set(self[btag_weights].produced_columns) + + weight_columns = sorted([col.string_nano_column for col in weight_columns]) + + # mc weight times correction weight (with variations) without any selection + for name in weight_columns: + if "weight" not in name: + # skip non-weight columns here + continue + + # TODO: decide whether to keep mc_weight * weight or just weight + weight_map[f"sum_mc_weight_{name}"] = events.mc_weight * events[name] + # weight_map[f"sum_{name}"] = events[name] + + # initialize histograms if not already done + # (NOTE: this only works as long as this is the only selector that adds histograms) + if not hists: + logger.debug("Initializing histograms for weight normalization...") + for key, weight in weight_map.items(): + if "btag_weight" not in key: + hists[key] = create_hist_from_variables(self.steps_variable) + hists[f"{key}_per_process"] = create_hist_from_variables(self.steps_variable, self.process_variable) + if key == "sum_mc_weight" or "btag_weight" in key: + hists[f"{key}_per_process_ht_njet_nhf"] = create_hist_from_variables( + self.steps_variable, + self.process_variable, + self.ht_variable, + self.njet_variable, + self.nhf_variable, + ) + + # fill histograms + for key, weight in weight_map.items(): + logger.debug(f"Filling histogram for {key}...") + for step, mask in event_masks.items(): + logger.debug(f"Filling histogram for {key} and step {step}...") + # TODO: can I fill with single value instead of array of strings? + step_arr = np.array([step] * ak.sum(mask)) + if "btag_weight" not in key: + logger.debug(f"Filling histogram for {key} and step {step} without btag weights...") + hists[key].fill(steps=step_arr, weight=weight[mask]) + hists[f"{key}_per_process"].fill(steps=step_arr, process=events.process_id[mask], weight=weight[mask]) + if step == "selected_no_bjet" and (key == "sum_mc_weight" or "btag_weight" in key): + logger.debug(f"Filling histogram for {key} and step {step} with btag weights...") + # to reduce computing time, only fill the selected_no_bjet mask for btag weights + hists[f"{key}_per_process_ht_njet_nhf"].fill( + steps=step_arr, + process=events.process_id[mask], + ht=ht[mask], + njet=njet[mask], + nhf=nhf[mask], + weight=weight[mask], + ) + + return events + + +@topsf_selection_hists.setup +def topsf_selection_hists_setup( + self: Selector, task: law.Task, reqs: dict, inputs: dict, reader_targets: dict, +) -> None: + self.process_variable = od.Variable( + name="process", + expression="process_id", + aux={"axis_type": "intcategory"}, + ) + self.steps_variable = od.Variable( + name="steps", + aux={"axis_type": "strcategory"}, + ) + self.ht_variable = od.Variable( + name="ht", + binning=[0, 100, 200, 300, 400, 500, 600, 700, 800, 900, 1000, 1100, 1200, 1300, 1450, 1700, 2400], + aux={"axis_type": "variable"}, + ) + self.njet_variable = od.Variable( + name="njet", + binning=[0, 1, 2, 3, 4, 5, 6, 7, 8], + aux={ + "axis_type": "integer", + "axis_kwargs": {"growth": True}, + }, + ) + self.nhf_variable = od.Variable( + name="nhf", + binning=[0, 1, 2, 3, 4], + aux={ + "axis_type": "integer", + "axis_kwargs": {"growth": True}, + }, + ) + + +@topsf_selection_hists.init +def topsf_selection_hists_init(self: Selector) -> None: + if not getattr(self, "dataset_inst", None): + return + + if self.dataset_inst.is_data: + return + + if not has_tag("skip_btag_weights", self.config_inst, self.dataset_inst, operator=any): + self.uses |= {btag_weights} + + if self.dataset_inst.is_mc: + self.uses |= {"mc_weight", "Jet.hadronFlavour"} + self.produces |= {"ht", "njet", "nhf"} diff --git a/topsf/selection/jet.py b/topsf/selection/jet.py index 2d21bc9..41d59c2 100644 --- a/topsf/selection/jet.py +++ b/topsf/selection/jet.py @@ -8,6 +8,7 @@ from columnflow.columnar_util import set_ak_column from columnflow.selection import Selector, SelectionResult, selector +# from columnflow.production.cms.jet import jet_id. # FIXME recalculate jetId in Nano version > v12 from topsf.selection.util import masked_sorted_indices from topsf.production.lepton import choose_lepton @@ -35,7 +36,8 @@ def jet_selection( # select jets jet_mask = ( (abs(jet.eta) < self.cfg.max_abseta) & - (jet.pt > self.cfg.min_pt) + (jet.pt > self.cfg.min_pt) & + (jet.jetId & self.cfg.jetId == self.cfg.jetId) # jetId bitmask ) jet_indices = masked_sorted_indices(jet_mask, jet.pt) @@ -62,6 +64,12 @@ def jet_selection( "LightJet": lightjet_indices, }, }, + aux={ + "jet_mask": jet_mask, + "bjet_mask": bjet_mask, + "ht": ak.sum(events.Jet.pt[jet_mask], axis=1), + "n_central_jets": ak.num(jet_indices), + } ) @@ -84,6 +92,7 @@ def jet_selection_init(self: Selector) -> None: f"{column}.phi", f"{column}.mass", f"{column}.{self.cfg.btag_column}", + f"{column}.jetId", } # Add shift dependencies diff --git a/topsf/selection/met.py b/topsf/selection/met.py index c0bc747..d05be21 100644 --- a/topsf/selection/met.py +++ b/topsf/selection/met.py @@ -26,7 +26,7 @@ def met_selection( met = events[self.cfg.column] # select met - sel_met = (met.pt > self.cfg.min_pt) + sel_met = (met["pt"] > self.cfg.min_pt) # return selection result return events, SelectionResult( diff --git a/topsf/selection/stats.py b/topsf/selection/stats.py new file mode 100644 index 0000000..166846e --- /dev/null +++ b/topsf/selection/stats.py @@ -0,0 +1,178 @@ +# coding: utf-8 + +""" +Custom increment_stats function to keep track of normalized weights. +""" +import law + +from columnflow.selection import Selector, SelectionResult, selector +from columnflow.selection.stats import increment_stats +from columnflow.production.cms.btag import btag_weights +from columnflow.columnar_util import optional_column as optional + +from columnflow.util import maybe_import + +from topsf.production.weights import event_weights_to_normalize +from topsf.util import has_tag + +np = maybe_import("numpy") +ak = maybe_import("awkward") + +logger = law.logger.get_logger(__name__) + + +@selector( + categorizers=None, # pass list of categorizers to evaluate number of (selected) events that fall in this category + uses={increment_stats, optional("mc_weight")}, +) +def topsf_selection_step_stats( + self: Selector, + events: ak.Array, + results: SelectionResult, + stats: dict, + **kwargs, +) -> ak.Array: + """ + Selector to increment stats + """ + weight_map = {} + for step, mask in results.steps.items(): + weight_map[f"num_events_step_{step}"] = mask + if self.dataset_inst.is_mc: + for step, mask in results.steps.items(): + weight_map[f"sum_mc_weight_step_{step}"] = (events.mc_weight, mask) + + if self.categorizers: + # apply list of categorizers and store number of (selected) events in each category + event_mask = results.event + for categorizer in self.categorizers: + if not self.has_dep(categorizer): + # skip categorizer if it is not used + continue + events, mask = self[categorizer](events, results, **kwargs) + weight_map[f"num_events_cat_{categorizer.cls_name}"] = mask + weight_map[f"num_events_selected_cat_{categorizer.cls_name}"] = mask & event_mask + + self[increment_stats]( + events, + results, + stats, + weight_map=weight_map, + group_map={}, + **kwargs, + ) + + return events + + +@topsf_selection_step_stats.init +def topsf_selection_step_stats_init(self: Selector) -> None: + if self.categorizers: + for categorizer in self.categorizers: + self.uses |= {categorizer} + + +@selector( + uses={increment_stats, event_weights_to_normalize}, +) +def topsf_increment_stats( + self: Selector, + events: ak.Array, + results: SelectionResult, + stats: dict, + **kwargs, +) -> ak.Array: + """ + Main selector to increment stats needed for weight normalization + """ + raw_met_column_name = f"Raw{self.config_inst.x.met_selection.default.column}" + # collect important information from the results + no_sel_mask = results.steps.no_sel_mask + event_mask = results.event + if not self.config_inst.has_tag("skip_btag_weights"): + event_mask_no_bjet = results.steps.all_but_bjet + + # weight map definition + weight_map = { + # "num" operations + "num_events_pre_bad_mask": Ellipsis, # all events + "num_events": no_sel_mask, # all events after base mask + "num_events_selected": event_mask, # selected events only + } + if not has_tag("skip_btag_weights", self.config_inst, self.dataset_inst, operator=any): + weight_map["num_events_selected_no_bjet"] = event_mask_no_bjet + + if self.dataset_inst.is_mc: + weight_map["num_negative_weights"] = (events.mc_weight < 0) + weight_map["num_pu_0"] = (events.pu_weight == 0) + weight_map["num_pu_100"] = (events.pu_weight >= 100) + + raw_met_column = events[raw_met_column_name] + weight_map["num_raw_met_isinf"] = (~np.isfinite(raw_met_column.pt)) + weight_map["num_raw_met_isinf_selected"] = (~np.isfinite(raw_met_column.pt) & event_mask) + # "sum" operations + weight_map["sum_mc_weight_pre_bad_mask"] = events.mc_weight # weights of all events + weight_map["sum_mc_weight"] = (events.mc_weight, no_sel_mask) # weights of all events after base mask + weight_map["sum_mc_weight_selected"] = (events.mc_weight, event_mask) # weights of selected events + if not has_tag("skip_btag_weights", self.config_inst, self.dataset_inst, operator=any): + weight_map["sum_mc_weight_no_bjet"] = (events.mc_weight, event_mask_no_bjet) + weight_map["sum_mc_weight_selected_no_bjet"] = (events.mc_weight, event_mask_no_bjet) + + weight_columns = set(self[event_weights_to_normalize].produced_columns) + if not has_tag("skip_btag_weights", self.config_inst, self.dataset_inst, operator=any): + # btag_weights are not produced and therefore need some manual care + weight_columns |= set(self[btag_weights].produced_columns) + + weight_columns = sorted([col.string_nano_column for col in weight_columns]) + + # mc weight times correction weight (with variations) without any selection + for name in weight_columns: + if "weight" not in name: + # skip non-weight columns here + continue + elif name.startswith("btag_weight"): + # btag weights are handled via histograms + continue + + weight_map[f"sum_mc_weight_{name}"] = (events.mc_weight * events[name], no_sel_mask) + + # weights for selected events + weight_map[f"sum_mc_weight_{name}_selected"] = (events.mc_weight * events[name], event_mask) + + if name.startswith("btag_weight"): + # weights for selected events, excluding the bjet selection + weight_map[f"sum_mc_weight_{name}_selected_no_bjet"] = ( + (events.mc_weight * events[name], event_mask_no_bjet) + ) + + group_map = { + "process": { + "values": events.process_id, + "mask_fn": (lambda v: events.process_id == v), + }, + } + + self[increment_stats]( + events, + results, + stats, + weight_map=weight_map, + group_map=group_map, + **kwargs, + ) + + return events + + +@topsf_increment_stats.init +def topsf_increment_stats_init(self: Selector) -> None: + raw_met_column = f"Raw{self.config_inst.x.met_selection.default.column}" + self.uses |= {f"{raw_met_column}.pt", f"{raw_met_column}.phi"} + if not getattr(self, "dataset_inst", None): + return + + if not has_tag("skip_btag_weights", self.config_inst, self.dataset_inst, operator=any): + self.uses |= {btag_weights} + + if self.dataset_inst.is_mc: + self.uses |= {"mc_weight"} diff --git a/topsf/selection/w_lep.py b/topsf/selection/w_lep.py index 71a9434..a8f4108 100644 --- a/topsf/selection/w_lep.py +++ b/topsf/selection/w_lep.py @@ -17,7 +17,7 @@ @selector( - uses={choose_lepton, "MET.pt", "MET.phi"}, + uses={choose_lepton}, ) def w_lep_selection( self: Selector, @@ -32,7 +32,10 @@ def w_lep_selection( events = self[choose_lepton](events, **kwargs) # get leptonic W pt from lepton and missing energy - met = events.MET + if self.config_inst.x.year == 2024: + met = events.PuppiMET + else: + met = events.MET lep = events.Lepton w_lep_pt = np.sqrt( (met.pt * np.cos(met.phi) + lep.pt * np.cos(lep.phi))**2 + @@ -49,3 +52,11 @@ def w_lep_selection( }, objects={}, ) + + +@w_lep_selection.init +def w_lep_selection_init(self: Selector) -> None: + if self.config_inst.x.year == 2024: + self.uses |= {"PuppiMET.pt", "PuppiMET.phi"} + else: + self.uses |= {"MET.pt", "MET.phi"} diff --git a/topsf/selection/wp.py b/topsf/selection/wp.py index 2b01a93..61a56db 100644 --- a/topsf/selection/wp.py +++ b/topsf/selection/wp.py @@ -3,31 +3,35 @@ """ Selection methods related to WP analysis. """ +from __future__ import annotations +import law from operator import and_ from functools import reduce from collections import defaultdict from columnflow.util import maybe_import +from columnflow.columnar_util import EMPTY_FLOAT from columnflow.selection import Selector, SelectionResult, selector -from columnflow.selection.cms.met_filters import met_filters -from columnflow.selection.cms.jets import jet_veto_map -from columnflow.production.cms.mc_weight import mc_weight -from columnflow.production.util import attach_coffea_behavior +# from columnflow.production.cms.jet import jet_id. # FIXME recalculate jetId in Nano version > v12 from columnflow.production.processes import process_ids from topsf.selection.util import masked_sorted_indices -from topsf.selection.default import increment_stats +from topsf.selection.stats import topsf_increment_stats from topsf.production.wp import wp_category_ids from topsf.production.gen_top import gen_parton_top +from topsf.selection.common import get_weights_and_no_sel_mask, pre_selection +from topsf.util import has_tag, record_calls np = maybe_import("numpy") ak = maybe_import("awkward") +logger = law.logger.get_logger(__name__) + @selector( uses={ @@ -41,9 +45,8 @@ def wp_fatjet_selection( **kwargs, ) -> tuple[ak.Array, SelectionResult]: """ - Select AK8 jets that are well separated from the lepton. + Select AK8 jets. """ - # get selection parameters from the config self.cfg = self.config_inst.x.jet_selection.get("ak8", "FatJet") @@ -53,7 +56,8 @@ def wp_fatjet_selection( # select jets fatjet_mask = ( (abs(fatjet.eta) < self.cfg.max_abseta) & - (fatjet.pt > self.cfg.min_pt) + (fatjet.pt > self.cfg.min_pt) & + (fatjet.jetId & self.cfg.jetId == self.cfg.jetId) # jetId bitmask ) # resolve optional msoftdrop range @@ -110,6 +114,7 @@ def wp_fatjet_selection_init(self: Selector) -> None: f"{column}.phi", f"{column}.mass", f"{column}.msoftdrop", + f"{column}.jetId", } # if ttbar, produce parton-level top quarks @@ -117,27 +122,25 @@ def wp_fatjet_selection_init(self: Selector) -> None: dataset_inst = getattr(self, "dataset_inst", None) if dataset_inst is not None and dataset_inst.has_tag("is_ttbar"): self.uses.add(gen_parton_top) + self.produces.add(gen_parton_top) @selector( uses={ - attach_coffea_behavior, - mc_weight, + pre_selection, wp_category_ids, process_ids, - met_filters, wp_fatjet_selection, - jet_veto_map, - increment_stats, + topsf_increment_stats, + get_weights_and_no_sel_mask, }, produces={ + pre_selection, wp_category_ids, - mc_weight, process_ids, - met_filters, wp_fatjet_selection, - jet_veto_map, - increment_stats, + topsf_increment_stats, + get_weights_and_no_sel_mask, }, exposed=True, ) @@ -148,51 +151,73 @@ def wp( msoftdrop_range=None, **kwargs, ) -> tuple[ak.Array, SelectionResult]: - # ensure coffea behavior - events = self[attach_coffea_behavior](events, **kwargs) - - # prepare the selection results that are updated at every step - results = SelectionResult() - - # MET filters - events, met_filters_results = self[met_filters](events, **kwargs) - results += met_filters_results - - # fatjet selection - events, wp_fatjet_results = self[wp_fatjet_selection]( - events, - msoftdrop_range=msoftdrop_range, - **kwargs, + run_list = [] + with record_calls(self, run_list): + events, results = self[pre_selection](events, stats, **kwargs) + + # fatjet selection + events, wp_fatjet_results = self[wp_fatjet_selection]( + events, + msoftdrop_range=msoftdrop_range, + **kwargs, + ) + results += wp_fatjet_results + + # derive event weights and add base mask of all events that are not considered bad to "cleanup" step + events, results = self[get_weights_and_no_sel_mask](events, results, **kwargs) + results.steps["cleanup"] = results.steps.cleanup & results.steps["no_sel_mask"] + + results.steps["all"] = ( + results.steps.cleanup & + results.steps.FatJet + ) + + # combined event selection after all steps + event_sel = reduce(and_, results.steps.values()) + results.event = event_sel + + for step, sel in results.steps.items(): + n_sel = ak.sum(sel, axis=-1) + logger.debug(f"{step}: {n_sel}") + + n_sel = ak.sum(event_sel, axis=-1) + logger.debug(f"__all__: {n_sel}") + + # produce features relevant for selection and event weights + if self.dataset_inst.has_tag("is_ttbar"): + events = self[gen_parton_top](events, **kwargs) + + # build categories + events = self[wp_category_ids](events, **kwargs) + + # create process ids + events = self[process_ids](events, **kwargs) + + # increment stats + events = self[topsf_increment_stats](events, results, stats, **kwargs) + # no custom hists needed, because we don't do b tagging in wp analysis + + def log_fraction(stats_key: str, msg: str | None = None): + if not stats.get(stats_key): + return + if not msg: + msg = "Fraction of {stats_key}" + logger.info(f"{msg}: {(100 * stats[stats_key] / stats['num_events']):.2f}%") + + log_fraction("num_negative_weights", "Fraction of negative weights") + log_fraction("num_pu_0", "Fraction of events with pu_weight == 0") + log_fraction("num_pu_100", "Fraction of events with pu_weight >= 100") + + # temporary fix for optional types from Calibration (e.g. events.Jet.pt --> ?float32) + # TODO: remove as soon as possible as it might lead to weird bugs when there are none entries in inputs + events = ak.fill_none(events, EMPTY_FLOAT) + + logger.info(f"Selected {ak.sum(results.event)} from {len(events)} events") + + logger.info_once( + "Finished WP selection steps:\n" + + "\n".join(run_list) ) - results += wp_fatjet_results - - # apply jet veto map - events, jet_veto_results = self[jet_veto_map](events, **kwargs) - results += jet_veto_results - - # combined event selection after all steps - event_sel = reduce(and_, results.steps.values()) - results.event = event_sel - - for step, sel in results.steps.items(): - n_sel = ak.sum(sel, axis=-1) - print(f"{step}: {n_sel}") - - n_sel = ak.sum(event_sel, axis=-1) - print(f"__all__: {n_sel}") - - # produce features relevant for selection and event weights - if self.dataset_inst.has_tag("is_ttbar"): - events = self[gen_parton_top](events, **kwargs) - - # build categories - events = self[wp_category_ids](events, **kwargs) - - # create process ids - events = self[process_ids](events, **kwargs) - - # increment stats - self[increment_stats](events, results, stats, **kwargs) return events, results @@ -211,96 +236,6 @@ def wp_init(self: Selector): self.produces.add(gen_parton_top) -@selector( - uses={ - attach_coffea_behavior, - mc_weight, - wp_category_ids, - process_ids, - met_filters, - wp_fatjet_selection, - increment_stats, - }, - produces={ - wp_category_ids, - mc_weight, - process_ids, - met_filters, - wp_fatjet_selection, - increment_stats, - }, - exposed=True, -) -def wp_wo_jvm( - self: Selector, - events: ak.Array, - stats: defaultdict, - msoftdrop_range=None, - **kwargs, -) -> tuple[ak.Array, SelectionResult]: - # ensure coffea behavior - events = self[attach_coffea_behavior](events, **kwargs) - - # prepare the selection results that are updated at every step - results = SelectionResult() - - # MET filters - events, met_filters_results = self[met_filters](events, **kwargs) - results += met_filters_results - - # fatjet selection - events, wp_fatjet_results = self[wp_fatjet_selection]( - events, - msoftdrop_range=msoftdrop_range, - **kwargs, - ) - results += wp_fatjet_results - - # # apply jet veto map - # events, jet_veto_results = self[jet_veto_map](events, **kwargs) - # results += jet_veto_results - - # combined event selection after all steps - event_sel = reduce(and_, results.steps.values()) - results.event = event_sel - - for step, sel in results.steps.items(): - n_sel = ak.sum(sel, axis=-1) - print(f"{step}: {n_sel}") - - n_sel = ak.sum(event_sel, axis=-1) - print(f"__all__: {n_sel}") - - # produce features relevant for selection and event weights - if self.dataset_inst.has_tag("is_ttbar"): - events = self[gen_parton_top](events, **kwargs) - - # build categories - events = self[wp_category_ids](events, **kwargs) - - # create process ids - events = self[process_ids](events, **kwargs) - - # increment stats - self[increment_stats](events, results, stats, **kwargs) - - return events, results - - -@wp_wo_jvm.init -def wp_wo_jvm_init(self: Selector): - dataset_inst = getattr(self, "dataset_inst", None) - if dataset_inst is not None and dataset_inst.is_data: - raise RuntimeError("selector 'wp' should not be run on data") - - # if ttbar, produce parton-level top quarks - # (relevant for jet selection) - dataset_inst = getattr(self, "dataset_inst", None) - if dataset_inst and dataset_inst.has_tag("is_ttbar"): - self.uses.add(gen_parton_top) - self.produces.add(gen_parton_top) - - # selector for msoftdrop region around top mass msoftdrop_range = (105, 210) diff --git a/topsf/tasks/corrections.py b/topsf/tasks/corrections.py new file mode 100644 index 0000000..15000dd --- /dev/null +++ b/topsf/tasks/corrections.py @@ -0,0 +1,377 @@ +""" +Tasks for creating correctionlib files. +""" + +import law +import luigi + +from functools import cached_property + +from columnflow.tasks.framework.base import Requirements, ShiftTask +from columnflow.tasks.framework.mixins import ( + SelectorClassMixin, CalibratorClassesMixin, + DatasetsProcessesMixin, +) +from columnflow.tasks.framework.remote import RemoteWorkflow +from columnflow.tasks.selection import MergeSelectionStats +from columnflow.util import maybe_import, dev_sandbox + +# from columnflow.config_util import get_datasets_from_process +from topsf.tasks.base import TopSFTask + +ak = maybe_import("awkward") +np = maybe_import("numpy") + + +logger = law.logger.get_logger(__name__) +logger_dev = law.logger.get_logger(f"{__name__}-dev") + + +class GetBtagNormalizationSF( + TopSFTask, + SelectorClassMixin, + CalibratorClassesMixin, + DatasetsProcessesMixin, + ShiftTask, + law.LocalWorkflow, + RemoteWorkflow, +): + """ + This task computes btag re-normalization scale factors based on the selection statistics. + It merges the selection statistics across datasets and processes, and computes the scale factors + based on the provided rescaling mode (either "nevents" or "xs"). The scale factors are stored + as correctionlib evaluators in a JSON file. + + The scale factors are computed in different binnings, such as: + - `ht`, `njet`, `nhf` + - `ht`, `njet` + - `ht` + - `njet` + - `nhf` + + The output is a JSON file containing the scale factors, which can be used for re-normalization + of btag weights in the analysis. + + Resources: + - https://btv-wiki.docs.cern.ch/PerformanceCalibration/shapeCorrectionSFRecommendations/#effect-on-event-yields + """ + resolution_task_cls = MergeSelectionStats + + single_config = True + reqs = Requirements( + RemoteWorkflow.reqs, + MergeSelectionStats=MergeSelectionStats, + ) + + store_as_dict = False + + reweighting_step = "selected_no_bjet" + + rescale_mode = luigi.ChoiceParameter( + default="nevents", + choices=("nevents", "xs"), + ) + base_key = luigi.ChoiceParameter( + default="rescaled_sum_mc_weight", + # NOTE: "num_events" does not work because I did not store the corresponding key in the stats :/ + choices=("rescaled_sum_mc_weight", "sum_mc_weight", "num_events"), + ) + + # default sandbox, might be overwritten by selector function + sandbox = dev_sandbox(law.config.get("analysis", "default_columnar_sandbox")) + + processes = DatasetsProcessesMixin.processes.copy( + default=("tt", "st"), + description="Processes to consider for the scale factors", + add_default_to_description=True, + ) + + # njet_overflow = 7 + # nhf_overflow = 4 + njet_overflow = luigi.IntParameter( + default=6, + description="Maximum number of jets to consider for the scale factors. If None, no overflow bin is applied.", + ) + nhf_overflow = luigi.IntParameter( + default=4, + description="Maximum number of jets to consider for the scale factors. If None, no overflow bin is applied.", + ) + + def create_branch_map(self): + # single branch without payload + return {0: None} + + @cached_property + def process_insts(self): + process_insts = [self.config_inst.get_process(process) for process in self.processes] + return process_insts + + @cached_property + def dataset_insts(self): + dataset_insts = [self.config_inst.get_dataset(dataset) for dataset in self.datasets] + return dataset_insts + + def workflow_requires(self): + reqs = super().workflow_requires() + reqs["selection_stats"] = { + dataset.name: self.reqs.MergeSelectionStats.req_different_branching( + self, + dataset=dataset.name, + branch=-1, + ) + for dataset in self.dataset_insts + } + return reqs + + def requires(self): + reqs = {} + reqs["selection_stats"] = { + dataset.name: self.reqs.MergeSelectionStats.req_different_branching( + self, + dataset=dataset.name, + branch=-1, + ) + for dataset in self.dataset_insts + } + return reqs + + def store_parts(self): + parts = super().store_parts() + + processes_repr = "__".join(self.processes) + parts.insert_before("version", "processes", processes_repr) + + significant_params = (self.rescale_mode, self.base_key) + parts.insert_before("version", "params", "__".join(significant_params)) + + overflow_repr = "" + if self.njet_overflow > 0: + overflow_repr += f"njet{self.njet_overflow}" + if self.nhf_overflow > 0: + overflow_repr += f"_nhf{self.nhf_overflow}" + if overflow_repr: + parts.insert_before("version", "overflow", overflow_repr) + + return parts + + def output(self): + return { + "btag_renormalization_sf": self.target("btag_renormalization_sf.json"), + "btag_renormalization_sf_plot": self.target("btag_renormalization_sf_plot.pdf", optional=True), + "plots": self.target("plots", dir=True, optional=True), + } + + def reduce_hist(self, h, mode: list[str]): + """ + Helper function that reduces the histogram to the requested axes based on the mode. + """ + h = h[{"process": sum, "steps": self.reweighting_step}] + + # check validity of mode + ax_names = [ax.name for ax in h.axes] + if not all(ax in ax_names for ax in mode): + raise ValueError(f"Invalid mode {mode} for axes {ax_names}") + + # remove axes not in mode + for ax in h.axes: + if ax.name not in mode: + h = h[{ax.name: sum}] + + return h + + def apply_overflow_bin(self, h): + # from columnflow.plotting.plot_util import use_flow_bins + ax_names = [ax.name for ax in h.axes] + if self.njet_overflow > 0 and "njet" in ax_names: + h = h[{"njet": slice(0, self.njet_overflow + 1)}] + if self.nhf_overflow > 0 and "nhf" in ax_names: + h = h[{"nhf": slice(0, self.nhf_overflow + 1)}] + return h + + def run(self): + import correctionlib + import correctionlib.convert + import hist + outputs = self.output() + inputs = self.input() + + # load the selection merged_hists + hists_per_dataset = { + dataset: inp["collection"][0]["hists"].load(formatter="pickle") + for dataset, inp in inputs["selection_stats"].items() + } + + def safe_div(num, den): + return np.where( + (num > 0) & (den > 0), + num / den, + 1.0, + ) + + # rescale the histograms + if "rescaled" in self.base_key: + for dataset, hists in hists_per_dataset.items(): + process = self.config_inst.get_dataset(dataset).processes.get_first() + if self.rescale_mode == "xs": + # scale such that the sum of weights is the cross section + xs = process.get_xsec(self.config_inst.campaign.ecm).nominal + dataset_factor = xs / hists["sum_mc_weight"][{"steps": "Initial"}].value + elif self.rescale_mode == "nevents": + # scale such that mean weight is 1 + n_events = hists["num_events"][{"steps": self.reweighting_step}].value + dataset_factor = n_events / hists["sum_mc_weight"][{"steps": self.reweighting_step}].value + else: + raise ValueError(f"Invalid rescale mode {self.rescale_mode}") + for key in tuple(hists.keys()): + if "sum" not in key: + continue + h = hists[key].copy() * dataset_factor + hists[f"rescaled_{key}"] = h + + # if necessary, merge the histograms across datasets + if len(hists_per_dataset) > 1: + merged_hists = {} + from columnflow.tasks.selection import MergeSelectionStats + for dataset, hists in hists_per_dataset.items(): + MergeSelectionStats.merge_counts(merged_hists, hists) + else: + merged_hists = hists_per_dataset[self.dataset_insts[0].name] + + # initialize the scale factor map + sf_map = {} + + # TODO: mode "" (reduce everything to a single bin) not yet working + for mode in ( + ("ht", "njet", "nhf"), + ("ht", "njet"), + ("ht",), + ("njet",), + ("nhf",), + # ("",), + ): + mode_str = "_".join(mode) + numerator = merged_hists[f"{self.base_key}_per_process_ht_njet_nhf"] + numerator = self.reduce_hist(numerator, mode) + numerator = self.apply_overflow_bin(numerator).values() + + for key in merged_hists.keys(): + if ( + not key.startswith(f"{self.base_key}_btag_weight") or + not key.endswith("_per_process_ht_njet_nhf") + ): + continue + + # extract the weight name + weight_name = key.replace(f"{self.base_key}_", "").replace("_per_process_ht_njet_nhf", "") + + # create the scale factor histogram + h = merged_hists[key] + h = self.reduce_hist(h, mode) + h = self.apply_overflow_bin(h) + denominator = h.values() + + # get axes for the output histogram + out_axes = [] + for ax in h.axes: + if isinstance(ax, hist.axis.Variable): + out_axes.append(ax) + elif isinstance(ax, hist.axis.Integer): + # convert from Integer to Variable to allow clamping + out_axes.append(hist.axis.Variable(ax.edges, name=ax.name, label=ax.label)) + else: + raise ValueError(f"Unsupported axis type {type(ax)}") + + # calculate the scale factor and store it as a correctionlib evaluator + sf = safe_div(numerator, denominator) + sfhist = hist.Hist(*out_axes, data=sf) + sfhist.name = f"{mode_str}_{weight_name}" + sfhist.label = "out" + + if mode_str == "ht_njet_nhf": + self.plot_ht_njet_hft_btag_weight(sfhist) + + # import correctionlib.convert + btag_renormalization = correctionlib.convert.from_histogram(sfhist) + btag_renormalization.description = f"{weight_name} per {mode_str} re-normalization" + + # set overflow bins behavior (default is to raise an error when out of bounds) + if any(isinstance(ax, hist.axis.Variable) for ax in out_axes): + btag_renormalization.data.flow = "clamp" + + # store the evaluator + sf_map[sfhist.name] = btag_renormalization + + # create correction set and store it + logger.debug(f"Storing corrections with keys {sf_map.keys()}") + cset = correctionlib.schemav2.CorrectionSet( + schema_version=2, + description="btag re-normalization SFs", + corrections=list(sf_map.values()), + ) + cset_json = cset.json(exclude_unset=True) + if self.store_as_dict: + import json + cset_json = json.loads(cset_json) + + outputs["btag_renormalization_sf"].dump( + cset_json, + formatter="json", + ) + + skip_variations = True + + def plot_ht_njet_hft_btag_weight(self, sfhist): + """ + Plot the btag weight SFs for the ht_njet_nhf mode. + """ + if self.skip_variations and sfhist.name != "ht_njet_nhf_btag_weight": + return + logger.debug(f"Plotting btag weight SFs for sfhist {sfhist.name}") + output = self.output()["plots"] + import matplotlib as mpl + import matplotlib.pyplot as plt + mpl.use("Agg") # Use a non-interactive backend + + nhf_edges = sfhist.axes["nhf"].edges + njet_edges = sfhist.axes["njet"].edges + ht_edges = sfhist.axes["ht"].edges + + # # create one grid of figures with one figure per njet and nhf bin + fig, axs = plt.subplots( + len(njet_edges) - 1, len(nhf_edges) - 1, + figsize=(len(nhf_edges) * 3, len(njet_edges) * 3), + constrained_layout=True, + gridspec_kw={ + "left": 0.05, "right": 0.95, + "bottom": 0.05, "top": 0.95, + "hspace": 0.30, "wspace": 0.30, + }, + sharex=True, + # sharey=True, + ) + + # create one 1D plot per nhf and njet bin + for nhf_idx, nhf_edge in enumerate(nhf_edges[:-1]): + for njet_idx, njet_edge in enumerate(njet_edges[:-1]): + logger.info(f"Plotting SF for NHF: {nhf_edge}, NJet: {njet_edge}") + # fig, ax = plt.subplots(figsize=(10, 6)) + ax = axs[njet_idx, nhf_idx] + sfbin = sfhist[{"nhf": nhf_idx, "njet": njet_idx}] + sfbin.plot1d( + ax=ax, + # histtype="fill", + histtype="step", + yerr=False, + ) + ax.set_title(f"NHF: {nhf_edge}, NJet: {njet_edge}") + ax.set_xlabel("HT (GeV)") + ax.set_ylabel("SF") + ax.set(xlim=(ht_edges[0], ht_edges[-1])) + # output.child(f"btag_sf_ht_njet_nhf_{nhf_idx}_{njet_idx}.pdf", type="f").dump(fig, formatter="mpl") + # plt.close(fig) + + # # save the plot + plt.tight_layout() + output.child(f"{sfhist.name}_plot.pdf", type="f").dump(fig, formatter="mpl") + if sfhist.name == "ht_njet_nhf_btag_weight": + self.output()["btag_renormalization_sf_plot"].dump(fig, formatter="mpl") diff --git a/topsf/tasks/inference.py b/topsf/tasks/inference.py index f169cc7..9fed826 100644 --- a/topsf/tasks/inference.py +++ b/topsf/tasks/inference.py @@ -17,6 +17,8 @@ from topsf.tasks.base import TopSFTask +logger = law.logger.get_logger(__name__) + def multi_string_repr(strings, max_len=1, sep="_"): """Create a unique representation of a sequence of strings.""" @@ -67,28 +69,48 @@ def _requires_cat_obj(self, cat_obj: DotDict, merge_variables: bool = False, **r variables = (config_data.variable,) # add merged shifted histograms for mc - reqs[config_inst.name] = { - proc_obj.name: { - dataset: self.reqs.MergeShiftedHistograms.req_different_branching( - self, - config=config_inst.name, - dataset=dataset, - shift_sources=tuple( - param_obj.config_data[config_inst.name].shift_source - for param_obj in proc_obj.parameters + reqs[config_inst.name] = {} + + for proc_obj in cat_obj.processes: + if config_inst.name in proc_obj.config_data and not proc_obj.is_dynamic: + + # Create the sub-dict for this proc_obj + reqs[config_inst.name][proc_obj.name] = {} + + # Loop over datasets + for dataset in self.get_mc_datasets(config_inst, proc_obj): + + # Build shift_sources tuple + shift_sources = [] + for param_obj in proc_obj.parameters: + if config_inst.name not in param_obj.config_data: + continue if ( - config_inst.name in param_obj.config_data and - self.inference_model_inst.require_shapes_for_parameter(param_obj) - ) - ), - variables=variables, - **req_kwargs, - ) - for dataset in self.get_mc_datasets(config_inst, proc_obj) - } - for proc_obj in cat_obj.processes - if config_inst.name in proc_obj.config_data and not proc_obj.is_dynamic - } + (param_obj.type.is_shape and not param_obj.transformations.any_from_rate) or + (param_obj.type.is_rate and param_obj.transformations.any_from_shape) + ): + shift_sources.append(param_obj.config_data[config_inst.name].shift_source) + # if (config_inst.name in param_obj.config_data and + # self.inference_model_inst.require_shapes_for_parameter(param_obj)): + # print(f"Adding shift source for parameter: {param_obj.name}") + # shift_sources.append( + # param_obj.config_data[config_inst.name].shift_source + # ) + shift_sources = tuple(shift_sources) + ("nominal",) + + # Compute the value + value = self.reqs.MergeShiftedHistograms.req_different_branching( + self, + config=config_inst.name, + dataset=dataset, + shift_sources=shift_sources, + variables=variables, + **req_kwargs, + ) + + # Store it + reqs[config_inst.name][proc_obj.name][dataset] = value + # add merged histograms for data, but only if # - data in that category is not faked from mc, or # - at least one process object is dynamic (that usually means data-driven) @@ -97,11 +119,12 @@ def _requires_cat_obj(self, cat_obj: DotDict, merge_variables: bool = False, **r (data_datasets := self.get_data_datasets(config_inst, cat_obj)) ): reqs[config_inst.name]["data"] = { - dataset: self.reqs.MergeHistograms.req_different_branching( + dataset: self.reqs.MergeShiftedHistograms.req_different_branching( self, config=config_inst.name, dataset=dataset, variables=variables, + shift_sources=("nominal",), **req_kwargs, ) for dataset in data_datasets @@ -153,6 +176,31 @@ def workflow_requires(self): hist_reqs[config_name][proc_name].setdefault(dataset_name, set()).add(task) return reqs + # def requires(self): + # # retrieving reqs from parent class didn't work due to `bypass_branch_requirements` + # # in `PlotVariablesBaseSingleShift`; copy-pasted the relevant part of the code here + # reqs = {} + + # for config_inst, datasets in zip(self.config_insts, self.datasets): + # logger.debug(f"datasets to plot for config '{config_inst.name}': {datasets}") + # reqs[config_inst.name] = {} + # for d in datasets: + # if d not in config_inst.datasets: + # logger.warning( + # f"dataset '{d}' not found in config '{config_inst.name}', skipping it", + # ) + # continue + # reqs[config_inst.name][d] = self.reqs.MergeHistograms.req_different_branching( + # self, + # config=config_inst.name, + # shift=self.global_shift_insts[config_inst].name, + # dataset=d, + # branch=-1, + # _prefer_cli={"variables"}, + # ) + + # return reqs + def requires(self): cat_objs = list(self.branch_map.values())[0]["categories"] reqs = {} @@ -257,7 +305,11 @@ def run(self): if h_proc is None: h_proc = h else: - h_proc += h + try: + h_proc += h + except: + q = __import__('functools').partial(__import__('os')._exit, 0) + __import__('IPython').embed() # there must be a histogram if h_proc is None: @@ -274,7 +326,10 @@ def run(self): if proc_obj: for param_obj in proc_obj.parameters: # skip the parameter when varied hists are not needed - if not self.inference_model_inst.require_shapes_for_parameter(param_obj): + if not ( + (param_obj.type.is_shape and not param_obj.transformations.any_from_rate) or + (param_obj.type.is_rate and param_obj.transformations.any_from_shape) + ): continue # store the varied hists # hists[proc_obj_name] = {} diff --git a/topsf/tasks/plotting.py b/topsf/tasks/plotting.py index dcbec57..1793176 100644 --- a/topsf/tasks/plotting.py +++ b/topsf/tasks/plotting.py @@ -13,6 +13,7 @@ # from columnflow.util import dict_add_strict from topsf.tasks.base import TopSFTask +logger = law.logger.get_logger(__name__) class PlotVariables1D( @@ -29,6 +30,31 @@ class PlotVariables1D( description="Use pretty legend", ) + def requires(self): + # retrieving reqs from parent class didn't work due to `bypass_branch_requirements` + # in `PlotVariablesBaseSingleShift`; copy-pasted the relevant part of the code here + reqs = {} + + for config_inst, datasets in zip(self.config_insts, self.datasets): + logger.debug(f"datasets to plot for config '{config_inst.name}': {datasets}") + reqs[config_inst.name] = {} + for d in datasets: + if d not in config_inst.datasets: + logger.warning( + f"dataset '{d}' not found in config '{config_inst.name}', skipping it", + ) + continue + reqs[config_inst.name][d] = self.reqs.MergeHistograms.req_different_branching( + self, + config=config_inst.name, + shift=self.global_shift_insts[config_inst].name, + dataset=d, + branch=-1, + _prefer_cli={"variables"}, + ) + + return reqs + @law.decorator.log @view_output_plots def run(self): @@ -121,6 +147,22 @@ def run(self): " - requested variable requires columns that were missing during histogramming\n" + " - selected --processes did not match any value on the process axis of the input histogram", ) + if all(not _hists for _hists in hists.values()): + raise Exception( + "no histograms found to plot after processing; possible reasons:\n" + + " - requested variable requires columns that were missing during histogramming\n" + + " - selected --processes did not match any value on the process axis of the input histogram\n" + + "configs with no histograms: " + + ", ".join(config_inst.name for config_inst, _hists in hists.items() if not _hists) + ) + if any(not _hists for _hists in hists.values()): + logger.warning( + "some configs do not have any histograms to plot after processing; possible reasons:\n" + + " - requested variable requires columns that were missing during histogramming\n" + + " - selected --processes did not match any value on the process axis of the input histogram\n" + + "configs with no histograms: " + + ", ".join(config_inst.name for config_inst, _hists in hists.items() if not _hists) + ) # merge configs if multiconfig if len(self.config_insts) != 1: @@ -146,7 +188,7 @@ def run(self): for process_inst in sorted(hists, key=process_insts.index) ) - # temporarily use a merged luminostiy value, assigned to the first config + # temporarily use a merged luminosity value, assigned to the first config config_inst = self.config_insts[0] lumi = sum([_config_inst.x.luminosity for _config_inst in self.config_insts]) with law.util.patch_object(config_inst.x, "luminosity", lumi): diff --git a/topsf/tasks/selresults.py b/topsf/tasks/selresults.py new file mode 100644 index 0000000..b8e3c8b --- /dev/null +++ b/topsf/tasks/selresults.py @@ -0,0 +1,55 @@ +# coding: utf-8 +""" +Custom tasks checking selection results. +""" + +from columnflow.tasks.framework.mixins import ( + CalibratorsMixin, SelectorMixin, +) +from columnflow.tasks.selection import MergeSelectionStats +import json + +class CheckSelectionResults( + TopSFTask, + CalibratorsMixin, + SelectorMixin, +): + run_command_in_tmp = False + + # upstream requirements + reqs = Requirements( + RemoteWorkflow.reqs, + MergeSelectionStats=MergeSelectionStats, + ) + + def workflow_requires(self): + reqs = super().workflow_requires() + + reqs["selection_stats"] = self.requires_from_branch() + + return reqs + + def requires(self): + reqs = { + "selection_stats": self.reqs.MergeSelectionStats.req(self), + } + return reqs + + def create_branch_map(self): + cats = list(self.inference_model_inst.categories) + + return [ + { + "categories": cats, + }, + ] + + @law.decorator.log + @law.decorator.safe_output + def run(self): + + input_stats = self.input()["selection_stats"]["stats"].path + with open(input_stats, "r") as f: + stats = json.load(f) + q = __import__('functools').partial(__import__('os')._exit, 0) + __import__('IPython').embed() diff --git a/topsf/tasks/wp/efficiency.py b/topsf/tasks/wp/efficiency.py index d2fbd95..d72696f 100644 --- a/topsf/tasks/wp/efficiency.py +++ b/topsf/tasks/wp/efficiency.py @@ -18,6 +18,8 @@ np = maybe_import("numpy") +logger = law.logger.get_logger(__name__) + class EfficiencyVariablesMixin(VariablesMixin): @@ -134,11 +136,40 @@ class PlotEfficiencyBase( the *processes* parameter. """ + signal_tag = luigi.Parameter( + description="datasets marked with this tag are considered signal, otherwise background", + ) + # upstream requirements reqs = Requirements( PlotVariablesBaseSingleShift.reqs, ) + def requires(self): + # retrieving reqs from parent class didn't work due to `bypass_branch_requirements` + # in `PlotVariablesBaseSingleShift`; copy-pasted the relevant part of the code here + reqs = {} + + for config_inst, datasets in zip(self.config_insts, self.datasets): + logger.debug(f"datasets to plot for config '{config_inst.name}': {datasets}") + reqs[config_inst.name] = {} + for d in datasets: + if d not in config_inst.datasets: + logger.warning( + f"dataset '{d}' not found in config '{config_inst.name}', skipping it", + ) + continue + reqs[config_inst.name][d] = self.reqs.MergeHistograms.req_different_branching( + self, + config=config_inst.name, + shift=self.global_shift_insts[config_inst].name, + dataset=d, + branch=-1, + _prefer_cli={"variables"}, + ) + + return reqs + plot_function = PlotBase.plot_function.copy( default="topsf.plotting.plot_roc_curve.plot_efficiency", add_default_to_description=True, @@ -171,6 +202,8 @@ def process_hists(self, hists): @law.decorator.safe_output def run(self): import hist + if len(self.input().items()) == 0: + raise Exception("No input found.") # get the shifts to extract and plot plot_shifts = law.util.make_list(self.get_plot_shifts()) @@ -189,18 +222,24 @@ def run(self): # histogram data, summed up for background and # signal processes hists = {} + side = self.config_inst.get_variable(self.efficiency_variable).x("signal_side", "unknown") - with self.publish_step(f"plotting ROC curve for variable {self.branch_data.variable} in category {category_inst.name}"): # noqa + with self.publish_step(f"plotting ROC curve for variable {self.branch_data.variable} in category {category_inst.name}, signal side {side}"): # noqa for i, config_inst in enumerate(self.config_insts): + logger.debug(f"processing config '{config_inst.name}'") # histogram data per process hists_config = {} hists[config_inst] = hists_config - for config, ds_inp in self.input().items(): - if config_inst.name == config: + for cfg, ds_inp in self.input().items(): + logger.debug(f"processing input for config '{cfg}'") + if config_inst.name == cfg: for dataset, inp in ds_inp.items(): dataset_inst = config_inst.get_dataset(dataset) # skip when the dataset does not contain any leaf process if not any(map(dataset_inst.has_process, leaf_process_insts)): + logger.warning( + f"dataset {dataset} does not contain any of the leaf processes {', '.join(p.name for p in leaf_process_insts)}, skipping it", + ) continue h_in = inp["collection"][0]["hists"].targets[self.branch_data.variable].load(formatter="pickle") # noqa: E501 @@ -241,8 +280,14 @@ def run(self): hists[config_inst][key] = hists_config[key] except KeyError: # if the key is not present, skip it + logger.warning( + f"histogram with key '{key}' not found for config '{config_inst.name}', skipping it", + ) continue - + else: + logger.debug( + f"config '{config_inst.name}' not currently requested for plotting, skipping it", + ) # there should be hists to plot if not hists: raise Exception( @@ -250,6 +295,22 @@ def run(self): " - requested variable requires columns that were missing during histogramming\n" + " - selected --processes did not match any value on the process axis of the input histogram", ) + if all(not _hists for _hists in hists.values()): + raise Exception( + "no histograms found to plot after processing; possible reasons:\n" + + " - requested variable requires columns that were missing during histogramming\n" + + " - selected --processes did not match any value on the process axis of the input histogram\n" + + "configs with no histograms: " + + ", ".join(config_inst.name for config_inst, _hists in hists.items() if not _hists) + ) + if any(not _hists for _hists in hists.values()): + logger.warning( + "some configs do not have any histograms to plot after processing; possible reasons:\n" + + " - requested variable requires columns that were missing during histogramming\n" + + " - selected --processes did not match any value on the process axis of the input histogram\n" + + "configs with no histograms: " + + ", ".join(config_inst.name for config_inst, _hists in hists.items() if not _hists) + ) # merge configs if multiconfig if len(self.config_insts) != 1: @@ -307,6 +368,7 @@ def run(self): totals=totals, config_inst=config_inst, category_inst=category_inst.copy_shallow(), + signal_side=side, **self.get_plot_parameters(), ) @@ -352,7 +414,13 @@ def plot_mode(self): return self.efficiency_type def get_hists_key(self, dataset_inst): - return self.efficiency_type + # return self.efficiency_type + return ( + "signal" + if dataset_inst.has_tag(self.signal_tag) + else "background" + ) + class PlotROCCurve( @@ -368,16 +436,21 @@ class PlotROCCurve( setting the *processes* parameter. """ - signal_tag = luigi.Parameter( - description="datasets marked with this tag are considered signal, otherwise background", - ) - def get_hists_key(self, dataset_inst): return ( "signal" if dataset_inst.has_tag(self.signal_tag) else "background" ) + + @property + def plot_mode(self): + return "roc" + + plot_function = PlotBase.plot_function.copy( + default="topsf.plotting.plot_roc_curve.plot_roc_curve", + add_default_to_description=True, + ) class PlotROCCurveByVariable( diff --git a/topsf/util.py b/topsf/util.py index 2da6512..04e2d0a 100644 --- a/topsf/util.py +++ b/topsf/util.py @@ -6,7 +6,13 @@ import law +from contextlib import contextmanager +from functools import wraps +import time + +from columnflow.types import Any from columnflow.util import maybe_import +from columnflow.columnar_util import ArrayFunction, deferred_column np = maybe_import("numpy") @@ -25,3 +31,54 @@ def has_tag(tag, *container, operator: callable = any) -> bool: """ values = [inst.has_tag(tag) for inst in container] return operator(values) + + +@deferred_column +def IF_MC(self: ArrayFunction.DeferredColumn, func: ArrayFunction) -> Any | set[Any]: + if getattr(func, "dataset_inst", None) is None: + return self.get() + + return self.get() if func.dataset_inst.is_mc else None + + +@contextmanager +def record_calls(inst, run_list): + cls = type(inst) + orig_getitem = cls.__getitem__ + + def wrapped_getitem(self, key): + prod = orig_getitem(self, key) + name = getattr(key, "__name__", None) or str(key) + + @wraps(prod) + def wrapped(*args, **kwargs): + start = time.perf_counter() + + result = prod(*args, **kwargs) + + duration = time.perf_counter() - start + run_list.append(f" {name:<30} {duration:7.3f}s") + + return result + + wrapped.__dict__.update(getattr(prod, "__dict__", {})) + return wrapped + + cls.__getitem__ = wrapped_getitem + + try: + yield + finally: + cls.__getitem__ = orig_getitem + + +def call_once_on_config(func=None, *, include_hash=False): + """ + Parametrized decorator to ensure that function *func* is only called once for the config *config*. + Can be used with or without parentheses. + """ + if func is None: + # If func is None, it means the decorator was called with arguments. + def wrapper(f): + return call_once_on_config(f, include_hash=include_hash) + return wrapper diff --git a/topsf/weights/__init__.py b/topsf/weights/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/topsf/weights/default.py b/topsf/weights/default.py new file mode 100644 index 0000000..9bd3cd1 --- /dev/null +++ b/topsf/weights/default.py @@ -0,0 +1,466 @@ +# coding: utf-8 + +""" +Event weight producer. +""" + +from topsf.util import call_once_on_config +import law + +from columnflow.util import maybe_import +from columnflow.histogramming import HistProducer +from columnflow.histogramming.default import cf_default +from columnflow.config_util import get_shifts_from_sources +from columnflow.columnar_util import Route, set_ak_column +from topsf.util import has_tag + +np = maybe_import("numpy") +ak = maybe_import("awkward") + +logger = law.logger.get_logger(__name__) + + +# extend columnflow's default hist producer +@cf_default.hist_producer(uses={"mc_weight"}, mc_only=True) +def mc_weight(self: HistProducer, events: ak.Array, **kwargs) -> ak.Array: + return events, events.mc_weight + + +@cf_default.hist_producer(uses={"normalization_weight"}, mc_only=True) +def norm(self: HistProducer, events: ak.Array, **kwargs) -> ak.Array: + return events, events.normalization_weight + + +@cf_default.hist_producer(mc_only=True) +def no_weights(self: HistProducer, events: ak.Array, **kwargs) -> ak.Array: + return events, ak.Array(np.ones(len(events), dtype=np.float32)) + + +@cf_default.hist_producer( + # uses={ + # "Jet.pt", "Jet.eta", "Jet.phi", "Jet.mass", + # "Jet.btagUParTAK4B", + # "FatJet.pt", + # "Muon.pt", + # "Electron.pt", + # }, + # both used columns and dependent shifts are defined in init below + weight_columns=None, + # only run on mc + mc_only=False, + # optional categorizer to obtain baseline event mask + categorizer_cls=None, + pre_label="", +) +def base(self: HistProducer, events: ak.Array, task: law.Task, **kwargs) -> ak.Array: + # apply behavior (for variable reconstruction) + + # jet = ak.without_parameters(events["Jet"]) + # fatjet = ak.without_parameters(events["FatJet"]) + # muon = ak.without_parameters(events["Muon"]) + # electron = ak.without_parameters(events["Electron"]) + + # apply mask + if self.categorizer_cls: + events, mask = self[self.categorizer_cls](events, **kwargs) + events = events[mask] + + if self.dataset_inst.is_data: + logger.debug(f"Dataset {self.dataset_inst} is data, not applying any weights") + return events, ak.Array(np.ones(len(events), dtype=np.float32)) + + # build the full event weight + weight = ak.Array(np.ones(len(events), dtype=np.float64)) + logger.info( + f"HistProducer '{self.cls_name}' (dataset {self.dataset_inst}) uses weight columns: \n" + f"{', '.join(self.local_weight_columns.keys())}", + ) + for column in self.local_weight_columns.keys(): + new_weight = weight * Route(column).apply(events) + if not ak.any(new_weight != weight): + logger.debug(f"Weight column {column} does not change the weight (all values are 1), skipping multiplication") + else: + weight = new_weight + + try: + wmin = float(ak.min(weight)) + wmax = float(ak.max(weight)) + wmean = float(ak.mean(weight)) + logger.debug(f"Applied weight columns {', '.join(self.local_weight_columns.keys())}; weight stats: min={wmin:.3g}, mean={wmean:.3g}, max={wmax:.3g}") + except Exception: + logger.warning("Applied weight columns, but couldn't compute summary stats for weights") + + # implement dummy shift by varying weight by factor of 2 + if "dummy" in task.local_shift_inst.name: + logger.warning("Applying dummy weight shift (should never be use for real analysis)") + variation = task.local_shift_inst.name.split("_")[-1] + weight = weight * {"up": 2.0, "down": 0.5}[variation] + + # special case: if only "weight_unweighted" is requested, we do not want to apply any weight at all + if ( + hasattr(task, "variables") and + len(task.variables) == 1 and + task.variables[0].startswith("weight_unweighted") + ): + events = set_ak_column(events, "weight", weight, value_type=np.float32) + return events, ak.Array(np.ones(len(events), dtype=np.float32)) + + return events, weight + + +# @base.setup +# def base_setup( +# self: HistProducer, +# reqs: dict, +# task: law.Task, +# inputs: dict, +# reader_targets: law.util.InsertableDict, +# ) -> None: +# logger.debug( +# f"HistProducer '{self.cls_name}' (dataset {self.dataset_inst}) uses weight columns: \n" +# f"{', '.join(self.weight_columns.keys())}", +# ) +# # if "dy_correction_weight" in self.local_weight_columns.keys(): +# # # add dy_correction_weight to the reader targets +# # reader_targets["dy_correction_weight"] = inputs["dy_correction_weight_producer"]["columns"] + +# # if self.require_hbb_sf_producer and not task.dataset_inst.is_data: +# # # add hbb_sf_weights to the reader targets +# # reader_targets["hbb_sf_weights"] = inputs["hbb_sf_weights"]["columns"] + +# # if self.require_msd_nonclosure_producer: +# # # add msd_nonclosure_uncertainty to the reader targets +# # reader_targets["msd_nonclosure_uncertainty"] = inputs["msd_nonclosure_uncertainty"]["columns"] + + +# @base.requires +# def base_requires(self: HistProducer, task: law.Task, reqs: law.util.InsertableDict) -> None: +# """ +# Define the requirements for the base HistProducer. +# This is called before the setup method. +# """ +# logger.debug("Currently no custom requirements for HistProducer base, only weight columns defined in init") +# # from columnflow.tasks.production import ProduceColumns +# # if "dy_correction_weight" in self.local_weight_columns.keys(): +# # if not self.dy_correction_weight_producer: +# # raise Exception( +# # "dy_correction_weight_producer must be set if dy_correction_weight is used " +# # "in weight_columns", +# # ) +# # reqs["dy_correction_weight_producer"] = ProduceColumns.req( +# # task, +# # producer=self.dy_correction_weight_producer, +# # ) + +# # if self.require_hbb_sf_producer and not self.dataset_inst.is_data: +# # reqs["hbb_sf_weights"] = ProduceColumns.req( +# # task, +# # producer="hbb_sf_weights", +# # ) + +# # self.require_msd_nonclosure_producer = False +# # if "msd_nonclosure_weight" in self.local_weight_columns.keys(): +# # msd_nonclosure_shifts = get_shifts_from_sources( +# # self.config_inst, +# # self.local_weight_columns["msd_nonclosure_weight"], +# # ) +# # if self.dataset_inst.is_mc and task.shift in msd_nonclosure_shifts: +# # reqs["msd_nonclosure_uncertainty"] = ProduceColumns.req( +# # task, +# # producer="msd_nonclosure_uncertainty", +# # ) +# # self.require_msd_nonclosure_producer = True +# # else: +# # self.local_weight_columns.pop("msd_nonclosure_weight", None) +# # self.uses.discard("msd_nonclosure_weight") + + +@base.init +def base_init(self: HistProducer) -> None: + + if not getattr(self, "config_inst"): + return + + def update_cat_label(config_inst, pre_label): + for cat_inst, _, _ in config_inst.walk_categories(): + if pre_label not in cat_inst.label: + cat_inst.label = "\n".join([cat_inst.label, pre_label]) + else: + logger.debug(f"Category {cat_inst} already includes pre_label '{pre_label}'") + + if self.pre_label: + update_cat_label(self.config_inst, self.pre_label) + + if self.categorizer_cls: + self.uses.add(self.categorizer_cls) + + dataset_inst = getattr(self, "dataset_inst", None) + if dataset_inst and dataset_inst.is_data: + # if we are on data, we do not need any weights + self.local_weight_columns = {} + return + + year = self.config_inst.campaign.x.year + cpn_tag = self.config_inst.x.cpn_tag + + if not self.weight_columns: + raise Exception("weight_columns not set") + self.local_weight_columns = self.weight_columns.copy() + + if has_tag("skip_btag_wp_weights", self.config_inst, self.dataset_inst): + logger.info_once(f"Config {self.config_inst.name} has tag 'skip_btag_wp_weights', removing btag weight columns and use normalized_ht_njet_nhf_btag_weight instead") + self.local_weight_columns.pop("btag_weight", None) + elif has_tag("skip_btag_weights", self.config_inst, self.dataset_inst): + logger.info_once(f"Config {self.config_inst.name} has tag 'skip_btag_weights', removing normalized_ht_njet_nhf_btag_weight columns and use btag weight instead") + self.local_weight_columns.pop("normalized_ht_njet_nhf_btag_weight", None) + else: + # throw error if no tag is set + if not any(has_tag(tag, self.config_inst, self.dataset_inst) for tag in ["skip_btag_wp_weights", "skip_btag_weights"]): + raise Exception( + "No tag set for btag weights. Please set either 'skip_btag_wp_weights' or 'skip_btag_weights' tag in the config or dataset.", + ) + + if dataset_inst and dataset_inst.has_tag("skip_scale"): + # remove dependency towards mur/muf weights + for column in [ + "normalized_mur_weight", "normalized_muf_weight", "normalized_murmuf_envelope_weight", + "mur_weight", "muf_weight", "murmuf_envelope_weight", + ]: + self.local_weight_columns.pop(column, None) + + if dataset_inst and dataset_inst.has_tag("no_ps_weights"): + self.local_weight_columns.pop("normalized_isr_weight", None) + self.local_weight_columns.pop("normalized_fsr_weight", None) + + if dataset_inst and dataset_inst.has_tag("skip_pdf"): + # remove dependency towards pdf weights + for column in ["pdf_weight", "normalized_pdf_weight"]: + self.local_weight_columns.pop(column, None) + + if dataset_inst and not dataset_inst.has_tag("is_ttbar"): + # remove dependency towards top pt weights + self.local_weight_columns.pop("top_pt_weight", None) + + if dataset_inst and not dataset_inst.has_tag("is_v_jets") and not dataset_inst.has_tag("is_vv"): + # remove dependency towards vjets weights + self.local_weight_columns.pop("vjets_weight", None) + + # if dataset_inst and not dataset_inst.has_tag("is_dy"): + # # remove dependency towards vjets weights + # self.local_weight_columns.pop("dy_correction_weight", None) + # self.local_weight_columns.pop("dy_weight", None) + + # if dataset_inst and not dataset_inst.has_tag("is_dy"): + # # remove dependency towards dy weights + # self.local_weight_columns.pop("dy_weight", None) + + self.shifts = set() + + # when jec sources are known btag SF source, then propagate the shift to the HistProducer + # TODO: we should do this somewhere centrally + btag_sf_jec_sources = ( + (set(self.config_inst.x.btag_sf_jec_sources) | {"Total"}) & + set(self.config_inst.x.jec.Jet["uncertainty_sources"]) + ) + self.shifts |= set(get_shifts_from_sources( + self.config_inst, + *[f"jec_{jec_source}" for jec_source in btag_sf_jec_sources], + )) + + for weight_column, shift_sources in self.local_weight_columns.items(): + shift_sources = law.util.make_list(shift_sources) + shift_sources = [s.format(year=year, cpn_tag=cpn_tag) for s in shift_sources] + shifts = get_shifts_from_sources(self.config_inst, *shift_sources) + for shift in shifts: + if weight_column not in shift.x("column_aliases").keys(): + # make sure that column aliases are implemented + raise Exception( + f"Weight column {weight_column} implements shift {shift}, but does not use it " + f"in 'column_aliases' aux {shift.x('column_aliases')}", + ) + + # declare shifts that the produced event weight depends on + self.shifts |= set(shifts) + + # remove dummy column from weight columns and uses + self.local_weight_columns.pop("dummy_weight", "") + + # store column names referring to weights to multiply + self.uses |= self.local_weight_columns.keys() + # self.uses = {"*"} + + +@base.post_init +def base_post_init(self: HistProducer, task: law.Task): + if self.dataset_inst.is_data: + return + if "isr" not in task.shift: + # no nominal ISR weight --> remove it from uses and local_weight_columns + self.uses.discard("normalized_isr_weight") + self.local_weight_columns.pop("normalized_isr_weight", None) + if "fsr" not in task.shift: + # no nominal FSR weight --> remove it from uses and local_weight_columns + self.uses.discard("normalized_fsr_weight") + self.local_weight_columns.pop("normalized_fsr_weight", None) + + +btag_uncs = [ + "hf", "lf", + "cferr1", "cferr2", + "hfstats1", "lfstats1", + "hfstats2", "lfstats2", +] + + +btag_uncs_bc = [ + "correlated", + "uncorrelated", + "bfragmentation", + "fsrdef", + "hdamp", + "isrdef", + "jer", + "jes", + "muf", + "mur", + "pdfas", + "pileup", + "statistic", + "topmass", + "type3p", +] +btag_uncs_bc_full = [f"{unc}_bc" for unc in btag_uncs_bc] + ["bc"] +btag_uncs_light = [ + "", + "correlated", "uncorrelated", +] +btag_uncs_light_full = [f"{unc}_light" for unc in btag_uncs_light] + ["light"] + +all_btag_uncs = btag_uncs_bc_full + btag_uncs_light_full + +default_correction_weights = { + "normalization_weight": [], + # "dummy_weight": ["dummy_{cpn_tag}"], + "normalized_pu_weight": ["minbias_xs"], + # muon SF + "muon_reco_weight": ["muon_reco"], + "muon_trigger_weight": ["muon_trigger"], + "muon_id_weight": ["muon_id"], + "muon_iso_weight": ["muon_iso"], + # electron SF + "electron_trigger_weight": ["electron_trigger"], + "electron_id_iso_weight": ["electron_id_iso"], + "electron_reco_weight": ["electron_reco"], + # store all btag weights, pop them later depending on the dataset, year, and cpn tag + "btag_weight": [f"btag_{unc}" for unc in all_btag_uncs], + "normalized_ht_njet_nhf_btag_weight": [f"btag_{unc}" for unc in btag_uncs], + # "normalized_murmuf_envelope_weight": ["murmuf_envelope"], + # "normalized_murmuf_weight": ["murmuf"] + "normalized_mur_weight": ["mur"], + "normalized_muf_weight": ["muf"], + "normalized_pdf_weight": ["pdf"], + "normalized_isr_weight": ["isr"], + "normalized_fsr_weight": ["fsr"], + "top_pt_weight": ["top_pt"], + # "vjets_weight": ["vjets"] +} + + +weight_columns_except_trigger = default_correction_weights.copy() +weight_columns_except_trigger.pop("muon_trigger_weight") +weight_columns_except_trigger.pop("electron_trigger_weight") + +weight_columns_except_btag = default_correction_weights.copy() +weight_columns_except_btag.pop("btag_weight") +weight_columns_except_btag.pop("normalized_ht_njet_nhf_btag_weight") + +weight_columns_except_btag_and_trigger = default_correction_weights.copy() +weight_columns_except_btag_and_trigger.pop("btag_weight") +weight_columns_except_btag_and_trigger.pop("normalized_ht_njet_nhf_btag_weight") +weight_columns_except_btag_and_trigger.pop("muon_trigger_weight") +weight_columns_except_btag_and_trigger.pop("electron_trigger_weight") + +default_hist_producer = base.derive("default", cls_dict={ + "weight_columns": default_correction_weights +}) + +default_fixed_hist_producer = base.derive("default_fixed", cls_dict={ + # use default weights plus the electron normalization fix weight, which applies a flat SF of 1.2 in the 1e channel for 2024 to account for the missing trigger SFs and other lepton SFs in that channel for 2024 + "weight_columns": { + **default_correction_weights, + "electron_norm_fix_weight": [], + }, +}) + +no_btag_weights = default_hist_producer.derive("no_btag_weights", cls_dict={ + "pre_label": "Without b-tagging weight", + # "weight_columns": weight_columns_except_btag_and_trigger, # no trigger weights available yet + "weight_columns": weight_columns_except_btag, +}) + +no_trigger_weights = default_hist_producer.derive("no_trigger_weights", cls_dict={ + "pre_label": "Without trigger SF", + "weight_columns": weight_columns_except_trigger, +}) + +no_lepton_weights = default_hist_producer.derive("no_lepton_weights", cls_dict={ + "pre_label": "Without lepton SFs", + "weight_columns": { + k: v for k, v in default_correction_weights.items() if not k.startswith("muon_") and not k.startswith("electron_") + }, +}) + +no_scale_weights = default_hist_producer.derive("no_scale_weights", cls_dict={ + "pre_label": "Without scale weights", + "weight_columns": { + k: v for k, v in default_correction_weights.items() if not k.startswith("normalized_mur") and not k.startswith("normalized_muf") and not k.startswith("normalized_murmuf") and not k.startswith("mur") and not k.startswith("muf") and not k.startswith("murmuf") + }, +}) + +no_pdf_weights = default_hist_producer.derive("no_pdf_weights", cls_dict={ + "pre_label": "Without PDF weights", + "weight_columns": { + k: v for k, v in default_correction_weights.items() if not k.startswith("normalized_pdf") and not k.startswith("pdf") + }, +}) + +no_ps_weights = default_hist_producer.derive("no_ps_weights", cls_dict={ + "pre_label": "Without PS weights", + "weight_columns": { + k: v for k, v in default_correction_weights.items() if not k.startswith("normalized_isr") and not k.startswith("normalized_fsr") and not k.startswith("isr") and not k.startswith("fsr") + }, +}) + +# +# HistProducers with masks via categorization +# not working, but kept as reference if we ever want to implement something like this +# + +# from hbw.categorization.categories import ( +# mask_fn_mbb80, catid_ge2b_loose, catid_njet2, mask_fn_met70, +# mask_fn_met_geq40, +# ) +# met_geq40_with_dy_corr = with_dy_corr.derive("met_geq40_with_dy_corr", cls_dict={ +# "pre_label": "\n".join([r"$p_{T}^{miss} \geq 40$ GeV"]), +# "nondy_hist_producer": "met_geq40_no_dycorr", +# "categorizer_cls": mask_fn_met_geq40, +# "dy_correction_weight_producer": "dy_correction_weight", +# }) + +# other btag normalization modes +# base.derive("btag_njet_normalized", cls_dict={"weight_columns": { +# **weight_columns_execpt_btag, +# "normalized_njet_btag_weight": [f"btag_{unc}" for unc in btag_uncs], +# }}) +# base.derive("btag_ht_njet_normalized", cls_dict={"weight_columns": { +# **weight_columns_execpt_btag, +# "normalized_ht_njet_btag_weight": [f"btag_{unc}" for unc in btag_uncs], +# }}) +# base.derive("btag_ht_njet_nhf_normalized", cls_dict={"weight_columns": { +# **weight_columns_execpt_btag, +# "normalized_ht_njet_nhf_btag_weight": [f"btag_{unc}" for unc in btag_uncs], +# }}) +# base.derive("btag_ht_normalized", cls_dict={"weight_columns": { +# **weight_columns_execpt_btag, +# "normalized_ht_btag_weight": [f"btag_{unc}" for unc in btag_uncs], +# }})