From a6cb3c1557beb9f43cc24e8910ef104ba437e354 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 07:04:41 +0200 Subject: [PATCH 01/39] catalog: metacal bitmasks use FITS format K, not I or the dead FLAGS_ key NGMIX_MCAL_FLAGS carries bit 30 (shapepipe#854's absent-measurement flag) alongside the native fitter bits. Format I truncates it to int16 and silently zeroes bit 30 on write, in both the FITS and the HDF5 branch of write_shape_catalog (the HDF5 path reuses the FITS column's array). The per-type NGMIX_FLAGS_{1P,1M,2P,2M,NOSHEAR} columns had the same problem from the other direction: their format entry was keyed as "FLAGS_{suffix}", which never matches the real column name "{prefix}_FLAGS_{suffix}", so the writer's float64 default masked the dead key rather than narrowing anything. Both parameter files now key and format all six metacal bitmasks the same way, at K, so they round- trip exactly and survive JointCat's optional memory-reduction pass (which only downcasts int32/float64, not int64). NGMIX_MCAL_TYPES_FAIL stays at I: it is a count in [0, 5], not a bitmask. Adds a round-trip test parametrized over both parameter files, FITS and HDF5, and reduce_mem on/off, checking 0, bit 30 alone, and bit 30 with a native bit together. Co-Authored-By: Claude Opus 5.5 --- scripts/calibration/params.py | 5 +- src/sp_validation/catalog.py | 6 ++ .../tests/test_catalog_flag_roundtrip.py | 63 +++++++++++++++++++ workflow/image_sims/params_im_sim.py | 5 +- 4 files changed, 75 insertions(+), 4 deletions(-) create mode 100644 src/sp_validation/tests/test_catalog_flag_roundtrip.py diff --git a/scripts/calibration/params.py b/scripts/calibration/params.py index ea871119..f20d3a1b 100644 --- a/scripts/calibration/params.py +++ b/scripts/calibration/params.py @@ -143,7 +143,6 @@ "NUMBER", "IMAFLAGS_ISO", "FLAGS", - "NGMIX_MCAL_FLAGS", "NGMIX_MCAL_TYPES_FAIL", "N_EPOCH", "NGMIX_N_EPOCH", @@ -152,6 +151,8 @@ add_cols_pre_cal_format["TILE_ID"] = "A7" add_cols_pre_cal_format["NUMBER"] = "J" +# Metacal bitmasks need bit 30 and native fitter bits, including after merging. +add_cols_pre_cal_format["NGMIX_MCAL_FLAGS"] = "K" # Create key names for metacal information prefix = "NGMIX" @@ -162,7 +163,7 @@ add_cols_pre_cal.append(f"{prefix}_{center}_{suffix}") for suffix in suffixes: - add_cols_pre_cal_format[f"FLAGS_{suffix}"] = "I" + add_cols_pre_cal_format[f"{prefix}_FLAGS_{suffix}"] = "K" # Catalog parameters diff --git a/src/sp_validation/catalog.py b/src/sp_validation/catalog.py index e39d2076..40d4d418 100644 --- a/src/sp_validation/catalog.py +++ b/src/sp_validation/catalog.py @@ -439,6 +439,12 @@ def write_shape_catalog( Write catalogue with galaxy shapes = shear estimates. + @sc [label:convention] metacal-flag-width + The data and image-simulation parameter files supply FITS ``K`` for + metacal bitmasks so bit 30 and native fitter flags survive FITS/HDF5 + output and optional ``JointCat`` memory reduction. + The number of failed metacal types is a count in [0, 5], not a bitmask. + Parameters ---------- output_path : str diff --git a/src/sp_validation/tests/test_catalog_flag_roundtrip.py b/src/sp_validation/tests/test_catalog_flag_roundtrip.py new file mode 100644 index 00000000..02e27580 --- /dev/null +++ b/src/sp_validation/tests/test_catalog_flag_roundtrip.py @@ -0,0 +1,63 @@ +"""Comprehensive catalogues preserve metacal failure bits exactly.""" + +import runpy +from pathlib import Path + +import h5py +import numpy as np +import pytest +from astropy.io import fits + +from sp_validation.catalog import write_shape_catalog +from sp_validation.catalog_builders import JointCat + +ROOT = Path(__file__).resolve().parents[3] + + +@pytest.mark.parametrize( + "params_path", + ["scripts/calibration/params.py", "workflow/image_sims/params_im_sim.py"], + ids=["data", "image-sims"], +) +@pytest.mark.parametrize("extension", [".fits", ".hdf5"]) +@pytest.mark.parametrize("reduce_mem", [False, True]) +def test_metacal_flags_roundtrip(tmp_path, params_path, extension, reduce_mem): + """Contract metacal-flag-width: neither output nor merging loses flag bits. + + Bit 30 marks an absent measurement; bit 3 alongside it must also survive. + Float64 inputs match ShapePipe's final catalogue. A float32 downcast loses + bit 3, while int16 and int8 conversions lose bit 30. + """ + with np.printoptions(): + params = runpy.run_path(str(ROOT / params_path)) + flags = np.array([0, 2**30, 2**30 | 8], dtype=np.float64) + bit_columns = ["NGMIX_MCAL_FLAGS"] + [ + f"NGMIX_FLAGS_{suffix}" for suffix in ("NOSHEAR", "1P", "1M", "2P", "2M") + ] + columns = {name: flags for name in bit_columns} + columns["NGMIX_MCAL_TYPES_FAIL"] = np.array([0, 5, 1]) + assert set(columns) <= set(params["add_cols_pre_cal"]) + path = tmp_path / f"comprehensive{extension}" + write_shape_catalog( + str(path), + np.zeros(3), + np.zeros(3), + np.ones(3), + add_cols=columns, + add_cols_format=params["add_cols_pre_cal_format"], + ) + if extension == ".fits": + written = fits.getdata(path, 1) + else: + with h5py.File(path, "r") as catalog: + written = catalog["data"][:] + + builder = JointCat() + builder._params["reduce_mem"] = reduce_mem + for name, expected in columns.items(): + values = written[name] + np.testing.assert_array_equal(values, expected, err_msg=name) + if name in bit_columns: + assert values.dtype.kind == "i" and values.dtype.itemsize == 8, name + reduced = values.astype(builder.dtype_out(name, values.dtype)) + np.testing.assert_array_equal(reduced, expected, err_msg=name) diff --git a/workflow/image_sims/params_im_sim.py b/workflow/image_sims/params_im_sim.py index 1cd395c6..c210dedf 100644 --- a/workflow/image_sims/params_im_sim.py +++ b/workflow/image_sims/params_im_sim.py @@ -148,7 +148,6 @@ for key in ( "NUMBER", "FLAGS", - "NGMIX_MCAL_FLAGS", "NGMIX_MCAL_TYPES_FAIL", "N_EPOCH", "NGMIX_N_EPOCH", @@ -157,6 +156,8 @@ add_cols_pre_cal_format["TILE_ID"] = "A7" add_cols_pre_cal_format["NUMBER"] = "J" +# Metacal bitmasks need bit 30 and native fitter bits, including after merging. +add_cols_pre_cal_format["NGMIX_MCAL_FLAGS"] = "K" # Create key names for metacal information prefix = "NGMIX" @@ -167,7 +168,7 @@ add_cols_pre_cal.append(f"{prefix}_{center}_{suffix}") for suffix in suffixes: - add_cols_pre_cal_format[f"FLAGS_{suffix}"] = "I" + add_cols_pre_cal_format[f"{prefix}_FLAGS_{suffix}"] = "K" # Catalog parameters From 469d82a676c377bb6006c6ccde0d875905118692 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 05:06:28 +0200 Subject: [PATCH 02/39] catalog: metacal bitmasks as int32; reduce_mem never narrows integers ShapePipe sets bit 2**30 in the metacal flags for a missing measurement, which needs 32 bits. The parameter files now write NGMIX_MCAL_FLAGS and the five NGMIX_FLAGS_* columns as FITS J (int32). JointCat.dtype_out with reduce_mem narrowed every int32 column outside a keep-list to int8, silently wrapping any value above 127. It now only narrows float64 to float32 (RA/Dec excepted) and leaves integer columns alone. Co-Authored-By: Claude Opus 5.5 --- scripts/calibration/params.py | 6 ++--- src/sp_validation/catalog.py | 6 ++--- src/sp_validation/catalog_builders.py | 25 +++++++------------ .../tests/test_catalog_flag_roundtrip.py | 22 +++++++++++++--- workflow/image_sims/params_im_sim.py | 6 ++--- 5 files changed, 37 insertions(+), 28 deletions(-) diff --git a/scripts/calibration/params.py b/scripts/calibration/params.py index f20d3a1b..621d1a4b 100644 --- a/scripts/calibration/params.py +++ b/scripts/calibration/params.py @@ -151,8 +151,8 @@ add_cols_pre_cal_format["TILE_ID"] = "A7" add_cols_pre_cal_format["NUMBER"] = "J" -# Metacal bitmasks need bit 30 and native fitter bits, including after merging. -add_cols_pre_cal_format["NGMIX_MCAL_FLAGS"] = "K" +# Metacal bitmasks: ShapePipe sets bit 30 for no measurement, so they need 32 bits. +add_cols_pre_cal_format["NGMIX_MCAL_FLAGS"] = "J" # Create key names for metacal information prefix = "NGMIX" @@ -163,7 +163,7 @@ add_cols_pre_cal.append(f"{prefix}_{center}_{suffix}") for suffix in suffixes: - add_cols_pre_cal_format[f"{prefix}_FLAGS_{suffix}"] = "K" + add_cols_pre_cal_format[f"{prefix}_FLAGS_{suffix}"] = "J" # Catalog parameters diff --git a/src/sp_validation/catalog.py b/src/sp_validation/catalog.py index 40d4d418..a3b5551a 100644 --- a/src/sp_validation/catalog.py +++ b/src/sp_validation/catalog.py @@ -440,9 +440,9 @@ def write_shape_catalog( Write catalogue with galaxy shapes = shear estimates. @sc [label:convention] metacal-flag-width - The data and image-simulation parameter files supply FITS ``K`` for - metacal bitmasks so bit 30 and native fitter flags survive FITS/HDF5 - output and optional ``JointCat`` memory reduction. + The data and image-simulation parameter files supply FITS ``J`` (int32) + for metacal bitmasks, so ShapePipe's no-measurement bit 30 and the ngmix + fitter bits survive FITS/HDF5 output; ``JointCat`` never narrows integers. The number of failed metacal types is a count in [0, 5], not a bitmask. Parameters diff --git a/src/sp_validation/catalog_builders.py b/src/sp_validation/catalog_builders.py index 78dc7838..b6abafc0 100644 --- a/src/sp_validation/catalog_builders.py +++ b/src/sp_validation/catalog_builders.py @@ -437,26 +437,19 @@ def dtype_out(self, name, dtype_in): output dtype """ - # Specify columns for which original (high-precision) format - # needs to be kept and not reduced to lower precision - cols_keep_dtype = [ - "RA", - "Dec", - "FLAGS", - "IMAFLAGS_ISO", - "NUMBER", - ] if dtype_in.kind == "U": # Transform unicode to string of equal length return np.dtype(f"S{dtype_in.itemsize // 4}") - if self._params["reduce_mem"] == False: - return dtype_in - elif name not in cols_keep_dtype: - if dtype_in.kind == "f" and dtype_in.itemsize == 8: - return np.float32 - if dtype_in.kind == "i" and dtype_in.itemsize == 4: - return np.int8 + # reduce_mem narrows float64 to float32, except coordinates. Integer + # columns (bitmasks, IDs, counts) are never narrowed. + if ( + self._params["reduce_mem"] + and dtype_in.kind == "f" + and dtype_in.itemsize == 8 + and name not in ("RA", "Dec") + ): + return np.dtype(np.float32) return dtype_in diff --git a/src/sp_validation/tests/test_catalog_flag_roundtrip.py b/src/sp_validation/tests/test_catalog_flag_roundtrip.py index 02e27580..580d3f64 100644 --- a/src/sp_validation/tests/test_catalog_flag_roundtrip.py +++ b/src/sp_validation/tests/test_catalog_flag_roundtrip.py @@ -25,8 +25,8 @@ def test_metacal_flags_roundtrip(tmp_path, params_path, extension, reduce_mem): """Contract metacal-flag-width: neither output nor merging loses flag bits. Bit 30 marks an absent measurement; bit 3 alongside it must also survive. - Float64 inputs match ShapePipe's final catalogue. A float32 downcast loses - bit 3, while int16 and int8 conversions lose bit 30. + Float64 inputs match ShapePipe's final catalogue. Bit 30 needs int32: + int16 or int8 storage drops it, and a float32 downcast drops bit 3. """ with np.printoptions(): params = runpy.run_path(str(ROOT / params_path)) @@ -58,6 +58,22 @@ def test_metacal_flags_roundtrip(tmp_path, params_path, extension, reduce_mem): values = written[name] np.testing.assert_array_equal(values, expected, err_msg=name) if name in bit_columns: - assert values.dtype.kind == "i" and values.dtype.itemsize == 8, name + assert values.dtype.kind == "i" and values.dtype.itemsize == 4, name reduced = values.astype(builder.dtype_out(name, values.dtype)) np.testing.assert_array_equal(reduced, expected, err_msg=name) + + +@pytest.mark.parametrize("reduce_mem", [False, True]) +def test_reduce_mem_never_narrows_integers(reduce_mem): + """reduce_mem narrows float64 (except RA/Dec) and leaves integers intact.""" + builder = JointCat() + builder._params["reduce_mem"] = reduce_mem + for dtype in (">i4", "i8", "f8")) == np.dtype(">f8") + expected = np.float32 if reduce_mem else np.dtype(">f8") + assert builder.dtype_out("NGMIX_G1_NOSHEAR", np.dtype(">f8")) == expected diff --git a/workflow/image_sims/params_im_sim.py b/workflow/image_sims/params_im_sim.py index c210dedf..9eae8585 100644 --- a/workflow/image_sims/params_im_sim.py +++ b/workflow/image_sims/params_im_sim.py @@ -156,8 +156,8 @@ add_cols_pre_cal_format["TILE_ID"] = "A7" add_cols_pre_cal_format["NUMBER"] = "J" -# Metacal bitmasks need bit 30 and native fitter bits, including after merging. -add_cols_pre_cal_format["NGMIX_MCAL_FLAGS"] = "K" +# Metacal bitmasks: ShapePipe sets bit 30 for no measurement, so they need 32 bits. +add_cols_pre_cal_format["NGMIX_MCAL_FLAGS"] = "J" # Create key names for metacal information prefix = "NGMIX" @@ -168,7 +168,7 @@ add_cols_pre_cal.append(f"{prefix}_{center}_{suffix}") for suffix in suffixes: - add_cols_pre_cal_format[f"{prefix}_FLAGS_{suffix}"] = "K" + add_cols_pre_cal_format[f"{prefix}_FLAGS_{suffix}"] = "J" # Catalog parameters From 9423e93adab4793952c50d5fec1bf3fc7a4784ef Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Wed, 9 Sep 2026 20:13:03 -0400 Subject: [PATCH 03/39] catalog: layout-agnostic campaign and star catalogue readers Replace read_hdf5_file's hardcoded patches// lookup with find_dataset_group(), which descends from the file root through single container groups until it reaches the per-unit datasets. This reads the legacy patches// layout that ShapePipe still writes as a compatibility shim, a future flat tiles/ layout, and the exposures/ layout of full_starcat_.hdf5 with the same code. read_star_catalogue() keeps the FITS path for files ending in .fits. Requested columns missing from the data now raise a clear KeyError. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01QbnPCyzuDNTgkg715pHhar --- src/sp_validation/catalog.py | 213 +++++++++++++++++++++++++++-------- 1 file changed, 168 insertions(+), 45 deletions(-) diff --git a/src/sp_validation/catalog.py b/src/sp_validation/catalog.py index a3b5551a..e96786ab 100644 --- a/src/sp_validation/catalog.py +++ b/src/sp_validation/catalog.py @@ -764,71 +764,194 @@ def read_param_file(path, verbose=False): return param_list_unique -def read_hdf5_file(file_path, name, stats_file, check_only=False, param_path=None): - """Read HDF5 File. +#: Columns written by ShapePipe v2's ``MergeStarCatPSFEX`` into +#: ``full_starcat_.hdf5`` (one dataset per exposure). +STAR_CAT_COLUMNS = ( + "X", + "Y", + "RA", + "DEC", + "HSM_G1_PSF", + "HSM_G2_PSF", + "HSM_T_PSF", + "HSM_G1_STAR", + "HSM_G2_STAR", + "HSM_T_STAR", + "HSM_FLAG_PSF", + "HSM_FLAG_STAR", + "MAG", + "SNR", + "ACCEPTED", + "CCD_NB", +) + + +def find_dataset_group(hdf5_file): + """Find Dataset Group. + + Descend from the root of an open HDF5 file to the single group whose + members are the per-unit datasets (one per tile, or one per exposure). + + This makes the reader independent of how deeply the products nest that + group: it walks down as long as the current node holds exactly one + sub-group, and stops as soon as the members are datasets. It therefore + reads both the legacy ``patches//`` layout (the + "patches" key is a ShapePipe-side compatibility shim, not a concept) and + a flat ``tiles/`` or ``exposures/`` layout. - Read hdf5 file and return contained data. + Parameters + ---------- + hdf5_file : h5py.File or h5py.Group + open input file + + Returns + ------- + h5py.Group + group whose members are the per-unit datasets + + Raises + ------ + ValueError + if the file is empty, or a level holds more than one sub-group + + """ + node = hdf5_file + while True: + keys = list(node) + if not keys: + raise ValueError(f"No data found under {node.name!r} in {hdf5_file.file.filename}") + if all(isinstance(node[key], h5py.Dataset) for key in keys): + return node + if len(keys) != 1: + raise ValueError( + f"Expected a single container group under {node.name!r} in" + + f" {hdf5_file.file.filename}, found {len(keys)}: {keys[:5]}" + ) + node = node[keys[0]] + + +def _check_columns(dtype, param_list, file_path): + """Raise a clear error if requested columns are absent from the data.""" + missing = [col for col in param_list if col not in (dtype.names or ())] + if missing: + raise KeyError( + f"Column(s) {missing} not found in catalogue {file_path}." + + f" Available columns: {sorted(dtype.names or ())}" + ) + + +def concatenate_datasets(group, param_list=None, file_path="", verbose=True): + """Concatenate Datasets. + + Concatenate every dataset of an HDF5 group into one structured array, + optionally restricted to a list of columns. + + Parameters + ---------- + group : h5py.Group + group whose members are structured datasets + param_list : list of str, optional + columns to keep; default is ``None`` (keep all) + file_path : str, optional + input file path, for error messages + verbose : bool, optional + verbose output if ``True`` + + Returns + ------- + numpy.ndarray + concatenated structured array + + """ + keys = list(group) + if param_list is not None: + _check_columns(group[keys[0]].dtype, param_list, file_path) + + n_rows = sum(group[key].shape[0] for key in keys) + if verbose: + n_cols = len(param_list) if param_list is not None else len(group[keys[0]].dtype) + print( + f"Reading {len(keys)} datasets," + + f" estimating {n_cols * n_rows * 8 / 1024**3:.1f}" + + f" Gb memory for the ({n_cols} x {n_rows}) data array ..." + ) + + data_list = [] + for key in tqdm.tqdm(keys, disable=not verbose): + data = group[key][()] + if param_list is not None: + data = data[param_list] + data_list.append(data) + + return np.concatenate(data_list, axis=0) + + +def read_campaign_catalogue( + file_path, + param_path=None, + param_list=None, + verbose=True, +): + """Read Campaign Catalogue. + + Read a campaign galaxy catalogue (``final_cat_.hdf5``) and + return its per-tile datasets concatenated into one structured array. Parameters ---------- file_path : str input file path - name : str - patch name - stats_file : file handler - summary statistics output file handler - check_only : bool, optional - If True only check, not return data + param_path : str, optional + path to a parameter file listing the columns to keep + param_list : list of str, optional + columns to keep; takes precedence over ``param_path`` + verbose : bool, optional + verbose output if ``True`` Returns ------- - dict - data + numpy.ndarray + catalogue data """ - param_list = read_param_file(param_path, verbose=True) if param_path else None + if param_list is None and param_path: + param_list = read_param_file(param_path, verbose=verbose) with h5py.File(file_path, "r") as hdf5_file: - # Find patch group in hierarchical structure - if f"patches/{name}" not in hdf5_file: - raise KeyError(f"Entry patches/{name} not found in file {file_path}") - patch_group = hdf5_file[f"patches/{name}"] - - # Get size of data array - num_rows = sum(patch_group[ID].shape[0] for ID in patch_group) - # num_cols = patch_group[next(iter(patch_group))].shape[1] - num_cols = len(param_list) - - print( - f"Estimating {num_cols * num_rows * 8 / 1024**3:.1f}" - + f" Gb memory for the ({num_cols} x {num_rows}) data array ..." + group = find_dataset_group(hdf5_file) + return concatenate_datasets( + group, param_list=param_list, file_path=file_path, verbose=verbose ) - # data_comb = np.memmap(output_file, dtype=patch_group[next(iter(patch_group))].dtype, - # mode="w+", shape=(num_rows, num_cols)) - data_list = [] - ID_pbl = set() - for ID in tqdm.tqdm(patch_group): - # Get data for this ID from file - data = patch_group[ID][()] - # Restrict to parameter list if given - data = data[param_list] if param_list is not None else data +def read_star_catalogue(file_path, hdu=1, verbose=True): + """Read Star Catalogue. - if not check_only: - # Add new to existing data - data_list.append(data) + Read a campaign star/PSF catalogue. Reads the ShapePipe v2 + ``full_starcat_.hdf5`` (one dataset per exposure), or a legacy + FITS star catalogue when ``file_path`` ends in ``.fits``. + + Parameters + ---------- + file_path : str + input file path + hdu : int, optional + HDU number for the FITS path; default is 1 + verbose : bool, optional + verbose output if ``True`` - print("Combine tile catalogues") - data_comb = np.concatenate(data_list, axis=0) - print("Done") + Returns + ------- + numpy.ndarray + star catalogue data - # Print problematic tile IDs - for ID in ID_pbl: - print("Tile IDs with missing keys:", file=stats_file) - print(ID, file=stats_file) + """ + if str(file_path).endswith(".fits"): + return fits.getdata(file_path, hdu) - return data_comb + with h5py.File(file_path, "r") as hdf5_file: + group = find_dataset_group(hdf5_file) + return concatenate_datasets(group, file_path=file_path, verbose=verbose) def get_maked_col(dat, col, mask): From c755cdd75592d1ef45a3f2b6476f61bcb2097b9d Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Wed, 9 Sep 2026 20:14:27 -0400 Subject: [PATCH 04/39] galaxy: replace IMAFLAGS_ISO cut with config-driven mask columns MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ShapePipe v2 drops IMAFLAGS_ISO for eleven boolean MASK_n* columns. galaxy.mask_cut() ORs a configurable list of them (default MASK_n4, MASK_n1, MASK_n2, MASK_n8, MASK_n1024 — stars, star halos, manual galaxy mask, MaxiMask) and returns the keep mask; a catalogue missing any of the requested columns raises a KeyError naming them. classification_galaxy_base takes mask_columns; extract_info.py passes the params.py mask_columns list and uses it for the star-sample cut too, and now reads both catalogues through the new readers. Column lists in params.py and masking.py's SPATIAL_CUTS updated to the MASK_n* names. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01QbnPCyzuDNTgkg715pHhar --- scripts/calibration/extract_info.py | 10 ++- scripts/calibration/params.py | 25 +++++- scripts/glass_mock/compute_leakage_harmony.py | 4 +- scripts/masking.py | 12 ++- src/sp_validation/galaxy.py | 90 ++++++++++++++++++- 5 files changed, 132 insertions(+), 9 deletions(-) diff --git a/scripts/calibration/extract_info.py b/scripts/calibration/extract_info.py index 123c7437..4d449416 100644 --- a/scripts/calibration/extract_info.py +++ b/scripts/calibration/extract_info.py @@ -38,6 +38,7 @@ # from sp_validation.catalog import * from sp_validation import catalog as spv_cat +from sp_validation import galaxy from sp_validation.calibration import * from sp_validation.calibration import metacal from sp_validation.galaxy import * @@ -69,8 +70,8 @@ dd = np.load(galaxy_cat_path, mmap_mode=mmap_mode) else: print("Loading galaxy .hdf5 file...") - dd = spv_cat.read_hdf5_file( - galaxy_cat_path, name, stats_file, param_path=param_list_path + dd = spv_cat.read_campaign_catalogue( + galaxy_cat_path, param_path=param_list_path ) n_obj = len(dd) @@ -116,7 +117,7 @@ # ### Load star catalogue if star_cat_path: - d_star = fits.getdata(star_cat_path, hdu_star_cat) + d_star = spv_cat.read_star_catalogue(star_cat_path, hdu=hdu_star_cat) if star_cat_path: print_stats("Stars:", stats_file, verbose=verbose) @@ -160,7 +161,7 @@ m_star = ( (dd["FLAGS"][ind_star] == 0) - & (dd["IMAFLAGS_ISO"][ind_star] == 0) + & galaxy.mask_cut(dd, mask_columns)[ind_star] & (dd["NGMIX_MCAL_FLAGS"][ind_star] == 0) & (dd["NGMIX_G1_PSF_ORIG_NOSHEAR"][ind_star] != -10) ) @@ -311,6 +312,7 @@ gal_mag_faint=gal_mag_faint, flags_keep=flags_keep, n_epoch_min=n_epoch_min, + mask_columns=mask_columns, ) if shape == "ngmix": m_gal = classification_galaxy_ngmix( diff --git a/scripts/calibration/params.py b/scripts/calibration/params.py index 621d1a4b..4b094540 100644 --- a/scripts/calibration/params.py +++ b/scripts/calibration/params.py @@ -121,11 +121,30 @@ "NGMIX_T_PSF_RECONV_NOSHEAR", ] +## ShapePipe v2 mask columns OR'd together for the galaxy selection cut +mask_columns = [ + "MASK_n4", + "MASK_n1", + "MASK_n2", + "MASK_n8", + "MASK_n1024", +] + ## Pre-calibration catalogue, including masked objects and mask flags add_cols_pre_cal = [ "TILE_ID", "NUMBER", - "IMAFLAGS_ISO", + "MASK_n1", + "MASK_n2", + "MASK_n4", + "MASK_n8", + "MASK_n16", + "MASK_n32", + "MASK_n64", + "MASK_n128", + "MASK_n256", + "MASK_n1024", + "MASK_n2048", "FLAGS", "NGMIX_MCAL_FLAGS", "NGMIX_MCAL_TYPES_FAIL", @@ -141,7 +160,6 @@ add_cols_pre_cal_format = {} for key in ( "NUMBER", - "IMAFLAGS_ISO", "FLAGS", "NGMIX_MCAL_TYPES_FAIL", "N_EPOCH", @@ -149,6 +167,9 @@ ): add_cols_pre_cal_format[key] = "I" +for key in mask_columns: + add_cols_pre_cal_format[key] = "L" + add_cols_pre_cal_format["TILE_ID"] = "A7" add_cols_pre_cal_format["NUMBER"] = "J" # Metacal bitmasks: ShapePipe sets bit 30 for no measurement, so they need 32 bits. diff --git a/scripts/glass_mock/compute_leakage_harmony.py b/scripts/glass_mock/compute_leakage_harmony.py index a63c6bd4..22c2439d 100644 --- a/scripts/glass_mock/compute_leakage_harmony.py +++ b/scripts/glass_mock/compute_leakage_harmony.py @@ -8,6 +8,8 @@ import numpy as np from astropy.io import fits +from sp_validation import catalog as spv_cat + from sp_validation.glass_mock import compute_leakage_harmony @@ -53,7 +55,7 @@ def get_parser(): print("Catalog data loaded successfully.") print("Loading the star catalog data...") - cat_star = fits.getdata(f"{args.star_cat_path}") + cat_star = spv_cat.read_star_catalogue(args.star_cat_path) print("Star catalog data loaded successfully.") print("Computing leakage in harmonic space...") diff --git a/scripts/masking.py b/scripts/masking.py index 32a970ed..0ef0491f 100644 --- a/scripts/masking.py +++ b/scripts/masking.py @@ -16,7 +16,17 @@ # the footprint definition. SPATIAL_CUTS = { "overlap", - "IMAFLAGS_ISO", + "MASK_n1", + "MASK_n2", + "MASK_n4", + "MASK_n8", + "MASK_n16", + "MASK_n32", + "MASK_n64", + "MASK_n128", + "MASK_n256", + "MASK_n1024", + "MASK_n2048", "N_EPOCH", "4_Stars", "8_Manual", diff --git a/src/sp_validation/galaxy.py b/src/sp_validation/galaxy.py index 3c49a96b..3eda60af 100644 --- a/src/sp_validation/galaxy.py +++ b/src/sp_validation/galaxy.py @@ -28,6 +28,87 @@ # required square root: FWHM = 2.35482 sqrt(T / 2) from sp_validation import io +#: All mask columns written by ShapePipe v2 (bool, ``True`` = masked). +#: n4 stars; n1/n2 faint/bright star halos; n8 manual galaxy mask; +#: n1024 MaxiMask; n16..n256 per-band coverage; n2048 no PS-z2 coverage. +#: These replace the single IMAFLAGS_ISO bitmask of ShapePipe v1. +MASK_COLUMNS = ( + "MASK_n1", + "MASK_n2", + "MASK_n4", + "MASK_n8", + "MASK_n16", + "MASK_n32", + "MASK_n64", + "MASK_n128", + "MASK_n256", + "MASK_n1024", + "MASK_n2048", +) + +#: Mask columns OR'd together for the default galaxy selection. Deliberately +#: not a blanket OR over MASK_COLUMNS: the per-band coverage columns +#: (n16..n256) and n2048 would mask essentially the whole catalogue. +DEFAULT_MASK_COLUMNS = ( + "MASK_n4", + "MASK_n1", + "MASK_n2", + "MASK_n8", + "MASK_n1024", +) + + +def _column_names(dd): + """Return the column names of a structured array or mapping.""" + dtype = getattr(dd, "dtype", None) + if dtype is not None and dtype.names is not None: + return tuple(dtype.names) + return tuple(dd.keys()) + + +def mask_cut(dd, mask_columns=None): + """Mask Cut. + + Return a boolean mask that is ``True`` for objects *not* flagged by any + of the requested ShapePipe mask columns. + + Parameters + ---------- + dd : numpy.ndarray or dict + input catalogue + mask_columns : list of str, optional + mask columns to OR together; default is ``DEFAULT_MASK_COLUMNS`` + + Returns + ------- + numpy.ndarray + boolean mask, ``True`` = keep + + Raises + ------ + KeyError + if any requested mask column is absent from the catalogue + + """ + columns = list(DEFAULT_MASK_COLUMNS if mask_columns is None else mask_columns) + + available = _column_names(dd) + missing = [col for col in columns if col not in available] + if missing: + raise KeyError( + f"Mask column(s) {missing} not found in catalogue." + + " ShapePipe v2 catalogues carry the boolean columns" + + f" {list(MASK_COLUMNS)}; ShapePipe v1 catalogues carry" + + " IMAFLAGS_ISO instead and are not supported." + + f" Available columns: {sorted(available)}" + ) + + masked = np.zeros(len(dd[columns[0]]), dtype=bool) + for col in columns: + masked |= np.asarray(dd[col], dtype=bool) + + return ~masked + def classification_galaxy_overlap_ra_dec(dd, ra_key="XWIN_WORLD", dec_key="YWIN_WORLD"): """Classification Galaxy Overlap Ra Dec. @@ -141,11 +222,18 @@ def classification_galaxy_base( gal_mag_faint=26, flags_keep=None, n_epoch_min=1, + mask_columns=None, ): """Classification Galaxy Base. Return mask corresponding to basic classification for galaxies. + Parameters + ---------- + mask_columns : list of str, optional + ShapePipe mask columns OR'd together to reject masked objects; + default is ``DEFAULT_MASK_COLUMNS`` + """ # SExtractor flags # Keep some flags if specified @@ -172,7 +260,7 @@ def classification_galaxy_base( & cut_flags & (dd["MAG_AUTO"] <= gal_mag_faint) & (dd["MAG_AUTO"] >= gal_mag_bright) - & (dd["IMAFLAGS_ISO"] == 0) + & mask_cut(dd, mask_columns) & (dd["N_EPOCH"] >= n_epoch_min) ) From afbc4576ec80a9bdf47350ded47e817207bba35e Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Wed, 9 Sep 2026 20:22:06 -0400 Subject: [PATCH 05/39] retire patch logic (#340) ShapePipe v2 processes a campaign (a tile list); P1-P7 no longer exist. - catalog_builders.JointCat: get_patches()/get_n_obj() and the per-patch FITS merge are replaced by a merge over a list of campaign hdf5 files (-i final_cat_A.hdf5+final_cat_B.hdf5), read through read_campaign_catalogue. The 'patch' int8 column becomes a 'campaign' string column; the hdf5 root attr becomes 'campaigns'. - survey.get_footprint() deleted: it was a lookup table of P1-P7 (plus W3) RA/Dec boundaries, meaningless for a campaign. Its only caller, catalog.check_matching, used it as an optional pre-filter via a 'name' argument that every caller passed as None; the argument goes too, along with the test_survey test that exercised P5. - merge_psf_cat.py, combine_results.py, stats_tile_id_gal_counts.py, compute_area.py: patch vocabulary and v1/v1.5/v1.6 P-name shortcuts generalised to an explicit list of campaigns. - params.py: 'name = "P7"' becomes 'campaign = None'. Deleted (only ever meaningful for the P1-P7 era): - scripts/prepare_patch_for_spval.sh: symlinks a v1 per-patch run tree (~/psfex/${patch}/output/run_sp_Ms/.../full_starcat-0000000.fits, tiles_${patch}.txt) into a working dir; neither the layout nor the file names exist in v2. - scripts/plot_rho_stats_patches.py: globs P* directories and reads P*/output/run_sp_Pl/mccd_plots_runner/output/rho_stats_id.fits, one curve per patch. No campaign analogue. - scripts/survey_stats_all.sh: hardcoded 'for patch in P1 ... P7' over v1 run-directory bookkeeping. - scripts/star_match_stats.py: sums stats_file.txt over the seven patches. - scripts/check_tile_IDs_SP_LF.py: compares per-patch ShapePipe tile IDs against CFIS3500_THELI_P.list Lensfit files; both sides P-named. The word 'patch' survives only in catalog.py's comment naming the legacy hdf5 group, and in the treecorr jackknife sense (npatch/patch_number). Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01QbnPCyzuDNTgkg715pHhar --- docs/source/post_processing.md | 18 +- docs/source/run_validation.md | 7 +- docs/source/using_the_catalogues.md | 4 +- scripts/calibration/extract_info.py | 3 +- scripts/calibration/params.py | 5 +- scripts/check_tile_IDs_SP_LF.py | 83 ------- scripts/combine_results.py | 272 ++++++++++----------- scripts/compute_area.py | 10 +- scripts/merge_psf_cat.py | 71 +++--- scripts/plot_rho_stats_patches.py | 98 -------- scripts/prepare_patch_for_spval.sh | 31 --- scripts/star_match_stats.py | 44 ---- scripts/stats_tile_id_gal_counts.py | 43 ++-- scripts/survey_stats_all.sh | 94 ------- src/sp_validation/catalog.py | 10 +- src/sp_validation/catalog_builders.py | 324 +++++++++---------------- src/sp_validation/galaxy.py | 4 + src/sp_validation/rho_tau.py | 2 +- src/sp_validation/survey.py | 71 ------ src/sp_validation/tests/test_survey.py | 8 - workflow/image_sims/params_im_sim.py | 10 +- 21 files changed, 336 insertions(+), 876 deletions(-) delete mode 100644 scripts/check_tile_IDs_SP_LF.py delete mode 100755 scripts/plot_rho_stats_patches.py delete mode 100755 scripts/prepare_patch_for_spval.sh delete mode 100644 scripts/star_match_stats.py delete mode 100644 scripts/survey_stats_all.sh diff --git a/docs/source/post_processing.md b/docs/source/post_processing.md index 6d2f2423..d16e2e33 100644 --- a/docs/source/post_processing.md +++ b/docs/source/post_processing.md @@ -3,8 +3,8 @@ ## Science-ready catalogue production Processing steps of `ShapePipe` output catalogues carried out by the `sp_validation` package to produce science-ready catalogues are: -1. Extract relevant information from a final `ShapePipe` output catalogue per patch; run basic diagnostic tests, create pre-calibration shear catalogues. -2. Merge pre-calibration catalogues created in the previous step, e.g. processed by individual patches, into one or more joint catalogues; +1. Extract relevant information from a final `ShapePipe` output catalogue per campaign; run basic diagnostic tests, create pre-calibration shear catalogues. +2. Merge pre-calibration catalogues created in the previous step, e.g. processed as individual campaigns, into one or more joint catalogues; 3. Apply external area and footprint masks. These are the "structural" and the coverage masks. 4. Create calibrated galaxy shear catalogue. This step includes the tasks: a. Mask objects using flags and criteria in `ShapePipe` output catalogues and external (e.g. mask) files; @@ -19,7 +19,7 @@ This is performed (version > v1.4.1, < v2.0) with the python script `scripts/cal This script creates three shear catalogues in FITS format: - _Basic_ catalogue containing - positions, shapes (calibrated + PSF-leakage corrected), weights (DES), magnitude, patch ID. Masking and galaxy selection are applied. + positions, shapes (calibrated + PSF-leakage corrected), weights (DES), magnitude, campaign ID. Masking and galaxy selection are applied. - _Extended_ catalogue containing **in addition** uncalibrated shapes inverse-variance weights, shear response matrices, SNR, flux, size, PSF quantities. Masking and galaxy selection are applied. - _Comprehensive_ catalogue containing **in addition** @@ -27,11 +27,11 @@ This script creates three shear catalogues in FITS format: This catalogue does not contain calibrated shear estimates, since the calibration is carried out after applying masking and selection. This is the main output catalogue that will be processed further. -This step is carried out per patch. Parameters have to be set via the python configuration file `params.py` (template at `scripts/calibration/params.py`). +This step is carried out per campaign. Parameters have to be set via the python configuration file `params.py` (template at `scripts/calibration/params.py`). ### 2. Merge catalogues -The patch-wise comprehensive catalogues extracted in the previous step are merged using the script `scripts/calibration/create_joint_comprehensive_cat.py`, which is a front-end +The per-campaign comprehensive catalogues extracted in the previous step are merged using the script `scripts/calibration/create_joint_comprehensive_cat.py`, which is a front-end of the `sp_validation` library class `catalog_builders:JointCat`. ### 3. Apply external masks @@ -57,8 +57,8 @@ The following describes the pre-v1.4.2 method to create a joint, calibrated shea Summary statistics created by shear validation runs of sub-areas of a survey can be combined to create joint summary statistics. This is useful in cases where the galaxy catalogue of an entire survey is too large to process, and -needs to be broken down in smaller patches. This step provides global summary -statistics from those patches. +needs to be broken down into smaller campaigns. This step provides global +summary statistics from those campaigns. Depending on the type of summary, their combination can be the sum (e.g. for number of objects), average, weighted average (e.g. for the additive bias), the @@ -66,7 +66,7 @@ weighted average of the square (e.g. the ellipticity dispersion), the weighted variance (to combine variance estimates), or the weighted variance of the mean (to combine mean variance estimates). -In a directory containing the subpatches as subdirectories, and within each +In a directory containing the campaigns as subdirectories, and within each their own output directory (`sp_output`by default in `params.py`) with results of the validation runs, type ```bash @@ -89,7 +89,7 @@ calibration outputs can be used to create a combined, globally calibrated shear catalogue. The calibration is obtained from the files `R.txt` and `c.txt` created above. -In the same directory containing the subpatches as above, type +In the same directory containing the campaigns as above, type ```bash create_joint_shape_cat.py ``` diff --git a/docs/source/run_validation.md b/docs/source/run_validation.md index 7ee06767..8d3fded1 100644 --- a/docs/source/run_validation.md +++ b/docs/source/run_validation.md @@ -12,7 +12,7 @@ including the sheared values for metacalibration. All inputs and settings are contained in the python configuration script `scripts/calibration/params.py`, that needs to be edited accordingly. The main parameters are: -- `name`: field or patch name, can be any string. E.g. `P3` for patch 3. +- `campaign`: campaign name (the ShapePipe tile list), can be any string. - `data_dir`: input directory for data. Set to `.` for validation run in current directory. - `galaxy_cat_path`: path to galaxy catalogue, format `.fits`. or `.hdf5`. @@ -27,8 +27,9 @@ Optional parameters are: - `mask_external_path`: path to external mask file, format `.reg`. Set to `None` if not required. -See the script `prepare_patch_for_spval.sh` for an example of copying -the required input files to where the validation is to be run. +Link or copy the campaign's merged products -- `final_cat_.hdf5` +and `full_starcat_.hdf5` -- into the directory where the validation +is to be run. ### Run diff --git a/docs/source/using_the_catalogues.md b/docs/source/using_the_catalogues.md index 72781c12..db408c65 100644 --- a/docs/source/using_the_catalogues.md +++ b/docs/source/using_the_catalogues.md @@ -16,7 +16,7 @@ how to apply the metacalibration corrections yourself. ```{note} The examples below target catalogue **v1.0** (April 2022), which is distributed as FITS. From ShapePipe catalogue **v1.4.1** onward the merged catalogues ship -as HDF5 instead; open those with {func}`sp_validation.io.read_hdf5_file` (or +as HDF5 instead; open those with {func}`sp_validation.catalog.read_campaign_catalogue` (or `h5py` / `astropy`) in place of `astropy.io.fits` below — the column names and the calibration recipe are unchanged. ``` @@ -167,7 +167,7 @@ mask = np.full(len(data_ext), True) # Other examples: # mask = data_ext['mask_extern'] == 0 # LensFit-unmasked regions -# mask = data_ext['patch'] == 3 # patch P3 +# mask = data_ext['campaign'] == b'W3' # objects from campaign W3 # mask = data_ext['mag'] < 23.5 # r-band magnitude cut n_kept, n_all = np.count_nonzero(mask), len(data_ext) diff --git a/scripts/calibration/extract_info.py b/scripts/calibration/extract_info.py index 4d449416..b65c1975 100644 --- a/scripts/calibration/extract_info.py +++ b/scripts/calibration/extract_info.py @@ -149,7 +149,6 @@ [col_name_ra, col_name_dec], thresh, stats_file, - name=None, verbose=verbose, ) @@ -492,7 +491,7 @@ write_tile_id_gal_counts(detection_IDs, galaxy_IDs, shape_IDs, fname) # + -# Add all weights (for combining weighted averages of subpatches) +# Add all weights (for combining weighted averages of sub-samples) w_tot = np.sum(w) diff --git a/scripts/calibration/params.py b/scripts/calibration/params.py index 4b094540..c156bae2 100644 --- a/scripts/calibration/params.py +++ b/scripts/calibration/params.py @@ -25,9 +25,8 @@ # Survey parameters -## Field or patch name. Put None if n/a -name = "P7" -print("Field name = {}".format(name)) +## Campaign name (the tile list processed by ShapePipe). Put None if n/a +campaign = None ## Area of a tile in deg^2 area_tile = 0.25 diff --git a/scripts/check_tile_IDs_SP_LF.py b/scripts/check_tile_IDs_SP_LF.py deleted file mode 100644 index 54f23fda..00000000 --- a/scripts/check_tile_IDs_SP_LF.py +++ /dev/null @@ -1,83 +0,0 @@ -import re -import sys - -import numpy as np - - -def main(argv=None): - - survey = "v1" - - if survey == "v1": - n_patch = 7 - patches = [f"P{x}" for x in np.arange(n_patch) + 1] - - IDs_sp_base = "found_ID_wshapes.txt" - tile_ID_gal_counts_sp_base = "tile_id_gal_counts_ngmix.txt" - - n_SP_not_in_LF_all = 0 - n_LF_not_in_SP_all = 0 - - for patch in patches: - print(patch) - - # ShapePipe - path = f"{patch}/sp_output/{IDs_sp_base}" - - with open(path) as f: - dat = f.readlines() - ID_SP = [] - for line in dat: - ID_SP.append(line.rstrip()) - print(f" #SP = {len(ID_SP)}") - - # LensFit - path = f"CFIS3500_THELI_{patch}.list" - with open(path) as f: - dat = f.readlines() - ID_LF = [] - for line in dat: - m = re.match(r".*CFIS\.(\d{3}\.\d{3})\.r", line) - if m: - ID_LF.append(m[1].rstrip()) - print(f" #LF = {len(ID_LF)}") - - # tile stats for SP - dat = np.loadtxt(f"{patch}/sp_output/{tile_ID_gal_counts_sp_base}") - tile_ID = [] - for my_ID in dat[:, 0]: - tile_ID.append(f"{my_ID:07.3f}") - - # ShapePipe tiles not contained in LensFit - n_SP_not_in_LF = 0 - for ID in ID_SP: - if ID not in ID_LF: - # Print number of galaxies on those tiles not contained in LF - n_SP_not_in_LF += 1 - if n_SP_not_in_LF == -1: - print(f" SP {ID} not in LF") - n_SP_not_in_LF_all += n_SP_not_in_LF - - # LensFit tiles not contained in ShapePipe - n_LF_not_in_SP = 0 - for ID in ID_LF: - if ID not in ID_SP: - n_LF_not_in_SP += 1 - if n_LF_not_in_SP == -1: - print(f" LF {ID} not in SP") - print(f" [{ID}]") - n_LF_not_in_SP_all += n_LF_not_in_SP - - print(f" # SP not in LF = {n_SP_not_in_LF}") - print(f" # LF not in SP = {n_LF_not_in_SP}") - - print() - print("All patches") - print(f" # SP not in LF = {n_SP_not_in_LF_all}") - print(f" # LF not in SP = {n_LF_not_in_SP_all}") - - return 0 - - -if __name__ == "__main__": - sys.exit(main(sys.argv)) diff --git a/scripts/combine_results.py b/scripts/combine_results.py index 636d5c07..50cc91c5 100755 --- a/scripts/combine_results.py +++ b/scripts/combine_results.py @@ -10,11 +10,11 @@ from astropy.io import ascii -def get_match(stats_files, patch, pattern, previous=None, n_previous=[1], typ=str): +def get_match(stats_files, campaign, pattern, previous=None, n_previous=[1], typ=str): prev_ok = False - for idx, line in enumerate(stats_files[patch]): + for idx, line in enumerate(stats_files[campaign]): m = re.search(pattern, line) if m: if (previous and prev_ok) or not previous: @@ -29,19 +29,19 @@ def get_match(stats_files, patch, pattern, previous=None, n_previous=[1], typ=st for prev, n_prev in zip(previous, n_previous): # Line index n_previous earlier, +1 since next line will be read in # next loop; look for pattern in previous line - m_prev = re.search(prev, stats_files[patch][idx - n_prev + 1]) + m_prev = re.search(prev, stats_files[campaign][idx - n_prev + 1]) if m_prev: prev_ok = True raise ValueError( - f"No match of '{pattern}' in patch {patch} (prev='{previous}'), n_prev={n_previous}" + f"No match of '{pattern}' in campaign {campaign} (prev='{previous}'), n_prev={n_previous}" ) -def read_stats_files(patches, path, verbose=False): +def read_stats_files(campaigns, path, verbose=False): stats_files = {} - for p in patches: + for p in campaigns: fname = f"{p}/{path}" if os.path.exists(fname): if verbose: @@ -115,7 +115,7 @@ def combine(results): # Weight values w = np.array(list(results["value"][key_w].values())) - # Patch mean values + # Campaign mean values m = np.array(list(results["value"][key_m].values())) # Overall mean @@ -142,7 +142,7 @@ def combine(results): # Weight values w = np.array(list(results["value"][key_w].values())) - # Patch mean values + # Campaign mean values m = np.array(list(results["value"][key_m].values())) # Overall mean @@ -177,18 +177,18 @@ def print_all( # Header if header: - print("# patch", " " * 3, end=" ", file=fout) + print("# campaign", " " * 3, end=" ", file=fout) for key in keys: print(f"{key:>11s}", end=" ", file=fout) print(file=fout) - # Loop over patches - for patch in stats_files: - print(f"{patch:11s}", end=" ", file=fout) + # Loop over campaigns + for campaign in stats_files: + print(f"{campaign:11s}", end=" ", file=fout) # Write value for each key for key in keys: - val = results["value"][key][patch] + val = results["value"][key][campaign] if key == "N_gal": print(f"{val:>11.0f}", end=" ", file=fout) elif key in ("w_tot", "n_gal_am2"): @@ -222,7 +222,7 @@ def get_area(fname): with open(fname) as f: lines = f.readlines() for line in lines: - m = re.search("nmasked patch area without overlap = (.*) deg", line) + m = re.search("nmasked campaign area without overlap = (.*) deg", line) if m: return float(m[1]) @@ -234,30 +234,30 @@ def get_area(fname): def get_values(results, stats_files, shape, use_keys, area_deg2=-1): """Get Values - Get values from stats files for all patches. + Get values from stats files for all campaigns. Parameters ---------- results : dict results dictionary stats_files : dict of array of str - stats files content for each patch + stats files content for each campaign shape : str shape measurement method use_keys : dict keys to include area_deg2 : float, optional area in square degree, optional is -1 (to be retrieved - for each patch) + for each campaign) """ # Number of galaxies key = "N_gal" if use_keys[key]: init_key(results, key, "sum") - for patch in stats_files: - results["value"][key][patch] = get_match( + for campaign in stats_files: + results["value"][key][campaign] = get_match( stats_files, - patch, + campaign, r"Number of galaxies after metacal = (\d+)/", previous=[f"^{shape}$"], typ=int, @@ -266,15 +266,15 @@ def get_values(results, stats_files, shape, use_keys, area_deg2=-1): area_deg2_tot = 0 key_der = "n_gal_am2" init_key(results, key_der, "w_avg", extra="N_gal") - for patch in stats_files: + for campaign in stats_files: if area_deg2 < 0: - area_deg2_patch = get_area(f"{patch}/area.txt") - area_deg2_tot += area_deg2_patch - print(f"area({patch}) = {area_deg2_patch} deg^2") + area_deg2_campaign = get_area(f"{campaign}/area.txt") + area_deg2_tot += area_deg2_campaign + print(f"area({campaign}) = {area_deg2_campaign} deg^2") else: - area_deg2_patch = area_deg2 - results["value"][key_der][patch] = results["value"]["N_gal"][patch] / ( - area_deg2_patch * 3600 + area_deg2_campaign = area_deg2 + results["value"][key_der][campaign] = results["value"]["N_gal"][campaign] / ( + area_deg2_campaign * 3600 ) if area_deg2 < 0: @@ -285,10 +285,10 @@ def get_values(results, stats_files, shape, use_keys, area_deg2=-1): key = "w_tot" if use_keys[key]: init_key(results, key, "sum") - for patch in stats_files: - results["value"][key][patch] = get_match( + for campaign in stats_files: + results["value"][key][campaign] = get_match( stats_files, - patch, + campaign, r"Sum of weights = (\S+)", typ=float, previous=[f"^{shape}$"], @@ -300,16 +300,16 @@ def get_values(results, stats_files, shape, use_keys, area_deg2=-1): for comp in (1, 2): key = f"{key_base}{comp}" init_key(results, key, "w_avg", extra="N_gal") - for patch in stats_files: + for campaign in stats_files: c = get_match( stats_files, - patch, + campaign, rf"{key_base}{comp} = (\S+)", previous=[f"^{shape}:$"], n_previous=[2 * comp - 1], typ=float, ) - results["value"][key][patch] = c + results["value"][key][campaign] = c # Additive bias (unweighted error) key_base = "dc_" @@ -317,16 +317,16 @@ def get_values(results, stats_files, shape, use_keys, area_deg2=-1): for comp in (1, 2): key = f"{key_base}{comp}" init_key(results, key, "var", extra=["N_gal", f"c_{comp}"]) - for patch in stats_files: + for campaign in stats_files: dc = get_match( stats_files, - patch, + campaign, rf"{key_base}{comp} = (\S+)", typ=float, previous=[f"^{shape}:$"], n_previous=[2 * comp + 7], ) - results["value"][key][patch] = dc + results["value"][key][campaign] = dc # Additive bias (unweighted error of mean) key_base = "dmc_" @@ -334,16 +334,16 @@ def get_values(results, stats_files, shape, use_keys, area_deg2=-1): for comp in (1, 2): key = f"{key_base}{comp}" init_key(results, key, "var_m", extra=["N_gal", f"c_{comp}"]) - for patch in stats_files: + for campaign in stats_files: dmc = get_match( stats_files, - patch, + campaign, rf"{key_base}{comp} = (\S+)", typ=float, previous=[f"^{shape}:$"], n_previous=[2 * comp + 7], ) - results["value"][key][patch] = dmc + results["value"][key][campaign] = dmc # Additive bias (weighted mean) key_base = "cw_" @@ -351,16 +351,16 @@ def get_values(results, stats_files, shape, use_keys, area_deg2=-1): for comp in (1, 2): key = f"{key_base}{comp}" init_key(results, key, "w_avg", extra="w_tot") - for patch in stats_files: + for campaign in stats_files: c = get_match( stats_files, - patch, + campaign, rf"{key_base}{comp} = (\S+)", previous=[f"^{shape}:$"], n_previous=[comp * 2], typ=float, ) - results["value"][key][patch] = c + results["value"][key][campaign] = c # Additive bias (weighted error of mean) key_base = "dmcw_" @@ -368,16 +368,16 @@ def get_values(results, stats_files, shape, use_keys, area_deg2=-1): for comp in (1, 2): key = f"{key_base}{comp}" init_key(results, key, "var_m", extra=["N_gal", f"cw_{comp}"]) - for patch in stats_files: + for campaign in stats_files: dmc = get_match( stats_files, - patch, + campaign, rf"{key_base}{comp} = (\S+)", typ=float, previous=[f"^{shape}:$"], n_previous=[2 * comp + 8], ) - results["value"][key][patch] = dmc + results["value"][key][campaign] = dmc # Additive bias (jackknife) key_base = "cjk_" @@ -393,26 +393,26 @@ def get_values(results, stats_files, shape, use_keys, area_deg2=-1): key_s = key init_key(results, key, "var_m", extra=["w_tot", f"cjk_{comp}"]) - for patch in stats_files: + for campaign in stats_files: c, dc = get_match( stats_files, - patch, + campaign, rf"{key_base}{comp} = (\S+)", previous=[f"^{shape}:$"], n_previous=[comp], typ="ufloat", ) - results["value"][key_m][patch] = c - results["value"][key_s][patch] = dc + results["value"][key_m][campaign] = c + results["value"][key_s][campaign] = dc # Ellipticity dispersion key = "sigma2_epsilon" if use_keys[key]: init_key(results, key, "w_avg", extra="N_gal") - for patch in stats_files: - results["value"][key][patch] = get_match( + for campaign in stats_files: + results["value"][key][campaign] = get_match( stats_files, - patch, + campaign, r"Dispersion of complex ellipticity = (\S+)", previous=[f"^{shape}$"], typ=float, @@ -424,60 +424,60 @@ def get_values(results, stats_files, shape, use_keys, area_deg2=-1): keys = [f"{key_base}11", f"{key_base}12", f"{key_base}21", f"{key_base}22"] for key in keys: init_key(results, key, "w_avg", extra="N_gal") - for patch in stats_files: + for campaign in stats_files: tmp = get_match( stats_files, - patch, + campaign, r"\[\[(\s?\S+)\s+\S+]", previous=["ngmix galaxies:", "total response matrix:"], n_previous=[2, 1], typ=float, ) - results["value"][keys[0]][patch] = tmp + results["value"][keys[0]][campaign] = tmp tmp = get_match( stats_files, - patch, + campaign, r"\[\[\s?\S+\s+(\S+)]", previous=["ngmix galaxies:", "total response matrix:"], n_previous=[2, 1], typ=float, ) - results["value"][keys[1]][patch] = tmp + results["value"][keys[1]][campaign] = tmp tmp = get_match( stats_files, - patch, + campaign, r"\[(\s?\S+)\s+\S+\]\]", previous=["ngmix galaxies", "total response matrix:"], n_previous=[3, 2], typ=float, ) - results["value"][keys[2]][patch] = tmp + results["value"][keys[2]][campaign] = tmp tmp = get_match( stats_files, - patch, + campaign, r" \[\s?\S+\s+(\S+)\]\]", previous=["ngmix galaxies:", "total response matrix:"], n_previous=[3, 2], typ=float, ) - results["value"][keys[3]][patch] = tmp + results["value"][keys[3]][campaign] = tmp # Normalised trace = mean diagonal key_der = "trN_R_tot" init_key(results, key_der, "w_avg", extra="N_gal") - for patch in stats_files: - results["value"][key_der][patch] = ( - results["value"]["R_tot_11"][patch] - + results["value"]["R_tot_22"][patch] + for campaign in stats_files: + results["value"][key_der][campaign] = ( + results["value"]["R_tot_11"][campaign] + + results["value"]["R_tot_22"][campaign] ) / 2 # Sum of absolute off-diagonal key_der = "abs_off_R_tot" init_key(results, key_der, "w_avg", extra="N_gal") - for patch in stats_files: - results["value"][key_der][patch] = np.abs( - results["value"]["R_tot_12"][patch] - ) + np.abs(results["value"]["R_tot_21"][patch]) + for campaign in stats_files: + results["value"][key_der][campaign] = np.abs( + results["value"]["R_tot_12"][campaign] + ) + np.abs(results["value"]["R_tot_21"][campaign]) # Galaxy shear response matrix key_base = "R_shear_" @@ -485,60 +485,60 @@ def get_values(results, stats_files, shape, use_keys, area_deg2=-1): keys = [f"{key_base}11", f"{key_base}12", f"{key_base}21", f"{key_base}22"] for key in keys: init_key(results, key, "w_avg", extra="N_gal") - for patch in stats_files: + for campaign in stats_files: tmp = get_match( stats_files, - patch, + campaign, r"\[\[(\s?\S+)\s+\S+]", previous=["ngmix galaxies:", "shear response matrix:"], n_previous=[5, 1], typ=float, ) - results["value"][keys[0]][patch] = tmp + results["value"][keys[0]][campaign] = tmp tmp = get_match( stats_files, - patch, + campaign, r"\[\[\s?\S+\s+(\S+)]", previous=["ngmix galaxies:", "shear response matrix:"], n_previous=[5, 1], typ=float, ) - results["value"][keys[1]][patch] = tmp + results["value"][keys[1]][campaign] = tmp tmp = get_match( stats_files, - patch, + campaign, r"\[(\s?\S+)\s+\S+\]\]", previous=["ngmix galaxies", "shear response matrix:"], n_previous=[6, 2], typ=float, ) - results["value"][keys[2]][patch] = tmp + results["value"][keys[2]][campaign] = tmp tmp = get_match( stats_files, - patch, + campaign, r" \[\s?\S+\s+(\S+)\]\]", previous=["ngmix galaxies:", "shear response matrix:"], n_previous=[6, 2], typ=float, ) - results["value"][keys[3]][patch] = tmp + results["value"][keys[3]][campaign] = tmp # Normalised trace = mean diagonal key_der = "trN_R_shear" init_key(results, key_der, "w_avg", extra="N_gal") - for patch in stats_files: - results["value"][key_der][patch] = ( - results["value"]["R_shear_11"][patch] - + results["value"]["R_shear_22"][patch] + for campaign in stats_files: + results["value"][key_der][campaign] = ( + results["value"]["R_shear_11"][campaign] + + results["value"]["R_shear_22"][campaign] ) / 2 # Sum of absolute off-diagonal key_der = "abs_off_R_shear" init_key(results, key_der, "w_avg", extra="N_gal") - for patch in stats_files: - results["value"][key_der][patch] = np.abs( - results["value"]["R_shear_12"][patch] - ) + np.abs(results["value"]["R_shear_21"][patch]) + for campaign in stats_files: + results["value"][key_der][campaign] = np.abs( + results["value"]["R_shear_12"][campaign] + ) + np.abs(results["value"]["R_shear_21"][campaign]) # Galaxy selection response matrix key_base = "R_select_" @@ -546,60 +546,60 @@ def get_values(results, stats_files, shape, use_keys, area_deg2=-1): keys = [f"{key_base}11", f"{key_base}12", f"{key_base}21", f"{key_base}22"] for key in keys: init_key(results, key, "w_avg", extra="N_gal") - for patch in stats_files: + for campaign in stats_files: tmp = get_match( stats_files, - patch, + campaign, r"\[\[(\s?\S+)\s+\S+]", previous=["ngmix galaxies:", "selection response matrix:"], n_previous=[8, 1], typ=float, ) - results["value"][keys[0]][patch] = tmp + results["value"][keys[0]][campaign] = tmp tmp = get_match( stats_files, - patch, + campaign, r"\[\[\s?\S+\s+(\S+)]", previous=["ngmix galaxies:", "selection response matrix:"], n_previous=[8, 1], typ=float, ) - results["value"][keys[1]][patch] = tmp + results["value"][keys[1]][campaign] = tmp tmp = get_match( stats_files, - patch, + campaign, r"\[(\s?\S+)\s+\S+\]\]", previous=["ngmix galaxies", "selection response matrix:"], n_previous=[9, 2], typ=float, ) - results["value"][keys[2]][patch] = tmp + results["value"][keys[2]][campaign] = tmp tmp = get_match( stats_files, - patch, + campaign, r" \[\s?\S+\s+(\S+)\]\]", previous=["ngmix galaxies:", "selection response matrix:"], n_previous=[9, 2], typ=float, ) - results["value"][keys[3]][patch] = tmp + results["value"][keys[3]][campaign] = tmp # Normalised trace = mean diagonal key_der = "trN_R_select" init_key(results, key_der, "w_avg", extra="N_gal") - for patch in stats_files: - results["value"][key_der][patch] = ( - results["value"]["R_select_11"][patch] - + results["value"]["R_select_22"][patch] + for campaign in stats_files: + results["value"][key_der][campaign] = ( + results["value"]["R_select_11"][campaign] + + results["value"]["R_select_22"][campaign] ) / 2 # Sum of absolute off-diagonal key_der = "abs_off_R_select" init_key(results, key_der, "w_avg", extra="N_gal") - for patch in stats_files: - results["value"][key_der][patch] = np.abs( - results["value"]["R_select_12"][patch] - ) + np.abs(results["value"]["R_select_21"][patch]) + for campaign in stats_files: + results["value"][key_der][campaign] = np.abs( + results["value"]["R_select_12"][campaign] + ) + np.abs(results["value"]["R_select_21"][campaign]) # Object-wise PSF leakage key_base = "m_" @@ -614,70 +614,70 @@ def get_values(results, stats_files, shape, use_keys, area_deg2=-1): ] for key in keys: init_key(results, key, "w_avg", extra="N_gal") - for patch in stats_files: + for campaign in stats_files: m, dm = get_match( stats_files, - patch, + campaign, "\\$e_\\{1\\}\\^\\{\\\\rm PSF\\}\\$: m_1=(\\S*)", previous=["ngmix"], n_previous=[1], typ="ufloat", ) - results["value"]["m_11"][patch] = m + results["value"]["m_11"][campaign] = m m, dm = get_match( stats_files, - patch, + campaign, "\\$e_\\{1\\}\\^\\{\\\\rm PSF\\}\\$: m_2=(\\S*)", previous=["ngmix"], n_previous=[2], typ="ufloat", ) - results["value"]["m_12"][patch] = m + results["value"]["m_12"][campaign] = m m, dm = get_match( stats_files, - patch, + campaign, "\\$e_\\{2\\}\\^\\{\\\\rm PSF\\}\\$: m_1=(\\S*)", previous=["ngmix"], n_previous=[3], typ="ufloat", ) - results["value"]["m_21"][patch] = m + results["value"]["m_21"][campaign] = m m, dm = get_match( stats_files, - patch, + campaign, "\\$e_\\{2\\}\\^\\{\\\\rm PSF\\}\\$: m_2=(\\S*)", previous=["ngmix"], n_previous=[4], typ="ufloat", ) - results["value"]["m_22"][patch] = m + results["value"]["m_22"][campaign] = m m, dm = get_match( stats_files, - patch, + campaign, "\\$\\\\mathrm\\{FWHM\\}\\^\\{\\\\rm PSF\\}\\$ \\[arcsec]: m_1=(\\S+)", previous=["ngmix"], n_previous=[5], typ="ufloat", ) - results["value"]["m_s1"][patch] = m + results["value"]["m_s1"][campaign] = m m, dm = get_match( stats_files, - patch, + campaign, "\\$\\\\mathrm\\{FWHM\\}\\^\\{\\\\rm PSF\\}\\$ \\[arcsec]: m_2=(\\S+)", previous=["ngmix"], n_previous=[6], typ="ufloat", ) - results["value"]["m_s2"][patch] = m + results["value"]["m_s2"][campaign] = m # Scale-dependent PSF leakage key = "alpha" if use_keys[key]: init_key(results, key, "w_avg", extra="N_gal") - for patch in stats_files: - results["value"][key][patch] = get_match( + for campaign in stats_files: + results["value"][key][campaign] = get_match( stats_files, - patch, + campaign, r"ngmix: Weighted average alpha =(\s?\S+)", typ=float, ) @@ -687,15 +687,15 @@ def get_values(results, stats_files, shape, use_keys, area_deg2=-1): if use_keys[key_base]: init_key(results, "xi_sys_p", "w_avg", extra="N_gal") init_key(results, "xi_sys_m", "w_avg", extra="N_gal") - for patch in stats_files: + for campaign in stats_files: tmp = get_match( - stats_files, patch, r"ngmix: <\|xi_sys_\+\|> = (\S*)", typ=float + stats_files, campaign, r"ngmix: <\|xi_sys_\+\|> = (\S*)", typ=float ) - results["value"]["xi_sys_p"][patch] = tmp + results["value"]["xi_sys_p"][campaign] = tmp tmp = get_match( - stats_files, patch, r"ngmix: <\|xi_sys_\-\|> = (\S*)", typ=float + stats_files, campaign, r"ngmix: <\|xi_sys_\-\|> = (\S*)", typ=float ) - results["value"]["xi_sys_m"][patch] = tmp + results["value"]["xi_sys_m"][campaign] = tmp def latex_table(file_base, cols=None, col_names=None): @@ -714,7 +714,7 @@ def latex_table(file_base, cols=None, col_names=None): print(r"}\hline\hline", file=fout) # Table header - str_line = "patch\t&" + str_line = "campaign\t&" for name in col_names: str_line = f"{str_line} ${name}$\t&" # str_line = f'{str_line} \\multicolumn{{2}}{{c}}{{${name}$}}\t&' @@ -724,7 +724,7 @@ def latex_table(file_base, cols=None, col_names=None): for nl in range(n_lines): str_line = "" - str_line = f"{str_line}{dat['patch'][nl]}\t&" + str_line = f"{str_line}{dat['campaign'][nl]}\t&" for col in cols: if len(col) == 2: @@ -767,25 +767,17 @@ def main(argv=None): if argv[1] == "snr": # All directories - patches = [f.path for f in os.scandir(".") if f.is_dir()] + campaigns = [f.path for f in os.scandir(".") if f.is_dir()] all = False - elif argv[1] == "v1": - n_patch = 7 - patches = [f"P{x}" for x in np.arange(n_patch) + 1] - elif argv[1] == "v1.5": - n_patch = 8 - patches = [f"P{x}" for x in np.arange(n_patch) + 1] - elif argv[1] == "test": - patches = ["P7", "W3", "S4"] elif argv[1] == "comb": # Validate with combined catalogue - patches = ["comb"] + campaigns = ["comb"] else: - patches = argv[1].split("+") + campaigns = argv[1].split("+") - n_patch = len(patches) + n_campaign = len(campaigns) - print("combine_results.py:", patches) + print("combine_results.py:", campaigns) directory = "sp_output/plots" fbase = "stats_file" @@ -796,7 +788,7 @@ def main(argv=None): verbose = False - stats_files = read_stats_files(patches, path, verbose=verbose) + stats_files = read_stats_files(campaigns, path, verbose=verbose) results = {"value": {}, "type": {}, "extra": {}, "all": {}} diff --git a/scripts/compute_area.py b/scripts/compute_area.py index afe797ad..1e9bfa50 100755 --- a/scripts/compute_area.py +++ b/scripts/compute_area.py @@ -21,7 +21,7 @@ def main(argv=None): random_log_path = "output/run_sp_Rc/random_cat_runner/logs" log_file_base = "process" - # Get expected number of tiles in patch + # Get expected number of tiles in campaign num_lines = sum(1 for _ in open(tile_ID_path)) print(f"Found {num_lines} tiles in ID file {tile_ID_path}") @@ -108,9 +108,9 @@ def main(argv=None): area_deg2_non_overl_tile = ufloat( np.mean(area_deg2_non_overl), np.std(area_deg2_non_overl) ) - print(f"Patch area without overlap = {area_deg2_non_overl_total:.3f} deg^2") + print(f"Campaign area without overlap = {area_deg2_non_overl_total:.3f} deg^2") print( - f"Patch area without overlap and no 0 gal = {area_deg2_non_overl_total_wgal:.3f} deg^2" + f"Campaign area without overlap and no 0 gal = {area_deg2_non_overl_total_wgal:.3f} deg^2" ) print(f"Tile area without overlap = {area_deg2_non_overl_tile:.3fP} deg^2") @@ -120,10 +120,10 @@ def main(argv=None): np.mean(area_deg2_eff_non_overl), np.std(area_deg2_eff_non_overl) ) print( - f"Unmasked patch area without overlap = {area_deg2_eff_non_overl_total:.3f} deg^2" + f"Unmasked campaign area without overlap = {area_deg2_eff_non_overl_total:.3f} deg^2" ) print( - f"Unmasked patch area without overlap and no 0 gal = {area_deg2_eff_non_overl_total_wgal:.3f} deg^2" + f"Unmasked campaign area without overlap and no 0 gal = {area_deg2_eff_non_overl_total_wgal:.3f} deg^2" ) print( f"Unmasked tile area without overlap = {area_deg2_eff_non_overl_tile:.3fP} deg^2" diff --git a/scripts/merge_psf_cat.py b/scripts/merge_psf_cat.py index 65c1e891..adda9d22 100644 --- a/scripts/merge_psf_cat.py +++ b/scripts/merge_psf_cat.py @@ -1,7 +1,7 @@ #!/usr/bin/env python3 """MERGE PSF CAT. -Merge PSF catalogues (psf_catalog_ngmix.fits) from different patches +Merge PSF catalogues (psf_catalog_ngmix.fits) from different campaigns into a single FITS file. :Author: Martin Kilbinger @@ -18,7 +18,7 @@ class MergePsfCat: """Merge Psf Cat. - Class to merge PSF catalogues from multiple patches. + Class to merge PSF catalogues from multiple campaigns. """ @@ -32,7 +32,7 @@ def params_default(self): """ self._params = { - "patches": "v1", + "campaigns": None, "sh": "ngmix", "survey": "unions", "year": "2024", @@ -43,7 +43,7 @@ def params_default(self): "verbose": False, } self._short_options = { - "patches": "-p", + "campaigns": "-p", "sh": "-g", "survey": "-s", "year": "-y", @@ -55,15 +55,12 @@ def params_default(self): "hdu": "int", } self._help_strings = { - "patches": ( - "list of patches separated by '+', or shortcut " - "(allowed are 'v1', 'v1.5'), default={}" - ), + "campaigns": "list of campaigns separated by '+'", "sh": "shape measurement method, default={}", "survey": "survey name, default={}", "year": "year of processing, default={}", "version": "catalogue version, default={}", - "base_path": "base path containing patch directories, default={}", + "base_path": "base path containing campaign directories, default={}", "hdu": "HDU number to read from input FITS files, default={}", } @@ -85,40 +82,32 @@ def set_params_from_command_line(self, args): logging.log_command(args) - def get_patches(self): - """Get Patches. + def get_campaigns(self): + """Get Campaigns. - Return list of patches according to option parameter value. + Return list of campaigns according to option parameter value. Returns ------- list - patches, list of str + campaigns, list of str """ - if self._params["patches"] == "v1": - n_patch = 7 - patches = [f"P{x}" for x in np.arange(n_patch) + 1] - elif self._params["patches"] == "v1.5": - n_patch = 8 - patches = [f"P{x}" for x in np.arange(n_patch) + 1] - elif self._params["patches"] == "v1.6": - n_patch = 9 - patches = [f"P{x}" for x in np.arange(n_patch) + 1] - else: - patches = self._params["patches"].split("+") - - return patches - - def merge_catalogues(self, patches): + campaigns = self._params["campaigns"] + if not campaigns: + raise ValueError("No campaigns given; set 'campaigns'") + + return campaigns.split("+") + + def merge_catalogues(self, campaigns): """Merge Catalogues. - Merge PSF catalogues from sub-patches into one FITS file. + Merge PSF catalogues from campaigns into one FITS file. Parameters ---------- - patches : list of str - list of patches/sub-directories + campaigns : list of str + list of campaigns / sub-directories """ base_path = self._params["base_path"] @@ -133,11 +122,11 @@ def merge_catalogues(self, patches): ) dat_all = {} - for idx, patch in enumerate(patches): + for idx, campaign in enumerate(campaigns): if verbose: - print(f" {patch}") + print(f" {campaign}") - input_path = f"{base_path}/{patch}/{input_sub_path}" + input_path = f"{base_path}/{campaign}/{input_sub_path}" try: dat = fits.getdata(input_path, hdu_in) except Exception: @@ -149,18 +138,18 @@ def merge_catalogues(self, patches): col_names = dat.dtype.names for name in col_names: dat_all[name] = [] - dat_all["patch"] = [] + dat_all["campaign"] = [] for name in col_names: dat_all[name] = np.append(dat_all[name], dat[name]) - dat_all["patch"] = np.append(dat_all["patch"], [idx + 1] * len(dat)) + dat_all["campaign"] = np.append(dat_all["campaign"], [idx + 1] * len(dat)) - col_names = col_names + ("patch",) + col_names = col_names + ("campaign",) column_all = [] for name in col_names: - if name != "patch": + if name != "campaign": my_format = "D" else: my_format = "I" @@ -177,12 +166,12 @@ def run(self): Main processing function. """ - patches = self.get_patches() + campaigns = self.get_campaigns() if self._params["verbose"]: - print("Merging PSF catalogues from patches:", patches) + print("Merging PSF catalogues from campaigns:", campaigns) - self.merge_catalogues(patches) + self.merge_catalogues(campaigns) def main(argv=None): diff --git a/scripts/plot_rho_stats_patches.py b/scripts/plot_rho_stats_patches.py deleted file mode 100755 index 915c07d1..00000000 --- a/scripts/plot_rho_stats_patches.py +++ /dev/null @@ -1,98 +0,0 @@ -# --- -# jupyter: -# jupytext: -# text_representation: -# extension: .py -# format_name: light -# format_version: '1.5' -# jupytext_version: 1.15.1 -# kernelspec: -# display_name: sp_validation -# language: python -# name: python3 -# --- - -import glob -import itertools -import os - -import matplotlib.pylab as plt -from cs_util import args -from shear_psf_leakage.rho_tau_stat import RhoStat - - -# + -# Set parameters from file or user input -class dummy(object): - def __init__(self): - - self._params = { - "in_dir_base": ".", - "title": None, - } - - -obj = dummy() -params_upd = args.read_param_script("params_rho.py", obj._params, verbose=True) -for key in params_upd: - obj._params[key] = params_upd[key] - -# patches = [f'P{x}' for x in np.arange(n_patch) + 1] -patches = glob.glob("P*") -print(patches) - -default_colors = plt.rcParams["axes.prop_cycle"].by_key()["color"] -color_cycle = itertools.cycle(default_colors) -col = {} -for patch in patches: - col[patch] = next(color_cycle) - -coord_units = "deg" -theta_min = 0.1 -theta_max = 250 -sep_units = "arcmin" -nbins = 20 - -# ## Set up -TreeCorrConfig = { - "ra_units": coord_units, - "dec_units": coord_units, - "min_sep": theta_min, - "max_sep": theta_max, - "sep_units": sep_units, - "nbins": nbins, - "var_method": "bootstrap", -} -# - - -rho_stat_handler = RhoStat( - output=obj._params["in_dir_base"], treecorr_config=TreeCorrConfig, verbose=True -) - -# + -filenames = [] -colors = [] - -for patch in patches: - path = f"{patch}/output/run_sp_Pl/mccd_plots_runner/output/rho_stats_id.fits" - if os.path.exists(f"{obj._params['in_dir_base']}/{path}"): - print(f"Reading rho stats {obj._params['in_dir_base']}/{path}...") - rho_stat_handler.load_rho_stats(path) - filenames.append(path) - colors.append(col[patch]) - else: - print( - f"File rho stats {obj._params['in_dir_base']}/{path} not found, skipping.." - ) -# - - -# Create plot -rho_stat_handler.plot_rho_stats( - filenames, - colors, - patches, - abs=False, - savefig="rho_stats.png", - legend="outside", - title=obj._params["title"], -) diff --git a/scripts/prepare_patch_for_spval.sh b/scripts/prepare_patch_for_spval.sh deleted file mode 100755 index 634d901f..00000000 --- a/scripts/prepare_patch_for_spval.sh +++ /dev/null @@ -1,31 +0,0 @@ -#!/usr/bin/env bash - -patch=$1 - -spdir=$HOME/astro/repositories/github/sp_validation - -# Galaxy catalogue -#cp ~/psfex/final_cat_${patch}.hdf5 . -ln -s ~/psfex/final_cat_${patch}.hdf5 - -# Parameter file, to avoid read errors for hdf5 file -ln -sf ~/shapepipe/example/cfis/final_cat.param - -# Star catalogue -## Ellipticities in pixel coordinates, MCCD output -# ln -sf $HOME/psfex/${patch}/output/run_sp_Ms/merge_starcat_runner/output/full_starcat-0000000.fits - -## Projected back to world coordinates -ln -sf $HOME/psfex/star_cat/${patch}/output/run_sp_Ms/merge_starcat_runner/output/full_starcat-0000000.fits - -# Tile number list -ln -sf ~/shapepipe/auxdir/CFIS/tiles_202106/tiles_${patch}.txt - -# Parameter file -#cp $spdir/scripts/calibration/params.py . -echo "Diff:" -diff $spdir/scripts/calibration/params.py params.py -echo "Run?" -echo "cp $spdir/scripts/calibration/params.py params.py" -echo "Run?" -echo "ipython $spdir/scripts/calibration/extract_info.py" diff --git a/scripts/star_match_stats.py b/scripts/star_match_stats.py deleted file mode 100644 index 5c74ed5c..00000000 --- a/scripts/star_match_stats.py +++ /dev/null @@ -1,44 +0,0 @@ -#!/usr/bin/env python - -import re -import sys - -import numpy as np - - -def main(argv=None): - - types = ["star", "gal", "other"] - text = "Number of stars selected as" - - n_patch = 7 - patches = [f"P{x}" for x in np.arange(n_patch) + 1] - - ntyp = {} - ntot = {} - for typ in types: - ntyp[typ] = 0 - ntot[typ] = 0 - - for patch in patches: - # print(patch) - path = f"{patch}/sp_output/plots/stats_file.txt" - with open(path, "r") as fin: - lines = fin.readlines() - for typ in types: - for line in lines: - pattern = rf"{text} {typ}.*= (\d+)/(\d+)" - m = re.search(pattern, line) - if m: - ntyp_patch = int(m.group(1)) - ntot_patch = int(m.group(2)) - # print(typ, m.group(1), m.group(2)) - ntyp[typ] += ntyp_patch - ntot[typ] += ntot_patch - - for typ in types: - print(f"{text} {typ} = {ntyp[typ]}/{ntot[typ]} = {ntyp[typ] / ntot[typ]:.2%}") - - -if __name__ == "__main__": - sys.exit(main(sys.argv)) diff --git a/scripts/stats_tile_id_gal_counts.py b/scripts/stats_tile_id_gal_counts.py index caed59ae..26adca7f 100755 --- a/scripts/stats_tile_id_gal_counts.py +++ b/scripts/stats_tile_id_gal_counts.py @@ -1,6 +1,7 @@ #!/usr/bin/env python3 import copy +import os import sys from optparse import OptionParser @@ -40,7 +41,7 @@ def params_default(): parameter values """ - p_def = param(survey="v1") + p_def = param(campaigns=None) return p_def @@ -66,7 +67,13 @@ def parse_options(p_def): # I/O parser.add_option("-i", "--input", dest="input", type="string", help="input file") - parser.add_option("-s", "--survey", dest="survey", type="string", help="survey") + parser.add_option( + "-c", + "--campaigns", + dest="campaigns", + type="string", + help="campaigns separated by '+'", + ) options, args = parser.parse_args() @@ -87,8 +94,8 @@ def check_options(options): Result of option check. False if invalid option value. """ - if not options.input and not options.survey: - print("Either input or survey need to be specified") + if not options.input and not options.campaigns: + print("Either input or campaigns need to be specified") return False return True @@ -148,13 +155,13 @@ def main(argv=None): # Get input file paths print("Retrieving input file paths") input_files = [] - if param.survey == "v1": - n_patch = 7 - patches = [f"P{x}" for x in np.arange(n_patch) + 1] - for patch in patches: - path = f"{patch}/sp_output/tile_id_gal_counts_{sh}.txt" + if param.campaigns: + campaigns = param.campaigns.split("+") + for campaign in campaigns: + path = f"{campaign}/sp_output/tile_id_gal_counts_{sh}.txt" input_files.append(path) else: + campaigns = [os.path.dirname(param.input) or "."] input_files = [param.input] print(f"Found {len(input_files)} input files") @@ -166,11 +173,11 @@ def main(argv=None): n_gal_arr = [] n_shape_arr = [] - for patch, input_path in zip(patches, input_files): - dat[patch] = np.loadtxt(input_path) - n_det = dat[patch][:, 1] - n_gal = dat[patch][:, 2] - n_shape = dat[patch][:, 3] + for campaign, input_path in zip(campaigns, input_files): + dat[campaign] = np.loadtxt(input_path) + n_det = dat[campaign][:, 1] + n_gal = dat[campaign][:, 2] + n_shape = dat[campaign][:, 3] n_det_arr.extend(n_det) n_gal_arr.extend(n_gal) @@ -178,11 +185,11 @@ def main(argv=None): # Write tile IDs with number of shapes > 0 print("Writing tile IDs with n_shapes>0") - for patch in patches: - out_path = f"{patch}/sp_output/found_ID_wshapes.txt" + for campaign in campaigns: + out_path = f"{campaign}/sp_output/found_ID_wshapes.txt" with open(out_path, "w") as f_out: - mask_n_shape = dat[patch][:, 3] > 0 - tile_ID_masked = dat[patch][mask_n_shape, 0] + mask_n_shape = dat[campaign][:, 3] > 0 + tile_ID_masked = dat[campaign][mask_n_shape, 0] for ID in tile_ID_masked: print(f"{ID:07.3f}", file=f_out) diff --git a/scripts/survey_stats_all.sh b/scripts/survey_stats_all.sh deleted file mode 100644 index f6c4cd5f..00000000 --- a/scripts/survey_stats_all.sh +++ /dev/null @@ -1,94 +0,0 @@ -#!/usr/bin/bash - -fbase_found='found_ID' -fbase_found_wsh='found_ID_wshapes' -fbase_lf='CFIS3500_THELI' -rm -f ${fbase_found}_all.txt -rm -f ${fbase_found_wsh}_all.txt -rm -f ${fbase_found_random}_all.txt -rm -f ${fbase_lf}_all.txt - -for patch in P1 P2 P3 P4 P5 P6 P7; do - - # Patch - - ## Total number of tiles - wc -l $patch/tiles_P?.txt - - - # Final catalogue - #echo "Final catalogue" - - ## Number of final catalogues (.tgz) - ntgz=`ls -rtl $patch/final*.tgz | wc -l` - echo "$ntgz final .tgz cats" - - ## Number of final catalogues (.fits) - nfits=`ls -rtl $patch/output/run_sp_combined/make_catalog_runner/output/final* | wc -l` - echo "$nfits final .fits cats" - - ## Number of merged final catalogues - if [ -e $patch/log_merge_final_gal_cat ]; then - tail -n 1 $patch/log_merge_final_gal_cat - fi - - ## Number of tile IDs found in merged catalogue - if [ -e $patch/sp_output/${fbase_found}.txt ]; then - wc -l $patch/sp_output/${fbase_found}.txt - wc -l $patch/sp_output/${fbase_found}.txt >> ${fbase_found}_all.txt - fi - - ## Number of tile IDs found in merged catalogue with shapes - if [ -e $patch/sp_output/${fbase_found_wsh}.txt ]; then - wc -l $patch/sp_output/${fbase_found_wsh}.txt - wc -l $patch/sp_output/${fbase_found_wsh}.txt >> ${fbase_found_wsh}_all.txt - fi - - - # Random catalogue - - ## Number of final catalogues (.tgz) - ntgz=`ls -rtl $patch/pipeline_flag*.tgz | wc -l` - echo "$ntgz random .tgz cats" - - - ## Number of random catalogues (.fits) - if [ -d $patch/output/run_sp_combined_flag ]; then - nfits=`ls -rtl $patch/output/run_sp_combined_flag/mask_runner/output/pip* | wc -l` - echo "$nfits random .fits cats" - fi - - ## Number of merged random catalogues - if [ -e $patch/log_merge_final_rand_cat ]; then - tail -n 1 $patch/log_merge_final_rand_cat - fi - - ### Number of tiles for random catalogue validation - if [ -e $patch/sp_output_random/${fbase_found}.txt ]; then - wc -l $patch/sp_output_random/${fbase_found}.txt - wc -l $patch/sp_output_random/${fbase_found}.txt >> ${fbase_found_random}_all.txt - fi - - # LensFit tile IDs - if [ -e ${fbase_lf}_$patch.list ]; then - wc -l ${fbase_lf}_$patch.list - wc -l ${fbase_lf}_$patch.list >> ${fbase_lf}_all.txt - fi - - echo - -done - -echo -n "number of tiles in ${fbase_found}_all.txt = " -summe.pl ${fbase_found}_all.txt 0 - -echo -n "number of tiles in ${fbase_found_wsh}_all.txt = " -summe.pl ${fbase_found_wsh}_all.txt 0 - -echo -n "number of tiles in ${fbase_found_random}_all.txt = " -summe.pl ${fbase_found_random}_all.txt 0 - -if [ -e ${fbase_lf} ]; then - echo -n "number of tiles in ${fbase_lf}_all.txt = " - summe.pl ${fbase_lf}_all.txt 0 -fi diff --git a/src/sp_validation/catalog.py b/src/sp_validation/catalog.py index e96786ab..08249587 100644 --- a/src/sp_validation/catalog.py +++ b/src/sp_validation/catalog.py @@ -23,7 +23,6 @@ from cs_util import cat from sp_validation import format, io -from sp_validation.survey import get_footprint from sp_validation.version import __version__ @@ -154,7 +153,6 @@ def check_matching( keys_2, thresh, stats_file, - name=None, verbose=False, ): """Check matching. @@ -182,13 +180,7 @@ def check_matching( index list of tiles in footprint """ - if name is not None: - # Filter stars outside footprint for efficiency - mask_area_tiles = get_footprint(name, d1[keys_1[0]], d1[keys_1[1]]) - if len(np.where(mask_area_tiles)[0]) == 0: - raise ValueError(f"Error: no object found in field '{name}'") - else: - mask_area_tiles = np.arange(len(d1)) + mask_area_tiles = np.arange(len(d1)) # Match stars from exposure (PSF) catalogue to total catalogue ind = match_stars2( diff --git a/src/sp_validation/catalog_builders.py b/src/sp_validation/catalog_builders.py index b6abafc0..b78637fc 100644 --- a/src/sp_validation/catalog_builders.py +++ b/src/sp_validation/catalog_builders.py @@ -279,147 +279,84 @@ def params_default(self): """ self._params = { - "patches": "v1", + "input_paths": None, "sh": "ngmix", "survey": "unions", "year": "2024", "version": "1.4.2", "pipeline": "shapepipe", - "hdu": 1, + "param_path": None, "reduce_mem": False, "verbose": False, } self._short_options = { - "patches": "-p", + "input_paths": "-i", "sh": "-g", "survey": "-s", "year": "-y", "version": "-V", + "param_path": "-p", "reduce_mem": "-r", } self._types = { - "hdu": "int", "reduce_mem": "bool", } self._help_strings = { - "patches": "list of patches separated by '+', or shortcut (allowed are 'v1'), default={}", + "input_paths": ( + "campaign catalogue files (final_cat_.hdf5) to merge," + + " separated by '+'" + ), "sh": "shape measurement method, default={}", "survey": "survey name, default={}", "year": "year of processing, default={}", "version": "catalogue version, default={}", + "param_path": "path to parameter file listing columns to keep", "reduce_mem": "output some columns in lower precision to reduce memory", } - def get_patches(self): - """Get Patches. + def get_input_paths(self): + """Get Input Paths. - Return list of patches according to option parameter value. + Return the list of campaign catalogue files to merge. Returns ------- - list - patches, list of str + list of str + input file paths """ - if self._params["patches"] == "v1": - n_patch = 7 - patches = [f"P{x}" for x in np.arange(n_patch) + 1] - elif self._params["patches"] == "v1.5": - n_patch = 8 - patches = [f"P{x}" for x in np.arange(n_patch) + 1] - - else: - patches = self._params["patches"].split("+") - - return patches - - def get_n_obj(self, patches, base_path, input_sub_path): - """Get N Obj. - - Get number of objects from FITS file headers. - - Parameters - ---------- - patches : list - input patches, type is str - base_path : str - input base directory, root dir of patches - input_sub_path : str - input file name; input path is base_path/patch/input_sub_path - - Raises: - ValueError: if input file canont be read - - Returns: - list - HDUs - list - number of objects per file - int - total number of objects - - """ - if self._params["verbose"]: - print("Getting number of objects") - n_obj_list = [] - n_obj = 0 - hdu_lists = [] - for patch in patches: - input_path = f"{base_path}/{patch}/{input_sub_path}" - try: - hdu_list = fits.open(input_path) - except Exception as err: - raise ValueError( - f"Could not open file {input_path} at HDU" - + f" #{self._params['hdu']}" - ) from err - hdu_lists.append(hdu_list) - - this_n = int(hdu_list[self._params["hdu"]].header["NAXIS2"]) - n_obj_list.append(this_n) - n_obj += this_n - - if self._params["verbose"]: - print(f"Found a total of {n_obj} (~{format.millify(n_obj)}) objects.") + input_paths = self._params["input_paths"] + if not input_paths: + raise ValueError( + "No input campaign catalogues given; set 'input_paths' to one" + + " or more final_cat_.hdf5 files separated by '+'" + ) + if isinstance(input_paths, str): + input_paths = input_paths.split("+") - return hdu_lists, n_obj_list, n_obj + return [path.strip() for path in input_paths if path.strip()] - def get_col_info(self, dat): - """Get Col Info. + @staticmethod + def campaign_name(input_path): + """Campaign Name. - Return information of input columns. + Return the campaign name encoded in a catalogue file name, + ``final_cat_.hdf5`` -> ````. Parameters ---------- - dat : numpy.ndarray - input data + input_path : str + input file path Returns ------- - list - column names - list - column formats - int - number of columns + str + campaign name """ - col_names = dat.dtype.names - - n_col = 0 - formats = {} - ndim = {} - for name in col_names: - formats[name] = dat.dtype.fields[name][0] - ndim[name] = dat[name].ndim - n_col += ndim[name] - # Add one for patch - n_col += 1 - - if self._params["verbose"]: - print(f"Number of input (output) columns = {len(col_names)} ({n_col})") - - return col_names, formats, ndim, n_col + stem = os.path.splitext(os.path.basename(input_path))[0] + prefix = "final_cat_" + return stem[len(prefix) :] if stem.startswith(prefix) else stem def dtype_out(self, name, dtype_in): """Set output dtype. @@ -447,65 +384,39 @@ def dtype_out(self, name, dtype_in): self._params["reduce_mem"] and dtype_in.kind == "f" and dtype_in.itemsize == 8 - and name not in ("RA", "Dec") + and name not in ("RA", "Dec", "DEC") ): return np.dtype(np.float32) return dtype_in - def init_data(self, n_col, n_obj, ndim, dat): - """Init Data. + def output_dtype(self, dtype_in, n_char_campaign): + """Output Dtype. - Initialize empty structured data. + Return the merged-catalogue dtype: the input columns (possibly + reduced in precision) plus a ``campaign`` column. Parameters ---------- - n_col : int - number of columns - n_obj : int - number of objects (rows) - ndim : dict - dimension of input columns - dat : numpy.ndarray - example data + dtype_in : numpy.dtype + structured dtype of an input campaign catalogue + n_char_campaign : int + width of the campaign name column Returns ------- - numpy.ndarray - combined structure data, (n_col x n_obj) array + numpy.dtype + output structured dtype """ - # Create dtypes from input column names and types. - # Reduce memory if flag set. - # Transform multi-D columns into 1D columns - dtype_tmp_list = [] - for name in ndim: - if ndim[name] == 1: - dtype_tmp_list.append((name, self.dtype_out(name, dat[name].dtype))) - else: - for jdx in range(ndim[name]): - dtype_tmp_list.append( - (f"{name}_{jdx}", self.dtype_out(name, dat[name].dtype)) - ) - dtype_tmp_list.append(("patch", np.int8)) - dtype_tmp_struct = np.dtype(dtype_tmp_list) - - if self._params["verbose"]: - memory = n_obj * dtype_tmp_struct.itemsize - print( - f"Allocating <= {memory / 1024**3:.1f}" - + f" Gb memory for the ({n_col} x {n_obj}) input data array ...", - end="", - ) - - dat_all = np.empty((n_obj,), dtype=dtype_tmp_struct) + fields = [ + (name, self.dtype_out(name, dtype_in[name])) for name in dtype_in.names + ] + fields.append(("campaign", np.dtype(f"S{n_char_campaign}"))) - if self._params["verbose"]: - print("done") + return np.dtype(fields) - return dat_all - - def write_hdf5_file(self, dat_all, patches): + def write_hdf5_file(self, dat_all, campaigns=None): """Write HDF5 File. Write data to HDF5 file. @@ -514,8 +425,8 @@ def write_hdf5_file(self, dat_all, patches): ---------- dat_all : numpy.ndarray input data - patches : list - input patches, list of str + campaigns : list, optional + input campaign names, list of str """ output_path = ( @@ -525,12 +436,12 @@ def write_hdf5_file(self, dat_all, patches): ) with h5py.File(output_path, "w") as f: - self.write_hdf5_header(f) + self.write_hdf5_header(f, campaigns=campaigns) dset = f.create_dataset("data", data=dat_all) dset[:] = dat_all - def write_hdf5_header(self, hd5file, patches=None): + def write_hdf5_header(self, hd5file, campaigns=None): """Write HDF5 Header. Write header information to HDF5 file. @@ -539,92 +450,82 @@ def write_hdf5_header(self, hd5file, patches=None): ---------- hd5file : h5py.File input HDF5 file - patches : list, optional - input patches, list of str, default is ``None`` + campaigns : list, optional + input campaign names, list of str, default is ``None`` """ super().write_hdf5_header(hd5file) - if patches is not None: - patches_str = " ".join(patches) - hd5file.attrs["patches"] = patches_str + if campaigns is not None: + hd5file.attrs["campaigns"] = " ".join(campaigns) - def merge_catalogues(self, patches, base_path="."): + def merge_catalogues(self, input_paths): """Merge Catalogues. - Merge individual patch-based catalogues. + Merge a list of campaign catalogues into one joint catalogue, adding + a ``campaign`` column that records each object's origin. Parameters ---------- - patches : list - input patches; list of `str` - base_path : str, optional - input base directory path; default is "." + input_paths : list of str + campaign catalogue files (final_cat_.hdf5) + + Returns + ------- + numpy.ndarray + merged catalogue """ - input_sub_path = ( - f"sp_output/shape_catalog_comprehensive_{self._params['sh']}.fits" + param_list = ( + sp_cat.read_param_file( + self._params["param_path"], verbose=self._params["verbose"] + ) + if self._params["param_path"] + else None ) - # Get input FITS files - hdu_lists, n_obj_list, n_obj = self.get_n_obj( - patches, - base_path, - input_sub_path, - ) + campaigns = [self.campaign_name(path) for path in input_paths] + n_char_campaign = max(len(name) for name in campaigns) - # Read data - start = end = 0 - for idx, patch in enumerate(patches): - input_path = f"{base_path}/{patch}/{input_sub_path}" - try: - dat = fits.getdata(input_path, self._params["hdu"]) - # dat = hdu_lists[idx][self._params["hdu"]].data + data_list = [] + dtype_out = None + for input_path, campaign in zip(input_paths, campaigns): + dat = sp_cat.read_campaign_catalogue( + input_path, + param_list=param_list, + verbose=self._params["verbose"], + ) - hdu_lists[idx].close() - except Exception as err: - raise ValueError( - f"Could not read data of file {input_path} at HDU" - + f" #{self._params['hdu']}" - ) from err - - # Create empty lists if first patch - if idx == 0: - col_names, formats, ndim, n_col = self.get_col_info(dat) - dat_all = self.init_data(n_col, n_obj, ndim, dat) - - # Append new data for that patch (between start and end) - end += n_obj_list[idx] - - # Copy data - i_col = 0 - names_out = dat_all.dtype.names - for name in col_names: - if ndim[name] == 1: - # Copy 1D column - dat_all[names_out[i_col]][start:end] = dat[name] - else: - # Copy all components of multi-D column - for jdx in range(ndim[name]): - dat_all[names_out[i_col + jdx]][start:end] = dat[name][:, jdx] - i_col += ndim[name] - # Add patch number - dat_all["patch"][start:end] = patch[1:] - - if i_col + 1 != n_col: + if dtype_out is None: + dtype_out = self.output_dtype(dat.dtype, n_char_campaign) + elif set(dat.dtype.names) != set(dtype_out.names) - {"campaign"}: raise ValueError( - "Inconsistent number of columns, {i_col + 1}" + f" != {n_col}" + f"Campaign catalogue {input_path} has columns" + + f" {sorted(dat.dtype.names)}, incompatible with" + + f" {sorted(set(dtype_out.names) - {'campaign'})}" ) + + dat_out = np.empty(len(dat), dtype=dtype_out) + for name in dat.dtype.names: + dat_out[name] = dat[name] + dat_out["campaign"] = campaign.encode() + data_list.append(dat_out) + if self._params["verbose"]: print( - f"{patch}: Added {len(dat)} (~{format.millify(len(dat))})" - + f" objects (from {start} to {end - 1})." + f"{campaign}: added {len(dat)}" + + f" (~{format.millify(len(dat))}) objects." ) - start = end - del dat + dat_all = np.concatenate(data_list, axis=0) - self.write_hdf5_file(dat_all, patches) + if self._params["verbose"]: + print( + f"Merged {len(dat_all)} (~{format.millify(len(dat_all))})" + + f" objects from {len(campaigns)} campaign(s)." + ) + + return dat_all def run(self): """Run. @@ -632,11 +533,12 @@ def run(self): Main processing function. """ - patches = self.get_patches() + input_paths = self.get_input_paths() if self._params["verbose"]: - print("Merging patches", patches) + print("Merging campaigns", input_paths) - self.merge_catalogues(patches) + dat_all = self.merge_catalogues(input_paths) + self.write_hdf5_file(dat_all, [self.campaign_name(p) for p in input_paths]) class ApplyHspMasks(BaseCat): diff --git a/src/sp_validation/galaxy.py b/src/sp_validation/galaxy.py index 3eda60af..7a8b8b1c 100644 --- a/src/sp_validation/galaxy.py +++ b/src/sp_validation/galaxy.py @@ -91,6 +91,10 @@ def mask_cut(dd, mask_columns=None): """ columns = list(DEFAULT_MASK_COLUMNS if mask_columns is None else mask_columns) + if not columns: + # No masking requested (e.g. the image simulations, which run no + # imaging-flag masking stage and carry no mask columns). + return np.ones(len(dd[_column_names(dd)[0]]), dtype=bool) available = _column_names(dd) missing = [col for col in columns if col not in available] diff --git a/src/sp_validation/rho_tau.py b/src/sp_validation/rho_tau.py index 62ae8bf8..16f69565 100644 --- a/src/sp_validation/rho_tau.py +++ b/src/sp_validation/rho_tau.py @@ -342,7 +342,7 @@ def get_jackknife_cov( tau_chunk = outdir + f"/cov_tau_{version}{i}.npy" rho_chunk = outdir + f"/cov_rho_{version}{i}.npy" if not (os.path.exists(tau_chunk) and os.path.exists(rho_chunk)): - print(f"Computing rho-statistics for {version} (patch {i + 1}/{ncov})") + print(f"Computing rho-statistics for {version} (jackknife realisation {i + 1}/{ncov})") if f"psf_{version}{i}" not in rho_stat_handler.catalogs.catalogs_dict: # Build catalogues diff --git a/src/sp_validation/survey.py b/src/sp_validation/survey.py index fe35cde9..e9d61025 100644 --- a/src/sp_validation/survey.py +++ b/src/sp_validation/survey.py @@ -183,77 +183,6 @@ def write_tile_id_gal_counts(detection_IDs, galaxy_IDs, shape_IDs, fname): print(file=f) -def get_footprint(patch, ra, dec): - """Get Footprint. - - Return coordinates within footprint of patch. - - Parameters - ---------- - patch : str - patch name - ra : array of float - R,A, coordintates - dec : array of float - DEC coordinates - - Returns - ------- - list of float - list of coordinates withint footprint - - """ - # Set boundary coordinates between some of the patches - ra_14 = 157.5 - ra_45 = 207 - ra2_45 = 220 - ra_36 = 230 - dec_3456 = 48 - - ra2_34 = 190 - - dec_min = 29 - dec_max = 60 - - # Check whether input matches one of the seven CFIS patch name. - # Return coordinates within the patch - if patch == "P1": - return (ra > 100) & (ra < ra_14) & (dec > dec_min) & (dec < dec_max) - - elif patch == "P2": - # -30 < ra < 60 - return ((ra > 0) & (ra < 60)) | ((ra > 330) & (ra < 360)) & (dec > dec_min) & ( - dec < dec_max - ) - - elif patch == "P3": - return (ra > ra2_34) & (ra < ra_36) & (dec > dec_3456) & (dec < 70) - - elif patch == "P4": - return ( - ((ra > ra_14) & (ra < ra_45) & (dec > dec_min) & (dec < dec_3456)) - | ((ra > ra_14) & (ra < ra2_34) & (dec > dec_min) & (dec < 70)) - | ((ra > ra_45) & (ra < ra2_45) * (dec > dec_min) & (dec < 36)) - ) - - elif patch == "P5": - return ((ra > ra2_45) & (ra < 330) & (dec > dec_min) & (dec < dec_3456)) | ( - (ra > ra_45) & (ra < ra2_45) & (dec > 36) & (dec < dec_3456) - ) - - elif patch == "P6": - return (ra > ra_36) & (ra < 330) & (dec > dec_3456) & (dec > 70) - - elif patch == "P7": - return (ra > 60) & (ra < 180) & (dec > 60) & (dec < 90) - - elif patch == "W3": - return (ra > 208) & (ra < 221) & (dec > 51) & (dec < 58) - - else: - return dec > dec_min - - def area_from_coords(ra, dec, nside): """Survey area from galaxy coordinates via HEALPix pixel counting. diff --git a/src/sp_validation/tests/test_survey.py b/src/sp_validation/tests/test_survey.py index f42f884c..c31f75cc 100644 --- a/src/sp_validation/tests/test_survey.py +++ b/src/sp_validation/tests/test_survey.py @@ -28,9 +28,6 @@ def setUp(self): self._area_amin2 = 3600 self._tile_IDs = (270.283, 188.308) - self._ra = np.array([240.0]) - self._dec = np.array([32.0]) - self._patch = ["P5"] def tearDown(self): @@ -58,8 +55,3 @@ def test_get_area(self): msg=f"{tile_IDs}!={self._tile_IDs}", ) - def test_get_footprint(self): - """Test ``sp_validation.survey_get_footprint`` method.""" - for patch in self._patch: - coords = survey.get_footprint(patch, self._ra, self._dec) - self.assertTrue(coords[0]) diff --git a/workflow/image_sims/params_im_sim.py b/workflow/image_sims/params_im_sim.py index 9eae8585..76fdd64b 100644 --- a/workflow/image_sims/params_im_sim.py +++ b/workflow/image_sims/params_im_sim.py @@ -27,7 +27,7 @@ # Survey parameters -## Field or patch name -- derived from the run directory, which is named +## Field name -- derived from the run directory, which is named ## after the simulation (e.g. '1z2z_grid_1'), so one shared params file ## serves every sim. name = os.path.basename(os.getcwd()) @@ -125,10 +125,14 @@ ## Pre-calibration catalogue, including masked objects and mask flags. ## ShapePipe-v2 (post-#761) ngmix grammar: ellipticity in named scalar ## components NGMIX_G{1,2}_*, PSF size split into NGMIX_T_PSF_ORIG/RECONV. -## IMAFLAGS_ISO (present in the data-path params) is omitted: the simulation -## pipeline runs no imaging-flag masking stage, so the column does not exist. +## The MASK_n* columns (present in the data-path params) are omitted: the +## simulation pipeline runs no imaging-flag masking stage, so they do not +## exist -- hence the empty mask_columns below. ## NGMIX_MCAL_TYPES_FAIL is kept -- it is the metacal moments-failure flag the ## calibration mask cuts on, identically to the data path. +## No mask columns to OR: no masking stage in the simulation pipeline +mask_columns = [] + add_cols_pre_cal = [ "TILE_ID", "NUMBER", From df6f1aca27e5da41c7e181c11cff1ceee0e3c9cf Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Wed, 9 Sep 2026 20:25:50 -0400 Subject: [PATCH 06/39] tests: campaign readers, mask cut, and campaign merge Seventeen unit tests on tiny synthetic hdf5 fixtures built in a temp dir: both campaign layouts (legacy patches// and flat tiles/) read identically, param-list restriction, missing-column and ambiguous-layout errors; the star reader on exposures/ hdf5 and on FITS; galaxy.mask_cut defaults, explicit list, empty list, missing column, and a v1 IMAFLAGS_ISO-only catalogue; JointCat.merge_catalogues across two campaigns of different layouts, plus its error paths. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01QbnPCyzuDNTgkg715pHhar --- .../tests/test_campaign_readers.py | 259 ++++++++++++++++++ 1 file changed, 259 insertions(+) create mode 100644 src/sp_validation/tests/test_campaign_readers.py diff --git a/src/sp_validation/tests/test_campaign_readers.py b/src/sp_validation/tests/test_campaign_readers.py new file mode 100644 index 00000000..ccba4fe5 --- /dev/null +++ b/src/sp_validation/tests/test_campaign_readers.py @@ -0,0 +1,259 @@ +"""Tests for the ShapePipe v2 campaign catalogue readers and mask cut.""" + +import unittest + +import h5py +import numpy as np +import numpy.testing as npt + +from sp_validation import catalog, galaxy +from sp_validation.catalog_builders import JointCat + +import tempfile +from pathlib import Path + +GAL_DTYPE = np.dtype( + [ + ("RA", "f8"), + ("Dec", "f8"), + ("MAG_AUTO", "f4"), + ("MASK_n4", "?"), + ("MASK_n1", "?"), + ("MASK_n2", "?"), + ("MASK_n8", "?"), + ("MASK_n1024", "?"), + ] +) + + +def make_galaxy_data(n_obj, offset=0): + dat = np.zeros(n_obj, dtype=GAL_DTYPE) + dat["RA"] = np.arange(n_obj) + offset + dat["Dec"] = np.arange(n_obj) + offset + 0.5 + dat["MAG_AUTO"] = 22.0 + return dat + + +def write_campaign(path, layout, tiles): + """Write a campaign hdf5 file in the legacy or flat layout.""" + with h5py.File(path, "w") as f: + if layout == "legacy": + group = f.create_group("patches").create_group("CAMPAIGN") + else: + group = f.create_group("tiles") + for tile_id, dat in tiles.items(): + group.create_dataset(tile_id, data=dat) + f.attrs["n_tiles"] = len(tiles) + + +class TestCampaignReader(unittest.TestCase): + """Campaign galaxy catalogue reader.""" + + def setUp(self): + self._tmp = tempfile.TemporaryDirectory() + self._dir = Path(self._tmp.name) + self._tiles = { + "000.000": make_galaxy_data(3), + "001.000": make_galaxy_data(2, offset=100), + } + + def tearDown(self): + self._tmp.cleanup() + + def _expected(self): + return np.concatenate(list(self._tiles.values())) + + def test_legacy_layout(self): + path = self._dir / "final_cat_CAMPAIGN.hdf5" + write_campaign(path, "legacy", self._tiles) + + dat = catalog.read_campaign_catalogue(path, verbose=False) + + self.assertEqual(len(dat), 5) + npt.assert_array_equal(np.sort(dat["RA"]), np.sort(self._expected()["RA"])) + + def test_flat_layout(self): + path = self._dir / "final_cat_CAMPAIGN.hdf5" + write_campaign(path, "flat", self._tiles) + + dat = catalog.read_campaign_catalogue(path, verbose=False) + + self.assertEqual(len(dat), 5) + npt.assert_array_equal(np.sort(dat["RA"]), np.sort(self._expected()["RA"])) + + def test_layouts_agree(self): + legacy = self._dir / "legacy.hdf5" + flat = self._dir / "flat.hdf5" + write_campaign(legacy, "legacy", self._tiles) + write_campaign(flat, "flat", self._tiles) + + npt.assert_array_equal( + catalog.read_campaign_catalogue(legacy, verbose=False), + catalog.read_campaign_catalogue(flat, verbose=False), + ) + + def test_param_list_restriction(self): + path = self._dir / "final_cat_CAMPAIGN.hdf5" + write_campaign(path, "flat", self._tiles) + + dat = catalog.read_campaign_catalogue( + path, param_list=["RA", "Dec"], verbose=False + ) + + self.assertEqual(tuple(dat.dtype.names), ("RA", "Dec")) + + def test_missing_column_raises(self): + path = self._dir / "final_cat_CAMPAIGN.hdf5" + write_campaign(path, "flat", self._tiles) + + with self.assertRaises(KeyError) as ctx: + catalog.read_campaign_catalogue( + path, param_list=["RA", "NOT_A_COLUMN"], verbose=False + ) + self.assertIn("NOT_A_COLUMN", str(ctx.exception)) + + def test_ambiguous_layout_raises(self): + path = self._dir / "ambiguous.hdf5" + with h5py.File(path, "w") as f: + f.create_group("tiles") + f.create_group("other") + + with self.assertRaises(ValueError): + catalog.read_campaign_catalogue(path, verbose=False) + + +class TestStarCatalogueReader(unittest.TestCase): + """Campaign star catalogue reader.""" + + def setUp(self): + self._tmp = tempfile.TemporaryDirectory() + self._dir = Path(self._tmp.name) + + dtype = np.dtype([(name, "f8") for name in catalog.STAR_CAT_COLUMNS]) + self._exposures = { + "2110000p": np.zeros(4, dtype=dtype), + "2110001p": np.ones(6, dtype=dtype), + } + + self._path = self._dir / "full_starcat_CAMPAIGN.hdf5" + with h5py.File(self._path, "w") as f: + group = f.create_group("exposures") + for exp, dat in self._exposures.items(): + group.create_dataset(exp, data=dat) + + def tearDown(self): + self._tmp.cleanup() + + def test_read_hdf5(self): + dat = catalog.read_star_catalogue(self._path, verbose=False) + + self.assertEqual(len(dat), 10) + self.assertEqual(tuple(dat.dtype.names), catalog.STAR_CAT_COLUMNS) + npt.assert_array_equal(dat["MAG"][:4], np.zeros(4)) + npt.assert_array_equal(dat["MAG"][4:], np.ones(6)) + + def test_read_fits(self): + from astropy.io import fits + + fits_path = self._dir / "full_starcat-0000000.fits" + dat = np.concatenate(list(self._exposures.values())) + fits.BinTableHDU(data=dat).writeto(fits_path) + + out = catalog.read_star_catalogue(str(fits_path), verbose=False) + + self.assertEqual(len(out), 10) + npt.assert_array_equal(np.asarray(out["MAG"]), dat["MAG"]) + + +class TestMaskCut(unittest.TestCase): + """Mask-column galaxy selection cut.""" + + def setUp(self): + self._dat = make_galaxy_data(5) + + def test_default_columns(self): + self._dat["MASK_n4"][0] = True + self._dat["MASK_n1024"][3] = True + + npt.assert_array_equal( + galaxy.mask_cut(self._dat), + np.array([False, True, True, False, True]), + ) + + def test_explicit_column_list(self): + self._dat["MASK_n4"][0] = True + self._dat["MASK_n8"][1] = True + + npt.assert_array_equal( + galaxy.mask_cut(self._dat, ["MASK_n8"]), + np.array([True, False, True, True, True]), + ) + + def test_empty_column_list_keeps_everything(self): + self._dat["MASK_n4"][:] = True + + npt.assert_array_equal( + galaxy.mask_cut(self._dat, []), np.ones(len(self._dat), dtype=bool) + ) + + def test_missing_column_raises(self): + with self.assertRaises(KeyError) as ctx: + galaxy.mask_cut(self._dat, ["MASK_n16"]) + self.assertIn("MASK_n16", str(ctx.exception)) + + def test_v1_catalogue_raises(self): + dat = np.zeros(3, dtype=[("IMAFLAGS_ISO", "i2")]) + + with self.assertRaises(KeyError): + galaxy.mask_cut(dat) + + +class TestCampaignMerge(unittest.TestCase): + """Merge of several campaign catalogues into a joint catalogue.""" + + def setUp(self): + self._tmp = tempfile.TemporaryDirectory() + self._dir = Path(self._tmp.name) + + self._paths = [] + for name, layout, n_obj in (("W3", "legacy", 3), ("SGC", "flat", 2)): + path = self._dir / f"final_cat_{name}.hdf5" + write_campaign(path, layout, {"000.000": make_galaxy_data(n_obj)}) + self._paths.append(str(path)) + + self._obj = JointCat() + self._obj._params["verbose"] = False + + def tearDown(self): + self._tmp.cleanup() + + def test_campaign_name(self): + self.assertEqual( + JointCat.campaign_name("/some/dir/final_cat_W3.hdf5"), "W3" + ) + + def test_merge(self): + dat = self._obj.merge_catalogues(self._paths) + + self.assertEqual(len(dat), 5) + self.assertIn("campaign", dat.dtype.names) + npt.assert_array_equal( + dat["campaign"], np.array([b"W3"] * 3 + [b"SGC"] * 2) + ) + npt.assert_array_equal(dat["RA"][:3], np.arange(3)) + + def test_merge_incompatible_columns_raises(self): + other = self._dir / "final_cat_X.hdf5" + dat = np.zeros(2, dtype=[("RA", "f8")]) + write_campaign(other, "flat", {"000.000": dat}) + + with self.assertRaises(ValueError): + self._obj.merge_catalogues(self._paths + [str(other)]) + + def test_no_input_raises(self): + with self.assertRaises(ValueError): + self._obj.get_input_paths() + + +if __name__ == "__main__": + unittest.main() From db93958038598e2d837879c1574b8f428ca69b37 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Wed, 9 Sep 2026 20:41:51 -0400 Subject: [PATCH 07/39] catalog: safe, low-memory campaign reading and merging - concatenate_datasets preallocates the output and fills it column by column, so peak memory is the packed output plus one tile instead of the full-width uncut catalogue; the verbose estimate uses the real itemsize. - validate the requested columns against every tile dataset, not only the first, and name the offending dataset in the error. - read_campaign_catalogue checks the root n_tiles attribute and refuses a truncated file; campaign_shape reports row count and dtype from metadata. - JointCat.merge_catalogues preallocates the merged array from that first pass (no more accumulate-then-concatenate, which doubled peak memory), promotes each column's dtype across all campaigns so a wider string or integer column in a later file is no longer silently truncated, and refuses multi-dimensional columns explicitly. - reduce_mem reduces int32 to int16 (int8 wrapped N_EPOCH/CCD_NB values) and every assignment is range-checked, raising instead of wrapping. - check_matching drops the identity index over d1 that the retired footprint prefilter left behind, and extract_info applies mask_cut to the matched subset instead of the whole catalogue. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01QbnPCyzuDNTgkg715pHhar --- scripts/calibration/extract_info.py | 4 +- src/sp_validation/catalog.py | 125 ++++++++++++++++++++++---- src/sp_validation/catalog_builders.py | 116 ++++++++++++++++++------ 3 files changed, 197 insertions(+), 48 deletions(-) diff --git a/scripts/calibration/extract_info.py b/scripts/calibration/extract_info.py index b65c1975..ed4fb145 100644 --- a/scripts/calibration/extract_info.py +++ b/scripts/calibration/extract_info.py @@ -142,7 +142,7 @@ # #### Match to all objects if star_cat_path: - ind_star, mask_area_tiles, n_star_tot = spv_cat.check_matching( + ind_star, n_star_tot = spv_cat.check_matching( d_star, dd, ["RA", "DEC"], @@ -160,7 +160,7 @@ m_star = ( (dd["FLAGS"][ind_star] == 0) - & galaxy.mask_cut(dd, mask_columns)[ind_star] + & galaxy.mask_cut(dd[ind_star], mask_columns) & (dd["NGMIX_MCAL_FLAGS"][ind_star] == 0) & (dd["NGMIX_G1_PSF_ORIG_NOSHEAR"][ind_star] != -10) ) diff --git a/src/sp_validation/catalog.py b/src/sp_validation/catalog.py index 08249587..8296ecee 100644 --- a/src/sp_validation/catalog.py +++ b/src/sp_validation/catalog.py @@ -176,22 +176,20 @@ def check_matching( ------- ind : array of int index list of d2 of objects that were matched to d1 - mask_area_tiles : array of int - index list of tiles in footprint + n_tot : int + number of objects in d1 """ - mask_area_tiles = np.arange(len(d1)) - # Match stars from exposure (PSF) catalogue to total catalogue ind = match_stars2( d2[keys_2[0]], d2[keys_2[1]], - d1[keys_1[0]][mask_area_tiles], - d1[keys_1[1]][mask_area_tiles], + d1[keys_1[0]], + d1[keys_1[1]], thresh=thresh, ) - n_tot = len(d1[keys_1[0]][mask_area_tiles]) + n_tot = len(d1[keys_1[0]]) msg = ( "Number of matched stars from exposures to total catalogue = " + f"{len(ind)}/{n_tot} = {len(ind) / n_tot:.1%}" @@ -207,7 +205,7 @@ def check_matching( ) io.print_stats(msg, stats_file, verbose=verbose) - return ind, mask_area_tiles, n_tot + return ind, n_tot def check_invalid(dd, key, val, stats_file, name=None, verbose=False): @@ -822,12 +820,13 @@ def find_dataset_group(hdf5_file): node = node[keys[0]] -def _check_columns(dtype, param_list, file_path): +def _check_columns(dtype, param_list, file_path, dataset_key=None): """Raise a clear error if requested columns are absent from the data.""" missing = [col for col in param_list if col not in (dtype.names or ())] if missing: + where = f" (dataset {dataset_key!r})" if dataset_key is not None else "" raise KeyError( - f"Column(s) {missing} not found in catalogue {file_path}." + f"Column(s) {missing} not found in catalogue {file_path}{where}." + f" Available columns: {sorted(dtype.names or ())}" ) @@ -838,6 +837,10 @@ def concatenate_datasets(group, param_list=None, file_path="", verbose=True): Concatenate every dataset of an HDF5 group into one structured array, optionally restricted to a list of columns. + The output array is preallocated and filled slice by slice, so peak + memory is the output catalogue plus one tile, not the full-width + uncut catalogue. + Parameters ---------- group : h5py.Group @@ -855,27 +858,110 @@ def concatenate_datasets(group, param_list=None, file_path="", verbose=True): concatenated structured array """ - keys = list(group) + keys = sorted(group) + if not keys: + raise ValueError(f"No datasets found in catalogue {file_path}") + + # Validate every dataset up front: a column may be missing from any tile, + # not only the first one. + for key in keys: + if param_list is not None: + _check_columns(group[key].dtype, param_list, file_path, key) + + dtype_in = group[keys[0]].dtype if param_list is not None: - _check_columns(group[keys[0]].dtype, param_list, file_path) + dtype_out = np.dtype([(name, dtype_in[name]) for name in param_list]) + else: + dtype_out = dtype_in n_rows = sum(group[key].shape[0] for key in keys) if verbose: - n_cols = len(param_list) if param_list is not None else len(group[keys[0]].dtype) print( f"Reading {len(keys)} datasets," - + f" estimating {n_cols * n_rows * 8 / 1024**3:.1f}" - + f" Gb memory for the ({n_cols} x {n_rows}) data array ..." + + f" estimating {dtype_out.itemsize * n_rows / 1024**3:.1f}" + + f" Gb memory for the ({len(dtype_out.names)} x {n_rows}) data array ..." ) - data_list = [] + data_out = np.empty(n_rows, dtype=dtype_out) + start = 0 for key in tqdm.tqdm(keys, disable=not verbose): data = group[key][()] + end = start + len(data) + for name in dtype_out.names: + data_out[name][start:end] = data[name] + start = end + del data + + return data_out + + +def check_n_tiles(hdf5_file, group, file_path): + """Check Number of Tiles. + + Compare the number of datasets found against the ``n_tiles`` root + attribute the ShapePipe v2 product carries, to catch a catalogue that + was truncated by an interrupted merge job or file transfer. + + Parameters + ---------- + hdf5_file : h5py.File + open input file + group : h5py.Group + group holding the per-tile datasets + file_path : str + input file path, for the error message + + Raises + ------ + ValueError + if the number of datasets differs from the ``n_tiles`` attribute + + """ + n_tiles = hdf5_file.attrs.get("n_tiles") + if n_tiles is None: + return + n_found = len(group) + if int(n_tiles) != n_found: + raise ValueError( + f"Catalogue {file_path} declares n_tiles = {int(n_tiles)} but holds" + + f" {n_found} tile dataset(s); the file is incomplete." + ) + + +def campaign_shape(file_path, param_list=None): + """Campaign Shape. + + Return the number of rows and the structured dtype of a campaign + catalogue without reading its data, so that a merged output array can + be preallocated. + + Parameters + ---------- + file_path : str + input file path + param_list : list of str, optional + columns to keep; default is ``None`` (keep all) + + Returns + ------- + tuple + number of rows (int) and dtype (numpy.dtype) + + """ + with h5py.File(file_path, "r") as hdf5_file: + group = find_dataset_group(hdf5_file) + check_n_tiles(hdf5_file, group, file_path) + keys = sorted(group) + if not keys: + raise ValueError(f"No datasets found in catalogue {file_path}") + n_rows = sum(group[key].shape[0] for key in keys) + dtype_in = group[keys[0]].dtype if param_list is not None: - data = data[param_list] - data_list.append(data) + for key in keys: + _check_columns(group[key].dtype, param_list, file_path, key) + dtype_in = np.dtype([(name, dtype_in[name]) for name in param_list]) - return np.concatenate(data_list, axis=0) + return n_rows, dtype_in def read_campaign_catalogue( @@ -911,6 +997,7 @@ def read_campaign_catalogue( with h5py.File(file_path, "r") as hdf5_file: group = find_dataset_group(hdf5_file) + check_n_tiles(hdf5_file, group, file_path) return concatenate_datasets( group, param_list=param_list, file_path=file_path, verbose=verbose ) diff --git a/src/sp_validation/catalog_builders.py b/src/sp_validation/catalog_builders.py index b78637fc..e06ee105 100644 --- a/src/sp_validation/catalog_builders.py +++ b/src/sp_validation/catalog_builders.py @@ -240,6 +240,37 @@ def close_hd5(self): self._hd5file.close() +def _promote(dtype_a, dtype_b): + """Return a dtype that holds both input dtypes without truncation.""" + if dtype_a == dtype_b: + return dtype_a + if dtype_a.kind in "SU" and dtype_b.kind in "SU": + kind = "U" if "U" in (dtype_a.kind, dtype_b.kind) else "S" + size_a = dtype_a.itemsize // (4 if dtype_a.kind == "U" else 1) + size_b = dtype_b.itemsize // (4 if dtype_b.kind == "U" else 1) + return np.dtype(f"{kind}{max(size_a, size_b)}") + + return np.promote_types(dtype_a, dtype_b) + + +def _checked_assign(target, start, end, values, name): + """Assign `values` into `target[start:end]`, refusing a lossy cast.""" + target[start:end] = values + written = target[start:end] + if written.dtype == values.dtype: + return + if written.dtype.kind in "fc": + bad = np.isfinite(values) & ~np.isfinite(written) + else: + bad = written != values + if np.any(bad): + raise ValueError( + f"Column {name!r} cannot be stored as {written.dtype}:" + + f" {int(np.sum(bad))} value(s) overflow or are truncated." + + " Disable reduce_mem or widen the output dtype." + ) + + class JointCat(BaseCat): """Joint Cat. @@ -390,16 +421,17 @@ def dtype_out(self, name, dtype_in): return dtype_in - def output_dtype(self, dtype_in, n_char_campaign): + def output_dtype(self, dtypes_in, n_char_campaign): """Output Dtype. Return the merged-catalogue dtype: the input columns (possibly - reduced in precision) plus a ``campaign`` column. + reduced in precision, and promoted to a common type across all input + campaigns) plus a ``campaign`` column. Parameters ---------- - dtype_in : numpy.dtype - structured dtype of an input campaign catalogue + dtypes_in : numpy.dtype or list of numpy.dtype + structured dtype(s) of the input campaign catalogues n_char_campaign : int width of the campaign name column @@ -408,10 +440,40 @@ def output_dtype(self, dtype_in, n_char_campaign): numpy.dtype output structured dtype + Raises + ------ + ValueError + if the inputs have different column sets, or a column is + multi-dimensional (campaign catalogues are scalar-column only) + """ - fields = [ - (name, self.dtype_out(name, dtype_in[name])) for name in dtype_in.names - ] + if isinstance(dtypes_in, np.dtype): + dtypes_in = [dtypes_in] + + names = dtypes_in[0].names + for dtype_in in dtypes_in[1:]: + if set(dtype_in.names) != set(names): + raise ValueError( + "Campaign catalogues have incompatible column sets:" + + f" {sorted(names)} vs {sorted(dtype_in.names)}" + ) + + fields = [] + for name in names: + subs = [dtype_in[name] for dtype_in in dtypes_in] + for sub in subs: + if sub.subdtype is not None: + raise ValueError( + f"Column {name!r} is multi-dimensional (shape" + + f" {sub.subdtype[1]}); campaign catalogues are" + + " expected to hold scalar columns only." + ) + # Promote across campaigns so a wider string or integer column in + # a later file is not silently truncated or overflowed. + promoted = subs[0] + for sub in subs[1:]: + promoted = _promote(promoted, sub) + fields.append((name, self.dtype_out(name, promoted))) fields.append(("campaign", np.dtype(f"S{n_char_campaign}"))) return np.dtype(fields) @@ -487,8 +549,18 @@ def merge_catalogues(self, input_paths): campaigns = [self.campaign_name(path) for path in input_paths] n_char_campaign = max(len(name) for name in campaigns) - data_list = [] - dtype_out = None + # First pass over file metadata only (row counts and dtypes), so the + # merged array is allocated once and filled in place, instead of + # concatenating per-campaign copies (peak memory 2x the output). + shapes = [ + sp_cat.campaign_shape(path, param_list=param_list) + for path in input_paths + ] + n_total = sum(n_rows for n_rows, _ in shapes) + dtype_out = self.output_dtype([dtype for _, dtype in shapes], n_char_campaign) + + dat_all = np.empty(n_total, dtype=dtype_out) + start = 0 for input_path, campaign in zip(input_paths, campaigns): dat = sp_cat.read_campaign_catalogue( input_path, @@ -496,28 +568,18 @@ def merge_catalogues(self, input_paths): verbose=self._params["verbose"], ) - if dtype_out is None: - dtype_out = self.output_dtype(dat.dtype, n_char_campaign) - elif set(dat.dtype.names) != set(dtype_out.names) - {"campaign"}: - raise ValueError( - f"Campaign catalogue {input_path} has columns" - + f" {sorted(dat.dtype.names)}, incompatible with" - + f" {sorted(set(dtype_out.names) - {'campaign'})}" - ) - - dat_out = np.empty(len(dat), dtype=dtype_out) + end = start + len(dat) for name in dat.dtype.names: - dat_out[name] = dat[name] - dat_out["campaign"] = campaign.encode() - data_list.append(dat_out) + _checked_assign(dat_all[name], start, end, dat[name], name) + dat_all["campaign"][start:end] = campaign.encode() + start = end if self._params["verbose"]: print( f"{campaign}: added {len(dat)}" + f" (~{format.millify(len(dat))}) objects." ) - - dat_all = np.concatenate(data_list, axis=0) + del dat if self._params["verbose"]: print( @@ -943,8 +1005,8 @@ def write_hdf5_header(self, hd5file): ---------- hd5file : h5py.File input HDF5 file - patches : list, optional - input patches, list of str, default is ``None`` + campaigns : list, optional + input campaign names, list of str, default is ``None`` """ super().write_hdf5_header(hd5file) @@ -1020,7 +1082,7 @@ def read_cat(self, load_into_memory=False): verbose = self._params["verbose"] # Image-simulation path: a single per-run comprehensive catalogue in - # FITS, not the joined multi-patch HDF5 the data path builds. Read the + # FITS, not the joined multi-campaign HDF5 the data path builds. Read the # FITS table directly into memory; there is no separate data_ext group. extension = os.path.splitext(fpath)[1] if extension == ".fits": From 03e1dac32d4495c395079e07520c5b2ee7928b28 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Wed, 9 Sep 2026 20:42:07 -0400 Subject: [PATCH 08/39] config: v2 mask configuration with the MASK_n* columns Add config/calibration/mask_v2.0.yaml: the v1.X.11 cut set with IMAFLAGS_ISO and the v1 post-processing masks replaced by the boolean MASK_n columns (True = masked, hence kind: equal, value: False), the coverage bits and MASK_n2048 listed but commented out. The v1.X configs are left untouched: each describes a legacy catalogue that really has IMAFLAGS_ISO, and rewriting them would break reproducing published versions. plots.sky_plots no longer hardcodes the v1 label set; it combines whichever of IMAFLAGS_ISO / MASK_n* / npoint3 / 1024_Maximask the config declared. The comprehensive-to-minimal demo asks for the v2 mask labels. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01QbnPCyzuDNTgkg715pHhar --- config/calibration/mask_v2.0.yaml | 122 ++++++++++++++++++ .../demo_comprehensive_to_minimal_cat.py | 9 +- src/sp_validation/plots.py | 20 ++- 3 files changed, 145 insertions(+), 6 deletions(-) create mode 100644 config/calibration/mask_v2.0.yaml diff --git a/config/calibration/mask_v2.0.yaml b/config/calibration/mask_v2.0.yaml new file mode 100644 index 00000000..db824f59 --- /dev/null +++ b/config/calibration/mask_v2.0.yaml @@ -0,0 +1,122 @@ +# Config file for masking and calibration, ShapePipe v2 catalogues. +# +# ShapePipe v2 replaces the single IMAFLAGS_ISO column with per-reason boolean +# mask columns MASK_n, where True means MASKED. They are therefore cut +# with `kind: equal, value: False` (keep the un-masked objects): +# +# MASK_n1, MASK_n2 bright/faint star halos +# MASK_n4 stars +# MASK_n8 manual galaxy mask +# MASK_n16 .. MASK_n256 per-band coverage +# MASK_n1024 MaxiMask +# MASK_n2048 no Pan-STARRS z2 coverage +# +# The default selection (see sp_validation.galaxy.DEFAULT_MASK_COLUMNS) is +# n4 + n1 + n2 + n8 + n1024; the coverage bits and n2048 are listed here but +# commented out, since ORing all of them masks essentially everything. + +# General parameters (can also given on command line) +params: + input_path: unions_shapepipe_comprehensive_2025_v2.0.hdf5 + cmatrices: False + sky_regions: False + verbose: True + +# Masks +## Using columns in 'dat' group (ShapePipe flags) +dat: + # SExtractor flags + - col_name: FLAGS + label: SE FLAGS + kind: smaller_equal + value: 3 + + # Duplicate objects + - col_name: overlap + label: tile overlap + kind: equal + value: True + + # ShapePipe masks (boolean, True = masked) + - col_name: MASK_n4 + label: "stars" + kind: equal + value: False + - col_name: MASK_n1 + label: "faint star halos" + kind: equal + value: False + - col_name: MASK_n2 + label: "bright star halos" + kind: equal + value: False + - col_name: MASK_n8 + label: "manual mask" + kind: equal + value: False + - col_name: MASK_n1024 + label: "maximask" + kind: equal + value: False + + # Per-band coverage and Pan-STARRS z2; enable as required + # - col_name: MASK_n16 + # label: "coverage (n16)" + # kind: equal + # value: False + # - col_name: MASK_n2048 + # label: "no PS-z2" + # kind: equal + # value: False + + # Number of epochs + - col_name: N_EPOCH + label: r"$n_{\rm epoch}$" + kind: greater_equal + value: 2 + + # Magnitude range + - col_name: mag + label: mag range + kind: range + value: [15, 30] + + # ngmix flags + - col_name: NGMIX_MCAL_TYPES_FAIL + label: "ngmix moments failure" + kind: equal + value: 0 + + # invalid PSF ellipticities + - col_name: NGMIX_G1_PSF_ORIG_NOSHEAR + label: "bad PSF ellipticity comp 1" + kind: not_equal + value: -10 + - col_name: NGMIX_G2_PSF_ORIG_NOSHEAR + label: "bad PSF ellipticity comp 2" + kind: not_equal + value: -10 + +## Using columns in 'dat_ext' group (post-processing flags) +## ShapePipe v2 carries the imaging masks in the 'dat' group above, so this +## group is empty unless external masks are added. +dat_ext: [] + +# Metacal parameters +metacal: + # Ellipticity dispersion + sigma_eps_prior: 0.34 + + # Signal-to-noise range + gal_snr_min: 10 + gal_snr_max: 500 + + # Relative-size (hlr / hlr_psf) range + gal_rel_size_min: 0.707 + gal_rel_size_max: 3 + + # Correct relative size for ellipticity? + gal_size_corr_ell: False + + # Weight for global response matrix, None for unweighted mean + global_R_weight: w diff --git a/scripts/examples/demo_comprehensive_to_minimal_cat.py b/scripts/examples/demo_comprehensive_to_minimal_cat.py index 11e5ed5a..30b6b1df 100644 --- a/scripts/examples/demo_comprehensive_to_minimal_cat.py +++ b/scripts/examples/demo_comprehensive_to_minimal_cat.py @@ -50,13 +50,18 @@ # + # List of masks to apply +# (labels as declared in config/calibration/mask_v2.0.yaml; ShapePipe v2 +# replaces IMAFLAGS_ISO and the v1 post-processing masks by MASK_n) masks_to_apply = [ "overlap", - "IMAFLAGS_ISO", + "MASK_n4", + "MASK_n1", + "MASK_n2", + "MASK_n8", + "MASK_n1024", "NGMIX_MCAL_TYPES_FAIL", "NGMIX_G1_PSF_ORIG_NOSHEAR", "NGMIX_G2_PSF_ORIG_NOSHEAR", - "8_Manual", ] # List of masks not to apply and not to copy to minimal catalogue diff --git a/src/sp_validation/plots.py b/src/sp_validation/plots.py index b21599c5..60b5b2b4 100644 --- a/src/sp_validation/plots.py +++ b/src/sp_validation/plots.py @@ -20,6 +20,7 @@ # Imported for its import-time side effect: sets matplotlib rcParams (plot style). import sp_validation.plot_style # noqa: F401 +from sp_validation.galaxy import MASK_COLUMNS from sp_validation.masks import Mask @@ -477,8 +478,13 @@ def sky_plots(dat, masks, labels, zoom_ra, zoom_dec): # No mask plot_area_mask(ra, dec, zoom) - # SExtractor and SP flags - m_flags = masks[labels["FLAGS"]]._mask & masks[labels["IMAFLAGS_ISO"]]._mask + # SExtractor and SP flags. ShapePipe v2 splits the single IMAFLAGS_ISO + # mask into per-reason MASK_n columns, so combine whichever of them + # the mask config declared. + m_flags = masks[labels["FLAGS"]]._mask + for col in ("IMAFLAGS_ISO",) + tuple(MASK_COLUMNS): + if col in labels: + m_flags = m_flags & masks[labels[col]]._mask plot_area_mask(ra, dec, zoom, mask=m_flags) # Overlap regions @@ -486,11 +492,17 @@ def sky_plots(dat, masks, labels, zoom_ra, zoom_dec): plot_area_mask(ra, dec, zoom, mask=m_over) # Coverage mask - m_point = masks[labels["npoint3"]]._mask & m_over + # Rough pointing coverage; v2 encodes coverage in the MASK_n* columns + m_point = m_over + if "npoint3" in labels: + m_point = masks[labels["npoint3"]]._mask & m_over plot_area_mask(ra, dec, zoom, mask=m_point) # Maximask - m_maxi = masks[labels["1024_Maximask"]]._mask & m_point + m_maxi = m_point + for maxi_key in ("1024_Maximask", "MASK_n1024"): + if maxi_key in labels: + m_maxi = masks[labels[maxi_key]]._mask & m_point plot_area_mask(ra, dec, zoom, mask=m_maxi) # Combined mask over all supplied masks (was passed in by the caller before From e75f5cba749a6f421a347b1a41b44cbcc15ae021 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Wed, 9 Sep 2026 20:42:08 -0400 Subject: [PATCH 09/39] scripts: finish the campaign rename in configs and entry points - params.py interpolated the removed 'name' into two paths (NameError on import); use 'campaign', give it a real default, and point star_cat_path at full_starcat_.hdf5 (hdu_star_cat now documented as legacy FITS only). params_im_sim.py renames 'name' to 'campaign' likewise. - combine_results.get_area matches both the campaign and the legacy patch wording, and raises on a missing file or unmatched pattern instead of returning None / a 1 deg^2 placeholder that silently rescales densities. - merge_psf_cat writes the campaign *name* as a string column (FITS 'A'), matching JointCat; the old 1-based ordinal depended on -p argument order. - compute_m_bias_image_sims counts tiles via the n_tiles attribute or find_dataset_group, not a hardcoded legacy group inside a bare except. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01QbnPCyzuDNTgkg715pHhar --- scripts/calibration/params.py | 14 ++++++----- scripts/combine_results.py | 28 ++++++++++++++-------- scripts/compute_m_bias_image_sims.py | 36 +++++++++++++++------------- scripts/merge_psf_cat.py | 9 +++++-- workflow/image_sims/params_im_sim.py | 10 ++++---- 5 files changed, 57 insertions(+), 40 deletions(-) diff --git a/scripts/calibration/params.py b/scripts/calibration/params.py index c156bae2..15d8f9f8 100644 --- a/scripts/calibration/params.py +++ b/scripts/calibration/params.py @@ -25,8 +25,9 @@ # Survey parameters -## Campaign name (the tile list processed by ShapePipe). Put None if n/a -campaign = None +## Campaign name (the tile list processed by ShapePipe); required, it names +## the ShapePipe v2 products (final_cat_.hdf5, ...) +campaign = "W3" ## Area of a tile in deg^2 area_tile = 0.25 @@ -43,19 +44,20 @@ data_dir = "." ### Tile IDs -path_tile_ID = f"{data_dir}/tiles_{name}.txt" +path_tile_ID = f"{data_dir}/tiles_{campaign}.txt" ### Weak-lensing galaxy catalog name -galaxy_cat_path = f"{data_dir}/final_cat_{name}.hdf5" +galaxy_cat_path = f"{data_dir}/final_cat_{campaign}.hdf5" print(f"Galaxy catalogue = {galaxy_cat_path}") ## Parameter list; optional, set to `None` if not required param_list_path = f"{data_dir}/cfis/final_cat.param" ### Star and PSF catalog name; optional, set to `None` if not required -star_cat_path = f"{data_dir}/full_starcat-0000000.fits" +star_cat_path = f"{data_dir}/full_starcat_{campaign}.hdf5" -# HDU number of star and PSF catalogue +# HDU number of star and PSF catalogue; only used for the legacy FITS star +# catalogue (a path ending in .fits), ignored for the v2 HDF5 product hdu_star_cat = 1 ### External mask; optional, set to `None` if not required diff --git a/scripts/combine_results.py b/scripts/combine_results.py index 50cc91c5..1b3b4f85 100755 --- a/scripts/combine_results.py +++ b/scripts/combine_results.py @@ -217,18 +217,26 @@ def print_all( def get_area(fname): + """Return the unmasked area in deg^2 read from an area.txt file. - if os.path.exists(fname): - with open(fname) as f: - lines = f.readlines() - for line in lines: - m = re.search("nmasked campaign area without overlap = (.*) deg", line) - if m: - return float(m[1]) + Accepts both the v2 wording ("campaign") and the legacy one ("patch"), + so results computed before the campaign rename can still be combined. + Raises rather than returning a placeholder: a wrong area silently + rescales every density. + """ + if not os.path.exists(fname): + raise FileNotFoundError(f"No file {fname} found to obtain area") + + with open(fname) as f: + lines = f.readlines() + for line in lines: + m = re.search( + r"nmasked (?:campaign|patch) area without overlap = (.*) deg", line + ) + if m: + return float(m[1]) - else: - print(f"Warning: No file {fname} found to obtain area") - return 1 + raise ValueError(f"No unmasked area found in file {fname}") def get_values(results, stats_files, shape, use_keys, area_deg2=-1): diff --git a/scripts/compute_m_bias_image_sims.py b/scripts/compute_m_bias_image_sims.py index 04ca49b7..eb0a76a8 100644 --- a/scripts/compute_m_bias_image_sims.py +++ b/scripts/compute_m_bias_image_sims.py @@ -57,23 +57,25 @@ def parse_args(): def get_n_tiles(grids_dir, num): - """Detect number of tiles from final_cat HDF5 files.""" - try: - import h5py - - # Count tiles in first sim's final_cat - for sim in ["1z2z_grid", "1m2z_grid", "1p2z_grid", "1z2m_grid", "1z2p_grid"]: - sim_name = f"{sim}_{num}" - final_cat = os.path.join(grids_dir, sim_name, f"final_cat_{sim_name}.hdf5") - if os.path.isfile(final_cat): - with h5py.File(final_cat, "r") as hf: - if "patches" in hf: - n_tiles = sum( - 1 for patch in hf["patches"] for _ in hf[f"patches/{patch}"] - ) - return n_tiles - except Exception: - pass + """Detect number of tiles from final_cat HDF5 files. + + Layout-agnostic: uses ``sp_validation.catalog.find_dataset_group``, so it + works on both the legacy nested and the flat per-tile HDF5 layouts. + """ + import h5py + + from sp_validation.catalog import find_dataset_group + + for sim in ["1z2z_grid", "1m2z_grid", "1p2z_grid", "1z2m_grid", "1z2p_grid"]: + sim_name = f"{sim}_{num}" + final_cat = os.path.join(grids_dir, sim_name, f"final_cat_{sim_name}.hdf5") + if os.path.isfile(final_cat): + with h5py.File(final_cat, "r") as hf: + n_tiles = hf.attrs.get("n_tiles") + if n_tiles is not None: + return int(n_tiles) + return len(find_dataset_group(hf)) + return None diff --git a/scripts/merge_psf_cat.py b/scripts/merge_psf_cat.py index adda9d22..8ca34cf4 100644 --- a/scripts/merge_psf_cat.py +++ b/scripts/merge_psf_cat.py @@ -143,7 +143,9 @@ def merge_catalogues(self, campaigns): for name in col_names: dat_all[name] = np.append(dat_all[name], dat[name]) - dat_all["campaign"] = np.append(dat_all["campaign"], [idx + 1] * len(dat)) + dat_all["campaign"] = np.append( + dat_all["campaign"], [campaign] * len(dat) + ) col_names = col_names + ("campaign",) @@ -152,7 +154,10 @@ def merge_catalogues(self, campaigns): if name != "campaign": my_format = "D" else: - my_format = "I" + # Store the campaign name, matching JointCat's string + # `campaign` column; an ordinal would depend on the order of + # the -p argument and could not be joined back. + my_format = f"A{max(len(c) for c in campaigns)}" column = fits.Column(name=name, array=dat_all[name], format=my_format) column_all.append(column) diff --git a/workflow/image_sims/params_im_sim.py b/workflow/image_sims/params_im_sim.py index 76fdd64b..1865fe0b 100644 --- a/workflow/image_sims/params_im_sim.py +++ b/workflow/image_sims/params_im_sim.py @@ -27,11 +27,11 @@ # Survey parameters -## Field name -- derived from the run directory, which is named +## Campaign name -- derived from the run directory, which is named ## after the simulation (e.g. '1z2z_grid_1'), so one shared params file ## serves every sim. -name = os.path.basename(os.getcwd()) -print("Field name = {}".format(name)) +campaign = os.path.basename(os.getcwd()) +print("Campaign name = {}".format(campaign)) ## Area of a tile in deg^2 area_tile = 0.25 @@ -49,10 +49,10 @@ data_dir = "." ### Tile IDs -path_tile_ID = f"{data_dir}/tiles_{name}.txt" +path_tile_ID = f"{data_dir}/tiles_{campaign}.txt" ### Weak-lensing galaxy catalog name -galaxy_cat_path = f"{data_dir}/final_cat_{name}.hdf5" +galaxy_cat_path = f"{data_dir}/final_cat_{campaign}.hdf5" print(f"Galaxy catalogue = {galaxy_cat_path}") ## Parameter list; optional, set to `None` if not required From 7f73c83a34c312afb8a89accda0d46a809af75bd Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Wed, 9 Sep 2026 20:42:08 -0400 Subject: [PATCH 10/39] tests, docs: pin row order, cover the new failure modes Tests: assert the exact concatenation instead of sorted values, add a fixture whose keys are inserted out of order, and cover the truncated n_tiles file, a column missing from a later tile, dtype promotion across campaigns in both argument orders, reduce_mem overflow, and a multi-dimensional column. Docs: CLAUDE.md, scripts/calibration/README.md, homogenize_cat_extended.py and the catalog_builders docstrings now say campaign. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01QbnPCyzuDNTgkg715pHhar --- CLAUDE.md | 4 +- scripts/calibration/README.md | 9 +- scripts/homogenize_cat_extended.py | 2 +- .../tests/test_campaign_readers.py | 119 +++++++++++++++++- 4 files changed, 123 insertions(+), 11 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index b4447758..606ed340 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -67,10 +67,10 @@ snakemake --profile workflow/profiles/candide -s workflow/Snakefile \ ### Configuration Main configuration in `scripts/calibration/params.py` with parameters: -- `name`: Field/patch identifier +- `campaign`: Campaign name (the ShapePipe tile list); names the input products - `data_dir`: Input data directory - `galaxy_cat_path`: Galaxy catalogue path (.fits/.hdf5) -- `star_cat_path`: Star catalogue path (.fits) +- `star_cat_path`: Star catalogue path (.hdf5, or legacy .fits) ### Key Dependencies - astropy, numpy, scipy for core calculations diff --git a/scripts/calibration/README.md b/scripts/calibration/README.md index 870c87ce..cd0a49e3 100644 --- a/scripts/calibration/README.md +++ b/scripts/calibration/README.md @@ -6,14 +6,13 @@ in order. See `docs/source/post_processing.md` for the full prose. | Step | Script | Does | |------|--------|------| -| 1 | `extract_info.py` | Extract metacal + diagnostic info per patch; create pre-calibration shear catalogues. Configured via `params.py`. | -| 2 | `create_joint_comprehensive_cat.py` | Merge the patch-wise comprehensive catalogues into one joint catalogue (front-end of `catalog_builders.JointCat`). | +| 1 | `extract_info.py` | Extract metacal + diagnostic info for one campaign; create pre-calibration shear catalogues. Configured via `params.py`. | +| 2 | `create_joint_comprehensive_cat.py` | Merge the per-campaign comprehensive catalogues into one joint catalogue (front-end of `catalog_builders.JointCat`). | | 3 | `demo_apply_hsp_masks.py` | Add the structural and coverage (HealSparse) masks. | | 4 | `calibrate_comprehensive_cat.py` | Galaxy selection + metacalibration. Uses the mask configs in `config/calibration/`. | `params.py` is the shared parameter template (paths, column names, survey constants) imported by `extract_info.py`; copy and edit it per run. -> **v2.0 note (Martin):** this clustering reflects the current (v1.4.x) reduction -> flow. v2.0 no longer has patches, so step 2 (and the per-patch structure of -> steps 1/3) will change. +> **v2.0 note:** ShapePipe v2 has no patches. Step 1 runs per campaign and +> step 2 merges a list of campaign catalogues (`final_cat_.hdf5`). diff --git a/scripts/homogenize_cat_extended.py b/scripts/homogenize_cat_extended.py index dd54922e..f197a941 100644 --- a/scripts/homogenize_cat_extended.py +++ b/scripts/homogenize_cat_extended.py @@ -3,7 +3,7 @@ Overwrite the extended catalog to replace the columns e1 and e2 from the extended catalog with the columns e1 and e2 from the non-extended one. Currently, the calibration of the columns e1 and e2 of the extended catalog -are calibrated per patch, while the non-extended catalog is calibrated on the whole footprint. +are calibrated per campaign, while the non-extended catalog is calibrated on the whole footprint. :Authors: Sacha Guerrini, Martin Kilbinger """ diff --git a/src/sp_validation/tests/test_campaign_readers.py b/src/sp_validation/tests/test_campaign_readers.py index ccba4fe5..b5714ec1 100644 --- a/src/sp_validation/tests/test_campaign_readers.py +++ b/src/sp_validation/tests/test_campaign_readers.py @@ -61,7 +61,8 @@ def tearDown(self): self._tmp.cleanup() def _expected(self): - return np.concatenate(list(self._tiles.values())) + """Expected concatenation: datasets in sorted-key order.""" + return np.concatenate([self._tiles[key] for key in sorted(self._tiles)]) def test_legacy_layout(self): path = self._dir / "final_cat_CAMPAIGN.hdf5" @@ -70,7 +71,7 @@ def test_legacy_layout(self): dat = catalog.read_campaign_catalogue(path, verbose=False) self.assertEqual(len(dat), 5) - npt.assert_array_equal(np.sort(dat["RA"]), np.sort(self._expected()["RA"])) + npt.assert_array_equal(dat["RA"], self._expected()["RA"]) def test_flat_layout(self): path = self._dir / "final_cat_CAMPAIGN.hdf5" @@ -79,7 +80,7 @@ def test_flat_layout(self): dat = catalog.read_campaign_catalogue(path, verbose=False) self.assertEqual(len(dat), 5) - npt.assert_array_equal(np.sort(dat["RA"]), np.sort(self._expected()["RA"])) + npt.assert_array_equal(dat["RA"], self._expected()["RA"]) def test_layouts_agree(self): legacy = self._dir / "legacy.hdf5" @@ -112,6 +113,54 @@ def test_missing_column_raises(self): ) self.assertIn("NOT_A_COLUMN", str(ctx.exception)) + + def test_row_order_is_tile_name_order(self): + """Keys inserted out of order still concatenate in name order.""" + path = self._dir / "unordered.hdf5" + tiles = { + "222.000": make_galaxy_data(2, offset=200), + "000.000": make_galaxy_data(2, offset=0), + "111.000": make_galaxy_data(2, offset=100), + } + with h5py.File(path, "w") as f: + group = f.create_group("tiles") + for tile_id, dat in tiles.items(): + group.create_dataset(tile_id, data=dat) + f.attrs["n_tiles"] = len(tiles) + + dat = catalog.read_campaign_catalogue(path, verbose=False) + + npt.assert_array_equal(dat["RA"], [0, 1, 100, 101, 200, 201]) + + def test_truncated_file_raises(self): + """n_tiles attribute larger than the number of datasets is fatal.""" + path = self._dir / "truncated.hdf5" + write_campaign(path, "flat", self._tiles) + with h5py.File(path, "a") as f: + f.attrs["n_tiles"] = 10 + + with self.assertRaises(ValueError) as ctx: + catalog.read_campaign_catalogue(path, verbose=False) + self.assertIn("incomplete", str(ctx.exception)) + + def test_column_missing_from_later_tile_raises_clear_error(self): + """A column absent from a non-first tile is named, with its dataset.""" + path = self._dir / "ragged.hdf5" + with h5py.File(path, "w") as f: + group = f.create_group("tiles") + group.create_dataset("000.000", data=make_galaxy_data(2)) + group.create_dataset( + "001.000", data=np.zeros(2, dtype=[("RA", "f8"), ("Dec", "f8")]) + ) + + with self.assertRaises(KeyError) as ctx: + catalog.read_campaign_catalogue( + path, param_list=["RA", "MAG_AUTO"], verbose=False + ) + message = str(ctx.exception) + self.assertIn("MAG_AUTO", message) + self.assertIn("001.000", message) + def test_ambiguous_layout_raises(self): path = self._dir / "ambiguous.hdf5" with h5py.File(path, "w") as f: @@ -242,6 +291,70 @@ def test_merge(self): ) npt.assert_array_equal(dat["RA"][:3], np.arange(3)) + + def test_merge_promotes_column_widths(self): + """A wider string/int column in a later file is not truncated.""" + dtype_narrow = np.dtype([("RA", "f8"), ("TILE_ID", "S7"), ("N", "i4")]) + dtype_wide = np.dtype([("RA", "f8"), ("TILE_ID", "S12"), ("N", "i8")]) + + narrow = np.zeros(1, dtype=dtype_narrow) + narrow["TILE_ID"] = b"123.456" + narrow["N"] = 7 + wide = np.zeros(1, dtype=dtype_wide) + wide["TILE_ID"] = b"999888.7776" + wide["N"] = 2**40 + + path_a = self._dir / "final_cat_AA.hdf5" + path_b = self._dir / "final_cat_BBBBBBBB.hdf5" + write_campaign(path_a, "flat", {"000.000": narrow}) + write_campaign(path_b, "flat", {"000.000": wide}) + + for paths in ([path_a, path_b], [path_b, path_a]): + dat = self._obj.merge_catalogues([str(path) for path in paths]) + by_campaign = { + name: row for name, row in zip(dat["campaign"], dat) + } + self.assertEqual(by_campaign[b"BBBBBBBB"]["TILE_ID"], b"999888.7776") + self.assertEqual(by_campaign[b"BBBBBBBB"]["N"], 2**40) + self.assertEqual(by_campaign[b"AA"]["TILE_ID"], b"123.456") + + def test_merge_reduce_mem_keeps_integers(self): + """reduce_mem leaves integer columns at their input width.""" + dat = np.zeros(3, dtype=[("RA", "f8"), ("N_EPOCH", "i4"), ("w", "f8")]) + dat["N_EPOCH"] = [3, 200, 300000] + dat["w"] = [0.5, 1.0, 2.0] + path = self._dir / "final_cat_Y.hdf5" + write_campaign(path, "flat", {"000.000": dat}) + + self._obj._params["reduce_mem"] = True + out = self._obj.merge_catalogues([str(path)]) + + npt.assert_array_equal(out["N_EPOCH"], [3, 200, 300000]) + self.assertEqual(out.dtype["N_EPOCH"], np.dtype("i4")) + self.assertEqual(out.dtype["RA"], np.dtype("f8")) # RA keeps precision + self.assertEqual(out.dtype["w"], np.dtype("f4")) + + def test_merge_reduce_mem_float_overflow_raises(self): + """A float64 value beyond float32 range is refused, not stored as inf.""" + dat = np.zeros(2, dtype=[("RA", "f8"), ("w", "f8")]) + dat["w"] = [1.0, 1e300] + path = self._dir / "final_cat_Z.hdf5" + write_campaign(path, "flat", {"000.000": dat}) + + self._obj._params["reduce_mem"] = True + with self.assertRaises(ValueError) as ctx: + self._obj.merge_catalogues([str(path)]) + self.assertIn("'w'", str(ctx.exception)) + + def test_merge_rejects_multidimensional_column(self): + dat = np.zeros(2, dtype=[("RA", "f8"), ("XY", "f8", (2,))]) + path = self._dir / "final_cat_M.hdf5" + write_campaign(path, "flat", {"000.000": dat}) + + with self.assertRaises(ValueError) as ctx: + self._obj.merge_catalogues([str(path)]) + self.assertIn("multi-dimensional", str(ctx.exception)) + def test_merge_incompatible_columns_raises(self): other = self._dir / "final_cat_X.hdf5" dat = np.zeros(2, dtype=[("RA", "f8")]) From 5c4a6367f8fc9ee1fdfad0863bb9b14aee190a5c Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Wed, 9 Sep 2026 21:31:49 -0400 Subject: [PATCH 11/39] catalog: promote dtypes across tiles, stream the campaign merge Three defects in the campaign reading path, found in review: - concatenate_datasets and campaign_shape both took the output dtype from the first dataset alone, so a campaign whose tiles differ (S7 next to S12 tile IDs, i2 next to i4, f4 next to f8 after a partial reprocessing) had the later tiles silently truncated, downcast or wrapped. np.concatenate, which this code replaced, promoted. Both now build the dtype with group_dtype(), promoting every column across every dataset; catalog_builders._promote becomes an alias of the shared catalog.promote_dtypes rather than a second copy of it. - merge_catalogues still held a whole campaign in memory next to the preallocated output, the very thing its comment claimed the rewrite avoided -- and with one campaign per merge in v2, that is the normal case, ~2x the merged catalogue at DR6 scale. It now fills the output tile by tile through the new catalog.iter_campaign_tiles(), so peak memory is the output plus a single tile. - write_hdf5_file wrote the merged array twice, create_dataset(data=...) followed by an immediate dset[:] = dat_all. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01QbnPCyzuDNTgkg715pHhar --- src/sp_validation/catalog.py | 113 ++++++++++++++++-- src/sp_validation/catalog_builders.py | 41 +++---- .../tests/test_campaign_readers.py | 61 ++++++++++ 3 files changed, 183 insertions(+), 32 deletions(-) diff --git a/src/sp_validation/catalog.py b/src/sp_validation/catalog.py index 8296ecee..c06efac4 100644 --- a/src/sp_validation/catalog.py +++ b/src/sp_validation/catalog.py @@ -820,6 +820,72 @@ def find_dataset_group(hdf5_file): node = node[keys[0]] +def promote_dtypes(dtype_a, dtype_b): + """Promote Dtypes. + + Return a scalar dtype that holds both input dtypes without truncation + or overflow. + + Parameters + ---------- + dtype_a : numpy.dtype + first input dtype + dtype_b : numpy.dtype + second input dtype + + Returns + ------- + numpy.dtype + promoted dtype + + """ + if dtype_a == dtype_b: + return dtype_a + if dtype_a.kind in "SU" and dtype_b.kind in "SU": + kind = "U" if "U" in (dtype_a.kind, dtype_b.kind) else "S" + size_a = dtype_a.itemsize // (4 if dtype_a.kind == "U" else 1) + size_b = dtype_b.itemsize // (4 if dtype_b.kind == "U" else 1) + return np.dtype(f"{kind}{max(size_a, size_b)}") + + return np.promote_types(dtype_a, dtype_b) + + +def group_dtype(group, keys, param_list=None): + """Group Dtype. + + Build the structured output dtype of a group of per-tile datasets, + promoting each column across *every* dataset. A campaign that was + partially reprocessed can carry e.g. ``S7`` tile IDs in one tile and + ``S12`` in another, or ``f4`` next to ``f8``; taking the dtype of the + first dataset alone would silently truncate the others. + + Parameters + ---------- + group : h5py.Group + group whose members are structured datasets + keys : list of str + dataset names to consider + param_list : list of str, optional + columns to keep; default is ``None`` (keep all) + + Returns + ------- + numpy.dtype + structured output dtype + + """ + names = param_list if param_list is not None else list(group[keys[0]].dtype.names) + + fields = [] + for name in names: + promoted = group[keys[0]].dtype[name] + for key in keys[1:]: + promoted = promote_dtypes(promoted, group[key].dtype[name]) + fields.append((name, promoted)) + + return np.dtype(fields) + + def _check_columns(dtype, param_list, file_path, dataset_key=None): """Raise a clear error if requested columns are absent from the data.""" missing = [col for col in param_list if col not in (dtype.names or ())] @@ -868,11 +934,7 @@ def concatenate_datasets(group, param_list=None, file_path="", verbose=True): if param_list is not None: _check_columns(group[key].dtype, param_list, file_path, key) - dtype_in = group[keys[0]].dtype - if param_list is not None: - dtype_out = np.dtype([(name, dtype_in[name]) for name in param_list]) - else: - dtype_out = dtype_in + dtype_out = group_dtype(group, keys, param_list=param_list) n_rows = sum(group[key].shape[0] for key in keys) if verbose: @@ -955,13 +1017,48 @@ def campaign_shape(file_path, param_list=None): if not keys: raise ValueError(f"No datasets found in catalogue {file_path}") n_rows = sum(group[key].shape[0] for key in keys) - dtype_in = group[keys[0]].dtype if param_list is not None: for key in keys: _check_columns(group[key].dtype, param_list, file_path, key) - dtype_in = np.dtype([(name, dtype_in[name]) for name in param_list]) + dtype_out = group_dtype(group, keys, param_list=param_list) + + return n_rows, dtype_out + + +def iter_campaign_tiles(file_path, param_list=None, verbose=True): + """Iter Campaign Tiles. + + Yield the per-tile datasets of a campaign catalogue one at a time, so a + caller filling a preallocated output array never holds more than one + tile in memory in addition to that output. - return n_rows, dtype_in + Parameters + ---------- + file_path : str + input file path + param_list : list of str, optional + columns to keep; default is ``None`` (keep all) + verbose : bool, optional + verbose output if ``True`` + + Yields + ------ + numpy.ndarray + one tile as a structured array + + """ + with h5py.File(file_path, "r") as hdf5_file: + group = find_dataset_group(hdf5_file) + check_n_tiles(hdf5_file, group, file_path) + keys = sorted(group) + if not keys: + raise ValueError(f"No datasets found in catalogue {file_path}") + for key in keys: + if param_list is not None: + _check_columns(group[key].dtype, param_list, file_path, key) + for key in tqdm.tqdm(keys, disable=not verbose): + data = group[key][()] + yield data if param_list is None else data[param_list] def read_campaign_catalogue( diff --git a/src/sp_validation/catalog_builders.py b/src/sp_validation/catalog_builders.py index e06ee105..8868595e 100644 --- a/src/sp_validation/catalog_builders.py +++ b/src/sp_validation/catalog_builders.py @@ -240,17 +240,8 @@ def close_hd5(self): self._hd5file.close() -def _promote(dtype_a, dtype_b): - """Return a dtype that holds both input dtypes without truncation.""" - if dtype_a == dtype_b: - return dtype_a - if dtype_a.kind in "SU" and dtype_b.kind in "SU": - kind = "U" if "U" in (dtype_a.kind, dtype_b.kind) else "S" - size_a = dtype_a.itemsize // (4 if dtype_a.kind == "U" else 1) - size_b = dtype_b.itemsize // (4 if dtype_b.kind == "U" else 1) - return np.dtype(f"{kind}{max(size_a, size_b)}") - - return np.promote_types(dtype_a, dtype_b) +# Column-dtype promotion is shared with the per-campaign reader in ``catalog``. +_promote = sp_cat.promote_dtypes def _checked_assign(target, start, end, values, name): @@ -500,8 +491,7 @@ def write_hdf5_file(self, dat_all, campaigns=None): with h5py.File(output_path, "w") as f: self.write_hdf5_header(f, campaigns=campaigns) - dset = f.create_dataset("data", data=dat_all) - dset[:] = dat_all + f.create_dataset("data", data=dat_all) def write_hdf5_header(self, hd5file, campaigns=None): """Write HDF5 Header. @@ -562,24 +552,27 @@ def merge_catalogues(self, input_paths): dat_all = np.empty(n_total, dtype=dtype_out) start = 0 for input_path, campaign in zip(input_paths, campaigns): - dat = sp_cat.read_campaign_catalogue( + # Fill tile by tile: peak memory is the merged output plus a + # single tile, never a whole campaign copy on top of it. + n_campaign = 0 + for dat in sp_cat.iter_campaign_tiles( input_path, param_list=param_list, verbose=self._params["verbose"], - ) - - end = start + len(dat) - for name in dat.dtype.names: - _checked_assign(dat_all[name], start, end, dat[name], name) - dat_all["campaign"][start:end] = campaign.encode() - start = end + ): + end = start + len(dat) + for name in dat.dtype.names: + _checked_assign(dat_all[name], start, end, dat[name], name) + dat_all["campaign"][start:end] = campaign.encode() + start = end + n_campaign += len(dat) + del dat if self._params["verbose"]: print( - f"{campaign}: added {len(dat)}" - + f" (~{format.millify(len(dat))}) objects." + f"{campaign}: added {n_campaign}" + + f" (~{format.millify(n_campaign)}) objects." ) - del dat if self._params["verbose"]: print( diff --git a/src/sp_validation/tests/test_campaign_readers.py b/src/sp_validation/tests/test_campaign_readers.py index b5714ec1..f237d4f0 100644 --- a/src/sp_validation/tests/test_campaign_readers.py +++ b/src/sp_validation/tests/test_campaign_readers.py @@ -161,6 +161,67 @@ def test_column_missing_from_later_tile_raises_clear_error(self): self.assertIn("MAG_AUTO", message) self.assertIn("001.000", message) + def test_dtype_promoted_across_tiles(self): + """A per-tile dtype difference within a campaign is not truncated.""" + narrow = np.zeros( + 2, dtype=[("N_EPOCH", "i2"), ("TILE_ID", "S7"), ("RA", "f4")] + ) + narrow["N_EPOCH"] = [1, 2] + narrow["TILE_ID"] = [b"123.456", b"123.457"] + narrow["RA"] = [1.5, 2.5] + + wide = np.zeros( + 2, dtype=[("N_EPOCH", "i4"), ("TILE_ID", "S12"), ("RA", "f8")] + ) + wide["N_EPOCH"] = [70000, 3] + wide["TILE_ID"] = [b"999888.7776", b"123.458"] + wide["RA"] = [3.123456789, 4.0] + + path = self._dir / "final_cat_MIX.hdf5" + write_campaign(path, "flat", {"000.000": narrow, "000.001": wide}) + param_list = ["N_EPOCH", "TILE_ID", "RA"] + + dat = catalog.read_campaign_catalogue( + str(path), param_list=param_list, verbose=False + ) + + self.assertEqual(dat.dtype["N_EPOCH"], np.dtype("i4")) + self.assertEqual(dat.dtype["TILE_ID"], np.dtype("S12")) + self.assertEqual(dat.dtype["RA"], np.dtype("f8")) + npt.assert_array_equal(dat["N_EPOCH"], [1, 2, 70000, 3]) + npt.assert_array_equal( + dat["TILE_ID"], + [b"123.456", b"123.457", b"999888.7776", b"123.458"], + ) + self.assertEqual(dat["RA"][2], 3.123456789) + + # campaign_shape must report the same promoted dtype, since the merge + # preallocates from it. + n_rows, dtype_out = catalog.campaign_shape( + str(path), param_list=param_list + ) + self.assertEqual(n_rows, 4) + self.assertEqual(dtype_out, dat.dtype) + + def test_iter_campaign_tiles(self): + """The streaming reader yields one restricted tile at a time.""" + path = self._dir / "final_cat_CAMPAIGN.hdf5" + write_campaign(path, "legacy", self._tiles) + + tiles = list( + catalog.iter_campaign_tiles( + str(path), param_list=["RA"], verbose=False + ) + ) + + self.assertEqual([len(tile) for tile in tiles], [3, 2]) + for tile in tiles: + self.assertEqual(tile.dtype.names, ("RA",)) + npt.assert_array_equal( + np.concatenate([tile["RA"] for tile in tiles]), + self._expected()["RA"], + ) + def test_ambiguous_layout_raises(self): path = self._dir / "ambiguous.hdf5" with h5py.File(path, "w") as f: From 335b1a75ef87ae085d07199618b548a791979b78 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Wed, 9 Sep 2026 21:31:49 -0400 Subject: [PATCH 12/39] config: restore the r-band coverage cut in the v2 mask config mask_v2.0.yaml dropped v1.X's '64_r' r-band imaging cut without a replacement, so a v2 calibration run admitted objects outside the r-band footprint that v1 rejected -- a silent change of effective area, n(z) and galaxy-density normalisation. Enable MASK_n64, v1's '64_r' equivalent. v1's other coverage cut, npoint3 >= 3, came from an external post-processing catalogue and has no v2 counterpart; say so in the config rather than leaving its absence unexplained. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01QbnPCyzuDNTgkg715pHhar --- config/calibration/mask_v2.0.yaml | 16 +++++++++++++++- 1 file changed, 15 insertions(+), 1 deletion(-) diff --git a/config/calibration/mask_v2.0.yaml b/config/calibration/mask_v2.0.yaml index db824f59..201388eb 100644 --- a/config/calibration/mask_v2.0.yaml +++ b/config/calibration/mask_v2.0.yaml @@ -59,7 +59,15 @@ dat: kind: equal value: False - # Per-band coverage and Pan-STARRS z2; enable as required + # r-band coverage: the v2 equivalent of v1's '64_r' cut, kept so that the + # v2 selection covers the same imaging footprint as v1.X. + - col_name: MASK_n64 + label: "r-band imaging" + kind: equal + value: False + + # Remaining per-band coverage bits and Pan-STARRS z2; enable as required. + # Cutting on all of them at once leaves essentially no objects. # - col_name: MASK_n16 # label: "coverage (n16)" # kind: equal @@ -68,6 +76,12 @@ dat: # label: "no PS-z2" # kind: equal # value: False + # + # v1.X also applied a rough pointing-coverage cut, 'npoint3' >= 3, from an + # external post-processing catalogue. ShapePipe v2 emits no such column and + # no MASK_n* bit encodes it, so that cut has no v2 counterpart; add it back + # here if an external pointing-coverage map is ever joined onto the + # catalogue (it would belong in the 'dat_ext' group below). # Number of epochs - col_name: N_EPOCH From 1fbc27a2617afb5d89b5b201b7d7324e54cacca1 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Wed, 9 Sep 2026 21:31:49 -0400 Subject: [PATCH 13/39] scripts: cheaper star mask cut, campaign-derived leakage labels extract_info fancy-indexed the full-width catalogue, dd[ind_star], only to read ~5 boolean mask columns from it -- a copy of every column for every matched star (~GB at DR6 scale). Mask first, index the resulting bool array. plot_leakage still labelled its curves "all", "P1" ... "P7": a P-named survivor of the #340 retirement, and a fixed length that silently mismatched the number of input files. Labels and colours now follow the input files. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01QbnPCyzuDNTgkg715pHhar --- scripts/calibration/extract_info.py | 2 +- scripts/plot_leakage.py | 14 +++++++++++++- 2 files changed, 14 insertions(+), 2 deletions(-) diff --git a/scripts/calibration/extract_info.py b/scripts/calibration/extract_info.py index ed4fb145..d0994f07 100644 --- a/scripts/calibration/extract_info.py +++ b/scripts/calibration/extract_info.py @@ -160,7 +160,7 @@ m_star = ( (dd["FLAGS"][ind_star] == 0) - & galaxy.mask_cut(dd[ind_star], mask_columns) + & galaxy.mask_cut(dd, mask_columns)[ind_star] & (dd["NGMIX_MCAL_FLAGS"][ind_star] == 0) & (dd["NGMIX_G1_PSF_ORIG_NOSHEAR"][ind_star] != -10) ) diff --git a/scripts/plot_leakage.py b/scripts/plot_leakage.py index 629d81dc..c5b8f956 100755 --- a/scripts/plot_leakage.py +++ b/scripts/plot_leakage.py @@ -1,6 +1,7 @@ #!/usr/bin/env python3 import copy +import os import sys from optparse import OptionParser @@ -152,6 +153,7 @@ def plot_alpha_leakage( xmin, xmax, ylim=None, + labels=None, ): """Plot Alpha Leakage. @@ -173,6 +175,9 @@ def plot_alpha_leakage( largest angular scale, interpreted in arcmin ylim : list, optional y-axis plot limits, default is `Ǹone` + labels : list, optional + curve labels, one per input curve; default is ``None``, for + unlabelled curves """ theta = meanr @@ -187,7 +192,7 @@ def plot_alpha_leakage( linewidths[0] = 3 colors = ["grey", "k", "b", "r", "c", "m", "g", "orange"] - labels = ["all", "P1", "P2", "P3", "P4", "P5", "P6", "P7"] + colors = [colors[idx % len(colors)] for idx in range(len(meanr))] plot_data_1d( theta, @@ -239,6 +244,12 @@ def main(argv=None): if param.verbose: print("Input files: ", fnames) + # Curve labels come from the input file names: the first file is the + # reference (all objects), the others whatever selection they hold. + labels = ["all"] + [ + os.path.splitext(os.path.basename(fn))[0] for fn in fnames[1:] + ] + # read input files, append data theta = [] alpha_leak = [] @@ -265,6 +276,7 @@ def main(argv=None): config.theta_min_amin, config.theta_max_amin, config.leakage_alpha_ylim, + labels=labels, ) return 0 From 81fc03e66cfc007e77340500c83f7edf83d61964 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Wed, 9 Sep 2026 21:55:17 -0400 Subject: [PATCH 14/39] masks: treat n64 as a reason bit, not r-band coverage The v2 branch inferred from v1's '64_r' naming that MASK_n64 flags r-band imaging coverage. It does not: bit 64 is an undocumented *reason* bit of the r-band default bitmask, and OR{n1,n2,n4,n8,n64,n1024} reproduces mask_r, the v1 r-band mask, exactly on the P3 region. Add MASK_n64 to DEFAULT_MASK_COLUMNS so the default galaxy cut is exactly that set, documented as reproducing mask_r, and mirror it in the calibration params, the v2 mask config and the minimal-catalogue demo. Document n16/n32/n128/n256 as the u/g/i/z coverage flags (no r flag: the catalogue is r-selected) and n2048 as absent Pan-STARRS z2. The faint vs bright assignment of n1/n2 is unconfirmed for the Aug-2026 products, so the labels no longer claim one. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01QbnPCyzuDNTgkg715pHhar --- config/calibration/mask_v2.0.yaml | 34 +++++++++++++------ scripts/calibration/params.py | 4 ++- .../demo_comprehensive_to_minimal_cat.py | 1 + src/sp_validation/galaxy.py | 23 +++++++++---- .../tests/test_campaign_readers.py | 1 + 5 files changed, 45 insertions(+), 18 deletions(-) diff --git a/config/calibration/mask_v2.0.yaml b/config/calibration/mask_v2.0.yaml index 201388eb..37b0f5c7 100644 --- a/config/calibration/mask_v2.0.yaml +++ b/config/calibration/mask_v2.0.yaml @@ -4,16 +4,26 @@ # mask columns MASK_n, where True means MASKED. They are therefore cut # with `kind: equal, value: False` (keep the un-masked objects): # -# MASK_n1, MASK_n2 bright/faint star halos +# Reason bits of the ShapePipe r-band default bitmask: +# +# MASK_n1, MASK_n2 star halos (which is faint and which is bright is +# unconfirmed for the Aug-2026 products) # MASK_n4 stars # MASK_n8 manual galaxy mask -# MASK_n16 .. MASK_n256 per-band coverage +# MASK_n64 undocumented reason bit # MASK_n1024 MaxiMask +# +# Per-band coverage flags and Pan-STARRS: +# +# MASK_n16, MASK_n32, MASK_n128, MASK_n256 u, g, i, z coverage (no r +# flag: the catalogue is r-selected) # MASK_n2048 no Pan-STARRS z2 coverage # # The default selection (see sp_validation.galaxy.DEFAULT_MASK_COLUMNS) is -# n4 + n1 + n2 + n8 + n1024; the coverage bits and n2048 are listed here but -# commented out, since ORing all of them masks essentially everything. +# n1 + n2 + n4 + n8 + n64 + n1024, whose OR reproduces mask_r, the v1 r-band +# mask, exactly on the P3 region. The coverage flags and n2048 are listed +# here but commented out, since ORing all of them masks essentially +# everything. # General parameters (can also given on command line) params: @@ -42,12 +52,14 @@ dat: label: "stars" kind: equal value: False + # n1/n2 are the two star-halo bits; the faint/bright assignment is + # unconfirmed for the Aug-2026 products. - col_name: MASK_n1 - label: "faint star halos" + label: "star halos (n1)" kind: equal value: False - col_name: MASK_n2 - label: "bright star halos" + label: "star halos (n2)" kind: equal value: False - col_name: MASK_n8 @@ -59,17 +71,17 @@ dat: kind: equal value: False - # r-band coverage: the v2 equivalent of v1's '64_r' cut, kept so that the - # v2 selection covers the same imaging footprint as v1.X. + # Undocumented reason bit of the r-band default bitmask. Required for the + # OR above to reproduce mask_r; it is not a coverage flag. - col_name: MASK_n64 - label: "r-band imaging" + label: "reason bit n64" kind: equal value: False - # Remaining per-band coverage bits and Pan-STARRS z2; enable as required. + # Per-band coverage flags and Pan-STARRS z2; enable as required. # Cutting on all of them at once leaves essentially no objects. # - col_name: MASK_n16 - # label: "coverage (n16)" + # label: "u coverage" # kind: equal # value: False # - col_name: MASK_n2048 diff --git a/scripts/calibration/params.py b/scripts/calibration/params.py index 15d8f9f8..558fcc67 100644 --- a/scripts/calibration/params.py +++ b/scripts/calibration/params.py @@ -122,12 +122,14 @@ "NGMIX_T_PSF_RECONV_NOSHEAR", ] -## ShapePipe v2 mask columns OR'd together for the galaxy selection cut +## ShapePipe v2 mask columns OR'd together for the galaxy selection cut: +## the reason bits of the r-band default bitmask, whose OR reproduces mask_r mask_columns = [ "MASK_n4", "MASK_n1", "MASK_n2", "MASK_n8", + "MASK_n64", "MASK_n1024", ] diff --git a/scripts/examples/demo_comprehensive_to_minimal_cat.py b/scripts/examples/demo_comprehensive_to_minimal_cat.py index 30b6b1df..5c99970e 100644 --- a/scripts/examples/demo_comprehensive_to_minimal_cat.py +++ b/scripts/examples/demo_comprehensive_to_minimal_cat.py @@ -58,6 +58,7 @@ "MASK_n1", "MASK_n2", "MASK_n8", + "MASK_n64", "MASK_n1024", "NGMIX_MCAL_TYPES_FAIL", "NGMIX_G1_PSF_ORIG_NOSHEAR", diff --git a/src/sp_validation/galaxy.py b/src/sp_validation/galaxy.py index 7a8b8b1c..df901b3e 100644 --- a/src/sp_validation/galaxy.py +++ b/src/sp_validation/galaxy.py @@ -29,9 +29,16 @@ from sp_validation import io #: All mask columns written by ShapePipe v2 (bool, ``True`` = masked). -#: n4 stars; n1/n2 faint/bright star halos; n8 manual galaxy mask; -#: n1024 MaxiMask; n16..n256 per-band coverage; n2048 no PS-z2 coverage. -#: These replace the single IMAFLAGS_ISO bitmask of ShapePipe v1. +#: They replace the single IMAFLAGS_ISO bitmask of ShapePipe v1. +#: +#: Reason bits making up the r-band default bitmask: n1/n2 star halos +#: (which of the two is faint and which bright is unconfirmed for the +#: Aug-2026 products), n4 stars, n8 manual galaxy mask, n64 (an +#: undocumented reason bit), n1024 MaxiMask. +#: +#: Per-band coverage flags: n16 (u), n32 (g), n128 (i), n256 (z). There is +#: no r coverage flag because the catalogue is r-selected. n2048 is ``True`` +#: where Pan-STARRS z2 coverage is absent. MASK_COLUMNS = ( "MASK_n1", "MASK_n2", @@ -46,14 +53,18 @@ "MASK_n2048", ) -#: Mask columns OR'd together for the default galaxy selection. Deliberately -#: not a blanket OR over MASK_COLUMNS: the per-band coverage columns -#: (n16..n256) and n2048 would mask essentially the whole catalogue. +#: Mask columns OR'd together for the default galaxy selection. This set is +#: exactly the reason bits of the ShapePipe r-band default bitmask: their OR +#: reproduces ``mask_r``, the v1 r-band mask, on the P3 region. Deliberately +#: not a blanket OR over MASK_COLUMNS: the per-band coverage flags +#: (n16, n32, n128, n256) and n2048 would mask essentially the whole +#: catalogue. DEFAULT_MASK_COLUMNS = ( "MASK_n4", "MASK_n1", "MASK_n2", "MASK_n8", + "MASK_n64", "MASK_n1024", ) diff --git a/src/sp_validation/tests/test_campaign_readers.py b/src/sp_validation/tests/test_campaign_readers.py index f237d4f0..99a0f6e7 100644 --- a/src/sp_validation/tests/test_campaign_readers.py +++ b/src/sp_validation/tests/test_campaign_readers.py @@ -21,6 +21,7 @@ ("MASK_n1", "?"), ("MASK_n2", "?"), ("MASK_n8", "?"), + ("MASK_n64", "?"), ("MASK_n1024", "?"), ] ) From 98f68a0e2a0e723cb8de7cf984cfef5bade4ba05 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Thu, 10 Sep 2026 01:57:08 +0000 Subject: [PATCH 15/39] ruff autofix (format + safe lint fixes) Pushed by the lint gate. --- scripts/calibration/extract_info.py | 5 +-- scripts/combine_results.py | 6 ++-- scripts/glass_mock/compute_leakage_harmony.py | 1 - scripts/merge_psf_cat.py | 4 +-- scripts/plot_leakage.py | 4 +-- src/sp_validation/catalog.py | 4 ++- src/sp_validation/catalog_builders.py | 3 +- src/sp_validation/rho_tau.py | 4 ++- .../tests/test_campaign_readers.py | 35 +++++-------------- src/sp_validation/tests/test_survey.py | 2 -- 10 files changed, 22 insertions(+), 46 deletions(-) diff --git a/scripts/calibration/extract_info.py b/scripts/calibration/extract_info.py index d0994f07..d9de957e 100644 --- a/scripts/calibration/extract_info.py +++ b/scripts/calibration/extract_info.py @@ -34,7 +34,6 @@ import h5py import numpy as np -from astropy.io import fits # from sp_validation.catalog import * from sp_validation import catalog as spv_cat @@ -70,9 +69,7 @@ dd = np.load(galaxy_cat_path, mmap_mode=mmap_mode) else: print("Loading galaxy .hdf5 file...") - dd = spv_cat.read_campaign_catalogue( - galaxy_cat_path, param_path=param_list_path - ) + dd = spv_cat.read_campaign_catalogue(galaxy_cat_path, param_path=param_list_path) n_obj = len(dd) print_stats( diff --git a/scripts/combine_results.py b/scripts/combine_results.py index 1b3b4f85..5b679b98 100755 --- a/scripts/combine_results.py +++ b/scripts/combine_results.py @@ -281,9 +281,9 @@ def get_values(results, stats_files, shape, use_keys, area_deg2=-1): print(f"area({campaign}) = {area_deg2_campaign} deg^2") else: area_deg2_campaign = area_deg2 - results["value"][key_der][campaign] = results["value"]["N_gal"][campaign] / ( - area_deg2_campaign * 3600 - ) + results["value"][key_der][campaign] = results["value"]["N_gal"][ + campaign + ] / (area_deg2_campaign * 3600) if area_deg2 < 0: with open("area_deg2_tot.txt", "w") as f: diff --git a/scripts/glass_mock/compute_leakage_harmony.py b/scripts/glass_mock/compute_leakage_harmony.py index 22c2439d..302f822e 100644 --- a/scripts/glass_mock/compute_leakage_harmony.py +++ b/scripts/glass_mock/compute_leakage_harmony.py @@ -9,7 +9,6 @@ from astropy.io import fits from sp_validation import catalog as spv_cat - from sp_validation.glass_mock import compute_leakage_harmony diff --git a/scripts/merge_psf_cat.py b/scripts/merge_psf_cat.py index 8ca34cf4..5bf089f6 100644 --- a/scripts/merge_psf_cat.py +++ b/scripts/merge_psf_cat.py @@ -143,9 +143,7 @@ def merge_catalogues(self, campaigns): for name in col_names: dat_all[name] = np.append(dat_all[name], dat[name]) - dat_all["campaign"] = np.append( - dat_all["campaign"], [campaign] * len(dat) - ) + dat_all["campaign"] = np.append(dat_all["campaign"], [campaign] * len(dat)) col_names = col_names + ("campaign",) diff --git a/scripts/plot_leakage.py b/scripts/plot_leakage.py index c5b8f956..ea6194aa 100755 --- a/scripts/plot_leakage.py +++ b/scripts/plot_leakage.py @@ -246,9 +246,7 @@ def main(argv=None): # Curve labels come from the input file names: the first file is the # reference (all objects), the others whatever selection they hold. - labels = ["all"] + [ - os.path.splitext(os.path.basename(fn))[0] for fn in fnames[1:] - ] + labels = ["all"] + [os.path.splitext(os.path.basename(fn))[0] for fn in fnames[1:]] # read input files, append data theta = [] diff --git a/src/sp_validation/catalog.py b/src/sp_validation/catalog.py index c06efac4..033c64fd 100644 --- a/src/sp_validation/catalog.py +++ b/src/sp_validation/catalog.py @@ -809,7 +809,9 @@ def find_dataset_group(hdf5_file): while True: keys = list(node) if not keys: - raise ValueError(f"No data found under {node.name!r} in {hdf5_file.file.filename}") + raise ValueError( + f"No data found under {node.name!r} in {hdf5_file.file.filename}" + ) if all(isinstance(node[key], h5py.Dataset) for key in keys): return node if len(keys) != 1: diff --git a/src/sp_validation/catalog_builders.py b/src/sp_validation/catalog_builders.py index 8868595e..c16efb21 100644 --- a/src/sp_validation/catalog_builders.py +++ b/src/sp_validation/catalog_builders.py @@ -543,8 +543,7 @@ def merge_catalogues(self, input_paths): # merged array is allocated once and filled in place, instead of # concatenating per-campaign copies (peak memory 2x the output). shapes = [ - sp_cat.campaign_shape(path, param_list=param_list) - for path in input_paths + sp_cat.campaign_shape(path, param_list=param_list) for path in input_paths ] n_total = sum(n_rows for n_rows, _ in shapes) dtype_out = self.output_dtype([dtype for _, dtype in shapes], n_char_campaign) diff --git a/src/sp_validation/rho_tau.py b/src/sp_validation/rho_tau.py index 16f69565..65b9b863 100644 --- a/src/sp_validation/rho_tau.py +++ b/src/sp_validation/rho_tau.py @@ -342,7 +342,9 @@ def get_jackknife_cov( tau_chunk = outdir + f"/cov_tau_{version}{i}.npy" rho_chunk = outdir + f"/cov_rho_{version}{i}.npy" if not (os.path.exists(tau_chunk) and os.path.exists(rho_chunk)): - print(f"Computing rho-statistics for {version} (jackknife realisation {i + 1}/{ncov})") + print( + f"Computing rho-statistics for {version} (jackknife realisation {i + 1}/{ncov})" + ) if f"psf_{version}{i}" not in rho_stat_handler.catalogs.catalogs_dict: # Build catalogues diff --git a/src/sp_validation/tests/test_campaign_readers.py b/src/sp_validation/tests/test_campaign_readers.py index 99a0f6e7..ac5984ee 100644 --- a/src/sp_validation/tests/test_campaign_readers.py +++ b/src/sp_validation/tests/test_campaign_readers.py @@ -1,6 +1,8 @@ """Tests for the ShapePipe v2 campaign catalogue readers and mask cut.""" +import tempfile import unittest +from pathlib import Path import h5py import numpy as np @@ -9,9 +11,6 @@ from sp_validation import catalog, galaxy from sp_validation.catalog_builders import JointCat -import tempfile -from pathlib import Path - GAL_DTYPE = np.dtype( [ ("RA", "f8"), @@ -114,7 +113,6 @@ def test_missing_column_raises(self): ) self.assertIn("NOT_A_COLUMN", str(ctx.exception)) - def test_row_order_is_tile_name_order(self): """Keys inserted out of order still concatenate in name order.""" path = self._dir / "unordered.hdf5" @@ -164,16 +162,12 @@ def test_column_missing_from_later_tile_raises_clear_error(self): def test_dtype_promoted_across_tiles(self): """A per-tile dtype difference within a campaign is not truncated.""" - narrow = np.zeros( - 2, dtype=[("N_EPOCH", "i2"), ("TILE_ID", "S7"), ("RA", "f4")] - ) + narrow = np.zeros(2, dtype=[("N_EPOCH", "i2"), ("TILE_ID", "S7"), ("RA", "f4")]) narrow["N_EPOCH"] = [1, 2] narrow["TILE_ID"] = [b"123.456", b"123.457"] narrow["RA"] = [1.5, 2.5] - wide = np.zeros( - 2, dtype=[("N_EPOCH", "i4"), ("TILE_ID", "S12"), ("RA", "f8")] - ) + wide = np.zeros(2, dtype=[("N_EPOCH", "i4"), ("TILE_ID", "S12"), ("RA", "f8")]) wide["N_EPOCH"] = [70000, 3] wide["TILE_ID"] = [b"999888.7776", b"123.458"] wide["RA"] = [3.123456789, 4.0] @@ -198,9 +192,7 @@ def test_dtype_promoted_across_tiles(self): # campaign_shape must report the same promoted dtype, since the merge # preallocates from it. - n_rows, dtype_out = catalog.campaign_shape( - str(path), param_list=param_list - ) + n_rows, dtype_out = catalog.campaign_shape(str(path), param_list=param_list) self.assertEqual(n_rows, 4) self.assertEqual(dtype_out, dat.dtype) @@ -210,9 +202,7 @@ def test_iter_campaign_tiles(self): write_campaign(path, "legacy", self._tiles) tiles = list( - catalog.iter_campaign_tiles( - str(path), param_list=["RA"], verbose=False - ) + catalog.iter_campaign_tiles(str(path), param_list=["RA"], verbose=False) ) self.assertEqual([len(tile) for tile in tiles], [3, 2]) @@ -339,21 +329,16 @@ def tearDown(self): self._tmp.cleanup() def test_campaign_name(self): - self.assertEqual( - JointCat.campaign_name("/some/dir/final_cat_W3.hdf5"), "W3" - ) + self.assertEqual(JointCat.campaign_name("/some/dir/final_cat_W3.hdf5"), "W3") def test_merge(self): dat = self._obj.merge_catalogues(self._paths) self.assertEqual(len(dat), 5) self.assertIn("campaign", dat.dtype.names) - npt.assert_array_equal( - dat["campaign"], np.array([b"W3"] * 3 + [b"SGC"] * 2) - ) + npt.assert_array_equal(dat["campaign"], np.array([b"W3"] * 3 + [b"SGC"] * 2)) npt.assert_array_equal(dat["RA"][:3], np.arange(3)) - def test_merge_promotes_column_widths(self): """A wider string/int column in a later file is not truncated.""" dtype_narrow = np.dtype([("RA", "f8"), ("TILE_ID", "S7"), ("N", "i4")]) @@ -373,9 +358,7 @@ def test_merge_promotes_column_widths(self): for paths in ([path_a, path_b], [path_b, path_a]): dat = self._obj.merge_catalogues([str(path) for path in paths]) - by_campaign = { - name: row for name, row in zip(dat["campaign"], dat) - } + by_campaign = {name: row for name, row in zip(dat["campaign"], dat)} self.assertEqual(by_campaign[b"BBBBBBBB"]["TILE_ID"], b"999888.7776") self.assertEqual(by_campaign[b"BBBBBBBB"]["N"], 2**40) self.assertEqual(by_campaign[b"AA"]["TILE_ID"], b"123.456") diff --git a/src/sp_validation/tests/test_survey.py b/src/sp_validation/tests/test_survey.py index c31f75cc..e786c8f8 100644 --- a/src/sp_validation/tests/test_survey.py +++ b/src/sp_validation/tests/test_survey.py @@ -28,7 +28,6 @@ def setUp(self): self._area_amin2 = 3600 self._tile_IDs = (270.283, 188.308) - def tearDown(self): self.number_tile = None @@ -54,4 +53,3 @@ def test_get_area(self): sorted(tile_IDs) == sorted(self._tile_IDs), msg=f"{tile_IDs}!={self._tile_IDs}", ) - From aacaca88956181eda53f0187036c988b5a382665 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Thu, 10 Sep 2026 00:49:47 -0400 Subject: [PATCH 16/39] galaxy: cut never-fit objects on NGMIX_N_EPOCH explicitly ShapePipe's make_cat pre-fills the NGMIX_* columns with sentinels (G1/G2 = -10, T/FLUX = 0) and overwrites them only for objects present in the ngmix output, so an object ngmix never fit keeps NGMIX_MCAL_FLAGS == 0 and passes a flag-only cut. In final_cat_smk-g7.hdf5 that is 18,983 of 1,851,100 objects (1.03%); admitting them drags mean e1 to -0.096 (std 0.98) from +0.0001. classification_galaxy_ngmix already rejected all 18,983 via the NGMIX_G1_PSF_ORIG_NOSHEAR != -10 guard, so the production selection was never affected -- verified on the real file, which gives 1,105,851 rows out with and without the new cut. But that protection was incidental: it is an exact float equality against a sentinel ShapePipe may change, and the coadd N_EPOCH >= 2 cut in classification_galaxy_base does not substitute for it (18,750 of the 18,983 have N_EPOCH >= 1). Cut on NGMIX_N_EPOCH > 0 explicitly so the guarantee is stated, not inferred. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01QbnPCyzuDNTgkg715pHhar --- src/sp_validation/galaxy.py | 11 +++++++++ src/sp_validation/tests/test_galaxy.py | 34 ++++++++++++++++++++++++++ 2 files changed, 45 insertions(+) diff --git a/src/sp_validation/galaxy.py b/src/sp_validation/galaxy.py index df901b3e..5baf6c1a 100644 --- a/src/sp_validation/galaxy.py +++ b/src/sp_validation/galaxy.py @@ -292,8 +292,19 @@ def classification_galaxy_ngmix( Return mask corresponding to ngmix classification of galaxies """ + # NGMIX_N_EPOCH == 0 marks objects ngmix never fit: ShapePipe's make_cat + # pre-fills every NGMIX_* column with sentinels (G1/G2 = -10, T/FLUX = 0) + # and only overwrites them for objects present in the ngmix output, so a + # never-fit object keeps NGMIX_MCAL_FLAGS == 0 and passes a flag-only cut. + # In final_cat_smk-g7.hdf5 this is 18,983 / 1,851,100 objects (1.03%); + # admitting them drags mean e1 to -0.096 (std 0.98) from +0.0001 (std + # 0.24). The coadd N_EPOCH cut in classification_galaxy_base does not + # catch them (18,750 of the 18,983 have N_EPOCH >= 1). Cut on N_EPOCH + # explicitly rather than relying on the -10 sentinel comparison below, + # which is an exact float equality against a value ShapePipe may change. m_gal_ngmix = ( cut_common + & (dd["NGMIX_N_EPOCH"] > 0) & (dd["NGMIX_MCAL_FLAGS"] == 0) & (dd["NGMIX_G1_PSF_ORIG_NOSHEAR"] != -10) & (dd["NGMIX_MCAL_TYPES_FAIL"] == 0) diff --git a/src/sp_validation/tests/test_galaxy.py b/src/sp_validation/tests/test_galaxy.py index 529fd0ad..5ea6a2b7 100644 --- a/src/sp_validation/tests/test_galaxy.py +++ b/src/sp_validation/tests/test_galaxy.py @@ -37,3 +37,37 @@ def test_T_to_fwhm_is_dimensionally_correct(self): npt.assert_allclose(sigma_to_fwhm(1.0), 2.3548200450, rtol=1e-6) # the old linear form would give 4.0 here npt.assert_allclose(T_to_fwhm(2.0), sigma_to_fwhm(1.0)) + + def test_never_fit_objects_are_rejected(self): + """Test that NGMIX_N_EPOCH == 0 objects are cut. + + ShapePipe's make_cat pre-fills the NGMIX_* columns with + sentinels (G1/G2 = -10, T/FLUX = 0) and overwrites them only + for objects present in the ngmix output. An object ngmix never + fit therefore keeps NGMIX_MCAL_FLAGS == 0 and passes a + flag-only selection, carrying e1 = -10 into the shear + statistics. In final_cat_smk-g7.hdf5 that is 18,983 of + 1,851,100 objects (1.03%). + """ + import numpy as np + + from sp_validation.galaxy import classification_galaxy_ngmix + + # row 0: never fit (sentinels, flags clean); row 1: a good fit + dd = { + "NGMIX_N_EPOCH": np.array([0.0, 3.0]), + "NGMIX_MCAL_FLAGS": np.array([0.0, 0.0]), + "NGMIX_G1_PSF_ORIG_NOSHEAR": np.array([-10.0, 0.02]), + "NGMIX_MCAL_TYPES_FAIL": np.array([0.0, 0.0]), + "NGMIX_G1_NOSHEAR": np.array([-10.0, 0.1]), + } + cut_common = np.array([True, True]) + + keep = classification_galaxy_ngmix(dd, cut_common) + npt.assert_array_equal(keep, [False, True]) + + # the N_EPOCH cut must stand on its own: even if the -10 + # sentinel were to change, the never-fit row stays rejected + dd["NGMIX_G1_PSF_ORIG_NOSHEAR"] = np.array([-99.0, 0.02]) + keep = classification_galaxy_ngmix(dd, cut_common) + npt.assert_array_equal(keep, [False, True]) From dc6eef8e7b67180366f752a28ca5fbb470c80bc1 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Thu, 10 Sep 2026 00:55:21 -0400 Subject: [PATCH 17/39] readers: NaN-safe mask cut, star-catalogue provenance and count check mask_cut decided on truthiness via astype(bool). ShapePipe writes the MASK_n* columns as float64 {0, 1} rather than bool (being fixed upstream), and astype(bool) reads NaN as True, so an incomplete mask column would have silently deleted sky. Decide on the value instead (masked iff > 0.5), accept bool, int and float alike, and treat NaN as "no verdict recorded" -- keep the object, but count and warn, since a nonzero count means the product is defective. final_cat_smk-g7.hdf5 carries no NaNs and only exact 0.0/1.0, so this is defensive: the real file gives 1,105,851 rows out before and after. read_star_catalogue silently accepted a truncated file and threw away exposure provenance. It now validates the n_exposures root attribute against the datasets found, as the galaxy reader validates n_tiles (check_n_tiles generalised to check_n_units), and adds an EXPID column carrying the exposure number each star came from. The datasets are named by that number and concatenating them discarded it, leaving no way to group stars by exposure downstream. Names may be bare ("2086324", as smk-g7 writes them) or carry the CFIS suffix ("2110000p"), so EXPID takes the leading digits. On the real star catalogue: 53,264 stars over 127 exposures, matching n_exposures. Also note in group_dtype that ShapePipe writes TILE_ID as f8, so the string-promotion branch is for a future string-valued TILE_ID, with a TODO recording that as an open schema decision. No behaviour change. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01QbnPCyzuDNTgkg715pHhar --- src/sp_validation/catalog.py | 107 ++++++++++++++---- src/sp_validation/galaxy.py | 30 ++++- .../tests/test_campaign_readers.py | 103 ++++++++++++++++- 3 files changed, 219 insertions(+), 21 deletions(-) diff --git a/src/sp_validation/catalog.py b/src/sp_validation/catalog.py index 033c64fd..7adb6a15 100644 --- a/src/sp_validation/catalog.py +++ b/src/sp_validation/catalog.py @@ -13,6 +13,7 @@ import getpass import os +import re import h5py import numpy as np @@ -861,6 +862,16 @@ def group_dtype(group, keys, param_list=None): ``S12`` in another, or ``f4`` next to ``f8``; taking the dtype of the first dataset alone would silently truncate the others. + Note that ShapePipe currently writes ``TILE_ID`` as ``f8`` (the tile + ``183.307`` arrives as the float ``183.307``), not as a string, so the + string-promotion branch above is for a future string-valued ``TILE_ID`` + and is not exercised by today's products. + + TODO: whether ``TILE_ID`` should be a string is an open schema decision. + A float cannot represent the ID exactly and cannot be compared for + equality safely; changing it is a breaking product change, so it is left + as-is here and this reader deliberately handles both. + Parameters ---------- group : h5py.Group @@ -899,7 +910,9 @@ def _check_columns(dtype, param_list, file_path, dataset_key=None): ) -def concatenate_datasets(group, param_list=None, file_path="", verbose=True): +def concatenate_datasets( + group, param_list=None, file_path="", verbose=True, key_column=None +): """Concatenate Datasets. Concatenate every dataset of an HDF5 group into one structured array, @@ -919,12 +932,23 @@ def concatenate_datasets(group, param_list=None, file_path="", verbose=True): input file path, for error messages verbose : bool, optional verbose output if ``True`` + key_column : str, optional + name of an extra integer column to add, filled per row with the + integer-valued name of the dataset the row came from. Concatenating + the group otherwise throws that name away; the star catalogue needs + it to keep exposure provenance. Default is ``None`` (add no column) Returns ------- numpy.ndarray concatenated structured array + Raises + ------ + ValueError + if ``key_column`` is given but a dataset name is not an integer, or + collides with an existing column + """ keys = sorted(group) if not keys: @@ -938,6 +962,25 @@ def concatenate_datasets(group, param_list=None, file_path="", verbose=True): dtype_out = group_dtype(group, keys, param_list=param_list) + if key_column is not None: + if key_column in (dtype_out.names or ()): + raise ValueError( + f"Cannot add column {key_column!r} to catalogue {file_path}:" + + " a column of that name is already present." + ) + # Dataset names are exposure numbers, either bare ("2086324", as the + # smk-g7 products write them) or carrying the CFIS processed-exposure + # suffix ("2110000p"), so take the leading run of digits. + matches = {key: re.match(r"\d+", key) for key in keys} + bad = [key for key, match in matches.items() if match is None] + if bad: + raise ValueError( + f"Cannot derive {key_column!r} for catalogue {file_path}:" + + f" dataset name(s) {bad} do not begin with an integer." + ) + key_values = {key: int(match.group()) for key, match in matches.items()} + dtype_out = np.dtype(dtype_out.descr + [(key_column, "i8")]) + n_rows = sum(group[key].shape[0] for key in keys) if verbose: print( @@ -952,43 +995,54 @@ def concatenate_datasets(group, param_list=None, file_path="", verbose=True): data = group[key][()] end = start + len(data) for name in dtype_out.names: - data_out[name][start:end] = data[name] + if key_column is not None and name == key_column: + data_out[name][start:end] = key_values[key] + else: + data_out[name][start:end] = data[name] start = end del data return data_out -def check_n_tiles(hdf5_file, group, file_path): - """Check Number of Tiles. +def check_n_units(hdf5_file, group, file_path, attr="n_tiles", unit="tile"): + """Check Number of Units. - Compare the number of datasets found against the ``n_tiles`` root - attribute the ShapePipe v2 product carries, to catch a catalogue that - was truncated by an interrupted merge job or file transfer. + Compare the number of datasets found against the count the ShapePipe v2 + product declares in a root attribute, to catch a catalogue that was + truncated by an interrupted merge job or file transfer. Galaxy + catalogues declare ``n_tiles`` and hold one dataset per tile; star + catalogues declare ``n_exposures`` and hold one per exposure. + + A missing attribute is not an error: older products carry none. Parameters ---------- hdf5_file : h5py.File open input file group : h5py.Group - group holding the per-tile datasets + group holding the per-unit datasets file_path : str input file path, for the error message + attr : str, optional + root attribute holding the declared count; default is ``n_tiles`` + unit : str, optional + name of one unit, for the error message; default is ``tile`` Raises ------ ValueError - if the number of datasets differs from the ``n_tiles`` attribute + if the number of datasets differs from the declared count """ - n_tiles = hdf5_file.attrs.get("n_tiles") - if n_tiles is None: + n_declared = hdf5_file.attrs.get(attr) + if n_declared is None: return n_found = len(group) - if int(n_tiles) != n_found: + if int(n_declared) != n_found: raise ValueError( - f"Catalogue {file_path} declares n_tiles = {int(n_tiles)} but holds" - + f" {n_found} tile dataset(s); the file is incomplete." + f"Catalogue {file_path} declares {attr} = {int(n_declared)} but" + + f" holds {n_found} {unit} dataset(s); the file is incomplete." ) @@ -1014,7 +1068,7 @@ def campaign_shape(file_path, param_list=None): """ with h5py.File(file_path, "r") as hdf5_file: group = find_dataset_group(hdf5_file) - check_n_tiles(hdf5_file, group, file_path) + check_n_units(hdf5_file, group, file_path) keys = sorted(group) if not keys: raise ValueError(f"No datasets found in catalogue {file_path}") @@ -1051,7 +1105,7 @@ def iter_campaign_tiles(file_path, param_list=None, verbose=True): """ with h5py.File(file_path, "r") as hdf5_file: group = find_dataset_group(hdf5_file) - check_n_tiles(hdf5_file, group, file_path) + check_n_units(hdf5_file, group, file_path) keys = sorted(group) if not keys: raise ValueError(f"No datasets found in catalogue {file_path}") @@ -1096,7 +1150,7 @@ def read_campaign_catalogue( with h5py.File(file_path, "r") as hdf5_file: group = find_dataset_group(hdf5_file) - check_n_tiles(hdf5_file, group, file_path) + check_n_units(hdf5_file, group, file_path) return concatenate_datasets( group, param_list=param_list, file_path=file_path, verbose=verbose ) @@ -1109,6 +1163,16 @@ def read_star_catalogue(file_path, hdu=1, verbose=True): ``full_starcat_.hdf5`` (one dataset per exposure), or a legacy FITS star catalogue when ``file_path`` ends in ``.fits``. + The HDF5 path adds an ``EXPID`` column carrying the exposure number each + star came from. The datasets are named by that number and concatenating + them would otherwise discard it, leaving no way to group stars by + exposure downstream (e.g. for per-exposure PSF residuals). + + The dataset count is validated against the ``n_exposures`` root + attribute, as the galaxy reader validates ``n_tiles``, so a catalogue + truncated by an interrupted merge is caught on read rather than showing + up as a quietly short star sample. + Parameters ---------- file_path : str @@ -1121,7 +1185,7 @@ def read_star_catalogue(file_path, hdu=1, verbose=True): Returns ------- numpy.ndarray - star catalogue data + star catalogue data, with an added ``EXPID`` column on the HDF5 path """ if str(file_path).endswith(".fits"): @@ -1129,7 +1193,12 @@ def read_star_catalogue(file_path, hdu=1, verbose=True): with h5py.File(file_path, "r") as hdf5_file: group = find_dataset_group(hdf5_file) - return concatenate_datasets(group, file_path=file_path, verbose=verbose) + check_n_units( + hdf5_file, group, file_path, attr="n_exposures", unit="exposure" + ) + return concatenate_datasets( + group, file_path=file_path, verbose=verbose, key_column="EXPID" + ) def get_maked_col(dat, col, mask): diff --git a/src/sp_validation/galaxy.py b/src/sp_validation/galaxy.py index 5baf6c1a..60e170fa 100644 --- a/src/sp_validation/galaxy.py +++ b/src/sp_validation/galaxy.py @@ -11,6 +11,7 @@ """ import re +import warnings import numpy as np import regions @@ -119,8 +120,35 @@ def mask_cut(dd, mask_columns=None): ) masked = np.zeros(len(dd[columns[0]]), dtype=bool) + n_undefined = 0 for col in columns: - masked |= np.asarray(dd[col], dtype=bool) + values = np.asarray(dd[col]) + if values.dtype == bool: + flagged = values + else: + # ShapePipe's writer emits the MASK_n* columns as float64 {0, 1} + # rather than bool (being fixed upstream), so decide on the value + # rather than on truthiness: a bare astype(bool) would silently + # read NaN as True, i.e. masked. Threshold at 0.5 so an integer, + # a float and a bool column all behave identically. + values = values.astype(float) + undefined = np.isnan(values) + n_undefined += int(undefined.sum()) + flagged = np.where(undefined, False, values > 0.5) + masked |= flagged + + if n_undefined: + # NaN means the masking stage recorded no verdict for this object. + # Treat it as un-masked (keep the object) so an incomplete mask + # column cannot silently delete sky, but say so loudly: a nonzero + # count here means the input product is defective. + warnings.warn( + f"{n_undefined} NaN value(s) in mask column(s) {columns};" + + " treated as not masked. The mask columns of a complete" + + " ShapePipe product hold only 0 and 1.", + RuntimeWarning, + stacklevel=2, + ) return ~masked diff --git a/src/sp_validation/tests/test_campaign_readers.py b/src/sp_validation/tests/test_campaign_readers.py index ac5984ee..dfd75bdc 100644 --- a/src/sp_validation/tests/test_campaign_readers.py +++ b/src/sp_validation/tests/test_campaign_readers.py @@ -238,6 +238,7 @@ def setUp(self): self._path = self._dir / "full_starcat_CAMPAIGN.hdf5" with h5py.File(self._path, "w") as f: + f.attrs["n_exposures"] = len(self._exposures) group = f.create_group("exposures") for exp, dat in self._exposures.items(): group.create_dataset(exp, data=dat) @@ -249,10 +250,61 @@ def test_read_hdf5(self): dat = catalog.read_star_catalogue(self._path, verbose=False) self.assertEqual(len(dat), 10) - self.assertEqual(tuple(dat.dtype.names), catalog.STAR_CAT_COLUMNS) + # the reader appends EXPID to the ShapePipe star columns + self.assertEqual( + tuple(dat.dtype.names), catalog.STAR_CAT_COLUMNS + ("EXPID",) + ) npt.assert_array_equal(dat["MAG"][:4], np.zeros(4)) npt.assert_array_equal(dat["MAG"][4:], np.ones(6)) + def test_expid_records_exposure_provenance(self): + """Test that EXPID keeps the exposure each star came from. + + The star catalogue holds one dataset per exposure, named by the + exposure number; concatenating them otherwise throws that number + away, leaving no way to group stars by exposure downstream (for + per-exposure PSF residuals, say). The name may be bare + ("2086324", as the smk-g7 products write it) or carry the CFIS + processed-exposure suffix ("2110000p"). + """ + dat = catalog.read_star_catalogue(self._path, verbose=False) + + self.assertEqual(dat.dtype["EXPID"].kind, "i") + npt.assert_array_equal(dat["EXPID"][:4], np.full(4, 2110000)) + npt.assert_array_equal(dat["EXPID"][4:], np.full(6, 2110001)) + + def test_truncated_star_catalogue_is_rejected(self): + """Test that n_exposures is validated against the datasets found. + + A merge job killed part-way through leaves a readable file with + fewer exposures than it declares. Without this check that shows + up only as a quietly short star sample, never as an error. The + galaxy reader already validates n_tiles this way. + """ + path = self._dir / "truncated.hdf5" + with h5py.File(path, "w") as f: + f.attrs["n_exposures"] = 7 # but only two datasets written + group = f.create_group("exposures") + for exp, dat in self._exposures.items(): + group.create_dataset(exp, data=dat) + + with self.assertRaises(ValueError) as ctx: + catalog.read_star_catalogue(path, verbose=False) + self.assertIn("n_exposures", str(ctx.exception)) + + def test_missing_n_exposures_attr_is_allowed(self): + """Test that a product carrying no n_exposures still reads. + + Older products declare no count; that is not a defect. + """ + path = self._dir / "no_attr.hdf5" + with h5py.File(path, "w") as f: + group = f.create_group("exposures") + for exp, dat in self._exposures.items(): + group.create_dataset(exp, data=dat) + + self.assertEqual(len(catalog.read_star_catalogue(path, verbose=False)), 10) + def test_read_fits(self): from astropy.io import fits @@ -308,6 +360,55 @@ def test_v1_catalogue_raises(self): with self.assertRaises(KeyError): galaxy.mask_cut(dat) + def test_float_and_int_mask_columns(self): + """Test that float and int mask columns cut like bool ones. + + ShapePipe's writer emits the real MASK_n* columns as float64 + {0.0, 1.0} rather than bool (being fixed upstream), so the cut + must decide on the value, not on the dtype. + """ + for dtype in ("f8", "i4"): + dat = np.zeros( + 4, dtype=[(col, dtype) for col in galaxy.DEFAULT_MASK_COLUMNS] + ) + dat["MASK_n4"][0] = 1 + dat["MASK_n1024"][2] = 1 + + npt.assert_array_equal( + galaxy.mask_cut(dat), + np.array([False, True, False, True]), + err_msg=f"mask column dtype {dtype}", + ) + + def test_nan_mask_value_is_not_masked_and_warns(self): + """Test that a NaN mask value keeps the object, loudly. + + A bare astype(bool) reads NaN as True, i.e. masked, so an + incomplete mask column would silently delete sky. NaN means the + masking stage recorded no verdict, so keep the object and warn: + a nonzero count means the input product is defective. The real + smk-g7 catalogue carries no NaNs, so this is defensive. + """ + dat = np.zeros(3, dtype=[(col, "f8") for col in galaxy.DEFAULT_MASK_COLUMNS]) + dat["MASK_n4"][0] = np.nan + dat["MASK_n4"][1] = 1.0 + + with self.assertWarns(RuntimeWarning) as ctx: + keep = galaxy.mask_cut(dat) + + # row 0 NaN -> kept, row 1 masked, row 2 clean -> kept + npt.assert_array_equal(keep, np.array([True, False, True])) + self.assertIn("NaN", str(ctx.warning)) + + def test_no_warning_when_no_nan(self): + dat = np.zeros(2, dtype=[(col, "f8") for col in galaxy.DEFAULT_MASK_COLUMNS]) + + import warnings + + with warnings.catch_warnings(): + warnings.simplefilter("error", RuntimeWarning) + npt.assert_array_equal(galaxy.mask_cut(dat), np.array([True, True])) + class TestCampaignMerge(unittest.TestCase): """Merge of several campaign catalogues into a joint catalogue.""" From da0979146e32a6204ce9ba870f99128d071c3005 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Thu, 10 Sep 2026 04:56:41 +0000 Subject: [PATCH 18/39] ruff autofix (format + safe lint fixes) Pushed by the lint gate. --- src/sp_validation/catalog.py | 4 +--- src/sp_validation/tests/test_campaign_readers.py | 4 +--- 2 files changed, 2 insertions(+), 6 deletions(-) diff --git a/src/sp_validation/catalog.py b/src/sp_validation/catalog.py index 7adb6a15..2a192d41 100644 --- a/src/sp_validation/catalog.py +++ b/src/sp_validation/catalog.py @@ -1193,9 +1193,7 @@ def read_star_catalogue(file_path, hdu=1, verbose=True): with h5py.File(file_path, "r") as hdf5_file: group = find_dataset_group(hdf5_file) - check_n_units( - hdf5_file, group, file_path, attr="n_exposures", unit="exposure" - ) + check_n_units(hdf5_file, group, file_path, attr="n_exposures", unit="exposure") return concatenate_datasets( group, file_path=file_path, verbose=verbose, key_column="EXPID" ) diff --git a/src/sp_validation/tests/test_campaign_readers.py b/src/sp_validation/tests/test_campaign_readers.py index dfd75bdc..01f789f2 100644 --- a/src/sp_validation/tests/test_campaign_readers.py +++ b/src/sp_validation/tests/test_campaign_readers.py @@ -251,9 +251,7 @@ def test_read_hdf5(self): self.assertEqual(len(dat), 10) # the reader appends EXPID to the ShapePipe star columns - self.assertEqual( - tuple(dat.dtype.names), catalog.STAR_CAT_COLUMNS + ("EXPID",) - ) + self.assertEqual(tuple(dat.dtype.names), catalog.STAR_CAT_COLUMNS + ("EXPID",)) npt.assert_array_equal(dat["MAG"][:4], np.zeros(4)) npt.assert_array_equal(dat["MAG"][4:], np.ones(6)) From eab83e57b7bc33e565e39d91c407eb7d14d89eef Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Thu, 10 Sep 2026 00:57:50 -0400 Subject: [PATCH 19/39] config: cut never-fit objects in the v2 mask config too mask_v2.0.yaml, applied downstream to the comprehensive catalogue, cut on neither NGMIX_MCAL_FLAGS nor any epoch column. It rejected the never-fit objects only through its NGMIX_G1/G2_PSF_ORIG_NOSHEAR != -10 cuts, and only because make_cat happens to fill the PSF columns from the same -10 literal it uses for the galaxy ellipticities. That is the same accidental immunity just removed from classification_galaxy_ngmix, one stage further downstream. Add NGMIX_N_EPOCH >= 1. The column is already carried into the comprehensive catalogue via add_cols_pre_cal in params.py. On final_cat_smk-g7.hdf5 the cut keeps 1,832,117 of 1,851,100 objects, removing exactly the 18,983 (1.03%) never-fit rows. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01QbnPCyzuDNTgkg715pHhar --- config/calibration/mask_v2.0.yaml | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/config/calibration/mask_v2.0.yaml b/config/calibration/mask_v2.0.yaml index 37b0f5c7..83aa72b1 100644 --- a/config/calibration/mask_v2.0.yaml +++ b/config/calibration/mask_v2.0.yaml @@ -113,6 +113,22 @@ dat: kind: equal value: 0 + # Objects ngmix never fit. ShapePipe's make_cat pre-fills the NGMIX_* + # columns with sentinels (G1/G2 = -10, T/FLUX = 0) and overwrites them only + # for objects present in the ngmix output, so a never-fit object also keeps + # NGMIX_MCAL_FLAGS = 0 and is not caught by any flag cut. In + # final_cat_smk-g7.hdf5 that is 18,983 of 1,851,100 objects (1.03%); + # admitting them gives mean e1 = -0.107 (std 1.03) against -0.004 (std + # 0.22). The -10 cuts below do reject them today, but only because make_cat + # happens to use the same literal for the PSF columns: that is an exact + # float equality against a sentinel ShapePipe may change, so state the + # condition directly. Mirrors the guard in + # sp_validation.galaxy.classification_galaxy_ngmix. + - col_name: NGMIX_N_EPOCH + label: "ngmix never fit" + kind: greater_equal + value: 1 + # invalid PSF ellipticities - col_name: NGMIX_G1_PSF_ORIG_NOSHEAR label: "bad PSF ellipticity comp 1" From 8d2e15daa5ac1c94af12ebd7b4089f2521d3c989 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Thu, 17 Sep 2026 12:59:03 +0200 Subject: [PATCH 20/39] mask_v2.0: FLAGS <= 2, matching the image-sims selection Co-Authored-By: Claude Fable 5.1 --- config/calibration/mask_v2.0.yaml | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/config/calibration/mask_v2.0.yaml b/config/calibration/mask_v2.0.yaml index 83aa72b1..d9800e81 100644 --- a/config/calibration/mask_v2.0.yaml +++ b/config/calibration/mask_v2.0.yaml @@ -35,11 +35,13 @@ params: # Masks ## Using columns in 'dat' group (ShapePipe flags) dat: - # SExtractor flags + # SExtractor flags: keep 0 (clean), 1 (neighbours), 2 (deblended); drop + # 3 (both) and above. Same as the image-sims config, so data and sims + # select identically (shapepipe#885). - col_name: FLAGS label: SE FLAGS kind: smaller_equal - value: 3 + value: 2 # Duplicate objects - col_name: overlap From 12178183960655d47622e5ce8afe0da641d0e030 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Fri, 25 Sep 2026 00:15:47 +0200 Subject: [PATCH 21/39] combine_results: drop unused n_campaign Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01VXzqmMYw7Kp8QMtoiVHjVq --- scripts/combine_results.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/scripts/combine_results.py b/scripts/combine_results.py index 5b679b98..8244c41c 100755 --- a/scripts/combine_results.py +++ b/scripts/combine_results.py @@ -783,8 +783,6 @@ def main(argv=None): else: campaigns = argv[1].split("+") - n_campaign = len(campaigns) - print("combine_results.py:", campaigns) directory = "sp_output/plots" From 1ea6ef5d675427425ac90afd5ca02fbafcac0c78 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 04:03:45 +0200 Subject: [PATCH 22/39] grammar: ShapePipe v1->v2 column-grammar adapter, wired into rho/tau sp_validation reads only the v2 column grammar, while every catalogue on disk (v1.3.x-v1.6.x) is v1, so e.g. rho/tau raised KeyError on the v1.4.a PSF file. New module sp_validation.grammar is the one place that knows the difference: - V1_RULES: the v1->v2 map of the shape-measurement columns as data (HSM renames, SIGMA -> T = 2 sigma^2 via cs_util.size.sigma_to_T, 2-vector or flattened NGMIX_ELL* split into G1/G2, PSFo/Tpsf/MOM_FAIL renames), applied to tables detect_generation calls v1. Mapping is by name, so v1 values reach the names the code reads. - MASK_RULES: the healsparse mask bits {b}_{label} (the names ApplyHspMasks gave them in the comprehensive HDF5's data_ext) -> MASK_n{b}, applied whatever the generation: they are the same bits of the same UNIONS bitmask ShapePipe v2 writes as MASK_n{b} (MASK_LABELS; 512 = outside the tile's unique region). IMAFLAGS_ISO is not mapped: its v1 bits mean different things. - adapt(table, *tables): tables with nothing to rename come back unchanged; otherwise a V2View, a lazy column view over numpy, FITS_rec or h5py that joins row-aligned tables (data + data_ext), computes derived columns on access, and composes row selections as ranges or selected indices, reading only the window of rows they span. - read_catalogue(path, hdu), materialise, v2_names, read_column_names. rho_tau.get_rho_tau / get_jackknife_cov / get_theory_cov read each catalogue once through read_catalogue and hand the table to shear_psf_leakage (needs its loaded-catalogue support). Tests: v1 twins adapt to their v2 twins for numpy (vector and flattened ELL), FITS_rec and h5py; row selection commutes; dtype matches every column read; data_ext mask names rename with or without a generation and conflict with their new names; h5py selections read only their window; the psf_size_error field and full get_rho_tau outputs agree between a v1 PSF catalogue and its v2 twin (and a rename-only sigma is caught). Co-Authored-By: Claude Opus 5.5 --- CLAUDE.md | 1 + docs/ngmix_psf_column_migration.md | 34 +- src/sp_validation/grammar.py | 434 ++++++++++++++++++++++++ src/sp_validation/rho_tau.py | 70 ++-- src/sp_validation/tests/test_grammar.py | 410 ++++++++++++++++++++++ 5 files changed, 897 insertions(+), 52 deletions(-) create mode 100644 src/sp_validation/grammar.py create mode 100644 src/sp_validation/tests/test_grammar.py diff --git a/CLAUDE.md b/CLAUDE.md index 606ed340..db1d1f02 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -43,6 +43,7 @@ is the container (full scientific stack pre-built). For a local dev environment: - `cosmo_val.py`: Cosmology validation routines - `cosmology.py`: Cosmological calculations and theory - `galaxy.py`: Galaxy-specific processing +- `grammar.py`: ShapePipe v1 -> v2 column-grammar adapter (`adapt`, `read_catalogue`) - `io.py`: Input/output utilities - `plots.py`: Plotting functions - `rho_tau.py`: Rho and tau statistics calculations diff --git a/docs/ngmix_psf_column_migration.md b/docs/ngmix_psf_column_migration.md index b24e9f83..71838372 100644 --- a/docs/ngmix_psf_column_migration.md +++ b/docs/ngmix_psf_column_migration.md @@ -1,23 +1,21 @@ # ShapePipe-v2 column-grammar migration (shapepipe → sp_validation) -**Status:** code-complete on `migrate/ngmix-psf-column-names` (draft PR -[#201](https://github.com/CosmoStat/sp_validation/pull/201)). Every shape-column -read in the live package, configs, calibration scripts, paper figures, and -notebooks uses the ShapePipe-v2 grammar; the σ→T units change and the -`spread_model` removal are in; the dead `galsim` estimator path is removed; and -the suite is green against synthetic catalogues carrying the new columns. - -**No code work is left waiting on a regenerated catalogue.** The one *value* -change — `*_PSF_ORIG` now holds a true original-PSF fit (shapepipe#749) rather -than the reconvolved-kernel alias the old columns silently held — is a straight -column rename at the code level and is already in place; the code does not care -that the numbers moved. All a v2 catalogue enables is a *look-at-the-numbers* -sanity check (do the α-leakage / size-ratio cuts still behave), which is analysis, -not code, and does not gate the PR. **The real merge gate is cutover timing:** -merging this branch makes `develop` *require* v2 columns and stop reading today's -catalogues, so #201 should land together with — or just after — -[shapepipe#761](https://github.com/CosmoStat/shapepipe/pull/761)→#741 and the -first v2 catalogue. +sp_validation reads the ShapePipe-v2 column grammar everywhere: live package, +configs, calibration scripts, paper figures and notebooks. Catalogues written in +the v1 grammar (every release up to v1.6.x) are read through +`sp_validation.grammar` (wired into the ρ/τ path, `rho_tau.py`, and the GLASS +leakage script): `adapt(table)` (or `read_catalogue(path, hdu)` for a +FITS file) presents a v1 table under the v2 names and units below, as a lazy +column view over a numpy array, FITS_rec or h5py dataset, and returns v2 or +grammar-neutral tables unchanged. `detect_generation` tells the grammars apart +from column names, and `v2_names` maps a header's names without reading data. +The adapter maps by *name*, so a v1 `*_PSFo` column is presented as +`*_PSF_ORIG` holding the v1 (reconvolved-alias) values; downstream code never +branches on the generation. The tables below (Old = v1, New = v2) are the rule +table in `grammar.RULES`. The adapter also presents the comprehensive HDF5's +`data_ext` mask flags (`1_Faint_star_halos`, …, `2048_z2`) as `MASK_n{b}`, and +passes `IMAFLAGS_ISO` through unmapped: its v1 bits (2 halo, 4 border, +16 Messier, 32 NGC, 128 spike) are not the v2 `MASK_n{b}` of the same value. shapepipe#761 turns the shape-measurement output into **one column grammar for the whole catalogue**: every estimator names its outputs diff --git a/src/sp_validation/grammar.py b/src/sp_validation/grammar.py new file mode 100644 index 00000000..9cc04582 --- /dev/null +++ b/src/sp_validation/grammar.py @@ -0,0 +1,434 @@ +"""ShapePipe column grammars. + +ShapePipe products name their columns in one of two grammars. ``v2`` is the +grammar the rest of sp_validation reads (``HSM_T_PSF``, ``NGMIX_G1_NOSHEAR``, +``MASK_n4``, ...; see ``docs/ngmix_psf_column_migration.md``). ``v1`` is the +grammar of the catalogues released up to and including v1.6.x (``SIGMA_PSF_HSM``, +the 2-vector ``NGMIX_ELL_NOSHEAR``, ...). + +This module is the single place that knows the difference. ``adapt`` hands a +table back as a lazy view presenting v2 names and units, so downstream code +reads one grammar and never branches on the generation. Two rule families +apply: + +- The v1 shape-measurement columns, applied when the table is v1 (see + ``detect_generation``): + + - ``E{1,2}_{PSF,STAR}_HSM`` and ``FLAG_{PSF,STAR}_HSM`` are renamed; + - ``SIGMA_{PSF,STAR}_HSM`` (sigma) is presented as ``HSM_T_{PSF,STAR}`` + holding ``T = 2 sigma^2``, via ``cs_util.size.sigma_to_T``; + - the 2-vector ngmix columns ``NGMIX_ELL[_ERR|_PSFo]_{shear}`` (or their + flattened ``_0``/``_1`` forms, as in the comprehensive HDF5) are split + into ``NGMIX_G{1,2}[_ERR|_PSF_ORIG]_{shear}``; + - ``NGMIX_T_PSFo_{shear}``, ``NGMIX_Tpsf_{shear}`` and ``NGMIX_MOM_FAIL`` + become ``NGMIX_T_PSF_ORIG_{shear}``, ``NGMIX_T_PSF_RECONV_{shear}`` and + ``NGMIX_MCAL_TYPES_FAIL``. + +- The UNIONS mask-bit columns, applied to any table: ``{b}_{label}`` + (``1_Faint_star_halos``, ``4_Stars``, ``2048_z2``, ...), the names under + which ``catalog_builders.ApplyHspMasks`` wrote the healsparse mask bits into + the ``data_ext`` dataset of the comprehensive HDF5, become ``MASK_n{b}``, + the name ShapePipe v2 and ``ApplyHspMasks`` give the same bit of the same + bitmask (``MASK_LABELS``). These columns belong to the mask product, not to + a ShapePipe generation, so they neither mark nor require one. + +``IMAFLAGS_ISO`` is not mapped: its v1 bits (2 halo, 4 border, 16 Messier, +32 NGC, 128 spike) differ in meaning from the ``MASK_n{b}`` of the same value, +so it passes through under its own name for the v1 mask configs that cut on it. + +The mapping is by *name*, deliberately, including where the v1 values mean +something different from their v2 namesakes: v1 ``NGMIX_*_PSFo_*`` hold the +reconvolved-PSF alias rather than a fit to the original PSF, and +``NGMIX_MOM_FAIL`` counts moments-guess failures rather than failed metacal +types. Reproducing a v1 calibration needs exactly the v1 values under the names +the code reads, so the view presents them unchanged. Every other column passes +through under its own name; source names that have a v2 equivalent are hidden. + +``adapt`` also joins row-aligned tables into one view, as the comprehensive +HDF5 splits one catalogue over its ``data`` and ``data_ext`` datasets. +""" + +import os +from dataclasses import dataclass + +import h5py +import numpy as np +from astropy.io import fits +from cs_util.size import sigma_to_T + +#: Metacal shear types carried by the ngmix columns. +SHEARS = ("NOSHEAR", "1P", "1M", "2P", "2M") + +#: Bits of the UNIONS healsparse mask product and what each flags. The column +#: for bit ``b`` is ``MASK_n{b}``; files written before that name carry +#: ``{b}_{label}``. +MASK_LABELS = { + 1: "Faint_star_halos", + 2: "Bright_star_halos", + 4: "Stars", + 8: "Manual", + 16: "u", + 32: "g", + 64: "r", + 128: "i", + 256: "z", + 512: "Tile_RA_DEC_cut", + 1024: "Maximask", + 2048: "z2", +} + + +def mask_column(bit): + """Return the catalogue column holding mask bit ``bit``: ``MASK_n{bit}``. + + Raises + ------ + KeyError + if ``bit`` is not a bit of the mask product + """ + if bit not in MASK_LABELS: + raise KeyError(f"{bit} is not a UNIONS mask bit; bits: {list(MASK_LABELS)}") + return f"MASK_n{bit}" + + +@dataclass(frozen=True) +class Rule: + """One v2 column presented from a source column. + + ``kind`` is ``"rename"`` (same values), ``"sigma_to_T"`` (T = 2 sigma^2) + or ``"component"`` (element ``arg`` of a 2-vector, stored either as one + vector column ``source`` or as the flattened ``{source}_{arg}``). + """ + + v2: str + source: str + kind: str = "rename" + arg: int | None = None + + def sources(self): + """Return the source names this rule can read from.""" + if self.kind == "component": + return (self.source, f"{self.source}_{self.arg}") + return (self.source,) + + def apply(self, column): + """Return the v2 column computed from source ``column``.""" + if self.kind == "rename": + return column + column = np.asarray(column) + if self.kind == "sigma_to_T": + return sigma_to_T(column) + return column[:, self.arg] if column.ndim == 2 else column + + +def _v1_rules(): + rules = [] + for obj in ("PSF", "STAR"): + rules += [ + Rule(f"HSM_G1_{obj}", f"E1_{obj}_HSM"), + Rule(f"HSM_G2_{obj}", f"E2_{obj}_HSM"), + Rule(f"HSM_T_{obj}", f"SIGMA_{obj}_HSM", "sigma_to_T"), + Rule(f"HSM_FLAG_{obj}", f"FLAG_{obj}_HSM"), + ] + for shear in SHEARS: + for i in (0, 1): + g = f"G{i + 1}" + rules += [ + Rule(f"NGMIX_{g}_{shear}", f"NGMIX_ELL_{shear}", "component", i), + Rule( + f"NGMIX_{g}_ERR_{shear}", f"NGMIX_ELL_ERR_{shear}", "component", i + ), + Rule( + f"NGMIX_{g}_PSF_ORIG_{shear}", + f"NGMIX_ELL_PSFo_{shear}", + "component", + i, + ), + ] + rules += [ + Rule(f"NGMIX_T_PSF_ORIG_{shear}", f"NGMIX_T_PSFo_{shear}"), + Rule(f"NGMIX_T_PSF_RECONV_{shear}", f"NGMIX_Tpsf_{shear}"), + ] + rules.append(Rule("NGMIX_MCAL_TYPES_FAIL", "NGMIX_MOM_FAIL")) + return tuple(rules) + + +#: The v1 -> v2 map of the shape-measurement columns, applied to v1 tables. +V1_RULES = _v1_rules() + +#: ``{b}_{label}`` -> ``MASK_n{b}``, applied to every table. +MASK_RULES = tuple( + Rule(mask_column(bit), f"{bit}_{label}") for bit, label in MASK_LABELS.items() +) + +#: Names whose presence marks a table as v1 / as v2. +V1_MARKERS = frozenset(name for rule in V1_RULES for name in rule.sources()) +V2_MARKERS = frozenset(rule.v2 for rule in V1_RULES) + + +def column_names(table): + """Return the column names of a structured array, FITS_rec or h5py Dataset.""" + names = getattr(getattr(table, "dtype", None), "names", None) + if names is None: + raise TypeError(f"cannot list the columns of a {type(table).__name__}") + return tuple(names) + + +def detect_generation(names): + """Return ``"v1"``, ``"v2"`` or ``None`` from the column names present. + + ``None`` means grammar-neutral: no shape-measurement column either grammar + names (e.g. a cut catalogue carrying only ``RA``, ``Dec``, ``e1``, ``e2``, + ``w``, or a table of mask columns). Mask columns mark no generation. + + Raises + ------ + ValueError + if columns of both grammars are present. + """ + names = set(names) + v1 = sorted(names & V1_MARKERS) + v2 = sorted(names & V2_MARKERS) + if v1 and v2: + raise ValueError( + "catalogue mixes ShapePipe column grammars:" + + f" v1 columns {v1[:5]} alongside v2 columns {v2[:5]}" + ) + if v1: + return "v1" + if v2: + return "v2" + return None + + +def _resolve(names): + """Return (presented names, {v2 name: (rule, source name)}) for ``names``.""" + rules = MASK_RULES + if detect_generation(names) == "v1": + rules = V1_RULES + MASK_RULES + present = set(names) + derived = {} + by_source = {} + for rule in rules: + source = next((s for s in rule.sources() if s in present), None) + if source is None: + continue + if rule.v2 in present: + raise ValueError( + f"catalogue carries both {source!r} and {rule.v2!r}," + + " two names for the same column" + ) + derived[rule.v2] = (rule, source) + by_source.setdefault(source, []).append(rule.v2) + + presented = [] + for name in names: + presented += by_source.get(name, [name]) + return tuple(presented), derived + + +def v2_names(names): + """Return the column names a table with columns ``names`` presents. + + Header-only counterpart of ``adapt``. + """ + return _resolve(tuple(names))[0] + + +def _read_window(table, name, start, stop): + """Read rows ``start:stop`` of column ``name``, and only those.""" + if isinstance(table, h5py.Dataset): + return table.fields(name)[start:stop] + return table[name][start:stop] + + +class V2View: + """Lazy v2-grammar view of one table, or of row-aligned tables joined. + + Wraps numpy structured arrays, FITS_recs or h5py Datasets without reading + them. ``view[name]`` returns one column as an array (computed on access for + derived columns); a list of names returns a structured array of those + columns; any other key (slice, boolean mask, index array) selects rows and + returns a view over them. Only the requested columns are ever read, and + only the window of rows spanning the selection, so an HDF5 dataset larger + than memory can be wrapped whole. + + Column names are case-sensitive, unlike a FITS_rec's. + """ + + def __init__(self, bases, rows=None): + self._bases = tuple(bases) + n_rows = {len(base) for base in self._bases} + if len(n_rows) != 1: + raise ValueError(f"cannot join tables of lengths {sorted(n_rows)}") + self._n_base = n_rows.pop() + self._owner = {} + for base in self._bases: + for name in column_names(base): + if name in self._owner: + raise ValueError(f"column {name!r} is in more than one table") + self._owner[name] = base + # None (all rows), a ``range`` or an integer index array. + self._rows = rows + self._names, self._derived = _resolve(tuple(self._owner)) + + @property + def names(self): + """Presented column names.""" + return self._names + + def keys(self): + """Presented column names.""" + return self._names + + @property + def dtype(self): + """Numpy dtype of the presented columns (computed without reading).""" + return np.dtype([(name, self._field_dtype(name)) for name in self._names]) + + @property + def shape(self): + return (len(self),) + + def __len__(self): + return self._n_base if self._rows is None else len(self._rows) + + def __contains__(self, name): + return name in self._names + + def __repr__(self): + return f"V2View({len(self)} rows, {len(self._names)} columns)" + + def __getitem__(self, key): + if isinstance(key, str): + return self._column(key) + if isinstance(key, list) and key and all(isinstance(k, str) for k in key): + return self.to_structured(key) + if isinstance(key, (int, np.integer)): + return self[np.array([key])].to_structured()[0] + return V2View(self._bases, rows=self._select(key)) + + def __array__(self, dtype=None, copy=None): + out = self.to_structured() + return out if dtype is None else out.astype(dtype) + + def to_structured(self, names=None): + """Materialise the presented columns as a numpy structured array.""" + names = self._names if names is None else list(names) + missing = [name for name in names if name not in self._names] + if missing: + raise KeyError(f"columns {missing} not in catalogue") + dtype = np.dtype([(name, self._field_dtype(name)) for name in names]) + out = np.empty(len(self), dtype=dtype) + for name in names: + out[name] = self._column(name) + return out + + def _select(self, key): + """Base rows of ``key`` applied to this view's rows.""" + n = len(self) + if isinstance(key, slice): + local = range(n)[key] + else: + key = np.asarray(key) + if key.dtype == bool: + if key.shape != (n,): + raise IndexError(f"boolean mask of shape {key.shape} for {n} rows") + local = np.flatnonzero(key) + elif key.size == 0: + local = np.empty(0, dtype=np.intp) + elif key.dtype.kind in "iu" and key.ndim == 1: + local = np.where(key < 0, key + n, key).astype(np.intp) + if local.min() < 0 or local.max() >= n: + raise IndexError(f"row index out of range for {n} rows") + else: + raise IndexError(f"cannot select rows with {key!r}") + + rows = self._rows + if rows is None: + return local + if isinstance(rows, range): + if isinstance(local, range): + return range( + rows.start + local.start * rows.step, + rows.start + local.stop * rows.step, + rows.step * local.step, + ) + return rows.start + local * rows.step + if isinstance(local, range): + local = np.arange(local.start, local.stop, local.step, dtype=np.intp) + return rows[local] + + def _read(self, name): + base = self._owner[name] + rows = self._rows + if rows is None: + return base[name] + if len(rows) == 0: + return _read_window(base, name, 0, 0) + if isinstance(rows, range): + lo, hi = min(rows[0], rows[-1]), max(rows[0], rows[-1]) + 1 + window = _read_window(base, name, lo, hi) + return window[rows.start - lo :: rows.step][: len(rows)] + lo, hi = int(rows.min()), int(rows.max()) + 1 + return _read_window(base, name, lo, hi)[rows - lo] + + def _column(self, name): + if name not in self._names: + raise KeyError(f"column {name!r} not in catalogue") + if name not in self._derived: + return self._read(name) + rule, source = self._derived[name] + return rule.apply(self._read(source)) + + def _field_dtype(self, name): + """Dtype (with any subarray shape) ``_column(name)`` returns.""" + if name in self._derived: + rule, source = self._derived[name] + column = rule.apply(_read_window(self._owner[source], source, 0, 0)) + else: + column = _read_window(self._owner[name], name, 0, 0) + return np.dtype((column.dtype, column.shape[1:])) + + +def adapt(table, *tables): + """Present ``table`` (joined with any further ``tables``) in the v2 grammar. + + A single table with nothing to rename is returned unchanged; otherwise the + result is a ``V2View``. Each table is anything whose ``dtype.names`` lists + its columns and whose ``table[name]`` reads one: a numpy structured array, + FITS_rec or h5py Dataset. Joined tables must have equal lengths and + disjoint column names. + """ + if not tables and not _resolve(column_names(table))[1]: + return table + return V2View((table,) + tables) + + +def materialise(table, names=None): + """Return ``adapt(table)`` as an in-memory numpy structured array. + + ``names`` restricts the output to those v2-grammar columns. + """ + view = adapt(table) + if isinstance(view, V2View): + return view.to_structured(names) + view = view[()] if isinstance(view, h5py.Dataset) else view + if names is None: + return view + out = np.empty(len(view), dtype=[(name, view.dtype[name]) for name in names]) + for name in names: + out[name] = view[name] + return out + + +def read_catalogue(path, hdu=1): + """Read a FITS catalogue HDU and present it in the v2 grammar.""" + return adapt(fits.getdata(os.fspath(path), ext=hdu)) + + +def read_column_names(path, hdu=1): + """Return the v2-grammar column names of a FITS HDU, reading only its header.""" + header = fits.getheader(os.fspath(path), ext=hdu) + names = [header[f"TTYPE{i}"] for i in range(1, header["TFIELDS"] + 1)] + return v2_names(names) diff --git a/src/sp_validation/rho_tau.py b/src/sp_validation/rho_tau.py index 65b9b863..b40ed9b1 100644 --- a/src/sp_validation/rho_tau.py +++ b/src/sp_validation/rho_tau.py @@ -6,7 +6,9 @@ from shear_psf_leakage.rho_tau_cov import CovTauTh from shear_psf_leakage.rho_tau_stat import RhoStat, TauStat -# SquareRootScale now lives in sp_validation.plots; re-exported here so that +from sp_validation.grammar import read_catalogue + +# SquareRootScale lives in sp_validation.plots; re-exported here so that # `from sp_validation.rho_tau import SquareRootScale` keeps working. from sp_validation.plots import SquareRootScale # noqa: F401 @@ -16,6 +18,27 @@ def _extract_xip(correlations): return np.array([corr.xip for corr in correlations]).flatten() +class _CatalogueLoader: + """Read a version's PSF and shear catalogues on first use, once each. + + Both are read through ``sp_validation.grammar.read_catalogue``, so a + ShapePipe v1 catalogue presents the v2 column names the configs declare. + """ + + def __init__(self, info): + self._info = info + self._cache = {} + + def __call__(self, block): + if block not in self._cache: + entry = self._info[block] + hdu = entry.get("hdu") + self._cache[block] = read_catalogue( + entry["path"], hdu=1 if hdu is None else hdu + ) + return self._cache[block] + + def get_params_rho_tau(cat, survey="other"): """Rho/tau parameters for one catalogue-config entry ``cat``. @@ -148,6 +171,8 @@ def get_rho_tau( output=outdir, treecorr_config=treecorr_config, verbose=True ) + load = _CatalogueLoader(config[version]) + rho_stats_exists = rho_path.exists() cov_exists = True if not cov_rho else cov_rho_path.exists() need_compute = (not rho_stats_exists) or (not cov_exists) @@ -158,14 +183,7 @@ def get_rho_tau( mask = version != "DES" rho_stat_handler.build_cat_to_compute_rho( - config[version]["psf"]["path"], - catalog_id=catalog_id, - mask=mask, - hdu=( - config[version]["psf"]["hdu"] - if config[version]["psf"]["hdu"] is not None - else 1 - ), + load("psf"), catalog_id=catalog_id, mask=mask ) rho_stat_handler.compute_rho_stats( @@ -200,23 +218,12 @@ def get_rho_tau( # Build the different catalogs if necessary if f"psf_{version}" not in tau_stat_handler.catalogs.catalogs_dict.keys(): tau_stat_handler.build_cat_to_compute_tau( - config[version]["psf"]["path"], - cat_type="psf", - catalog_id=version, - mask=mask, - hdu=( - config[version]["psf"]["hdu"] - if config[version]["psf"]["hdu"] is not None - else 1 - ), + load("psf"), cat_type="psf", catalog_id=version, mask=mask ) # Build the catalog of galaxies. PSF was computed above tau_stat_handler.build_cat_to_compute_tau( - config[version]["shear"]["path"], - cat_type="gal", - catalog_id=version, - mask=mask, + load("shear"), cat_type="gal", catalog_id=version, mask=mask ) # function to extract the tau_+ @@ -246,10 +253,6 @@ def get_theory_cov( n_e = info["cov_th"]["n_e"] n_psf = info["cov_th"]["n_psf"] - path_gal = info["shear"]["path"] - path_psf = info["psf"]["path"] - hdu_psf = info["psf"]["hdu"] - target_cov = Path(outdir) / f"cov_tau_{base}_th.npy" if target_cov.exists(): @@ -259,10 +262,11 @@ def get_theory_cov( print("Computing the covariance matrix for the version: ", version) start_time = time.time() + load = _CatalogueLoader(info) cov_tau_th = CovTauTh( - path_gal=path_gal, - path_psf=path_psf, - hdu_psf=hdu_psf, + path_gal=load("shear"), + path_psf=load("psf"), + hdu_psf=None, treecorr_config=treecorr_config, A=A, n_e=n_e, @@ -338,6 +342,7 @@ def get_jackknife_cov( tau_stat_handler.catalogs.set_params(params, outdir) + load = _CatalogueLoader(config[version]) for i in range(ncov): tau_chunk = outdir + f"/cov_tau_{version}{i}.npy" rho_chunk = outdir + f"/cov_rho_{version}{i}.npy" @@ -349,10 +354,7 @@ def get_jackknife_cov( if f"psf_{version}{i}" not in rho_stat_handler.catalogs.catalogs_dict: # Build catalogues rho_stat_handler.build_cat_to_compute_rho( - config[version]["psf"]["path"], - catalog_id=version + str(i), - mask=False, - hdu=config[version]["psf"]["hdu"], + load("psf"), catalog_id=version + str(i), mask=False ) tau_stat_handler.catalogs.catalogs_dict = ( @@ -361,7 +363,7 @@ def get_jackknife_cov( # Build the catalog of galaxies. PSF was computed above tau_stat_handler.build_cat_to_compute_tau( - config[version]["shear"]["path"], + load("shear"), cat_type="gal", catalog_id=version + str(i), mask=False, diff --git a/src/sp_validation/tests/test_grammar.py b/src/sp_validation/tests/test_grammar.py new file mode 100644 index 00000000..716f7df4 --- /dev/null +++ b/src/sp_validation/tests/test_grammar.py @@ -0,0 +1,410 @@ +"""The ShapePipe v1 -> v2 grammar adapter presents v1 products as their v2 twins. + +One synthetic catalogue is written in both grammars: v2 (T, split G1/G2, +boolean MASK_n*) and v1 (sigma, 2-vector or flattened ELL, {b}_{label} mask +flags, plus the IMAFLAGS_ISO bitmask the adapter leaves alone). +Adapting the v1 form must reproduce the v2 form column for column, for every +table flavour the pipeline reads (numpy, FITS_rec, h5py), and row selection +must commute with adaptation. +""" + +import h5py +import numpy as np +import pytest +from astropy.io import fits +from cs_util.size import T_to_sigma +from numpy.lib import recfunctions as rfn + +from sp_validation import grammar +from sp_validation.grammar import ( + MASK_LABELS, + SHEARS, + V2View, + adapt, + detect_generation, + read_catalogue, + read_column_names, + v2_names, +) + +N = 64 + +#: cat_config ``psf:`` block naming the v2 HSM columns. +PSF_BLOCK = { + "ra_col": "RA", + "dec_col": "DEC", + "e1_PSF_col": "HSM_G1_PSF", + "e2_PSF_col": "HSM_G2_PSF", + "e1_star_col": "HSM_G1_STAR", + "e2_star_col": "HSM_G2_STAR", + "PSF_size": "HSM_T_PSF", + "star_size": "HSM_T_STAR", + "PSF_flag": "HSM_FLAG_PSF", + "star_flag": "HSM_FLAG_STAR", +} +SHEAR_BLOCK = {"w_col": "w", "e1_col": "e1", "e2_col": "e2"} + + +def _twins(vector_ell=True): + """Return (v1, v2) structured arrays holding the same content.""" + rng = np.random.default_rng(42) + v2 = { + "RA": rng.uniform(0, 360, N), + "DEC": rng.uniform(-30, 30, N), + "MAG": rng.uniform(18, 24, N), + } + v1 = dict(v2) + for obj in ("PSF", "STAR"): + g1, g2 = rng.normal(0, 0.05, (2, N)) + T = rng.uniform(0.4, 0.9, N) + flag = rng.integers(0, 3, N).astype(np.int16) + v2 |= { + f"HSM_G1_{obj}": g1, + f"HSM_G2_{obj}": g2, + f"HSM_T_{obj}": T, + f"HSM_FLAG_{obj}": flag, + } + v1 |= { + f"E1_{obj}_HSM": g1, + f"E2_{obj}_HSM": g2, + f"SIGMA_{obj}_HSM": T_to_sigma(T), + f"FLAG_{obj}_HSM": flag, + } + for shear in SHEARS: + for v1_tag, v2_tag in (("", ""), ("_ERR", "_ERR"), ("_PSFo", "_PSF_ORIG")): + g = rng.normal(0, 0.2, (N, 2)) + v2[f"NGMIX_G1{v2_tag}_{shear}"] = g[:, 0] + v2[f"NGMIX_G2{v2_tag}_{shear}"] = g[:, 1] + name = f"NGMIX_ELL{v1_tag}_{shear}" + if vector_ell: + v1[name] = g + else: + v1[f"{name}_0"], v1[f"{name}_1"] = g[:, 0], g[:, 1] + T_orig, T_reconv = rng.uniform(0.3, 0.8, (2, N)) + v2[f"NGMIX_T_PSF_ORIG_{shear}"] = v1[f"NGMIX_T_PSFo_{shear}"] = T_orig + v2[f"NGMIX_T_PSF_RECONV_{shear}"] = v1[f"NGMIX_Tpsf_{shear}"] = T_reconv + fail = rng.integers(0, 2, N).astype(np.int16) + v2["NGMIX_MCAL_TYPES_FAIL"] = v1["NGMIX_MOM_FAIL"] = fail + v1["IMAFLAGS_ISO"] = rng.integers(0, 256, N).astype(np.int16) + for b, label in MASK_LABELS.items(): + v2[f"MASK_n{b}"] = v1[f"{b}_{label}"] = rng.integers(0, 2, N).astype(bool) + return _structured(v1), _structured(v2) + + +def _structured(columns): + dtype = [(k, v.dtype, v.shape[1:]) for k, v in columns.items()] + out = np.empty(N, dtype=dtype) + for key, value in columns.items(): + out[key] = value + return out + + +def _to_fits_rec(arr, tmp_path): + path = tmp_path / "cat.fits" + fits.BinTableHDU(arr).writeto(path) + return fits.getdata(path, 1), path + + +@pytest.fixture(params=["numpy-vector", "numpy-flat", "fits-vector", "h5py-flat"]) +def v1_and_v2(request, tmp_path): + """A v1 table of each flavour, plus its v2 twin as a numpy array.""" + flavour, layout = request.param.split("-") + v1, v2 = _twins(vector_ell=layout == "vector") + if flavour == "fits": + v1, _ = _to_fits_rec(v1, tmp_path) + elif flavour == "h5py": + handle = h5py.File(tmp_path / "cat.hdf5", "w") + handle.create_dataset("data", data=v1) + request.addfinalizer(handle.close) + v1 = handle["data"] + return v1, v2 + + +def _assert_same_columns(view, v2): + names = set(grammar.column_names(view)) + assert names - {"IMAFLAGS_ISO"} == set(v2.dtype.names) + for name in v2.dtype.names: + expected, got = v2[name], np.asarray(view[name]) + assert got.shape == expected.shape, name + if name.startswith("HSM_T_"): + np.testing.assert_allclose(got, expected, rtol=1e-12, err_msg=name) + else: + np.testing.assert_array_equal(got, expected, err_msg=name) + + +def test_v1_adapts_to_its_v2_twin(v1_and_v2): + v1, v2 = v1_and_v2 + view = adapt(v1) + assert isinstance(view, V2View) + assert len(view) == len(v2) + _assert_same_columns(view, v2) + assert view.dtype.names == view.names + _assert_same_columns(view.to_structured(), v2) + + +def test_v1_names_are_hidden_but_imaflags_passes_through(v1_and_v2): + view = adapt(v1_and_v2[0]) + assert "SIGMA_PSF_HSM" not in view + assert "NGMIX_ELL_NOSHEAR" not in view.names + assert "NGMIX_ELL_NOSHEAR_0" not in view.names + assert "4_Stars" not in view + with pytest.raises(KeyError): + view["E1_PSF_HSM"] + np.testing.assert_array_equal( + view["IMAFLAGS_ISO"], np.asarray(v1_and_v2[0]["IMAFLAGS_ISO"]) + ) + + +def test_row_selection_commutes_with_adaptation(v1_and_v2): + v1, v2 = v1_and_v2 + view = adapt(v1) + mask = np.asarray(v2["HSM_T_PSF"]) > 0.6 + _assert_same_columns(view[mask], v2[mask]) + _assert_same_columns(view[5:40:3], v2[5:40:3]) + _assert_same_columns(view[mask][::2], v2[mask][::2]) + _assert_same_columns(view[4:50][::-3], v2[4:50][::-3]) + _assert_same_columns(view[10:5], v2[10:5]) + if isinstance(v1, np.ndarray): + _assert_same_columns(adapt(v1[mask]), v2[mask]) + row = view[7] + assert row["HSM_T_STAR"] == pytest.approx(v2["HSM_T_STAR"][7], rel=1e-12) + + +def test_header_names_match_view_names(v1_and_v2): + v1, _ = v1_and_v2 + names = grammar.column_names(v1) + assert v2_names(names) == adapt(v1).names + + +def test_detect_generation(): + v1, v2 = _twins() + assert detect_generation(v1.dtype.names) == "v1" + assert detect_generation(v2.dtype.names) == "v2" + assert detect_generation(["RA", "Dec", "e1", "e2", "w"]) is None + with pytest.raises(ValueError, match="mixes"): + detect_generation(["SIGMA_PSF_HSM", "HSM_T_STAR"]) + + +def test_v2_and_neutral_tables_pass_through_unchanged(): + _, v2 = _twins() + assert adapt(v2) is v2 + neutral = np.zeros(3, dtype=[("e1", "f8"), ("e2", "f8"), ("w", "f8")]) + assert adapt(neutral) is neutral + assert v2_names(v2.dtype.names) == v2.dtype.names + + +def test_read_catalogue_and_header_names(tmp_path): + v1, v2 = _twins() + _, path = _to_fits_rec(v1, tmp_path) + view = read_catalogue(path, hdu=1) + _assert_same_columns(view, v2) + assert read_column_names(path, hdu=1) == view.names + + +def test_psf_size_error_matches_between_grammars(): + """shear_psf_leakage's size-residual field sees T, not sigma, from v1. + + ``psf_size_error`` is e_star (T_star - T_psf) / T_star. Computed through + the adapter from the v1 twin it must equal the v2 value; computed from a + v1 table merely renamed (sigma under the T names) it must not. + """ + from shear_psf_leakage.rho_tau_stat import Catalogs + + from sp_validation.rho_tau import get_params_rho_tau + + v1, v2 = _twins() + entry = {"patch_number": 2, "psf": PSF_BLOCK, "shear": SHEAR_BLOCK} + catalogs = Catalogs(params=get_params_rho_tau(entry)) + + def size_error(cat): + return np.stack(catalogs.get_cat_fields(cat, "psf_size_error")[2:4]) + + expected = size_error(v2) + np.testing.assert_allclose(size_error(adapt(v1)), expected, rtol=1e-12) + + renamed_only = {name: np.asarray(v2[name]) for name in v2.dtype.names} + for obj in ("PSF", "STAR"): + renamed_only[f"HSM_T_{obj}"] = np.asarray(v1[f"SIGMA_{obj}_HSM"]) + assert not np.allclose(size_error(renamed_only), expected) + + +def test_get_rho_tau_identical_for_v1_and_v2_psf_catalogues(tmp_path): + """ρ/τ from a v1 PSF catalogue equal those from its v2 twin, end to end.""" + from sp_validation.rho_tau import get_rho_tau + + v1, v2 = _twins() + v1_hsm = [n for n in v1.dtype.names if n in ("RA", "DEC") or "HSM" in n] + rng = np.random.default_rng(3) + shear = np.empty(N, dtype=[(n, "f8") for n in ("RA", "Dec", "e1", "e2", "w")]) + shear["RA"], shear["Dec"] = v2["RA"], v2["DEC"] + shear["e1"], shear["e2"] = rng.normal(0, 0.3, (2, N)) + shear["w"] = 1.0 + fits.BinTableHDU(shear).writeto(tmp_path / "shear.fits") + + stats = {} + for label, psf in (("v1", v1[v1_hsm]), ("v2", v2[list(PSF_BLOCK.values())])): + psf_path = tmp_path / f"psf_{label}.fits" + fits.BinTableHDU(np.array(psf)).writeto(psf_path) + config = { + label: { + "patch_number": 2, + "psf": PSF_BLOCK | {"path": str(psf_path), "hdu": 1}, + "shear": SHEAR_BLOCK | {"path": str(tmp_path / "shear.fits")}, + } + } + outdir = tmp_path / label + outdir.mkdir() + treecorr_config = { + "ra_units": "deg", + "dec_units": "deg", + "sep_units": "arcmin", + "min_sep": 10, + "max_sep": 600, + "nbins": 4, + } + get_rho_tau(config, label, treecorr_config, str(outdir), label) + stats[label] = [ + fits.getdata(outdir / f"{kind}_stats_{label}.fits") + for kind in ("rho", "tau") + ] + + for table_v1, table_v2 in zip(stats["v1"], stats["v2"], strict=True): + assert table_v1.dtype.names == table_v2.dtype.names + for name in table_v1.dtype.names: + np.testing.assert_allclose(table_v1[name], table_v2[name], rtol=1e-10) + + +# -- mask-bit columns: a rule family of their own --------------------------- + + +def _legacy_mask_table(n=N, extra=()): + """A data_ext-style table of {b}_{label} flags, and its MASK_n{b} twin.""" + rng = np.random.default_rng(7) + old = { + f"{b}_{label}": rng.integers(0, 2, n).astype(bool) + for b, label in MASK_LABELS.items() + } + new = {f"MASK_n{b}": old[f"{b}_{label}"] for b, label in MASK_LABELS.items()} + for name in extra: + old[name] = new[name] = rng.integers(0, 6, n) + return _structured_n(old), _structured_n(new) + + +def _structured_n(columns): + n = len(next(iter(columns.values()))) + out = np.empty(n, dtype=[(k, v.dtype, v.shape[1:]) for k, v in columns.items()]) + for key, value in columns.items(): + out[key] = value + return out + + +@pytest.mark.parametrize("bit", sorted(MASK_LABELS)) +def test_every_mask_bit_maps_to_mask_n(bit): + assert grammar.mask_column(bit) == f"MASK_n{bit}" + assert v2_names([f"{bit}_{MASK_LABELS[bit]}"]) == (f"MASK_n{bit}",) + + +def test_mask_column_rejects_unknown_bit(): + with pytest.raises(KeyError): + grammar.mask_column(4096) + + +def test_legacy_mask_flags_rename_without_a_generation(): + """data_ext mask names are renamed whatever the ShapePipe generation.""" + old, new = _legacy_mask_table(extra=("npoint3",)) + assert detect_generation(old.dtype.names) is None + view = adapt(old) + assert isinstance(view, V2View) + assert view.names == new.dtype.names + for name in new.dtype.names: + np.testing.assert_array_equal(view[name], new[name]) + + # Alongside v2 shape columns too: the mask family is not a v1 marker. + _, v2 = _twins() + v2_shape = v2[[n for n in v2.dtype.names if not n.startswith("MASK_n")]] + joined = adapt(rfn.repack_fields(v2_shape), old) + assert detect_generation(joined.names) == "v2" + np.testing.assert_array_equal(joined["MASK_n4"], new["MASK_n4"]) + np.testing.assert_array_equal(joined["HSM_T_PSF"], v2["HSM_T_PSF"]) + + +def test_new_mask_names_pass_through_and_both_names_conflict(): + old, new = _legacy_mask_table() + assert adapt(new) is new + with pytest.raises(ValueError, match="two names"): + adapt(old, new[["MASK_n4"]].copy()) + + +def test_join_requires_equal_lengths_and_disjoint_names(): + old, new = _legacy_mask_table() + with pytest.raises(ValueError, match="lengths"): + adapt(new, old[:10]) + with pytest.raises(ValueError, match="more than one table"): + adapt(new, new) + + +# -- V2View mechanics -------------------------------------------------------- + + +def test_dtype_matches_columns(v1_and_v2): + """dtype reports exactly what each column read returns.""" + view = adapt(v1_and_v2[0]) + for name in view.names: + column = np.asarray(view[name]) + assert view.dtype[name].base == column.dtype, name + assert view.dtype[name].shape == column.shape[1:], name + + +def test_sigma_to_T_keeps_float32(): + v1 = np.zeros(4, dtype=[("SIGMA_PSF_HSM", "f4"), ("E1_PSF_HSM", "f4")]) + v1["SIGMA_PSF_HSM"] = 0.5 + view = adapt(v1) + assert view.dtype["HSM_T_PSF"] == view["HSM_T_PSF"].dtype + assert view.to_structured()["HSM_T_PSF"].dtype == view["HSM_T_PSF"].dtype + + +def test_empty_list_selects_an_empty_view(v1_and_v2): + view = adapt(v1_and_v2[0]) + empty = view[[]] + assert isinstance(empty, V2View) + assert len(empty) == 0 + assert empty["HSM_T_PSF"].shape == (0,) + assert len(view[np.zeros(len(view), dtype=bool)]) == 0 + + +def test_boolean_selection_holds_only_the_selected_rows(): + """A mask becomes the selected indices, never a full-length index array.""" + old, _ = _legacy_mask_table(n=1000) + view = adapt(old) + mask = np.zeros(1000, dtype=bool) + mask[[3, 500, 998]] = True + assert list(view[mask]._rows) == [3, 500, 998] + window = view[100:900] + assert isinstance(window._rows, range) + sub = window[mask[100:900]] + assert list(sub._rows) == [500] + np.testing.assert_array_equal(sub["MASK_n4"], old["4_Stars"][[500]]) + + +def test_h5py_reads_only_the_selected_window(tmp_path, monkeypatch): + """Row selections read the window spanning them, not the whole column.""" + old, _ = _legacy_mask_table(n=1000) + with h5py.File(tmp_path / "ext.hdf5", "w") as handle: + handle.create_dataset("data_ext", data=old) + reads = [] + original = grammar._read_window + + def spy(table, name, start, stop): + reads.append((start, stop)) + return original(table, name, start, stop) + + monkeypatch.setattr(grammar, "_read_window", spy) + with h5py.File(tmp_path / "ext.hdf5", "r") as handle: + view = adapt(handle["data_ext"]) + mask = np.zeros(1000, dtype=bool) + mask[[200, 250, 300]] = True + got = view[mask]["MASK_n8"] + np.testing.assert_array_equal(got, old["8_Manual"][[200, 250, 300]]) + assert reads[-1] == (200, 301) + view[10:20:3]["MASK_n1"] + assert reads[-1] == (10, 20) From 0863142fd9fdc2f97f213111b055c5fd1a294894 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 04:03:47 +0200 Subject: [PATCH 23/39] cat_config: psf/shear columns match their files; guard it on candide test_cat_config_columns_exist_on_candide reads each cat_config catalogue's FITS header through grammar.read_column_names and checks the psf block's declared columns and the shear block's *_col columns are among the names the file presents. Known gaps are listed with reasons and fail the test once healed. Config fixes it surfaced: - psf dec_col Dec -> DEC: every PSF file (and ShapePipe v2) names it DEC; only FITS_rec's case-insensitive lookup hid the mismatch. - SP_v1.4.12.3 / SP_v1.4.13.3 psf blocks named v1 columns and the retired square_size flag; now the v2 names like every other entry. - SP_axel_v0.0 / SP_v1.4.5.A shear: declare their lowercase ra/dec; SP_v1.4.5.A's PSF shape columns are psf_g1/psf_g2. Co-Authored-By: Claude Opus 5.5 --- cosmo_val/cat_config.yaml | 96 ++++++++++--------- .../tests/test_config_paths_exist.py | 85 ++++++++++++++++ 2 files changed, 134 insertions(+), 47 deletions(-) diff --git a/cosmo_val/cat_config.yaml b/cosmo_val/cat_config.yaml index a4d34bee..848c9086 100644 --- a/cosmo_val/cat_config.yaml +++ b/cosmo_val/cat_config.yaml @@ -59,7 +59,7 @@ SP_axel_v0.0: hdu: 1 path: star_cat.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -67,6 +67,8 @@ SP_axel_v0.0: shear: R: 1.0 path: shapepipe_1500_goldshape_v1.fits + ra_col: ra + dec_col: dec w_col: w e1_col: g1 e1_PSF_col: HSM_G1_PSF @@ -94,7 +96,7 @@ SP_v0.1.1: hdu: 1 path: star_cat.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -128,7 +130,7 @@ SP_v1.3: hdu: 1 path: unions_shapepipe_star_2022_v1.3.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -170,7 +172,7 @@ SP_v1.3.6: hdu: 1 path: unions_shapepipe_psf_2022_v1.3.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -214,7 +216,7 @@ SP_v1.4.5: hdu: 1 path: /n17data/UNIONS/WL/v1.4.x/unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -257,7 +259,7 @@ SP_v1.4.5.A: hdu: 1 path: /n17data/UNIONS/WL/v1.4.x/unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -265,12 +267,14 @@ SP_v1.4.5.A: shear: R: 1.0 path: shapepipe_SPv1.fits + ra_col: ra + dec_col: dec redshift_path: /n17data/mkilbing/astro/data/CFIS/v1.0/nz/dndz_SP_A.txt w_col: w e1_col: g1 - e1_PSF_col: e1_PSF + e1_PSF_col: psf_g1 e2_col: g2 - e2_PSF_col: e2_PSF + e2_PSF_col: psf_g2 star: ra_col: RA dec_col: Dec @@ -298,7 +302,7 @@ SP_v1.4.5_bright: hdu: 1 path: /n17data/UNIONS/WL/v1.4.x/unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -339,7 +343,7 @@ SP_v1.4.5_faint: hdu: 1 path: /n17data/UNIONS/WL/v1.4.x/unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -381,7 +385,7 @@ SP_v1.4.6_glass_mock: hdu: 1 path: unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -423,7 +427,7 @@ SP_v1.4.5_intermediate: hdu: 1 path: /n17data/UNIONS/WL/v1.4.x/unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -465,7 +469,7 @@ SP_v1.4.6: hdu: 1 path: unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -509,7 +513,7 @@ SP_v1.4.6.3: hdu: 1 path: /n17data/UNIONS/WL/v1.4.x/unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -553,7 +557,7 @@ SP_v1.4.6.3_B: hdu: 1 path: unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -597,7 +601,7 @@ SP_v1.4.6.3_C: hdu: 1 path: unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -641,7 +645,7 @@ SP_v1.4.6.3_A: hdu: 1 path: unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -685,7 +689,7 @@ SP_v1.4.6_ecut07: hdu: 1 path: unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -729,7 +733,7 @@ SP_v1.4.6.1: hdu: 1 path: unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -773,7 +777,7 @@ SP_v1.4.7: hdu: 1 path: unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -819,7 +823,7 @@ SP_v1.4.8: hdu: 1 path: /n17data/UNIONS/WL/v1.4.x/unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -863,7 +867,7 @@ SP_v1.4.11.2: hdu: 1 path: /n17data/UNIONS/WL/v1.4.x/unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -907,7 +911,7 @@ SP_v1.4.11.3: hdu: 1 path: /n17data/UNIONS/WL/v1.4.x/unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -945,19 +949,18 @@ SP_v1.4.12.3: sigma_e: 0.379587601488189 mask: /home/guerrini/sp_validation/cosmo_inference/data/mask/mask_map_v1.4.6_nside_8192.fits psf: - PSF_flag: FLAG_PSF_HSM - PSF_size: SIGMA_PSF_HSM - square_size: true - star_flag: FLAG_STAR_HSM - star_size: SIGMA_STAR_HSM + PSF_flag: HSM_FLAG_PSF + PSF_size: HSM_T_PSF + star_flag: HSM_FLAG_STAR + star_size: HSM_T_STAR hdu: 1 path: /n17data/UNIONS/WL/v1.4.x/unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec - e1_PSF_col: E1_PSF_HSM - e1_star_col: E1_STAR_HSM - e2_PSF_col: E2_PSF_HSM - e2_star_col: E2_STAR_HSM + dec_col: DEC + e1_PSF_col: HSM_G1_PSF + e1_star_col: HSM_G1_STAR + e2_PSF_col: HSM_G2_PSF + e2_star_col: HSM_G2_STAR shear: R: 1.0 covmat_file: ./covs/shapepipe_A/cov_shapepipe_A.txt @@ -991,19 +994,18 @@ SP_v1.4.13.3: sigma_e: 0.379587601488189 mask: /home/guerrini/sp_validation/cosmo_inference/data/mask/mask_map_v1.4.6_nside_8192.fits psf: - PSF_flag: FLAG_PSF_HSM - PSF_size: SIGMA_PSF_HSM - square_size: true - star_flag: FLAG_STAR_HSM - star_size: SIGMA_STAR_HSM + PSF_flag: HSM_FLAG_PSF + PSF_size: HSM_T_PSF + star_flag: HSM_FLAG_STAR + star_size: HSM_T_STAR hdu: 1 path: /n17data/UNIONS/WL/v1.4.x/unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec - e1_PSF_col: E1_PSF_HSM - e1_star_col: E1_STAR_HSM - e2_PSF_col: E2_PSF_HSM - e2_star_col: E2_STAR_HSM + dec_col: DEC + e1_PSF_col: HSM_G1_PSF + e1_star_col: HSM_G1_STAR + e2_PSF_col: HSM_G2_PSF + e2_star_col: HSM_G2_STAR shear: R: 1.0 covmat_file: ./covs/shapepipe_A/cov_shapepipe_A.txt @@ -1044,7 +1046,7 @@ SP_v1.4.11.3_ecut07: hdu: 1 path: /n17data/UNIONS/WL/v1.4.x/unions_shapepipe_psf_2024_v1.4.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -1180,7 +1182,7 @@ SP_v1.4_LFmask_8k: hdu: 1 path: unions_shapepipe_psf_conv_2022_v1.4.0_mtheli8k.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -1221,7 +1223,7 @@ SP_v1.4_LFmask_8k_noalpha: hdu: 1 path: unions_shapepipe_psf_conv_2022_v1.4.0_mtheli8k.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF @@ -1270,7 +1272,7 @@ SP_v1.6.6: hdu: 1 path: unions_shapepipe_psf_2024_v1.6.a.fits ra_col: RA - dec_col: Dec + dec_col: DEC e1_PSF_col: HSM_G1_PSF e1_star_col: HSM_G1_STAR e2_PSF_col: HSM_G2_PSF diff --git a/src/sp_validation/tests/test_config_paths_exist.py b/src/sp_validation/tests/test_config_paths_exist.py index 2db03bc1..bc5fa662 100644 --- a/src/sp_validation/tests/test_config_paths_exist.py +++ b/src/sp_validation/tests/test_config_paths_exist.py @@ -220,3 +220,88 @@ def test_configured_paths_exist_on_candide(): f"{len(missing)} missing configured paths out of {len(candidates)} checked:\n" + "\n".join(missing[:25]) ) + + +# Columns each cat_config block declares for the rho/tau path. The psf block's +# are read from the PSF file (``rho_tau.get_params_rho_tau``); the shear +# block's are every ``*_col`` key, with the RA/Dec defaults rho/tau assumes. +PSF_COLUMN_KEYS = ( + "ra_col", + "dec_col", + "e1_PSF_col", + "e2_PSF_col", + "e1_star_col", + "e2_star_col", + "PSF_size", + "star_size", + "PSF_flag", + "star_flag", +) +SHEAR_COLUMN_DEFAULTS = {"ra_col": "RA", "dec_col": "Dec"} + +# Entries whose declared columns are known not to exist in their files, with +# why. The test fails if one of these starts passing, so the entry is dropped +# here once its config or data is fixed. +KNOWN_COLUMN_GAPS = { + ("SP_axel_v0.0", "shear"): "file carries no PSF-shape columns", + ("SP_v1.3", "psf"): "2022 star file stores T_{PSF,STAR}_HSM, in neither" + " ShapePipe grammar and of unconfirmed convention", + ("SP_v1.4.6_glass_mock", "shear"): "GLASS mock carries no PSF-shape columns", + ("SP_v1.4.6.3_uncal_w_1", "shear"): "w_col 'one' names no column", +} + + +def _declared_columns(block, kind): + if kind == "psf": + return {key: block[key] for key in PSF_COLUMN_KEYS if key in block} + declared = {key: val for key, val in block.items() if key.endswith("_col")} + return SHEAR_COLUMN_DEFAULTS | declared + + +def test_cat_config_columns_exist_on_candide(): + """Every cat_config psf/shear column is among those its file presents. + + Reads only FITS headers, through ``grammar.read_column_names``, so a v1 + file is checked against the v2 names it presents to the rho/tau path. + Entries whose file is absent are left to the path guard above. + """ + if not _on_candide(): + pytest.skip("Candide-local column guard skipped: no /automnt/n17data/cdaley") + + from sp_validation.grammar import read_column_names + + with (_repo_root() / "cosmo_val/cat_config.yaml").open() as handle: + config = yaml.safe_load(handle) + + checked, missing = 0, {} + for version, entry in config.items(): + if not isinstance(entry, dict) or "subdir" not in entry: + continue + for kind in ("psf", "shear"): + block = entry.get(kind) or {} + if "path" not in block: + continue + path = Path(block["path"]) + if not path.is_absolute(): + path = Path(entry["subdir"]) / path + if not path.exists(): + continue + hdu = block.get("hdu") or 1 + present = set(read_column_names(path, hdu=hdu)) + absent = { + key: col + for key, col in _declared_columns(block, kind).items() + if col not in present + } + checked += 1 + if absent: + missing[(version, kind)] = f"{path}: {absent}" + + assert checked, "no cat_config catalogue found to check" + unexpected = {k: v for k, v in missing.items() if k not in KNOWN_COLUMN_GAPS} + healed = sorted(set(KNOWN_COLUMN_GAPS) - set(missing)) + assert not unexpected, "declared columns absent from their files:\n" + "\n".join( + f"{version}.{kind} -> {detail}" + for (version, kind), detail in unexpected.items() + ) + assert not healed, f"known column gaps now pass; drop them: {healed}" From 39d909643f119b5df0780d177bc86718d34b89b0 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 04:04:08 +0200 Subject: [PATCH 24/39] rho_tau: build rho/tau catalogues without a scalar-bool mask build_catalog indexes each column with ``mask``; a scalar True only adds an axis treecorr reshapes away (no stars are cut), and a scalar False selects nothing, so treecorr raises "Input arrays have zero length". That broke get_rho_tau for DES and get_jackknife_cov for every catalogue. No flag cut was ever applied, so dropping the argument leaves the non-DES results unchanged. Co-Authored-By: Claude Opus 5.5 --- src/sp_validation/rho_tau.py | 15 ++++----------- 1 file changed, 4 insertions(+), 11 deletions(-) diff --git a/src/sp_validation/rho_tau.py b/src/sp_validation/rho_tau.py index b40ed9b1..cf04b09c 100644 --- a/src/sp_validation/rho_tau.py +++ b/src/sp_validation/rho_tau.py @@ -180,11 +180,7 @@ def get_rho_tau( if need_compute: rho_stat_handler.catalogs.set_params(params, outdir) - mask = version != "DES" - - rho_stat_handler.build_cat_to_compute_rho( - load("psf"), catalog_id=catalog_id, mask=mask - ) + rho_stat_handler.build_cat_to_compute_rho(load("psf"), catalog_id=catalog_id) rho_stat_handler.compute_rho_stats( catalog_id, @@ -213,17 +209,15 @@ def get_rho_tau( else: tau_stat_handler.catalogs.set_params(params, outdir) - mask = version != "DES" - # Build the different catalogs if necessary if f"psf_{version}" not in tau_stat_handler.catalogs.catalogs_dict.keys(): tau_stat_handler.build_cat_to_compute_tau( - load("psf"), cat_type="psf", catalog_id=version, mask=mask + load("psf"), cat_type="psf", catalog_id=version ) # Build the catalog of galaxies. PSF was computed above tau_stat_handler.build_cat_to_compute_tau( - load("shear"), cat_type="gal", catalog_id=version, mask=mask + load("shear"), cat_type="gal", catalog_id=version ) # function to extract the tau_+ @@ -354,7 +348,7 @@ def get_jackknife_cov( if f"psf_{version}{i}" not in rho_stat_handler.catalogs.catalogs_dict: # Build catalogues rho_stat_handler.build_cat_to_compute_rho( - load("psf"), catalog_id=version + str(i), mask=False + load("psf"), catalog_id=version + str(i) ) tau_stat_handler.catalogs.catalogs_dict = ( @@ -366,7 +360,6 @@ def get_jackknife_cov( load("shear"), cat_type="gal", catalog_id=version + str(i), - mask=False, ) else: From 2c6eb14657ec6c0f4f37dadeefd0adc44953a6d3 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 05:31:07 +0200 Subject: [PATCH 25/39] calibration: read v1 and v2 products through the grammar adapter Every reader on the calibration path now presents its tables in the v2 column grammar, so a ShapePipe v1 product runs through the same code and the same configs as a v2 one: - catalog.read_campaign_catalogue, campaign_shape, iter_campaign_tiles and the JointCat merge adapt each tile (param_list names v2 columns); read_star_catalogue adapts FITS and HDF5 star catalogues. - CalibrateCat.read_cat returns one table: the comprehensive HDF5's data and data_ext joined by grammar.adapt, so mask columns read the same whether they sit in data (v2) or data_ext (post-processed v1), and galaxy.mask_cut works on either. - get_masks_from_config takes that one table, and each mask config is one `dat` cut list: the v1 configs' dat_ext cuts move into it under their MASK_n{b} names (the selections are unchanged; IMAFLAGS_ISO stays in the v1 configs). The image-sim overlay and scripts/masking.py follow. - ApplyHspMasks writes mask bit b as MASK_n{b}, from grammar.MASK_LABELS. - cosmo_val/compute_theory_cov.py hands CovTauTh loaded, adapted tables. Tests: a v1 comprehensive HDF5 (v1 data + data_ext with the old mask names) and its v2 twin give identical mask_cut and per-config mask selections for every config in config/calibration, and identical metacal inputs and response; the campaign and star readers present v1 files in v2. Co-Authored-By: Claude Opus 5.5 --- .../calibration/mask_from_compr_v1.X.11.yaml | 14 +- .../calibration/mask_from_compr_v1.X.6.yaml | 14 +- config/calibration/mask_v1.X.10.yaml | 14 +- config/calibration/mask_v1.X.11.yaml | 14 +- config/calibration/mask_v1.X.2.yaml | 18 +- config/calibration/mask_v1.X.3.yaml | 14 +- config/calibration/mask_v1.X.4.yaml | 18 +- config/calibration/mask_v1.X.4_im_sim.yaml | 8 +- config/calibration/mask_v1.X.5.yaml | 14 +- config/calibration/mask_v1.X.6.yaml | 14 +- config/calibration/mask_v1.X.6_ppv1.yaml | 20 +- config/calibration/mask_v1.X.7.yaml | 16 +- config/calibration/mask_v1.X.8.yaml | 18 +- config/calibration/mask_v1.X.9.yaml | 14 +- .../mask_v1.X.9_im_sim.overlay.yaml | 20 +- config/calibration/mask_v1.X.9_im_sim.yaml | 2 +- config/calibration/mask_v2.0.yaml | 9 +- cosmo_val/compute_theory_cov.py | 9 +- docs/ngmix_psf_column_migration.md | 43 ++- papers/catalog/hist_mag.py | 28 +- .../calibrate_comprehensive_cat.py | 5 +- .../create_binned_mask_comprehensive.py | 5 +- .../examples/demo_calibrate_minimal_cat.py | 10 +- .../demo_comprehensive_to_minimal_cat.py | 28 +- .../examples/demo_create_footprint_mask.py | 27 +- scripts/masking.py | 37 +-- src/sp_validation/catalog.py | 95 +++--- src/sp_validation/catalog_builders.py | 79 ++--- src/sp_validation/galaxy.py | 13 +- src/sp_validation/masks.py | 73 ++--- src/sp_validation/plots.py | 23 +- .../tests/test_campaign_readers.py | 4 +- .../tests/test_grammar_calibration.py | 273 ++++++++++++++++++ 33 files changed, 585 insertions(+), 408 deletions(-) create mode 100644 src/sp_validation/tests/test_grammar_calibration.py diff --git a/config/calibration/mask_from_compr_v1.X.11.yaml b/config/calibration/mask_from_compr_v1.X.11.yaml index 3fa9ee3e..6d19cfc5 100644 --- a/config/calibration/mask_from_compr_v1.X.11.yaml +++ b/config/calibration/mask_from_compr_v1.X.11.yaml @@ -9,7 +9,7 @@ params: verbose: True # Masks -## Using columns in 'dat' group (ShapePipe flags) +## Cuts on catalogue columns, in the v2 grammar (sp_validation.grammar) dat: # Duplicate objects @@ -24,29 +24,29 @@ dat: kind: greater_equal value: 2 -## Using columns in 'dat_ext' group (post-processing flags) -dat_ext: + # Healsparse mask bits and post-processing columns (the comprehensive + # HDF5's data_ext dataset) # Stars - - col_name: 4_Stars + - col_name: MASK_n4 label: "Stars" kind: equal value: False # Manual mask - - col_name: 8_Manual + - col_name: MASK_n8 label: "manual mask" kind: equal value: False # r-band footprint - - col_name: 64_r + - col_name: MASK_n64 label: "r-band imaging" kind: equal value: False # Maximask - - col_name: 1024_Maximask + - col_name: MASK_n1024 label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_from_compr_v1.X.6.yaml b/config/calibration/mask_from_compr_v1.X.6.yaml index 1c9b08e5..e090141e 100644 --- a/config/calibration/mask_from_compr_v1.X.6.yaml +++ b/config/calibration/mask_from_compr_v1.X.6.yaml @@ -9,7 +9,7 @@ params: verbose: True # Masks -## Using columns in 'dat' group (ShapePipe flags) +## Cuts on catalogue columns, in the v2 grammar (sp_validation.grammar) dat: # Duplicate objects - col_name: overlap @@ -23,29 +23,29 @@ dat: kind: greater_equal value: 2 -## Using columns in 'dat_ext' group (post-processing flags) -dat_ext: + # Healsparse mask bits and post-processing columns (the comprehensive + # HDF5's data_ext dataset) # Stars - - col_name: 4_Stars + - col_name: MASK_n4 label: "Stars" kind: equal value: False # Manual mask - - col_name: 8_Manual + - col_name: MASK_n8 label: "manual mask" kind: equal value: False # r-band footprint - - col_name: 64_r + - col_name: MASK_n64 label: "r-band imaging" kind: equal value: False # Maximask - - col_name: 1024_Maximask + - col_name: MASK_n1024 label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.10.yaml b/config/calibration/mask_v1.X.10.yaml index 703ea91b..037bc2d9 100644 --- a/config/calibration/mask_v1.X.10.yaml +++ b/config/calibration/mask_v1.X.10.yaml @@ -9,7 +9,7 @@ params: verbose: True # Masks -## Using columns in 'dat' group (ShapePipe flags) +## Cuts on catalogue columns, in the v2 grammar (sp_validation.grammar) dat: # SExtractor flags - col_name: FLAGS @@ -57,29 +57,29 @@ dat: kind: not_equal value: -10 -## Using columns in 'dat_ext' group (post-processing flags) -dat_ext: + # Healsparse mask bits and post-processing columns (the comprehensive + # HDF5's data_ext dataset) # Stars - - col_name: 4_Stars + - col_name: MASK_n4 label: "Stars" kind: equal value: False # Manual mask - - col_name: 8_Manual + - col_name: MASK_n8 label: "manual mask" kind: equal value: False # r-band footprint - - col_name: 64_r + - col_name: MASK_n64 label: "r-band imaging" kind: equal value: False # Maximask - - col_name: 1024_Maximask + - col_name: MASK_n1024 label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.11.yaml b/config/calibration/mask_v1.X.11.yaml index 88e01643..df81d5bd 100644 --- a/config/calibration/mask_v1.X.11.yaml +++ b/config/calibration/mask_v1.X.11.yaml @@ -9,7 +9,7 @@ params: verbose: True # Masks -## Using columns in 'dat' group (ShapePipe flags) +## Cuts on catalogue columns, in the v2 grammar (sp_validation.grammar) dat: # SExtractor flags - col_name: FLAGS @@ -57,29 +57,29 @@ dat: kind: not_equal value: -10 -## Using columns in 'dat_ext' group (post-processing flags) -dat_ext: + # Healsparse mask bits and post-processing columns (the comprehensive + # HDF5's data_ext dataset) # Stars - - col_name: 4_Stars + - col_name: MASK_n4 label: "Stars" kind: equal value: False # Manual mask - - col_name: 8_Manual + - col_name: MASK_n8 label: "manual mask" kind: equal value: False # r-band footprint - - col_name: 64_r + - col_name: MASK_n64 label: "r-band imaging" kind: equal value: False # Maximask - - col_name: 1024_Maximask + - col_name: MASK_n1024 label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.2.yaml b/config/calibration/mask_v1.X.2.yaml index 846f14ff..4ef4c22f 100644 --- a/config/calibration/mask_v1.X.2.yaml +++ b/config/calibration/mask_v1.X.2.yaml @@ -9,7 +9,7 @@ params: verbose: True # Masks -## Using columns in 'dat' group (ShapePipe flags) +## Cuts on catalogue columns, in the v2 grammar (sp_validation.grammar) dat: # SExtractor flags - col_name: FLAGS @@ -57,40 +57,40 @@ dat: kind: not_equal value: -10 -## Using columns in 'dat_ext' group (post-processing flags) -dat_ext: + # Healsparse mask bits and post-processing columns (the comprehensive + # HDF5's data_ext dataset) # Faint star halos - - col_name: 1_Faint_star_halos + - col_name: MASK_n1 label: "Faint star halos" kind: equal value: False # Bright star halos - - col_name: 2_Bright_star_halos + - col_name: MASK_n2 label: "Bright star halos" kind: equal value: False # Stars - - col_name: 4_Stars + - col_name: MASK_n4 label: "Stars" kind: equal value: False # Manual mask - - col_name: 8_Manual + - col_name: MASK_n8 label: "manual mask" kind: equal value: False # r-band footprint - - col_name: 64_r + - col_name: MASK_n64 label: "r-band imaging" kind: equal value: False # Maximask - - col_name: 1024_Maximask + - col_name: MASK_n1024 label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.3.yaml b/config/calibration/mask_v1.X.3.yaml index bcbbc620..2d12a458 100644 --- a/config/calibration/mask_v1.X.3.yaml +++ b/config/calibration/mask_v1.X.3.yaml @@ -9,7 +9,7 @@ params: verbose: True # Masks -## Using columns in 'dat' group (ShapePipe flags) +## Cuts on catalogue columns, in the v2 grammar (sp_validation.grammar) dat: # SExtractor flags - col_name: FLAGS @@ -57,29 +57,29 @@ dat: kind: not_equal value: -10 -## Using columns in 'dat_ext' group (post-processing flags) -dat_ext: + # Healsparse mask bits and post-processing columns (the comprehensive + # HDF5's data_ext dataset) # Stars - - col_name: 4_Stars + - col_name: MASK_n4 label: "Stars" kind: equal value: False # Manual mask - - col_name: 8_Manual + - col_name: MASK_n8 label: "manual mask" kind: equal value: False # r-band footprint - - col_name: 64_r + - col_name: MASK_n64 label: "r-band imaging" kind: equal value: False # Maximask - - col_name: 1024_Maximask + - col_name: MASK_n1024 label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.4.yaml b/config/calibration/mask_v1.X.4.yaml index 903b9098..93779a44 100644 --- a/config/calibration/mask_v1.X.4.yaml +++ b/config/calibration/mask_v1.X.4.yaml @@ -9,7 +9,7 @@ params: verbose: True # Masks -## Using columns in 'dat' group (ShapePipe flags) +## Cuts on catalogue columns, in the v2 grammar (sp_validation.grammar) dat: # SExtractor flags - col_name: FLAGS @@ -57,40 +57,40 @@ dat: kind: not_equal value: -10 -## Using columns in 'dat_ext' group (post-processing flags) -dat_ext: + # Healsparse mask bits and post-processing columns (the comprehensive + # HDF5's data_ext dataset) # Faint star halos - - col_name: 1_Faint_star_halos + - col_name: MASK_n1 label: "Faint star halos" kind: equal value: False # Bright star halos - - col_name: 2_Bright_star_halos + - col_name: MASK_n2 label: "Bright star halos" kind: equal value: False # Stars - - col_name: 4_Stars + - col_name: MASK_n4 label: "Stars" kind: equal value: False # Manual mask - - col_name: 8_Manual + - col_name: MASK_n8 label: "manual mask" kind: equal value: False # r-band footprint - - col_name: 64_r + - col_name: MASK_n64 label: "r-band imaging" kind: equal value: False # Maximask - - col_name: 1024_Maximask + - col_name: MASK_n1024 label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.4_im_sim.yaml b/config/calibration/mask_v1.X.4_im_sim.yaml index 794566d4..8f3636c2 100644 --- a/config/calibration/mask_v1.X.4_im_sim.yaml +++ b/config/calibration/mask_v1.X.4_im_sim.yaml @@ -9,7 +9,7 @@ params: verbose: True # Masks -## Using columns in 'dat' group (ShapePipe flags) +## Cuts on catalogue columns, in the v2 grammar (sp_validation.grammar) dat: # SExtractor flags - col_name: FLAGS @@ -30,17 +30,17 @@ dat: value: [15, 30] # ngmix flags - - col_name: NGMIX_MOM_FAIL + - col_name: NGMIX_MCAL_TYPES_FAIL label: "ngmix moments failure" kind: equal value: 0 # invalid PSF ellipticities - - col_name: NGMIX_ELL_PSFo_NOSHEAR_0 + - col_name: NGMIX_G1_PSF_ORIG_NOSHEAR label: "bad PSF ellipticity comp 1" kind: not_equal value: -10 - - col_name: NGMIX_ELL_PSFo_NOSHEAR_1 + - col_name: NGMIX_G2_PSF_ORIG_NOSHEAR label: "bad PSF ellipticity comp 2" kind: not_equal value: -10 diff --git a/config/calibration/mask_v1.X.5.yaml b/config/calibration/mask_v1.X.5.yaml index 5f56b135..c80c5f04 100644 --- a/config/calibration/mask_v1.X.5.yaml +++ b/config/calibration/mask_v1.X.5.yaml @@ -9,7 +9,7 @@ params: verbose: True # Masks -## Using columns in 'dat' group (ShapePipe flags) +## Cuts on catalogue columns, in the v2 grammar (sp_validation.grammar) dat: # SExtractor flags - col_name: FLAGS @@ -57,29 +57,29 @@ dat: kind: not_equal value: -10 -## Using columns in 'dat_ext' group (post-processing flags) -dat_ext: + # Healsparse mask bits and post-processing columns (the comprehensive + # HDF5's data_ext dataset) # Stars - - col_name: 4_Stars + - col_name: MASK_n4 label: "Stars" kind: equal value: False # Manual mask - - col_name: 8_Manual + - col_name: MASK_n8 label: "manual mask" kind: equal value: False # r-band footprint - - col_name: 64_r + - col_name: MASK_n64 label: "r-band imaging" kind: equal value: False # Maximask - - col_name: 1024_Maximask + - col_name: MASK_n1024 label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.6.yaml b/config/calibration/mask_v1.X.6.yaml index 1b5f7da6..3eb33b95 100644 --- a/config/calibration/mask_v1.X.6.yaml +++ b/config/calibration/mask_v1.X.6.yaml @@ -9,7 +9,7 @@ params: verbose: True # Masks -## Using columns in 'dat' group (ShapePipe flags) +## Cuts on catalogue columns, in the v2 grammar (sp_validation.grammar) dat: # SExtractor flags - col_name: FLAGS @@ -57,29 +57,29 @@ dat: kind: not_equal value: -10 -## Using columns in 'dat_ext' group (post-processing flags) -dat_ext: + # Healsparse mask bits and post-processing columns (the comprehensive + # HDF5's data_ext dataset) # Stars - - col_name: 4_Stars + - col_name: MASK_n4 label: "Stars" kind: equal value: False # Manual mask - - col_name: 8_Manual + - col_name: MASK_n8 label: "manual mask" kind: equal value: False # r-band footprint - - col_name: 64_r + - col_name: MASK_n64 label: "r-band imaging" kind: equal value: False # Maximask - - col_name: 1024_Maximask + - col_name: MASK_n1024 label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.6_ppv1.yaml b/config/calibration/mask_v1.X.6_ppv1.yaml index a3d29cda..639dde4a 100644 --- a/config/calibration/mask_v1.X.6_ppv1.yaml +++ b/config/calibration/mask_v1.X.6_ppv1.yaml @@ -9,7 +9,7 @@ params: verbose: True # Masks -## Using columns in 'dat' group (ShapePipe flags) +## Cuts on catalogue columns, in the v2 grammar (sp_validation.grammar) dat: # SExtractor flags - col_name: FLAGS @@ -42,17 +42,17 @@ dat: value: [15, 30] # ngmix flags - - col_name: NGMIX_MOM_FAIL + - col_name: NGMIX_MCAL_TYPES_FAIL label: "ngmix moments failure" kind: equal value: 0 # invalid PSF ellipticities - - col_name: NGMIX_ELL_PSFo_NOSHEAR_0 + - col_name: NGMIX_G1_PSF_ORIG_NOSHEAR label: "bad PSF ellipticity comp 1" kind: not_equal value: -10 - - col_name: NGMIX_ELL_PSFo_NOSHEAR_1 + - col_name: NGMIX_G2_PSF_ORIG_NOSHEAR label: "bad PSF ellipticity comp 2" kind: not_equal value: -10 @@ -84,29 +84,29 @@ dat: col_name2: MAG_GAAP_0p7_z2 value: -99 -## Using columns in 'dat_ext' group (post-processing flags) -dat_ext: + # Healsparse mask bits and post-processing columns (the comprehensive + # HDF5's data_ext dataset) # Stars - - col_name: 4_Stars + - col_name: MASK_n4 label: "Stars" kind: equal value: False # Manual mask - - col_name: 8_Manual + - col_name: MASK_n8 label: "manual mask" kind: equal value: False # r-band footprint - - col_name: 64_r + - col_name: MASK_n64 label: "r-band imaging" kind: equal value: False # Maximask - - col_name: 1024_Maximask + - col_name: MASK_n1024 label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.7.yaml b/config/calibration/mask_v1.X.7.yaml index cbc802be..b92ddf58 100644 --- a/config/calibration/mask_v1.X.7.yaml +++ b/config/calibration/mask_v1.X.7.yaml @@ -9,7 +9,7 @@ params: verbose: True # Masks -## Using columns in 'dat' group (ShapePipe flags) +## Cuts on catalogue columns, in the v2 grammar (sp_validation.grammar) dat: # SExtractor flags - col_name: FLAGS @@ -57,34 +57,34 @@ dat: kind: not_equal value: -10 -## Using columns in 'dat_ext' group (post-processing flags) -dat_ext: + # Healsparse mask bits and post-processing columns (the comprehensive + # HDF5's data_ext dataset) # Bright star halos - - col_name: 2_Bright_star_halos + - col_name: MASK_n2 label: "Bright star halos" kind: equal value: False # Stars - - col_name: 4_Stars + - col_name: MASK_n4 label: "Stars" kind: equal value: False # Manual mask - - col_name: 8_Manual + - col_name: MASK_n8 label: "manual mask" kind: equal value: False # r-band footprint - - col_name: 64_r + - col_name: MASK_n64 label: "r-band imaging" kind: equal value: False # Maximask - - col_name: 1024_Maximask + - col_name: MASK_n1024 label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.8.yaml b/config/calibration/mask_v1.X.8.yaml index 58234d64..3f5e599d 100644 --- a/config/calibration/mask_v1.X.8.yaml +++ b/config/calibration/mask_v1.X.8.yaml @@ -9,7 +9,7 @@ params: verbose: True # Masks -## Using columns in 'dat' group (ShapePipe flags) +## Cuts on catalogue columns, in the v2 grammar (sp_validation.grammar) dat: # SExtractor flags - col_name: FLAGS @@ -57,40 +57,40 @@ dat: kind: not_equal value: -10 -## Using columns in 'dat_ext' group (post-processing flags) -dat_ext: + # Healsparse mask bits and post-processing columns (the comprehensive + # HDF5's data_ext dataset) # Faint star halos - - col_name: 1_Faint_star_halos + - col_name: MASK_n1 label: "Faint star halos" kind: equal value: False # Bright star halos - - col_name: 2_Bright_star_halos + - col_name: MASK_n2 label: "Bright star halos" kind: equal value: False # Stars - - col_name: 4_Stars + - col_name: MASK_n4 label: "Stars" kind: equal value: False # Manual mask - - col_name: 8_Manual + - col_name: MASK_n8 label: "manual mask" kind: equal value: False # r-band footprint - - col_name: 64_r + - col_name: MASK_n64 label: "r-band imaging" kind: equal value: False # Maximask - - col_name: 1024_Maximask + - col_name: MASK_n1024 label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.9.yaml b/config/calibration/mask_v1.X.9.yaml index 4a448ce7..0baef956 100644 --- a/config/calibration/mask_v1.X.9.yaml +++ b/config/calibration/mask_v1.X.9.yaml @@ -9,7 +9,7 @@ params: verbose: True # Masks -## Using columns in 'dat' group (ShapePipe flags) +## Cuts on catalogue columns, in the v2 grammar (sp_validation.grammar) dat: # SExtractor flags - col_name: FLAGS @@ -57,29 +57,29 @@ dat: kind: not_equal value: -10 -## Using columns in 'dat_ext' group (post-processing flags) -dat_ext: + # Healsparse mask bits and post-processing columns (the comprehensive + # HDF5's data_ext dataset) # Stars - - col_name: 4_Stars + - col_name: MASK_n4 label: "Stars" kind: equal value: False # Manual mask - - col_name: 8_Manual + - col_name: MASK_n8 label: "manual mask" kind: equal value: False # r-band footprint - - col_name: 64_r + - col_name: MASK_n64 label: "r-band imaging" kind: equal value: False # Maximask - - col_name: 1024_Maximask + - col_name: MASK_n1024 label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.9_im_sim.overlay.yaml b/config/calibration/mask_v1.X.9_im_sim.overlay.yaml index 94c5b9ae..e334d0c5 100644 --- a/config/calibration/mask_v1.X.9_im_sim.overlay.yaml +++ b/config/calibration/mask_v1.X.9_im_sim.overlay.yaml @@ -45,36 +45,36 @@ ops: with: |2 # invalid PSF ellipticities (ShapePipe-v2 grammar: scalar G1/G2 components) - # --- dat_ext (post-processing / coverage masks) ------------------------ + # --- mask bits (post-processing / coverage masks) ---------------------- - why: >- - No coverage masks on sims: the whole dat_ext group (Stars, manual mask, - r-band footprint, Maximask) is survey post-processing with no analogue - in the simulated tiles. + No coverage masks on sims: the healsparse mask-bit cuts (stars, manual + mask, n64, Maximask) are survey post-processing with no analogue in the + simulated tiles. drop: |2 - ## Using columns in 'dat_ext' group (post-processing flags) - dat_ext: + # Healsparse mask bits and post-processing columns (the comprehensive + # HDF5's data_ext dataset) # Stars - - col_name: 4_Stars + - col_name: MASK_n4 label: "Stars" kind: equal value: False # Manual mask - - col_name: 8_Manual + - col_name: MASK_n8 label: "manual mask" kind: equal value: False # r-band footprint - - col_name: 64_r + - col_name: MASK_n64 label: "r-band imaging" kind: equal value: False # Maximask - - col_name: 1024_Maximask + - col_name: MASK_n1024 label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.9_im_sim.yaml b/config/calibration/mask_v1.X.9_im_sim.yaml index 1c6ff672..3c28cc03 100644 --- a/config/calibration/mask_v1.X.9_im_sim.yaml +++ b/config/calibration/mask_v1.X.9_im_sim.yaml @@ -9,7 +9,7 @@ params: verbose: True # Masks -## Using columns in 'dat' group (ShapePipe flags) +## Cuts on catalogue columns, in the v2 grammar (sp_validation.grammar) dat: # SExtractor flags - col_name: FLAGS diff --git a/config/calibration/mask_v2.0.yaml b/config/calibration/mask_v2.0.yaml index d9800e81..22a5444e 100644 --- a/config/calibration/mask_v2.0.yaml +++ b/config/calibration/mask_v2.0.yaml @@ -33,7 +33,7 @@ params: verbose: True # Masks -## Using columns in 'dat' group (ShapePipe flags) +## Cuts on catalogue columns, in the v2 grammar (sp_validation.grammar) dat: # SExtractor flags: keep 0 (clean), 1 (neighbours), 2 (deblended); drop # 3 (both) and above. Same as the image-sims config, so data and sims @@ -95,7 +95,7 @@ dat: # external post-processing catalogue. ShapePipe v2 emits no such column and # no MASK_n* bit encodes it, so that cut has no v2 counterpart; add it back # here if an external pointing-coverage map is ever joined onto the - # catalogue (it would belong in the 'dat_ext' group below). + # catalogue. # Number of epochs - col_name: N_EPOCH @@ -141,11 +141,6 @@ dat: kind: not_equal value: -10 -## Using columns in 'dat_ext' group (post-processing flags) -## ShapePipe v2 carries the imaging masks in the 'dat' group above, so this -## group is empty unless external masks are added. -dat_ext: [] - # Metacal parameters metacal: # Ellipticity dispersion diff --git a/cosmo_val/compute_theory_cov.py b/cosmo_val/compute_theory_cov.py index 2868355f..4ca911e3 100644 --- a/cosmo_val/compute_theory_cov.py +++ b/cosmo_val/compute_theory_cov.py @@ -4,9 +4,10 @@ import yaml from shear_psf_leakage.rho_tau_cov import CovTauTh +from sp_validation.grammar import read_catalogue -def get_params_rho_tau(cat, survey="other"): +def get_params_rho_tau(cat, survey="other"): # Set parameters params = {} # TODO to yaml file @@ -86,9 +87,9 @@ def get_params_rho_tau(cat, survey="other"): print("Computing the covariance matrix for the version: ", ver) start_time = time.time() cov_tau_th = CovTauTh( - path_gal=path_gal, - path_psf=path_psf, - hdu_psf=hdu_psf, + path_gal=read_catalogue(path_gal, hdu=info["shear"].get("hdu") or 1), + path_psf=read_catalogue(path_psf, hdu=hdu_psf), + hdu_psf=None, treecorr_config=TreeCorrConfig_xi, A=A, n_e=n_e, diff --git a/docs/ngmix_psf_column_migration.md b/docs/ngmix_psf_column_migration.md index 71838372..ba6caf23 100644 --- a/docs/ngmix_psf_column_migration.md +++ b/docs/ngmix_psf_column_migration.md @@ -2,9 +2,11 @@ sp_validation reads the ShapePipe-v2 column grammar everywhere: live package, configs, calibration scripts, paper figures and notebooks. Catalogues written in -the v1 grammar (every release up to v1.6.x) are read through -`sp_validation.grammar` (wired into the ρ/τ path, `rho_tau.py`, and the GLASS -leakage script): `adapt(table)` (or `read_catalogue(path, hdu)` for a +the v1 grammar (every release up to v1.6.x) are presented in the v2 grammar at the +read boundary by `sp_validation.grammar`: the campaign and star readers +(`catalog.read_campaign_catalogue`, `read_star_catalogue`, the `JointCat` merge), +`CalibrateCat.read_cat`, the ρ/τ path (`rho_tau.py`) and the theory-covariance +script all read through it. `adapt(table)` (or `read_catalogue(path, hdu)` for a FITS file) presents a v1 table under the v2 names and units below, as a lazy column view over a numpy array, FITS_rec or h5py dataset, and returns v2 or grammar-neutral tables unchanged. `detect_generation` tells the grammars apart @@ -12,10 +14,29 @@ from column names, and `v2_names` maps a header's names without reading data. The adapter maps by *name*, so a v1 `*_PSFo` column is presented as `*_PSF_ORIG` holding the v1 (reconvolved-alias) values; downstream code never branches on the generation. The tables below (Old = v1, New = v2) are the rule -table in `grammar.RULES`. The adapter also presents the comprehensive HDF5's -`data_ext` mask flags (`1_Faint_star_halos`, …, `2048_z2`) as `MASK_n{b}`, and -passes `IMAFLAGS_ISO` through unmapped: its v1 bits (2 halo, 4 border, -16 Messier, 32 NGC, 128 spike) are not the v2 `MASK_n{b}` of the same value. +table in `grammar.V1_RULES`. + +Mask columns follow a separate rule family, applied whatever the generation. +The UNIONS healsparse mask product is one bitmask; ShapePipe v2 and +`catalog_builders.ApplyHspMasks` both write bit `b` as the boolean `MASK_n{b}`, +and comprehensive HDF5 files written before that name carry it in `data_ext` as +`{b}_{label}` (`grammar.MASK_LABELS`), which the adapter renames: + +| bit | old `data_ext` name | meaning | +|---|---|---| +| 1, 2 | `1_Faint_star_halos`, `2_Bright_star_halos` | star halos | +| 4 | `4_Stars` | star mask | +| 8 | `8_Manual` | manual mask (large galaxies) | +| 16–256 | `16_u`, `32_g`, `64_r`, `128_i`, `256_z` | per-band coverage | +| 512 | `512_Tile_RA_DEC_cut` | outside the tile's unique region | +| 1024 | `1024_Maximask` | MaxiMask | +| 2048 | `2048_z2` | no Pan-STARRS z2 | + +`adapt(data, data_ext)` joins the two datasets of a comprehensive HDF5 into one +table, so a v1 comprehensive catalogue and a v2 catalogue take the same mask +configuration and the same `galaxy.mask_cut`. `IMAFLAGS_ISO` passes through +unmapped: its v1 bits (2 halo, 4 border, 16 Messier, 32 NGC, 128 spike) are not +the `MASK_n{b}` of the same value. shapepipe#761 turns the shape-measurement output into **one column grammar for the whole catalogue**: every estimator names its outputs @@ -74,7 +95,7 @@ reconvolved-kernel alias the old `ELL_PSFo`/`T_PSFo` columns silently held. The rename is a straight column rename in sp_validation and is correct as-is — the code does not care that the numbers moved. The only thing a regenerated catalogue buys is a *look-at-the-numbers* check that the α-leakage / size-ratio cuts still -behave; that is analysis, not code, and it does not gate this PR (see Status). +behave; that is analysis, not code, and it gates nothing in the code. | Old | New | |---|---| @@ -166,7 +187,7 @@ outside `scratch/` instantiates `metacal(prefix="GALSIM")` or calls `col_1p = f"{prefix}_T_PSF_RECONV_1P"` read in `metacal._read_data` never matched the galsim producer output (`GALSIM_T_PSF_*`, not `..._T_PSF_RECONV_*`). Carrying an untestable, already-broken path onto the new grammar is a worse end state than -deleting it, so this branch **removes** it: +deleting it, so sp_validation has none. Absent by design: - `calibration.metacal._read_data_galsim`, the `prefix == "GALSIM"` dispatch branch (now `else: raise` — unknown prefixes fail loudly), and the two galsim @@ -189,8 +210,8 @@ grammar, and add coverage — not the dead stub that was removed. "Adopt ShapePipe-v2 HSM column grammar, retire square_size") is the sibling consumer migration. It lands on the same HSM grammar (`HSM_G1/G2_{PSF,STAR}`, `HSM_T_*`, `HSM_FLAG_*`) and removed the `square_size` - parameter from `build_cat_to_compute_{rho,tau}` and `CovTauTh`; this PR drops the - matching argument, so the two land together (see "`square_size` is retired"). + parameter from `build_cat_to_compute_{rho,tau}` and `CovTauTh`; sp_validation + passes no such argument (see "`square_size` is retired"). - **shapepipe#761** (producer) still renames the `GALSIM_*` family onto the grammar for columns nothing can create. If the goal is to simplify the grammar, retiring that serialization is a producer-side follow-up worth raising there. diff --git a/papers/catalog/hist_mag.py b/papers/catalog/hist_mag.py index b0bc33ef..a59f7994 100644 --- a/papers/catalog/hist_mag.py +++ b/papers/catalog/hist_mag.py @@ -53,15 +53,14 @@ def get_data(obj, test_only=False): """ # Get data. Set load_into_memory to False for very large files - dat, dat_ext = obj.read_cat(load_into_memory=False) + dat = obj.read_cat(load_into_memory=False) if test_only: n_max = 1_000_000 print(f"MKDEBUG testing only first {n_max} objects") dat = dat[:n_max] - dat_ext = dat_ext[:n_max] - return dat, dat_ext + return dat def read_hist_data(hist_data_path): @@ -219,7 +218,6 @@ def plot_all_hists( ax=None, out_path=None, ): - if ax is None: plt.figure() fig, (ax) = plt.subplots(1, 1, figsize=(figsize, figsize)) @@ -266,24 +264,23 @@ def plot_all_hists( print(f"Histogram data file {hist_data_path} found.") print("Reading and plotting.") - dat = dat_ext = None + dat = None hist_data = read_hist_data(hist_data_path) else: print(f"Histogram data file {hist_data_path} not found.") print("Reading UNIONS cat and computing.") - dat, dat_ext = get_data(obj, test_only=test_only) + dat = get_data(obj, test_only=test_only) hist_data = None # %% # Masking -# Get all masks, with or without dat, dat_ext +# Get all masks, with or without dat masks, labels = sp_joint.get_masks_from_config( config, dat, - dat_ext, verbose=obj._params["verbose"], ) @@ -291,7 +288,7 @@ def plot_all_hists( # Combine mask according to scenario # List of basic masks to apply to all cases -masks_labels_basic = ["overlap", "mag", "64_r"] +masks_labels_basic = ["overlap", "mag", "MASK_n64"] col_names = ["basic masks"] if scenario == 0: @@ -302,9 +299,9 @@ def plot_all_hists( "NGMIX_MCAL_TYPES_FAIL", "NGMIX_G1_PSF_ORIG_NOSHEAR", "NGMIX_G2_PSF_ORIG_NOSHEAR", - "4_Stars", - "8_Manual", - "1024_Maximask", + "MASK_n4", + "MASK_n8", + "MASK_n1024", ] ) @@ -318,9 +315,9 @@ def plot_all_hists( "NGMIX_MCAL_TYPES_FAIL", "NGMIX_G1_PSF_ORIG_NOSHEAR", "NGMIX_G2_PSF_ORIG_NOSHEAR", - "4_Stars", - "8_Manual", - "1024_Maximask", + "MASK_n4", + "MASK_n8", + "MASK_n1024", "N_EPOCH", "npoint3", "metacal", @@ -408,7 +405,6 @@ def plot_all_hists( # %% def get_info_for_metacal_masking(dat, mask, prefix="NGMIX", name_shear="NOSHEAR"): - res = {} res["flag"] = dat[mask][f"{prefix}_FLAGS_{name_shear}"] diff --git a/scripts/calibration/calibrate_comprehensive_cat.py b/scripts/calibration/calibrate_comprehensive_cat.py index 2605338f..4912b963 100644 --- a/scripts/calibration/calibrate_comprehensive_cat.py +++ b/scripts/calibration/calibrate_comprehensive_cat.py @@ -41,7 +41,7 @@ # %% # Get data. Set load_into_memory to False for very large files -dat, dat_ext = obj.read_cat(load_into_memory=False) +dat = obj.read_cat(load_into_memory=False) # %% n_test = -1 @@ -49,14 +49,13 @@ if n_test > 0: print(f"MKDEBUG testing only first {n_test} objects") dat = dat[:n_test] - dat_ext = dat_ext[:n_test] # ## Masking # %% # ### Pre-processing ShapePipe flags -masks, labels = sp_joint.get_masks_from_config(config, dat, dat_ext, verbose=True) +masks, labels = sp_joint.get_masks_from_config(config, dat, verbose=True) mask_combined = sp_joint.Mask.from_list( masks, diff --git a/scripts/examples/create_binned_mask_comprehensive.py b/scripts/examples/create_binned_mask_comprehensive.py index 68964120..03546bce 100644 --- a/scripts/examples/create_binned_mask_comprehensive.py +++ b/scripts/examples/create_binned_mask_comprehensive.py @@ -36,7 +36,7 @@ # %% # Get data. Set load_into_memory to False for very large files -dat, dat_ext = obj.read_cat(load_into_memory=False) +dat = obj.read_cat(load_into_memory=False) # %% n_test = -1 @@ -44,7 +44,6 @@ if n_test > 0: print(f"MKDEBUG testing only first {n_test} objects") dat = dat[:n_test] - dat_ext = dat_ext[:n_test] test = "_test" else: test = "" @@ -68,7 +67,7 @@ # %% # Load and initialise masks -masks, labels = sp_joint.get_masks_from_config(config, dat, dat_ext, verbose=True) +masks, labels = sp_joint.get_masks_from_config(config, dat, verbose=True) mask_combined = sp_joint.Mask.from_list( masks, diff --git a/scripts/examples/demo_calibrate_minimal_cat.py b/scripts/examples/demo_calibrate_minimal_cat.py index a90025dd..9d161ccc 100644 --- a/scripts/examples/demo_calibrate_minimal_cat.py +++ b/scripts/examples/demo_calibrate_minimal_cat.py @@ -35,7 +35,7 @@ # !pwd # Get data. Set load_into_memory to False for very large files -dat, _ = obj.read_cat(load_into_memory=False) +dat = obj.read_cat(load_into_memory=False) if True: n_max = 1_000_000 @@ -47,9 +47,9 @@ masks_to_apply = [ "FLAGS", - "4_Stars", - "64_r", - "1024_Maximask", + "MASK_n4", + "MASK_n64", + "MASK_n1024", "N_EPOCH", "mag", "NGMIX_MCAL_TYPES_FAIL", @@ -58,7 +58,7 @@ ] masks, labels = sp_joint.get_masks_from_config( - config, dat, dat, masks_to_apply=masks_to_apply, verbose=obj._params["verbose"] + config, dat, masks_to_apply=masks_to_apply, verbose=obj._params["verbose"] ) mask_combined = sp_joint.Mask.from_list( diff --git a/scripts/examples/demo_comprehensive_to_minimal_cat.py b/scripts/examples/demo_comprehensive_to_minimal_cat.py index 5c99970e..4a3d0919 100644 --- a/scripts/examples/demo_comprehensive_to_minimal_cat.py +++ b/scripts/examples/demo_comprehensive_to_minimal_cat.py @@ -38,20 +38,18 @@ # !pwd # Get data. Set load_into_memory to False for very large files -dat, dat_ext = obj.read_cat(load_into_memory=False) +dat = obj.read_cat(load_into_memory=False) if False: n_max = 1_000_000 print(f"MKDEBUG testing only first {n_max} objects") dat = dat[:n_max] - dat_ext = dat_ext[:n_max] # ## Masking # + # List of masks to apply -# (labels as declared in config/calibration/mask_v2.0.yaml; ShapePipe v2 -# replaces IMAFLAGS_ISO and the v1 post-processing masks by MASK_n) +# (column names as declared in config/calibration/mask_v2.0.yaml) masks_to_apply = [ "overlap", "MASK_n4", @@ -74,7 +72,6 @@ masks, labels = sp_joint.get_masks_from_config( config, dat, - dat_ext, masks_to_apply=masks_to_apply, verbose=obj._params["verbose"], ) @@ -123,9 +120,9 @@ del mask -def strip_h5py_metadata_dtype(dat_dtype, dat_ext_dtype): +def strip_h5py_metadata_dtype(dat_dtype): cleaned_fields = [] - for name, dt in dat_dtype.descr + dat_ext_dtype.descr: + for name, dt in dat_dtype.descr: # If dt is a tuple (e.g., ('S7', {'h5py_encoding': 'ascii'})) if isinstance(dt, tuple): cleaned_fields.append((name, dt[0])) # keep only the base dtype string @@ -138,26 +135,19 @@ def strip_h5py_metadata_dtype(dat_dtype, dat_ext_dtype): # Remove mask columns that were applied earlier # Columns to keep -names_to_keep = [ - name - for name in dat.dtype.names + dat_ext.dtype.names - if name not in masks_not_to_include -] +names_to_keep = [name for name in dat.dtype.names if name not in masks_not_to_include] # Remove metadata from the dtype (in particular, encoding for TILE_ID) -clean_dtype_descr = strip_h5py_metadata_dtype(dat.dtype, dat_ext.dtype) +clean_dtype_descr = strip_h5py_metadata_dtype(dat.dtype) new_dtype = [(name, dt) for name, dt in clean_dtype_descr if name in names_to_keep] # Create a new structured array new_dat = np.zeros(len(dat[mask_combined._mask]), dtype=new_dtype) -# Copy relevant columns and lines from each source array +# Copy relevant columns and lines for name in names_to_keep: - if name in dat.dtype.names: - new_dat[name] = dat[name][mask_combined._mask] - else: - new_dat[name] = dat_ext[name][mask_combined._mask] + new_dat[name] = dat[name][mask_combined._mask] # + # Add information to FITS header @@ -194,7 +184,6 @@ def strip_h5py_metadata_dtype(dat_dtype, dat_ext_dtype): def correlation_matrix(masks, confidence_level=0.9): - n_key = len(masks) print(n_key) @@ -238,7 +227,6 @@ def correlation_matrix(masks, confidence_level=0.9): def confusion_matrix(prediction, observation): - result = {} result["true_pos"] = sum(prediction & observation) diff --git a/scripts/examples/demo_create_footprint_mask.py b/scripts/examples/demo_create_footprint_mask.py index 8338e676..a1a1850d 100644 --- a/scripts/examples/demo_create_footprint_mask.py +++ b/scripts/examples/demo_create_footprint_mask.py @@ -28,6 +28,7 @@ from cs_util.plots import FootprintPlotter from sp_validation import catalog_builders as sp_joint +from sp_validation.grammar import MASK_LABELS from sp_validation.plots import hsp_map_logical_or # - @@ -46,28 +47,22 @@ # Read configuration file and set parameters config = obj.read_config_set_params("config_mask.yaml") -# Get data sections -config_data = {key: config[key] for key in ["dat", "dat_ext"] if key in config} - # Get names of possible mask names -all_masks_bits = {} -for bit in obj._labels_struct: - all_masks_bits[obj.get_mask_col_name(bit)] = bit +all_masks_bits = {obj.get_mask_col_name(bit): bit for bit in MASK_LABELS} # Identify masks specified in the config file bits = 0 auxiliary_masks = [] auxiliary_labels = [] -for section, mask_list in config_data.items(): - for mask_params in mask_list: - # Check bit-coded masks - if mask_params["col_name"] in all_masks_bits: - bits = bits | all_masks_bits[mask_params["col_name"]] - - # Check auxiliary masks - if "npoint3" == mask_params["col_name"]: - auxiliary_masks.append(obj._params["aux_mask_files"]) - auxiliary_labels.append("npoint3") +for mask_params in config["dat"]: + # Check bit-coded masks + if mask_params["col_name"] in all_masks_bits: + bits = bits | all_masks_bits[mask_params["col_name"]] + + # Check auxiliary masks + if "npoint3" == mask_params["col_name"]: + auxiliary_masks.append(obj._params["aux_mask_files"]) + auxiliary_labels.append("npoint3") # Update bits for following function call obj._params["bits"] = bits diff --git a/scripts/masking.py b/scripts/masking.py index 0ef0491f..021e5b6e 100644 --- a/scripts/masking.py +++ b/scripts/masking.py @@ -7,6 +7,7 @@ import numpy as np import yaml +from sp_validation.grammar import adapt from sp_validation.masks import apply_condition # ------------------------- @@ -28,39 +29,28 @@ "MASK_n1024", "MASK_n2048", "N_EPOCH", - "4_Stars", - "8_Manual", - "64_r", - "1024_Maximask", "npoint3", - "1_Faint_star_halos", - "2_Bright_star_halos", } # ------------------------- # Masking logic -def apply_masks(data, data_ext, mask_config, footprint_only=False): +def apply_masks(data, mask_config, footprint_only=False): """ Construct a boolean mask selecting galaxies that satisfy all masking criteria defined in the YAML configuration file. Parameters ---------- - data : numpy.ndarray or structured array - Slice of the HDF5 "data" group containing per-object - measurements (e.g. FLAGS, mag, NGMIX quantities). - - data_ext : numpy.ndarray or structured array - Slice of the HDF5 "data_ext" group containing external or - post-processing flags (e.g. star masks, footprint flags). + data : numpy.ndarray, structured array or grammar.V2View + Slice of the catalogue in the v2 column grammar: the HDF5 "data" + and "data_ext" datasets joined by ``grammar.adapt``. mask_config : dict Dictionary parsed from the YAML mask configuration file. Expected structure: - mask_config["dat"] : list of cuts applied to `data` - - mask_config["dat_ext"] : list of cuts applied to `data_ext` - mask_config["metacal"] : derived-quantity parameters (e.g. relative size limits) @@ -82,7 +72,6 @@ def apply_masks(data, data_ext, mask_config, footprint_only=False): # Initialize mask mask = np.ones(len(data), dtype=bool) - # --- dat group --- for cut in mask_config.get("dat", []): col = cut["col_name"] if footprint_only and col not in SPATIAL_CUTS: @@ -92,16 +81,6 @@ def apply_masks(data, data_ext, mask_config, footprint_only=False): mask &= apply_condition(data[col], kind, value) - # --- dat_ext group --- - for cut in mask_config.get("dat_ext", []): - col = cut["col_name"] - if footprint_only and col not in SPATIAL_CUTS: - continue - kind = cut["kind"] - value = cut["value"] - - mask &= apply_condition(data_ext[col], kind, value) - # --- metacal relative size (skip for footprint-only) --- if not footprint_only: rel_size = np.divide( @@ -155,10 +134,10 @@ def process_chunk(args): start, stop, filename, nside, mask_config, footprint_only = args with h5py.File(filename, "r") as f: - data = f["data"][start:stop] - data_ext = f["data_ext"][start:stop] + parts = [f[name][start:stop] for name in ("data", "data_ext") if name in f] + data = adapt(*parts) - mask = apply_masks(data, data_ext, mask_config, footprint_only=footprint_only) + mask = apply_masks(data, mask_config, footprint_only=footprint_only) ra = data["RA"][mask] dec = data["Dec"][mask] diff --git a/src/sp_validation/catalog.py b/src/sp_validation/catalog.py index 2a192d41..dd81ebf2 100644 --- a/src/sp_validation/catalog.py +++ b/src/sp_validation/catalog.py @@ -23,7 +23,7 @@ from astropy.io import fits from cs_util import cat -from sp_validation import format, io +from sp_validation import format, grammar, io from sp_validation.version import __version__ @@ -853,14 +853,14 @@ def promote_dtypes(dtype_a, dtype_b): return np.promote_types(dtype_a, dtype_b) -def group_dtype(group, keys, param_list=None): +def group_dtype(tables, param_list=None): """Group Dtype. - Build the structured output dtype of a group of per-tile datasets, - promoting each column across *every* dataset. A campaign that was + Build the structured output dtype of a group of per-tile tables, + promoting each column across *every* table. A campaign that was partially reprocessed can carry e.g. ``S7`` tile IDs in one tile and ``S12`` in another, or ``f4`` next to ``f8``; taking the dtype of the - first dataset alone would silently truncate the others. + first table alone would silently truncate the others. Note that ShapePipe currently writes ``TILE_ID`` as ``f8`` (the tile ``183.307`` arrives as the float ``183.307``), not as a string, so the @@ -874,10 +874,8 @@ def group_dtype(group, keys, param_list=None): Parameters ---------- - group : h5py.Group - group whose members are structured datasets - keys : list of str - dataset names to consider + tables : list + per-tile tables (h5py Datasets, or their ``grammar.adapt`` views) param_list : list of str, optional columns to keep; default is ``None`` (keep all) @@ -887,18 +885,34 @@ def group_dtype(group, keys, param_list=None): structured output dtype """ - names = param_list if param_list is not None else list(group[keys[0]].dtype.names) + names = param_list if param_list is not None else list(tables[0].dtype.names) fields = [] for name in names: - promoted = group[keys[0]].dtype[name] - for key in keys[1:]: - promoted = promote_dtypes(promoted, group[key].dtype[name]) + promoted = tables[0].dtype[name] + for table in tables[1:]: + promoted = promote_dtypes(promoted, table.dtype[name]) fields.append((name, promoted)) return np.dtype(fields) +def _unit_tables(group, file_path, param_list=None): + """Return {dataset name: v2-grammar table} for the datasets of ``group``. + + Each dataset is presented through ``grammar.adapt`` without reading it, + and validated to carry every column of ``param_list``, not only the first. + """ + keys = sorted(group) + if not keys: + raise ValueError(f"No datasets found in catalogue {file_path}") + tables = {key: grammar.adapt(group[key]) for key in keys} + if param_list is not None: + for key, table in tables.items(): + _check_columns(table.dtype, param_list, file_path, key) + return tables + + def _check_columns(dtype, param_list, file_path, dataset_key=None): """Raise a clear error if requested columns are absent from the data.""" missing = [col for col in param_list if col not in (dtype.names or ())] @@ -915,8 +929,9 @@ def concatenate_datasets( ): """Concatenate Datasets. - Concatenate every dataset of an HDF5 group into one structured array, - optionally restricted to a list of columns. + Concatenate every dataset of an HDF5 group into one structured array in + the v2 column grammar (each dataset passes through ``grammar.adapt``), + optionally restricted to a list of v2-grammar columns. The output array is preallocated and filled slice by slice, so peak memory is the output catalogue plus one tile, not the full-width @@ -950,17 +965,9 @@ def concatenate_datasets( collides with an existing column """ - keys = sorted(group) - if not keys: - raise ValueError(f"No datasets found in catalogue {file_path}") - - # Validate every dataset up front: a column may be missing from any tile, - # not only the first one. - for key in keys: - if param_list is not None: - _check_columns(group[key].dtype, param_list, file_path, key) - - dtype_out = group_dtype(group, keys, param_list=param_list) + tables = _unit_tables(group, file_path, param_list=param_list) + keys = list(tables) + dtype_out = group_dtype(list(tables.values()), param_list=param_list) if key_column is not None: if key_column in (dtype_out.names or ()): @@ -981,7 +988,7 @@ def concatenate_datasets( key_values = {key: int(match.group()) for key, match in matches.items()} dtype_out = np.dtype(dtype_out.descr + [(key_column, "i8")]) - n_rows = sum(group[key].shape[0] for key in keys) + n_rows = sum(len(table) for table in tables.values()) if verbose: print( f"Reading {len(keys)} datasets," @@ -992,7 +999,7 @@ def concatenate_datasets( data_out = np.empty(n_rows, dtype=dtype_out) start = 0 for key in tqdm.tqdm(keys, disable=not verbose): - data = group[key][()] + data = grammar.adapt(group[key][()]) end = start + len(data) for name in dtype_out.names: if key_column is not None and name == key_column: @@ -1069,14 +1076,9 @@ def campaign_shape(file_path, param_list=None): with h5py.File(file_path, "r") as hdf5_file: group = find_dataset_group(hdf5_file) check_n_units(hdf5_file, group, file_path) - keys = sorted(group) - if not keys: - raise ValueError(f"No datasets found in catalogue {file_path}") - n_rows = sum(group[key].shape[0] for key in keys) - if param_list is not None: - for key in keys: - _check_columns(group[key].dtype, param_list, file_path, key) - dtype_out = group_dtype(group, keys, param_list=param_list) + tables = list(_unit_tables(group, file_path, param_list=param_list).values()) + n_rows = sum(len(table) for table in tables) + dtype_out = group_dtype(tables, param_list=param_list) return n_rows, dtype_out @@ -1106,15 +1108,9 @@ def iter_campaign_tiles(file_path, param_list=None, verbose=True): with h5py.File(file_path, "r") as hdf5_file: group = find_dataset_group(hdf5_file) check_n_units(hdf5_file, group, file_path) - keys = sorted(group) - if not keys: - raise ValueError(f"No datasets found in catalogue {file_path}") - for key in keys: - if param_list is not None: - _check_columns(group[key].dtype, param_list, file_path, key) + keys = list(_unit_tables(group, file_path, param_list=param_list)) for key in tqdm.tqdm(keys, disable=not verbose): - data = group[key][()] - yield data if param_list is None else data[param_list] + yield grammar.materialise(group[key][()], param_list) def read_campaign_catalogue( @@ -1126,7 +1122,9 @@ def read_campaign_catalogue( """Read Campaign Catalogue. Read a campaign galaxy catalogue (``final_cat_.hdf5``) and - return its per-tile datasets concatenated into one structured array. + return its per-tile datasets concatenated into one structured array, in + the v2 column grammar (``sp_validation.grammar``), whichever ShapePipe + generation wrote it. ``param_list`` names v2-grammar columns. Parameters ---------- @@ -1160,8 +1158,9 @@ def read_star_catalogue(file_path, hdu=1, verbose=True): """Read Star Catalogue. Read a campaign star/PSF catalogue. Reads the ShapePipe v2 - ``full_starcat_.hdf5`` (one dataset per exposure), or a legacy - FITS star catalogue when ``file_path`` ends in ``.fits``. + ``full_starcat_.hdf5`` (one dataset per exposure), or a FITS + star catalogue when ``file_path`` ends in ``.fits``, presenting either in + the v2 column grammar (``sp_validation.grammar``). The HDF5 path adds an ``EXPID`` column carrying the exposure number each star came from. The datasets are named by that number and concatenating @@ -1189,7 +1188,7 @@ def read_star_catalogue(file_path, hdu=1, verbose=True): """ if str(file_path).endswith(".fits"): - return fits.getdata(file_path, hdu) + return grammar.materialise(fits.getdata(file_path, hdu)) with h5py.File(file_path, "r") as hdf5_file: group = find_dataset_group(hdf5_file) diff --git a/src/sp_validation/catalog_builders.py b/src/sp_validation/catalog_builders.py index c16efb21..065b7dff 100644 --- a/src/sp_validation/catalog_builders.py +++ b/src/sp_validation/catalog_builders.py @@ -31,7 +31,7 @@ print_mask_stats, ) -from . import calibration, format +from . import calibration, format, grammar from . import catalog as sp_cat # Names re-exported for external code that resolves them off this module. @@ -598,22 +598,6 @@ def run(self): class ApplyHspMasks(BaseCat): """Apply Hsp Masks.""" - # Labels of bit-coded structural masks - _labels_struct = { - 1: "Faint_star_halos", - 2: "Bright_star_halos", - 4: "Stars", - 8: "Manual", - 16: "u", - 32: "g", - 64: "r", - 128: "i", - 256: "z", - 512: "Tile_RA_DEC_cut", - 1024: "Maximask", - 2048: "z2", - } - def __init__(self): # Set default parameters self.params_default() @@ -635,13 +619,14 @@ def get_label_struct(cls, bit): label """ - return cls._labels_struct[bit] + return grammar.MASK_LABELS[bit] @classmethod def get_mask_col_name(cls, bit): """Get Mask Col Name. - Return column name of mask corresponding to input bit. + Return column name of mask corresponding to input bit: ``MASK_n{bit}``, + the name ShapePipe v2 gives the same bit. Parameters ---------- @@ -654,7 +639,7 @@ def get_mask_col_name(cls, bit): column name """ - return f"{bit}_{cls.get_label_struct(bit)}" + return grammar.mask_column(bit) def params_default(self): """Params Default. @@ -1054,7 +1039,11 @@ def params_default(self): def read_cat(self, load_into_memory=False): """Read Cat. - Read input HDF5 catalogue. + Read the input comprehensive catalogue as one table in the v2 column + grammar (``sp_validation.grammar``). An HDF5 catalogue's ``data`` and, + when present, ``data_ext`` datasets are joined column-wise, so mask + columns read the same whether they sit in ``data`` (ShapePipe v2) or + in ``data_ext`` (post-processed v1). Parameters ---------- @@ -1064,44 +1053,30 @@ def read_cat(self, load_into_memory=False): Returns ------- - list - Catalogue data - list_ext - Extended catalogue data if exists in input file + numpy.ndarray or grammar.V2View + catalogue data """ fpath = self._params["input_path"] verbose = self._params["verbose"] # Image-simulation path: a single per-run comprehensive catalogue in - # FITS, not the joined multi-campaign HDF5 the data path builds. Read the - # FITS table directly into memory; there is no separate data_ext group. + # FITS, not the joined multi-campaign HDF5 the data path builds. extension = os.path.splitext(fpath)[1] if extension == ".fits": if verbose: print(f"Reading FITS file {fpath}, HDU 1...") - dat = fits.getdata(fpath, 1) - dat_ext = None + dat = grammar.materialise(fits.getdata(fpath, 1)) + else: if verbose: - print( - f"Found {len(dat)} (~{format.millify(len(dat))}) objects" - + " in catalogue" - ) - return dat, dat_ext - - if verbose: - print(f"Reading HDF5 file {fpath}...") - - self._hd5file = h5py.File(fpath, "r") - try: - dat = self._hd5file["data"] + print(f"Reading HDF5 file {fpath}...") + self._hd5file = h5py.File(fpath, "r") + parts = [self._hd5file["data"]] if "data_ext" in self._hd5file: - dat_ext = self._hd5file["data_ext"] - else: - dat_ext = None - except: - print(f"Error while reading file {fpath}") - raise + parts.append(self._hd5file["data_ext"]) + dat = grammar.adapt(*parts) + if load_into_memory: + dat = grammar.materialise(dat) if verbose: print( @@ -1109,16 +1084,9 @@ def read_cat(self, load_into_memory=False): + " in catalogue" ) - if load_into_memory: - if dat_ext: - return dat[()], dat_ext[()] - else: - return dat[()] - else: - return dat, dat_ext + return dat def add_params_to_FITS_header(self, header, cm=None): - header_new = fits.Header() # General information @@ -1168,7 +1136,6 @@ def params_default(self): } def run(self): - pass diff --git a/src/sp_validation/galaxy.py b/src/sp_validation/galaxy.py index 60e170fa..f119ab18 100644 --- a/src/sp_validation/galaxy.py +++ b/src/sp_validation/galaxy.py @@ -29,8 +29,10 @@ # required square root: FWHM = 2.35482 sqrt(T / 2) from sp_validation import io -#: All mask columns written by ShapePipe v2 (bool, ``True`` = masked). -#: They replace the single IMAFLAGS_ISO bitmask of ShapePipe v1. +#: All mask columns written by ShapePipe v2 (bool, ``True`` = masked), one per +#: bit of the UNIONS healsparse mask product (``grammar.MASK_LABELS``; v2 does +#: not write n512). Post-processed v1 comprehensive catalogues carry the same +#: columns, presented under these names by ``sp_validation.grammar``. #: #: Reason bits making up the r-band default bitmask: n1/n2 star halos #: (which of the two is faint and which bright is unconfirmed for the @@ -114,9 +116,10 @@ def mask_cut(dd, mask_columns=None): raise KeyError( f"Mask column(s) {missing} not found in catalogue." + " ShapePipe v2 catalogues carry the boolean columns" - + f" {list(MASK_COLUMNS)}; ShapePipe v1 catalogues carry" - + " IMAFLAGS_ISO instead and are not supported." - + f" Available columns: {sorted(available)}" + + f" {list(MASK_COLUMNS)}, as do comprehensive catalogues whose" + + " data_ext holds the healsparse mask bits (read through" + + " sp_validation.grammar). Available columns:" + + f" {sorted(available)}" ) masked = np.zeros(len(dd[columns[0]]), dtype=bool) diff --git a/src/sp_validation/masks.py b/src/sp_validation/masks.py index ddd09ef0..64f7272b 100644 --- a/src/sp_validation/masks.py +++ b/src/sp_validation/masks.py @@ -57,7 +57,6 @@ def apply_condition(array, kind, value): def correlation_matrix(masks, confidence_level=0.9): - n_key = len(masks) print(n_key) @@ -75,7 +74,6 @@ def correlation_matrix(masks, confidence_level=0.9): def confusion_matrix(prediction, observation): - result = {} result["true_pos"] = sum(prediction & observation) @@ -143,7 +141,6 @@ def __init__( dat=None, verbose=False, ): - self._col_name = col_name self._label = label self._value = value @@ -161,7 +158,6 @@ def __init__( self.apply(dat) def __repr__(self): - return ( f"Mask(col_name={self._col_name}, label={self._label}, kind={self._kind}," + f" value={self._value})" @@ -169,7 +165,6 @@ def __repr__(self): @classmethod def from_list(cls, masks, label="combined", verbose=False): - if verbose: print(f"Combining {len(masks)} masks") @@ -180,7 +175,6 @@ def from_list(cls, masks, label="combined", verbose=False): return my_mask def apply(self, dat): - if self._kind == "not_equal_2bands": self._mask = apply_condition( dat[self._col_name], "not_equal", self._value @@ -189,7 +183,6 @@ def apply(self, dat): self._mask = apply_condition(dat[self._col_name], self._kind, self._value) def to_bool(self, hsp_mask): - if self._verbose: print("to_bool: get valid pixels") valid_pixels = hsp_mask.valid_pixels @@ -224,7 +217,6 @@ def print_stats(self, num_obj, f_out=None): self.print_strings(self._col_name, self._label, si, sf, f_out=f_out) def get_sign(self, latex=False): - sign = None if self._kind == "equal": sign = "$=$" if latex else "=" @@ -237,7 +229,6 @@ def get_sign(self, latex=False): return sign def print_condition(self, f_out, latex=False): - if self._value is None: return "" @@ -284,7 +275,6 @@ def create_descr(self): # Create description for FITS header def add_summary_to_FITS_header(self, header): - header_new = fits.Header() self.create_descr() @@ -311,19 +301,18 @@ def print_mask_stats(num_obj, masks, mask_combined): mask_combined.print_stats(num_obj) -def get_masks_from_config(config, dat, dat_ext, masks_to_apply=None, verbose=False): +def get_masks_from_config(config, dat, masks_to_apply=None, verbose=False): """Get Masks From Config. - Return mask information from yaml config structure. + Return the masks of the config's ``dat`` cut list, evaluated on ``dat``. Parameters ---------- config : dict config information - dat : numpy.ndarray - input data - det_ext : numpy.ndarray - input extended data + dat : numpy.ndarray or grammar.V2View + input catalogue, in the v2 column grammar (e.g. from + ``CalibrateCat.read_cat``) masks_to_apply: list, optional masks to apply exclusively; if `None` (default), use all masks verbose : bool, optional @@ -343,40 +332,22 @@ def get_masks_from_config(config, dat, dat_ext, masks_to_apply=None, verbose=Fal # Dict to associate labels with index in mask list labels = {} - # Loop over mask sections from config file - config_data = {key: config[key] for key in ["dat", "dat_ext"] if key in config} - idx = 0 - for section, mask_list in config_data.items(): - # Set data source - dat_source = dat if section == "dat" else dat_ext - - # Loop over mask information in this section - for mask_params in mask_list: - use_this_mask = False - if masks_to_apply is not None: - if mask_params["col_name"] in masks_to_apply: - use_this_mask = True - else: - use_this_mask = True - - if use_this_mask: - # Ensure 'range' kind has exactly two values - value = mask_params["value"] - if mask_params["kind"] == "range" and ( - not isinstance(value, list) or len(value) != 2 - ): - raise ValueError( - f"Range kind requires a list of two values, got {value}" - ) - - # Create mask instance and append to list - my_mask = Mask(**mask_params, dat=dat_source, verbose=verbose) - masks.append(my_mask) - labels[my_mask._col_name] = idx - idx += 1 - else: - if verbose: - print(f"Skipping mask {mask_params['col_name']}") - continue + for mask_params in config.get("dat", []): + if masks_to_apply is not None and mask_params["col_name"] not in masks_to_apply: + if verbose: + print(f"Skipping mask {mask_params['col_name']}") + continue + + # Ensure 'range' kind has exactly two values + value = mask_params["value"] + if mask_params["kind"] == "range" and ( + not isinstance(value, list) or len(value) != 2 + ): + raise ValueError(f"Range kind requires a list of two values, got {value}") + + # Create mask instance and append to list + my_mask = Mask(**mask_params, dat=dat, verbose=verbose) + labels[my_mask._col_name] = len(masks) + masks.append(my_mask) return masks, labels diff --git a/src/sp_validation/plots.py b/src/sp_validation/plots.py index 60b5b2b4..18a8a826 100644 --- a/src/sp_validation/plots.py +++ b/src/sp_validation/plots.py @@ -305,7 +305,6 @@ def plot_binned_one( xlabel=None, ylabel=None, ): - # Note: transpose R slice to match (y, x) shape required by pcolormesh pcm = ax.pcolormesh( bin_edges_x, bin_edges_y, quantity, vmin=vmin, vmax=vmax, shading="auto" @@ -332,7 +331,6 @@ def plot_binned( xlabel=None, ylabel=None, ): - len_shape = len(quantities[key].shape) fig_size = 2 * len_shape @@ -478,9 +476,8 @@ def sky_plots(dat, masks, labels, zoom_ra, zoom_dec): # No mask plot_area_mask(ra, dec, zoom) - # SExtractor and SP flags. ShapePipe v2 splits the single IMAFLAGS_ISO - # mask into per-reason MASK_n columns, so combine whichever of them - # the mask config declared. + # SExtractor and SP flags: whichever mask columns the config declared + # (the MASK_n columns, and the v1 configs' IMAFLAGS_ISO). m_flags = masks[labels["FLAGS"]]._mask for col in ("IMAFLAGS_ISO",) + tuple(MASK_COLUMNS): if col in labels: @@ -491,8 +488,7 @@ def sky_plots(dat, masks, labels, zoom_ra, zoom_dec): m_over = masks[labels["overlap"]]._mask & m_flags plot_area_mask(ra, dec, zoom, mask=m_over) - # Coverage mask - # Rough pointing coverage; v2 encodes coverage in the MASK_n* columns + # Rough pointing coverage, where the catalogue carries it m_point = m_over if "npoint3" in labels: m_point = masks[labels["npoint3"]]._mask & m_over @@ -500,9 +496,8 @@ def sky_plots(dat, masks, labels, zoom_ra, zoom_dec): # Maximask m_maxi = m_point - for maxi_key in ("1024_Maximask", "MASK_n1024"): - if maxi_key in labels: - m_maxi = masks[labels[maxi_key]]._mask & m_point + if "MASK_n1024" in labels: + m_maxi = masks[labels["MASK_n1024"]]._mask & m_point plot_area_mask(ra, dec, zoom, mask=m_maxi) # Combined mask over all supplied masks (was passed in by the caller before @@ -514,12 +509,8 @@ def sky_plots(dat, masks, labels, zoom_ra, zoom_dec): m_comb = mask_combined._mask plot_area_mask(ra, dec, zoom, mask=m_comb) - m_man = m_maxi & masks[labels["8_Manual"]]._mask + m_man = m_maxi & masks[labels["MASK_n8"]]._mask plot_area_mask(ra, dec, zoom, mask=m_man) - m_halos = ( - m_maxi - & masks[labels["1_Faint_star_halos"]]._mask - & masks[labels["2_Bright_star_halos"]]._mask - ) + m_halos = m_maxi & masks[labels["MASK_n1"]]._mask & masks[labels["MASK_n2"]]._mask plot_area_mask(ra, dec, zoom, mask=m_halos) diff --git a/src/sp_validation/tests/test_campaign_readers.py b/src/sp_validation/tests/test_campaign_readers.py index 01f789f2..ccb535bd 100644 --- a/src/sp_validation/tests/test_campaign_readers.py +++ b/src/sp_validation/tests/test_campaign_readers.py @@ -352,8 +352,8 @@ def test_missing_column_raises(self): galaxy.mask_cut(self._dat, ["MASK_n16"]) self.assertIn("MASK_n16", str(ctx.exception)) - def test_v1_catalogue_raises(self): - dat = np.zeros(3, dtype=[("IMAFLAGS_ISO", "i2")]) + def test_catalogue_without_mask_columns_raises(self): + dat = np.zeros(3, dtype=[("FLAGS", "i2")]) with self.assertRaises(KeyError): galaxy.mask_cut(dat) diff --git a/src/sp_validation/tests/test_grammar_calibration.py b/src/sp_validation/tests/test_grammar_calibration.py new file mode 100644 index 00000000..473992fd --- /dev/null +++ b/src/sp_validation/tests/test_grammar_calibration.py @@ -0,0 +1,273 @@ +"""A v1 comprehensive catalogue calibrates exactly like its v2 twin. + +One synthetic catalogue is written twice: as a post-processed ShapePipe v1 +comprehensive HDF5 (``data`` in the v1 grammar, ``data_ext`` holding the +healsparse mask bits under their ``{b}_{label}`` names) and as a ShapePipe v2 +catalogue (one ``data`` dataset, v2 grammar, ``MASK_n{b}``). Read through the +readers the pipeline uses, the two must give identical mask selections, for +``galaxy.mask_cut`` and for every mask config, and identical metacal inputs. +Also covers the campaign and star readers on v1-grammar files. +""" + +from pathlib import Path + +import h5py +import numpy as np +import pytest +import yaml +from astropy.io import fits +from cs_util.size import T_to_sigma + +from sp_validation import catalog, galaxy, grammar +from sp_validation.calibration import metacal +from sp_validation.catalog_builders import CalibrateCat, JointCat +from sp_validation.masks import Mask, get_masks_from_config + +N = 4000 +CONFIG_DIR = Path(__file__).resolve().parents[3] / "config" / "calibration" +BITS = (1, 2, 4, 8, 64, 1024) # the bits v1 post-processing wrote + + +def _twins(): + """Return (v1 data, v1 data_ext, v2 data) holding the same catalogue.""" + rng = np.random.default_rng(11) + common = { + "RA": rng.uniform(150, 160, N), + "Dec": rng.uniform(30, 40, N), + "FLAGS": (rng.random(N) < 0.1).astype(np.int16) * 2, + "overlap": rng.random(N) < 0.95, + "N_EPOCH": rng.integers(1, 8, N).astype(np.int16), + "mag": rng.uniform(14.5, 30.5, N), + "NGMIX_N_EPOCH": rng.integers(0, 6, N).astype(np.int16), + "NGMIX_MCAL_FLAGS": (rng.random(N) < 0.05).astype(np.int32), + "w_iv": rng.uniform(0.5, 2, N), + # passes through unmapped in both + "IMAFLAGS_ISO": (rng.random(N) < 0.1).astype(np.int16) * 2, + } + v1 = dict(common) + v2 = dict(common) + fail = (rng.random(N) < 0.05).astype(np.int16) + v1["NGMIX_MOM_FAIL"] = v2["NGMIX_MCAL_TYPES_FAIL"] = fail + for shear in grammar.SHEARS: + g = rng.normal(0, 0.3, (2, N)) + err = rng.uniform(0.05, 0.2, (2, N)) + Tpsf = rng.uniform(0.3, 0.6, N) + for i in (0, 1): + v1[f"NGMIX_ELL_{shear}_{i}"] = v2[f"NGMIX_G{i + 1}_{shear}"] = g[i] + v1[f"NGMIX_ELL_ERR_{shear}_{i}"] = v2[f"NGMIX_G{i + 1}_ERR_{shear}"] = err[ + i + ] + v1[f"NGMIX_Tpsf_{shear}"] = v2[f"NGMIX_T_PSF_RECONV_{shear}"] = Tpsf + columns = { + "T": Tpsf * rng.uniform(0.5, 3.5, N), + "T_ERR": rng.uniform(0.01, 0.1, N), + "FLUX": rng.uniform(5, 400, N), + "FLUX_ERR": rng.uniform(0.5, 2, N), + "FLAGS": (rng.random(N) < 0.03).astype(np.int32), + } + for key, values in columns.items(): + v1[f"NGMIX_{key}_{shear}"] = v2[f"NGMIX_{key}_{shear}"] = values + psf = rng.normal(0, 0.02, (2, N)) + psf[:, rng.random(N) < 0.02] = -10 + for i in (0, 1): + v1[f"NGMIX_ELL_PSFo_NOSHEAR_{i}"] = v2[f"NGMIX_G{i + 1}_PSF_ORIG_NOSHEAR"] = ( + psf[i] + ) + sigma = rng.uniform(0.3, 0.5, N) + v1["SIGMA_PSF_HSM"] = sigma + v2["HSM_T_PSF"] = 2 * sigma**2 + + ext = {} + for bit in BITS: + flag = rng.random(N) < 0.08 + ext[f"{bit}_{grammar.MASK_LABELS[bit]}"] = flag + v2[f"MASK_n{bit}"] = flag + npoint = rng.integers(0, 6, N).astype(np.int16) + ext["npoint3"] = v2["npoint3"] = npoint + return _structured(v1), _structured(ext), _structured(v2) + + +def _structured(columns): + out = np.empty(N, dtype=[(k, np.asarray(v).dtype) for k, v in columns.items()]) + for key, value in columns.items(): + out[key] = value + return out + + +@pytest.fixture +def comprehensive(tmp_path): + """Read the v1 and v2 comprehensive files through ``CalibrateCat``.""" + v1, ext, v2 = _twins() + with h5py.File(tmp_path / "v1.hdf5", "w") as handle: + handle.create_dataset("data", data=v1) + handle.create_dataset("data_ext", data=ext) + with h5py.File(tmp_path / "v2.hdf5", "w") as handle: + handle.create_dataset("data", data=v2) + + tables, readers = {}, [] + for label in ("v1", "v2"): + reader = CalibrateCat() + reader._params["input_path"] = str(tmp_path / f"{label}.hdf5") + tables[label] = reader.read_cat() + readers.append(reader) + yield tables + for reader in readers: + reader._hd5file.close() + + +def test_v1_comprehensive_presents_the_v2_columns(comprehensive): + v1, v2 = comprehensive["v1"], comprehensive["v2"] + assert isinstance(v1, grammar.V2View) + assert set(v1.dtype.names) == set(v2.dtype.names) + for name in v2.dtype.names: + np.testing.assert_allclose(v1[name], v2[name], rtol=1e-14, err_msg=name) + + +def test_mask_cut_is_identical(comprehensive): + v1, v2 = comprehensive["v1"], comprehensive["v2"] + for columns in (None, [f"MASK_n{bit}" for bit in BITS], ["MASK_n4"]): + np.testing.assert_array_equal( + galaxy.mask_cut(v1, columns), galaxy.mask_cut(v2, columns) + ) + assert 0 < galaxy.mask_cut(v1).sum() < N + + +@pytest.mark.parametrize( + "config_name", + sorted( + path.name + for path in CONFIG_DIR.glob("mask_*.yaml") + if not path.name.endswith("overlay.yaml") + ), +) +def test_config_selection_is_identical(comprehensive, config_name): + config = yaml.safe_load((CONFIG_DIR / config_name).read_text()) + available = set(comprehensive["v2"].dtype.names) + cuts = [cut for cut in config["dat"] if cut["col_name"] in available] + if not cuts: + pytest.skip(f"{config_name} cuts on no column of the twin catalogue") + config = {"dat": cuts} + + combined = {} + for label, table in comprehensive.items(): + masks, labels = get_masks_from_config(config, table) + for mask in masks: + assert mask._mask.shape == (N,) + combined[label] = ( + Mask.from_list(masks)._mask, + [m._mask for m in masks], + labels, + ) + assert combined["v1"][2] == combined["v2"][2] + for mask_v1, mask_v2 in zip(combined["v1"][1], combined["v2"][1], strict=True): + np.testing.assert_array_equal(mask_v1, mask_v2) + np.testing.assert_array_equal(combined["v1"][0], combined["v2"][0]) + + +def test_v1_configs_cut_every_mask_column_they_name(comprehensive): + """The v1 configs' mask cuts all resolve on a v1 comprehensive file.""" + names = set(comprehensive["v1"].dtype.names) + for path in CONFIG_DIR.glob("mask_v1.X.*.yaml"): + if path.name.endswith("overlay.yaml"): + continue + config = yaml.safe_load(path.read_text()) + for cut in config["dat"]: + if cut["col_name"].startswith("MASK_n"): + assert cut["col_name"] in names, (path.name, cut["col_name"]) + + +def test_metacal_inputs_are_identical(comprehensive): + config = yaml.safe_load((CONFIG_DIR / "mask_v1.X.6.yaml").read_text()) + cm = config["metacal"] + results = {} + for label, table in comprehensive.items(): + masks, _ = get_masks_from_config(config, table) + selection = Mask.from_list(masks)._mask + results[label] = metacal( + table, + selection, + snr_min=cm["gal_snr_min"], + snr_max=cm["gal_snr_max"], + rel_size_min=cm["gal_rel_size_min"], + rel_size_max=cm["gal_rel_size_max"], + size_corr_ell=cm["gal_size_corr_ell"], + sigma_eps=cm["sigma_eps_prior"], + global_R_weight=None, + ) + mc_v1, mc_v2 = results["v1"], results["v2"] + assert mc_v1._n_input == mc_v2._n_input > 0 + assert np.all(np.isfinite(mc_v1.R)) + for attr in ("m1", "p1", "m2", "p2", "ns"): + d1, d2 = getattr(mc_v1, attr), getattr(mc_v2, attr) + assert d1.keys() == d2.keys() + for key in d1: + np.testing.assert_array_equal(d1[key], d2[key], err_msg=f"{attr}.{key}") + np.testing.assert_array_equal(mc_v1.R, mc_v2.R) + + +def test_load_into_memory_matches_the_view(comprehensive, tmp_path): + reader = CalibrateCat() + reader._params["input_path"] = str(tmp_path / "v1.hdf5") + loaded = reader.read_cat(load_into_memory=True) + reader._hd5file.close() + assert isinstance(loaded, np.ndarray) + for name in loaded.dtype.names: + np.testing.assert_array_equal(loaded[name], comprehensive["v1"][name]) + + +# -- campaign and star readers on v1-grammar files ---------------------------- + + +def _write_campaign(path, tiles): + with h5py.File(path, "w") as handle: + group = handle.create_group("patches").create_group("P3") + for tile_id, dat in tiles.items(): + group.create_dataset(tile_id, data=dat) + handle.attrs["n_tiles"] = len(tiles) + + +def test_campaign_reader_presents_v1_tiles_in_v2(tmp_path): + v1, _, v2 = _twins() + _write_campaign( + tmp_path / "final_cat_P3.hdf5", {"000.000": v1[:1500], "001.000": v1[1500:]} + ) + wanted = ["RA", "NGMIX_G1_NOSHEAR", "NGMIX_MCAL_TYPES_FAIL", "HSM_T_PSF"] + dat = catalog.read_campaign_catalogue( + str(tmp_path / "final_cat_P3.hdf5"), param_list=wanted, verbose=False + ) + assert dat.dtype.names == tuple(wanted) + for name in wanted: + np.testing.assert_allclose(dat[name], v2[name], rtol=1e-14) + + n_rows, dtype = catalog.campaign_shape( + str(tmp_path / "final_cat_P3.hdf5"), param_list=wanted + ) + assert n_rows == N and dtype.names == tuple(wanted) + + merger = JointCat() + merger._params["verbose"] = False + merger._params["param_path"] = None + merged = merger.merge_catalogues([str(tmp_path / "final_cat_P3.hdf5")]) + np.testing.assert_array_equal(merged["NGMIX_G2_ERR_1P"], v2["NGMIX_G2_ERR_1P"]) + assert "NGMIX_ELL_1P_0" not in merged.dtype.names + + +def test_star_reader_presents_v1_fits_in_v2(tmp_path): + rng = np.random.default_rng(5) + T = rng.uniform(0.3, 0.6, 50) + v1 = np.zeros( + 50, + dtype=[ + ("RA", "f8"), + ("DEC", "f8"), + ("E1_STAR_HSM", "f8"), + ("SIGMA_STAR_HSM", "f8"), + ], + ) + v1["E1_STAR_HSM"] = rng.normal(0, 0.05, 50) + v1["SIGMA_STAR_HSM"] = T_to_sigma(T) + fits.BinTableHDU(v1).writeto(tmp_path / "stars.fits") + stars = catalog.read_star_catalogue(str(tmp_path / "stars.fits"), verbose=False) + assert isinstance(stars, np.ndarray) + np.testing.assert_allclose(stars["HSM_T_STAR"], T, rtol=1e-12) + np.testing.assert_array_equal(stars["HSM_G1_STAR"], v1["E1_STAR_HSM"]) From 2194a4bc2cb0ce843c0879de34fabfaefb8e25f1 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 06:18:41 +0200 Subject: [PATCH 26/39] uv.lock: shear_psf_leakage develop@00e38c4 (loaded-catalogue rho/tau builders, seeded patches) Pinned below develop's tip 372980f, whose scipy>=1.18 requirement cannot resolve against cs_util<0.3. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LsAQRxNg6yDsUX4y2SWxTJ --- uv.lock | 8 +------- 1 file changed, 1 insertion(+), 7 deletions(-) diff --git a/uv.lock b/uv.lock index 1460cbc9..ff624753 100644 --- a/uv.lock +++ b/uv.lock @@ -3584,28 +3584,22 @@ wheels = [ [[package]] name = "shear-psf-leakage" version = "0.2.1" -source = { git = "https://github.com/CosmoStat/shear_psf_leakage.git?rev=develop#0b3c17523b77760c9259be15ba5bbba99d575d0a" } +source = { git = "https://github.com/CosmoStat/shear_psf_leakage.git?rev=develop#00e38c421a3a3a16b26934f4c8cf5c388567f406" } dependencies = [ { name = "camb" }, { name = "cs-util" }, { name = "emcee" }, { name = "getdist" }, { name = "gsl" }, - { name = "jupyter" }, - { name = "jupyter-server" }, - { name = "jupyterlab" }, - { name = "jupytext" }, { name = "lenspack" }, { name = "lmfit" }, { name = "matplotlib" }, - { name = "notebook" }, { name = "pandas" }, { name = "pyccl" }, { name = "pyyaml" }, { name = "scipy" }, { name = "stats" }, { name = "swig" }, - { name = "tornado" }, { name = "tqdm" }, { name = "treecorr" }, { name = "uncertainties" }, From 34c497bedaccc13f58b38c386cc8b25d898dedbc Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 06:32:30 +0200 Subject: [PATCH 27/39] grammar: v1 no-shear reconvolved-PSF size from NGMIX_Tpsf_1P ShapePipe v1 wrote a wrong NGMIX_Tpsf_NOSHEAR: it differs by ~2% for nearly every object (v1.4, v1.5, v1.6 comprehensive) from the reconvolution kernel metacal applied, which NGMIX_Tpsf_{1P,1M,2P,2M} record (they agree to ~1e-5). The v1.4.6.3 release used the 1P value in its metacal size cut and wrote it as the no-shear column; #267 dropped that substitution, which is a no-op only on the v2 stack, so the branch's v1 calibration selected ~3% more objects than the release. The adapter now presents NGMIX_T_PSF_RECONV_NOSHEAR from NGMIX_Tpsf_1P when the table has it, falling back to NGMIX_Tpsf_NOSHEAR (a cut catalogue such as v1.4.6.3's already holds the 1P value there), and hides the raw column. Rules gain a fallback source; a derived column is presented where the first of its sources sits. The module docstring and the migration doc state the rule: name mapping, except this one documented v1 defect. Every consumer now sees the 1P value, including the w_des and PSF-leakage size-ratio binning, where the release used the raw value; those two columns therefore do not reproduce the release bit for bit. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LsAQRxNg6yDsUX4y2SWxTJ --- docs/ngmix_psf_column_migration.md | 25 +++++-- src/sp_validation/grammar.py | 67 ++++++++++++++----- src/sp_validation/tests/test_grammar.py | 32 +++++++++ .../tests/test_grammar_calibration.py | 3 + 4 files changed, 105 insertions(+), 22 deletions(-) diff --git a/docs/ngmix_psf_column_migration.md b/docs/ngmix_psf_column_migration.md index ba6caf23..171d9e27 100644 --- a/docs/ngmix_psf_column_migration.md +++ b/docs/ngmix_psf_column_migration.md @@ -13,13 +13,15 @@ grammar-neutral tables unchanged. `detect_generation` tells the grammars apart from column names, and `v2_names` maps a header's names without reading data. The adapter maps by *name*, so a v1 `*_PSFo` column is presented as `*_PSF_ORIG` holding the v1 (reconvolved-alias) values; downstream code never -branches on the generation. The tables below (Old = v1, New = v2) are the rule -table in `grammar.V1_RULES`. +branches on the generation. The one exception is a documented v1 defect, which +the adapter corrects rather than renames: `NGMIX_T_PSF_RECONV_NOSHEAR` is read +from `NGMIX_Tpsf_1P` (see the reconvolved-PSF table). The tables below +(Old = v1, New = v2) are the rule table in `grammar.V1_RULES`. Mask columns follow a separate rule family, applied whatever the generation. The UNIONS healsparse mask product is one bitmask; ShapePipe v2 and `catalog_builders.ApplyHspMasks` both write bit `b` as the boolean `MASK_n{b}`, -and comprehensive HDF5 files written before that name carry it in `data_ext` as +and the `data_ext` dataset of a v1 comprehensive HDF5 carries it as `{b}_{label}` (`grammar.MASK_LABELS`), which the adapter renames: | bit | old `data_ext` name | meaning | @@ -86,7 +88,22 @@ column is also carried through `params.add_cols_pre_cal` and every | Old | New | Note | |---|---|---| -| `NGMIX_Tpsf_{shear}` | `NGMIX_T_PSF_RECONV_{shear}` | same value (the `T/Tpsf` size-ratio cut) | +| `NGMIX_Tpsf_{shear}` | `NGMIX_T_PSF_RECONV_{shear}` | same value (the `T/Tpsf` size-ratio cut), for `{shear}` ≠ `NOSHEAR` | +| `NGMIX_Tpsf_1P`, else `NGMIX_Tpsf_NOSHEAR` | `NGMIX_T_PSF_RECONV_NOSHEAR` | **v1 defect corrected** (below) | + +ShapePipe v1 wrote a wrong `NGMIX_Tpsf_NOSHEAR`: in the v1.4, v1.5 and v1.6 +comprehensive catalogues it differs by ~2% for nearly every object from the +reconvolution kernel metacal applied, which the sheared types' +`NGMIX_Tpsf_{1P,1M,2P,2M}` record (they agree to ~1e-5). The v1.4.6.3 release +used the `1P` value in its metacal size cut and wrote it as +`NGMIX_Tpsf_NOSHEAR` (keeping the raw one as `NGMIX_Tpsf_NOSHEAR_orig`), so the +adapter reads `NGMIX_Tpsf_1P` whenever a v1 table has it and falls back to +`NGMIX_Tpsf_NOSHEAR` for a cut catalogue such as v1.4.6.3's, whose no-shear +column already holds the `1P` value. The release's `w_des` and PSF-leakage +binning used the raw size, so a calibration through the adapter reproduces the +release's selection and per-object columns but not those two bit for bit. +ShapePipe v2 reuses one reconvolved PSF across metacal types and needs no +correction. ### ngmix — original PSF (value change — shapepipe#749 fix — *not a code blocker*) diff --git a/src/sp_validation/grammar.py b/src/sp_validation/grammar.py index 9cc04582..43e7fd9a 100644 --- a/src/sp_validation/grammar.py +++ b/src/sp_validation/grammar.py @@ -22,7 +22,8 @@ into ``NGMIX_G{1,2}[_ERR|_PSF_ORIG]_{shear}``; - ``NGMIX_T_PSFo_{shear}``, ``NGMIX_Tpsf_{shear}`` and ``NGMIX_MOM_FAIL`` become ``NGMIX_T_PSF_ORIG_{shear}``, ``NGMIX_T_PSF_RECONV_{shear}`` and - ``NGMIX_MCAL_TYPES_FAIL``. + ``NGMIX_MCAL_TYPES_FAIL``, except that ``NGMIX_T_PSF_RECONV_NOSHEAR`` is + read from ``NGMIX_Tpsf_1P`` when the table has it (see below). - The UNIONS mask-bit columns, applied to any table: ``{b}_{label}`` (``1_Faint_star_halos``, ``4_Stars``, ``2048_z2``, ...), the names under @@ -40,9 +41,20 @@ something different from their v2 namesakes: v1 ``NGMIX_*_PSFo_*`` hold the reconvolved-PSF alias rather than a fit to the original PSF, and ``NGMIX_MOM_FAIL`` counts moments-guess failures rather than failed metacal -types. Reproducing a v1 calibration needs exactly the v1 values under the names -the code reads, so the view presents them unchanged. Every other column passes -through under its own name; source names that have a v2 equivalent are hidden. +types. Those values are presented unchanged. + +One v1 value is corrected rather than renamed. ShapePipe v1 wrote a wrong +no-shear reconvolved-PSF size: ``NGMIX_Tpsf_NOSHEAR`` differs by ~2% for +nearly every object from the reconvolution kernel metacal applied, which the +sheared types' ``NGMIX_Tpsf_{1P,1M,2P,2M}`` record (they agree to ~1e-5). +The v1.4.6.3 release cut on and wrote the ``1P`` value, so +``NGMIX_T_PSF_RECONV_NOSHEAR`` reads ``NGMIX_Tpsf_1P`` whenever the table has +it, and ``NGMIX_Tpsf_NOSHEAR`` otherwise (a cut catalogue such as v1.4.6.3's, +whose no-shear column already holds the ``1P`` value). ShapePipe v2 reuses one +reconvolved PSF across metacal types, so its columns need no correction. + +Every other column passes through under its own name; source names that have a +v2 equivalent are hidden. ``adapt`` also joins row-aligned tables into one view, as the comprehensive HDF5 splits one catalogue over its ``data`` and ``data_ext`` datasets. @@ -98,17 +110,21 @@ class Rule: ``kind`` is ``"rename"`` (same values), ``"sigma_to_T"`` (T = 2 sigma^2) or ``"component"`` (element ``arg`` of a 2-vector, stored either as one vector column ``source`` or as the flattened ``{source}_{arg}``). + ``fallback`` names a column read instead when ``source`` is absent. """ v2: str source: str kind: str = "rename" arg: int | None = None + fallback: str | None = None def sources(self): - """Return the source names this rule can read from.""" + """Return the source names this rule can read from, in preference order.""" if self.kind == "component": return (self.source, f"{self.source}_{self.arg}") + if self.fallback is not None: + return (self.source, self.fallback) return (self.source,) def apply(self, column): @@ -145,10 +161,18 @@ def _v1_rules(): i, ), ] - rules += [ - Rule(f"NGMIX_T_PSF_ORIG_{shear}", f"NGMIX_T_PSFo_{shear}"), - Rule(f"NGMIX_T_PSF_RECONV_{shear}", f"NGMIX_Tpsf_{shear}"), - ] + rules.append(Rule(f"NGMIX_T_PSF_ORIG_{shear}", f"NGMIX_T_PSFo_{shear}")) + if shear == "NOSHEAR": + # v1 wrote a wrong no-shear size; see the module docstring. + rules.append( + Rule( + "NGMIX_T_PSF_RECONV_NOSHEAR", + "NGMIX_Tpsf_1P", + fallback="NGMIX_Tpsf_NOSHEAR", + ) + ) + else: + rules.append(Rule(f"NGMIX_T_PSF_RECONV_{shear}", f"NGMIX_Tpsf_{shear}")) rules.append(Rule("NGMIX_MCAL_TYPES_FAIL", "NGMIX_MOM_FAIL")) return tuple(rules) @@ -206,24 +230,31 @@ def _resolve(names): rules = MASK_RULES if detect_generation(names) == "v1": rules = V1_RULES + MASK_RULES - present = set(names) + position = {name: i for i, name in enumerate(names)} derived = {} - by_source = {} + consumed = set() + by_anchor = {} for rule in rules: - source = next((s for s in rule.sources() if s in present), None) - if source is None: + sources = [s for s in rule.sources() if s in position] + if not sources: continue - if rule.v2 in present: + if rule.v2 in position: raise ValueError( - f"catalogue carries both {source!r} and {rule.v2!r}," + f"catalogue carries both {sources[0]!r} and {rule.v2!r}," + " two names for the same column" ) - derived[rule.v2] = (rule, source) - by_source.setdefault(source, []).append(rule.v2) + derived[rule.v2] = (rule, sources[0]) + consumed.update(sources) + # Presented where the first of its sources sits in the table. + anchor = min(sources, key=position.get) + by_anchor.setdefault(anchor, []).append(rule.v2) presented = [] for name in names: - presented += by_source.get(name, [name]) + if name in by_anchor: + presented += by_anchor[name] + elif name not in consumed: + presented.append(name) return tuple(presented), derived diff --git a/src/sp_validation/tests/test_grammar.py b/src/sp_validation/tests/test_grammar.py index 716f7df4..265fe661 100644 --- a/src/sp_validation/tests/test_grammar.py +++ b/src/sp_validation/tests/test_grammar.py @@ -83,6 +83,9 @@ def _twins(vector_ell=True): T_orig, T_reconv = rng.uniform(0.3, 0.8, (2, N)) v2[f"NGMIX_T_PSF_ORIG_{shear}"] = v1[f"NGMIX_T_PSFo_{shear}"] = T_orig v2[f"NGMIX_T_PSF_RECONV_{shear}"] = v1[f"NGMIX_Tpsf_{shear}"] = T_reconv + # v1's no-shear reconvolved-PSF size is wrong; the view reads the 1P kernel. + v1["NGMIX_Tpsf_NOSHEAR"] = 1.02 * v1["NGMIX_Tpsf_1P"] + v2["NGMIX_T_PSF_RECONV_NOSHEAR"] = v1["NGMIX_Tpsf_1P"] fail = rng.integers(0, 2, N).astype(np.int16) v2["NGMIX_MCAL_TYPES_FAIL"] = v1["NGMIX_MOM_FAIL"] = fail v1["IMAFLAGS_ISO"] = rng.integers(0, 256, N).astype(np.int16) @@ -408,3 +411,32 @@ def spy(table, name, start, stop): assert reads[-1] == (200, 301) view[10:20:3]["MASK_n1"] assert reads[-1] == (10, 20) + + +def test_v1_no_shear_reconv_psf_reads_the_1P_kernel(): + """v1 NGMIX_Tpsf_NOSHEAR is wrong: RECONV_NOSHEAR reads Tpsf_1P when present.""" + names = [f"NGMIX_Tpsf_{shear}" for shear in SHEARS] + table = np.zeros(4, dtype=[(name, "f8") for name in names]) + for i, name in enumerate(names): + table[name] = 0.5 + i + view = adapt(table) + np.testing.assert_array_equal( + view["NGMIX_T_PSF_RECONV_NOSHEAR"], table["NGMIX_Tpsf_1P"] + ) + np.testing.assert_array_equal(view["NGMIX_T_PSF_RECONV_1P"], table["NGMIX_Tpsf_1P"]) + assert "NGMIX_Tpsf_NOSHEAR" not in view + assert view.names[0] == "NGMIX_T_PSF_RECONV_NOSHEAR" + assert sorted(view.names) == sorted( + f"NGMIX_T_PSF_RECONV_{shear}" for shear in SHEARS + ) + + # A v1 cut catalogue carries only the (already corrected) no-shear size. + cut = np.zeros( + 3, dtype=[("NGMIX_Tpsf_NOSHEAR", "f8"), ("NGMIX_Tpsf_NOSHEAR_orig", "f8")] + ) + cut["NGMIX_Tpsf_NOSHEAR"], cut["NGMIX_Tpsf_NOSHEAR_orig"] = 0.4, 0.41 + view = adapt(cut) + assert view.names == ("NGMIX_T_PSF_RECONV_NOSHEAR", "NGMIX_Tpsf_NOSHEAR_orig") + np.testing.assert_array_equal( + view["NGMIX_T_PSF_RECONV_NOSHEAR"], cut["NGMIX_Tpsf_NOSHEAR"] + ) diff --git a/src/sp_validation/tests/test_grammar_calibration.py b/src/sp_validation/tests/test_grammar_calibration.py index 473992fd..0467363a 100644 --- a/src/sp_validation/tests/test_grammar_calibration.py +++ b/src/sp_validation/tests/test_grammar_calibration.py @@ -67,6 +67,9 @@ def _twins(): } for key, values in columns.items(): v1[f"NGMIX_{key}_{shear}"] = v2[f"NGMIX_{key}_{shear}"] = values + # v1's no-shear reconvolved-PSF size is wrong; the reader uses the 1P kernel. + v2["NGMIX_T_PSF_RECONV_NOSHEAR"] = v1["NGMIX_Tpsf_1P"] + v1["NGMIX_Tpsf_NOSHEAR"] = 1.02 * v1["NGMIX_Tpsf_1P"] psf = rng.normal(0, 0.02, (2, N)) psf[:, rng.random(N) < 0.02] = -10 for i in (0, 1): From f7aea5f98ed02152b8ea5465b6b7f01c24ef4054 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 06:32:39 +0200 Subject: [PATCH 28/39] grammar: select rows of an HDF5-backed view in one pass; cache its dtype metacal reads ~40 columns of data[mask]. Over h5py datasets the lazy view read each column over the whole span of the selection, so a full-sky v1 calibration re-read the 238 GB comprehensive file once per column. Selecting rows of a view over h5py Datasets now reads the selected rows of every column in one pass per dataset (whole rows, in 256 MB blocks over the selection's span, skipping empty blocks), as indexing the Dataset did on develop, and wraps them in a view over the in-memory arrays so renames and derived columns still apply. Views over in-memory tables stay lazy. to_structured reads each dataset once for all requested fields. On a cold 10M-row window of v1.4.c (4.8M rows selected): h5py Dataset[mask] 19.2 s; adapt(data, data_ext)[mask] plus the 40 metacal columns 23.5 s; the previous per-column path took 16.5 s for its first column alone, i.e. one full re-read per column. dtype is computed once per view and passed to row selections (group_dtype asked for it per column, quadratically). view[()] and view[...] select every row, as for an h5py Dataset; other tuples raise. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LsAQRxNg6yDsUX4y2SWxTJ --- src/sp_validation/grammar.py | 146 ++++++++++++++++++++---- src/sp_validation/tests/test_grammar.py | 83 ++++++++++---- 2 files changed, 186 insertions(+), 43 deletions(-) diff --git a/src/sp_validation/grammar.py b/src/sp_validation/grammar.py index 43e7fd9a..ae8b7061 100644 --- a/src/sp_validation/grammar.py +++ b/src/sp_validation/grammar.py @@ -72,8 +72,8 @@ SHEARS = ("NOSHEAR", "1P", "1M", "2P", "2M") #: Bits of the UNIONS healsparse mask product and what each flags. The column -#: for bit ``b`` is ``MASK_n{b}``; files written before that name carry -#: ``{b}_{label}``. +#: for bit ``b`` is ``MASK_n{b}``; the ``data_ext`` dataset of a v1 +#: comprehensive HDF5 carries it as ``{b}_{label}``. MASK_LABELS = { 1: "Faint_star_halos", 2: "Bright_star_halos", @@ -266,6 +266,10 @@ def v2_names(names): return _resolve(tuple(names))[0] +#: Bytes of whole HDF5 rows read per block when selecting rows of a dataset. +BLOCK_BYTES = 256 * 1024**2 + + def _read_window(table, name, start, stop): """Read rows ``start:stop`` of column ``name``, and only those.""" if isinstance(table, h5py.Dataset): @@ -273,21 +277,85 @@ def _read_window(table, name, start, stop): return table[name][start:stop] +def _take(table, rows, fields=None): + """Return rows ``rows`` of ``table`` as an in-memory structured array. + + ``rows`` is ``None`` (all rows), a ``range`` or an integer index array; + ``fields`` restricts the output to those columns. An h5py Dataset is read + in one pass: whole rows in blocks of ``BLOCK_BYTES`` over the span of the + selection, skipping blocks that hold no selected row, so each byte of the + file is read at most once however many columns are wanted. + """ + if not isinstance(table, h5py.Dataset): + out = table if rows is None else table[_row_key(rows)] + if fields is None: + return np.asarray(out) + taken = np.empty(len(out), dtype=[(f, out.dtype[f]) for f in fields]) + for f in fields: + taken[f] = out[f] + return taken + + if fields is None: + source, dtype = table, table.dtype + else: + fields = list(fields) + source = table.fields(fields) + dtype = np.dtype([(f, table.dtype[f]) for f in fields]) + if rows is None: + rows = range(len(table)) + if isinstance(rows, range) and rows.step == 1: + return source[rows.start : rows.stop] + + rows = np.asarray(rows, dtype=np.intp) + out = np.empty(len(rows), dtype=dtype) + if len(rows) == 0: + return out + order = None + if np.any(rows[1:] < rows[:-1]): + order = np.argsort(rows, kind="stable") + rows = rows[order] + block = max(1, BLOCK_BYTES // table.dtype.itemsize) + done = 0 + start, stop = int(rows[0]), int(rows[-1]) + 1 + while done < len(rows): + start = max(start, int(rows[done])) + end = min(start + block, stop) + upto = int(np.searchsorted(rows, end, side="left")) + chunk = source[start:end][rows[done:upto] - start] + if order is None: + out[done:upto] = chunk + else: + out[order[done:upto]] = chunk + done, start = upto, end + return out + + +def _row_key(rows): + """Return ``rows`` (a ``range`` or index array) as a numpy row index.""" + if isinstance(rows, range): + return slice(rows.start, rows.stop if rows.stop >= 0 else None, rows.step) + return rows + + class V2View: """Lazy v2-grammar view of one table, or of row-aligned tables joined. Wraps numpy structured arrays, FITS_recs or h5py Datasets without reading them. ``view[name]`` returns one column as an array (computed on access for derived columns); a list of names returns a structured array of those - columns; any other key (slice, boolean mask, index array) selects rows and - returns a view over them. Only the requested columns are ever read, and - only the window of rows spanning the selection, so an HDF5 dataset larger - than memory can be wrapped whole. + columns; ``view[()]``, ``view[...]`` and any other row key (slice, boolean + mask, index array) select rows and return a view over them. + + Selecting rows of a view over h5py Datasets reads the selected rows of + every column into memory in one pass per dataset, as indexing the Dataset + itself would, and wraps them in a view over the in-memory arrays: reading + many columns of a selection one at a time would otherwise re-read the whole + span of rows for each. Over in-memory tables, selection stays lazy. Column names are case-sensitive, unlike a FITS_rec's. """ - def __init__(self, bases, rows=None): + def __init__(self, bases, rows=None, dtype=None): self._bases = tuple(bases) n_rows = {len(base) for base in self._bases} if len(n_rows) != 1: @@ -302,6 +370,8 @@ def __init__(self, bases, rows=None): # None (all rows), a ``range`` or an integer index array. self._rows = rows self._names, self._derived = _resolve(tuple(self._owner)) + self._on_disk = any(isinstance(base, h5py.Dataset) for base in self._bases) + self._dtype = dtype @property def names(self): @@ -314,8 +384,12 @@ def keys(self): @property def dtype(self): - """Numpy dtype of the presented columns (computed without reading).""" - return np.dtype([(name, self._field_dtype(name)) for name in self._names]) + """Numpy dtype of the presented columns (found without reading rows).""" + if self._dtype is None: + self._dtype = np.dtype( + [(name, self._field_dtype(name)) for name in self._names] + ) + return self._dtype @property def shape(self): @@ -337,22 +411,50 @@ def __getitem__(self, key): return self.to_structured(key) if isinstance(key, (int, np.integer)): return self[np.array([key])].to_structured()[0] - return V2View(self._bases, rows=self._select(key)) + if key is Ellipsis or (isinstance(key, tuple) and not key): + key = slice(None) + elif isinstance(key, tuple): + raise IndexError(f"cannot select rows with the tuple {key!r}") + rows = self._select(key) + if self._on_disk: + bases = [_take(base, rows) for base in self._bases] + return V2View(bases, dtype=self._dtype) + return V2View(self._bases, rows=rows, dtype=self._dtype) def __array__(self, dtype=None, copy=None): out = self.to_structured() return out if dtype is None else out.astype(dtype) def to_structured(self, names=None): - """Materialise the presented columns as a numpy structured array.""" + """Materialise the presented columns as a numpy structured array. + + Over h5py Datasets, each dataset is read once, for all the columns + requested of it. + """ names = self._names if names is None else list(names) missing = [name for name in names if name not in self._names] if missing: raise KeyError(f"columns {missing} not in catalogue") - dtype = np.dtype([(name, self._field_dtype(name)) for name in names]) - out = np.empty(len(self), dtype=dtype) - for name in names: - out[name] = self._column(name) + sources = { + name: self._derived[name] if name in self._derived else (None, name) + for name in names + } + if self._on_disk: + fields = {} + for _, source in sources.values(): + fields.setdefault(id(self._owner[source]), {})[source] = None + taken = {} + for base in self._bases: + if id(base) in fields: + table = _take(base, self._rows, list(fields[id(base)])) + taken.update((f, table[f]) for f in table.dtype.names) + read = taken.__getitem__ + else: + read = self._read + out = np.empty(len(self), dtype=[(n, self.dtype[n]) for n in names]) + for name, (rule, source) in sources.items(): + column = read(source) + out[name] = column if rule is None else rule.apply(column) return out def _select(self, key): @@ -425,13 +527,15 @@ def _field_dtype(self, name): def adapt(table, *tables): """Present ``table`` (joined with any further ``tables``) in the v2 grammar. - A single table with nothing to rename is returned unchanged; otherwise the - result is a ``V2View``. Each table is anything whose ``dtype.names`` lists - its columns and whose ``table[name]`` reads one: a numpy structured array, - FITS_rec or h5py Dataset. Joined tables must have equal lengths and - disjoint column names. + A single table with nothing to rename, or already a ``V2View``, is returned + unchanged; otherwise the result is a ``V2View``. Each table is anything + whose ``dtype.names`` lists its columns and whose ``table[name]`` reads + one: a numpy structured array, FITS_rec or h5py Dataset. Joined tables + must have equal lengths and disjoint column names. """ - if not tables and not _resolve(column_names(table))[1]: + if not tables and ( + isinstance(table, V2View) or not _resolve(column_names(table))[1] + ): return table return V2View((table,) + tables) diff --git a/src/sp_validation/tests/test_grammar.py b/src/sp_validation/tests/test_grammar.py index 265fe661..535bdecf 100644 --- a/src/sp_validation/tests/test_grammar.py +++ b/src/sp_validation/tests/test_grammar.py @@ -280,7 +280,7 @@ def test_get_rho_tau_identical_for_v1_and_v2_psf_catalogues(tmp_path): # -- mask-bit columns: a rule family of their own --------------------------- -def _legacy_mask_table(n=N, extra=()): +def _data_ext_mask_table(n=N, extra=()): """A data_ext-style table of {b}_{label} flags, and its MASK_n{b} twin.""" rng = np.random.default_rng(7) old = { @@ -312,9 +312,9 @@ def test_mask_column_rejects_unknown_bit(): grammar.mask_column(4096) -def test_legacy_mask_flags_rename_without_a_generation(): +def test_data_ext_mask_flags_rename_without_a_generation(): """data_ext mask names are renamed whatever the ShapePipe generation.""" - old, new = _legacy_mask_table(extra=("npoint3",)) + old, new = _data_ext_mask_table(extra=("npoint3",)) assert detect_generation(old.dtype.names) is None view = adapt(old) assert isinstance(view, V2View) @@ -332,14 +332,14 @@ def test_legacy_mask_flags_rename_without_a_generation(): def test_new_mask_names_pass_through_and_both_names_conflict(): - old, new = _legacy_mask_table() + old, new = _data_ext_mask_table() assert adapt(new) is new with pytest.raises(ValueError, match="two names"): adapt(old, new[["MASK_n4"]].copy()) def test_join_requires_equal_lengths_and_disjoint_names(): - old, new = _legacy_mask_table() + old, new = _data_ext_mask_table() with pytest.raises(ValueError, match="lengths"): adapt(new, old[:10]) with pytest.raises(ValueError, match="more than one table"): @@ -377,7 +377,7 @@ def test_empty_list_selects_an_empty_view(v1_and_v2): def test_boolean_selection_holds_only_the_selected_rows(): """A mask becomes the selected indices, never a full-length index array.""" - old, _ = _legacy_mask_table(n=1000) + old, _ = _data_ext_mask_table(n=1000) view = adapt(old) mask = np.zeros(1000, dtype=bool) mask[[3, 500, 998]] = True @@ -389,28 +389,67 @@ def test_boolean_selection_holds_only_the_selected_rows(): np.testing.assert_array_equal(sub["MASK_n4"], old["4_Stars"][[500]]) -def test_h5py_reads_only_the_selected_window(tmp_path, monkeypatch): - """Row selections read the window spanning them, not the whole column.""" - old, _ = _legacy_mask_table(n=1000) +def _h5py_mask_table(tmp_path, n=1000): + old, new = _data_ext_mask_table(n=n, extra=("npoint3",)) with h5py.File(tmp_path / "ext.hdf5", "w") as handle: handle.create_dataset("data_ext", data=old) - reads = [] - original = grammar._read_window + return old, new - def spy(table, name, start, stop): - reads.append((start, stop)) - return original(table, name, start, stop) - monkeypatch.setattr(grammar, "_read_window", spy) +@pytest.mark.parametrize("block_rows", [1, 7, 10_000]) +def test_h5py_selection_reads_each_dataset_once(tmp_path, monkeypatch, block_rows): + """Selecting rows of an on-disk view reads every column in one pass.""" + old, new = _h5py_mask_table(tmp_path) + monkeypatch.setattr(grammar, "BLOCK_BYTES", block_rows * old.dtype.itemsize) + takes = [] + original = grammar._take + + def spy(table, rows, fields=None): + takes.append(fields) + return original(table, rows, fields) + + monkeypatch.setattr(grammar, "_take", spy) + rng = np.random.default_rng(3) + mask = rng.random(1000) < 0.3 + mask[400:700] = False + index = np.array([998, 3, 500, 500, 2, 250]) with h5py.File(tmp_path / "ext.hdf5", "r") as handle: view = adapt(handle["data_ext"]) - mask = np.zeros(1000, dtype=bool) - mask[[200, 250, 300]] = True - got = view[mask]["MASK_n8"] - np.testing.assert_array_equal(got, old["8_Manual"][[200, 250, 300]]) - assert reads[-1] == (200, 301) - view[10:20:3]["MASK_n1"] - assert reads[-1] == (10, 20) + for key in (mask, index, slice(10, 20, 3), slice(None, None, -7)): + takes.clear() + sub = view[key] + assert takes == [None] + assert not sub._on_disk + for name in new.dtype.names: + np.testing.assert_array_equal(sub[name], new[key][name], err_msg=name) + takes.clear() + got = view[mask].to_structured(["MASK_n8", "npoint3"]) + np.testing.assert_array_equal(got["MASK_n8"], new["MASK_n8"][mask]) + + # Materialising named columns reads only their fields, in one pass. + takes.clear() + got = view.to_structured(["MASK_n8", "npoint3", "MASK_n1"]) + assert takes == [["8_Manual", "npoint3", "1_Faint_star_halos"]] + np.testing.assert_array_equal(got["MASK_n1"], new["MASK_n1"]) + + +@pytest.mark.parametrize("key", [(), Ellipsis]) +def test_empty_tuple_and_ellipsis_select_every_row(tmp_path, key): + old, new = _h5py_mask_table(tmp_path) + with h5py.File(tmp_path / "ext.hdf5", "r") as handle: + for view in (adapt(old), adapt(handle["data_ext"])): + everything = view[key] + assert len(everything) == len(new) + np.testing.assert_array_equal(everything["MASK_n4"], new["MASK_n4"]) + with pytest.raises(IndexError, match="tuple"): + view[(0, 1)] + + +def test_dtype_is_computed_once(): + old, _ = _data_ext_mask_table() + view = adapt(old) + assert view.dtype is view.dtype + assert view[3:9].dtype is view.dtype def test_v1_no_shear_reconv_psf_reads_the_1P_kernel(): From ea09ed7681853277ff0ac41776372d1752f2a562 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 06:32:48 +0200 Subject: [PATCH 29/39] catalog.group_dtype: name the tables that lack a column A column missing from some tiles (several v1 tiles carry no SPREAD_MODEL) surfaced as numpy's bare "no field of name" KeyError. group_dtype now checks every table first and raises a KeyError naming how many tables lack which columns, and which ones; callers pass the tables keyed by dataset name so the message names tiles. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LsAQRxNg6yDsUX4y2SWxTJ --- src/sp_validation/catalog.py | 45 ++++++++++++++----- .../tests/test_campaign_readers.py | 10 +++++ 2 files changed, 45 insertions(+), 10 deletions(-) diff --git a/src/sp_validation/catalog.py b/src/sp_validation/catalog.py index dd81ebf2..127b1555 100644 --- a/src/sp_validation/catalog.py +++ b/src/sp_validation/catalog.py @@ -874,24 +874,49 @@ def group_dtype(tables, param_list=None): Parameters ---------- - tables : list - per-tile tables (h5py Datasets, or their ``grammar.adapt`` views) + tables : list or dict + per-tile tables (h5py Datasets, or their ``grammar.adapt`` views), + optionally keyed by dataset name for error messages param_list : list of str, optional - columns to keep; default is ``None`` (keep all) + columns to keep; default is ``None`` (all columns of the first table) Returns ------- numpy.dtype structured output dtype + Raises + ------ + KeyError + if a table lacks one of the columns, naming the tables that do + """ - names = param_list if param_list is not None else list(tables[0].dtype.names) + if not hasattr(tables, "items"): + tables = dict(enumerate(tables)) + dtypes = {key: table.dtype for key, table in tables.items()} + first = next(iter(dtypes.values())) + names = param_list if param_list is not None else list(first.names) + + missing = { + key: [name for name in names if name not in dtype.names] + for key, dtype in dtypes.items() + } + missing = {key: cols for key, cols in missing.items() if cols} + if missing: + key, cols = next(iter(missing.items())) + raise KeyError( + f"{len(missing)} of {len(tables)} tables lack columns" + + f" {'requested' if param_list is not None else 'of the first table'}," + + f" e.g. table {key!r} lacks {cols}; tables lacking some:" + + f" {list(missing)[:20]}. Pass a param_list without them." + ) + dtypes = list(dtypes.values()) fields = [] for name in names: - promoted = tables[0].dtype[name] - for table in tables[1:]: - promoted = promote_dtypes(promoted, table.dtype[name]) + promoted = dtypes[0][name] + for dtype in dtypes[1:]: + promoted = promote_dtypes(promoted, dtype[name]) fields.append((name, promoted)) return np.dtype(fields) @@ -967,7 +992,7 @@ def concatenate_datasets( """ tables = _unit_tables(group, file_path, param_list=param_list) keys = list(tables) - dtype_out = group_dtype(list(tables.values()), param_list=param_list) + dtype_out = group_dtype(tables, param_list=param_list) if key_column is not None: if key_column in (dtype_out.names or ()): @@ -1076,8 +1101,8 @@ def campaign_shape(file_path, param_list=None): with h5py.File(file_path, "r") as hdf5_file: group = find_dataset_group(hdf5_file) check_n_units(hdf5_file, group, file_path) - tables = list(_unit_tables(group, file_path, param_list=param_list).values()) - n_rows = sum(len(table) for table in tables) + tables = _unit_tables(group, file_path, param_list=param_list) + n_rows = sum(len(table) for table in tables.values()) dtype_out = group_dtype(tables, param_list=param_list) return n_rows, dtype_out diff --git a/src/sp_validation/tests/test_campaign_readers.py b/src/sp_validation/tests/test_campaign_readers.py index ccb535bd..bcfee833 100644 --- a/src/sp_validation/tests/test_campaign_readers.py +++ b/src/sp_validation/tests/test_campaign_readers.py @@ -514,3 +514,13 @@ def test_no_input_raises(self): if __name__ == "__main__": unittest.main() + + +class TestGroupDtype(unittest.TestCase): + def test_missing_column_names_the_tables_lacking_it(self): + full = np.zeros(2, dtype=[("RA", "f8"), ("SPREAD_MODEL", "f4")]) + tiles = {"t0": full, "t1": full[["RA"]], "t2": full} + with self.assertRaisesRegex(KeyError, r"1 of 3 tables.*'t1'.*SPREAD_MODEL"): + catalog.group_dtype(tiles) + dtype = catalog.group_dtype(tiles, param_list=["RA"]) + self.assertEqual(dtype.names, ("RA",)) From e3f2c3f8733169b7ff592e0d8277bd23926c4312 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 06:32:57 +0200 Subject: [PATCH 30/39] masks: refuse a mask config's dat_ext list; masking keeps IMAFLAGS_ISO Mask configs are one dat: list over the joined data + data_ext table, but get_masks_from_config and scripts/masking.py read only config["dat"], so an older config with a dat_ext: list (several sit under v1.4.x and v1.5.x) silently lost those cuts. masks.catalogue_cuts returns the dat list and raises on dat_ext, saying to merge it into dat with {b}_{label} renamed MASK_n{b}; the calibration, footprint and demo scripts all read cuts through it. scripts/masking.py's footprint cuts had dropped IMAFLAGS_ISO, which the v1 configs still cut on and which is spatial (halo, border, Messier, NGC, spike bits); it is back in SPATIAL_CUTS. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LsAQRxNg6yDsUX4y2SWxTJ --- .../examples/demo_create_footprint_mask.py | 3 +- scripts/masking.py | 5 ++- src/sp_validation/masks.py | 28 +++++++++++- src/sp_validation/tests/test_masks.py | 45 +++++++++++++++---- 4 files changed, 68 insertions(+), 13 deletions(-) diff --git a/scripts/examples/demo_create_footprint_mask.py b/scripts/examples/demo_create_footprint_mask.py index a1a1850d..de45101a 100644 --- a/scripts/examples/demo_create_footprint_mask.py +++ b/scripts/examples/demo_create_footprint_mask.py @@ -29,6 +29,7 @@ from sp_validation import catalog_builders as sp_joint from sp_validation.grammar import MASK_LABELS +from sp_validation.masks import catalogue_cuts from sp_validation.plots import hsp_map_logical_or # - @@ -54,7 +55,7 @@ bits = 0 auxiliary_masks = [] auxiliary_labels = [] -for mask_params in config["dat"]: +for mask_params in catalogue_cuts(config): # Check bit-coded masks if mask_params["col_name"] in all_masks_bits: bits = bits | all_masks_bits[mask_params["col_name"]] diff --git a/scripts/masking.py b/scripts/masking.py index 021e5b6e..c1acf7f0 100644 --- a/scripts/masking.py +++ b/scripts/masking.py @@ -8,7 +8,7 @@ import yaml from sp_validation.grammar import adapt -from sp_validation.masks import apply_condition +from sp_validation.masks import apply_condition, catalogue_cuts # ------------------------- # Spatially-structured cuts: these define the survey footprint. @@ -17,6 +17,7 @@ # the footprint definition. SPATIAL_CUTS = { "overlap", + "IMAFLAGS_ISO", "MASK_n1", "MASK_n2", "MASK_n4", @@ -72,7 +73,7 @@ def apply_masks(data, mask_config, footprint_only=False): # Initialize mask mask = np.ones(len(data), dtype=bool) - for cut in mask_config.get("dat", []): + for cut in catalogue_cuts(mask_config): col = cut["col_name"] if footprint_only and col not in SPATIAL_CUTS: continue diff --git a/src/sp_validation/masks.py b/src/sp_validation/masks.py index 64f7272b..f1efa2ae 100644 --- a/src/sp_validation/masks.py +++ b/src/sp_validation/masks.py @@ -301,10 +301,34 @@ def print_mask_stats(num_obj, masks, mask_combined): mask_combined.print_stats(num_obj) +def catalogue_cuts(config): + """Return the catalogue cuts of a mask config: its ``dat`` list. + + Every cut names a column of the one v2-grammar table + ``CalibrateCat.read_cat`` returns, mask columns included. + + Raises + ------ + ValueError + if the config also has a ``dat_ext`` list, whose cuts would otherwise + be silently dropped + """ + if "dat_ext" in config: + names = [cut.get("col_name") for cut in config["dat_ext"] or []] + raise ValueError( + f"mask config has a 'dat_ext' cut list {names}, which is not read:" + + " move its entries into 'dat', renaming the mask columns" + + " {b}_{label} to MASK_n{b} (e.g. 4_Stars -> MASK_n4;" + + " see sp_validation.grammar.MASK_LABELS)" + ) + return config.get("dat") or [] + + def get_masks_from_config(config, dat, masks_to_apply=None, verbose=False): """Get Masks From Config. - Return the masks of the config's ``dat`` cut list, evaluated on ``dat``. + Return the masks of the config's ``dat`` cut list (``catalogue_cuts``), + evaluated on ``dat``. Parameters ---------- @@ -332,7 +356,7 @@ def get_masks_from_config(config, dat, masks_to_apply=None, verbose=False): # Dict to associate labels with index in mask list labels = {} - for mask_params in config.get("dat", []): + for mask_params in catalogue_cuts(config): if masks_to_apply is not None and mask_params["col_name"] not in masks_to_apply: if verbose: print(f"Skipping mask {mask_params['col_name']}") diff --git a/src/sp_validation/tests/test_masks.py b/src/sp_validation/tests/test_masks.py index 76e97748..054bce70 100644 --- a/src/sp_validation/tests/test_masks.py +++ b/src/sp_validation/tests/test_masks.py @@ -9,11 +9,19 @@ """ +import importlib.util +from pathlib import Path + import numpy as np import numpy.testing as npt import pytest -from sp_validation.masks import Mask, apply_condition +from sp_validation.masks import ( + Mask, + apply_condition, + catalogue_cuts, + get_masks_from_config, +) pytestmark = pytest.mark.fast @@ -34,12 +42,10 @@ ], ) def test_apply_condition_kinds(kind, value, expected): - npt.assert_array_equal(apply_condition(_ARRAY, kind, value), expected) def test_smaller_equal_alias_matches_less_equal(): - npt.assert_array_equal( apply_condition(_ARRAY, "smaller_equal", 3), apply_condition(_ARRAY, "less_equal", 3), @@ -47,13 +53,11 @@ def test_smaller_equal_alias_matches_less_equal(): def test_unknown_kind_raises(): - with pytest.raises(ValueError): apply_condition(_ARRAY, "not_a_real_kind", 3) def test_mask_apply_matches_apply_condition(): - dat = np.array([(1,), (2,), (3,), (4,), (5,)], dtype=[("col", "i8")]) my_mask = Mask("col", "test_mask", kind="greater_equal", value=3, dat=dat) @@ -64,7 +68,6 @@ def test_mask_apply_matches_apply_condition(): def test_not_equal_2bands_is_two_column_or(): - # Keep an object if EITHER band column differs from the sentinel dat = np.array( [(-99, -99), (-99, 20.0), (21.0, -99), (21.0, 20.0)], @@ -84,16 +87,42 @@ def test_not_equal_2bands_is_two_column_or(): def test_not_equal_2bands_requires_col_name2(): - with pytest.raises(ValueError, match="col_name2"): Mask("mag_z", "zband", kind="not_equal_2bands", value=-99) def test_not_equal_2bands_descr_names_both_columns(): - my_mask = Mask( "mag_z", "zband", kind="not_equal_2bands", value=-99, col_name2="mag_z2" ) my_mask.create_descr() assert my_mask._descr == "!=-99 in mag_z or mag_z2" + + +def test_dat_ext_cut_list_is_refused_not_dropped(): + """A config's ``dat_ext`` cuts would be silently ignored; refuse them.""" + config = { + "dat": [{"col_name": "FLAGS", "label": "SE", "kind": "equal", "value": 0}], + "dat_ext": [{"col_name": "4_Stars", "kind": "equal", "value": False}], + } + dat = np.zeros(3, dtype=[("FLAGS", "i2"), ("MASK_n4", "?")]) + with pytest.raises(ValueError, match="dat_ext.*4_Stars.*MASK_n4"): + get_masks_from_config(config, dat) + del config["dat_ext"] + assert catalogue_cuts(config) == config["dat"] + masks, _ = get_masks_from_config(config, dat) + assert len(masks) == 1 + + +def test_masking_script_refuses_dat_ext(): + spec = importlib.util.spec_from_file_location( + "masking_script", Path(__file__).resolve().parents[3] / "scripts/masking.py" + ) + masking = importlib.util.module_from_spec(spec) + spec.loader.exec_module(masking) + config = {"dat": [], "dat_ext": [{"col_name": "8_Manual"}]} + dat = np.zeros(3, dtype=[("MASK_n8", "?")]) + with pytest.raises(ValueError, match="dat_ext"): + masking.apply_masks(dat, config, footprint_only=True) + assert "IMAFLAGS_ISO" in masking.SPATIAL_CUTS From b9046e7219f2181eb3ca33e30be1f15b092faa72 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 06:33:45 +0200 Subject: [PATCH 31/39] galaxy.mask_cut: a NaN mask value masks the object, as the config cuts do mask_cut kept objects whose mask column is NaN (with a warning), while the config-driven cut that calibration runs (kind: equal, value: False) drops them, so the two selections disagreed on a defective product. mask_cut now drops them too, still counting and warning. Keeping them was chosen to stop astype(bool) from dropping NaN rows silently; the warning covers that, and dropping an object with no mask verdict is the conservative choice for a shear catalogue. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LsAQRxNg6yDsUX4y2SWxTJ --- src/sp_validation/galaxy.py | 19 +++++++++---------- .../tests/test_campaign_readers.py | 19 ++++++++++--------- 2 files changed, 19 insertions(+), 19 deletions(-) diff --git a/src/sp_validation/galaxy.py b/src/sp_validation/galaxy.py index f119ab18..fbe688bd 100644 --- a/src/sp_validation/galaxy.py +++ b/src/sp_validation/galaxy.py @@ -129,25 +129,24 @@ def mask_cut(dd, mask_columns=None): if values.dtype == bool: flagged = values else: - # ShapePipe's writer emits the MASK_n* columns as float64 {0, 1} - # rather than bool (being fixed upstream), so decide on the value - # rather than on truthiness: a bare astype(bool) would silently - # read NaN as True, i.e. masked. Threshold at 0.5 so an integer, - # a float and a bool column all behave identically. + # ShapePipe's writer can emit the MASK_n* columns as float64 + # {0, 1} rather than bool, so decide on the value (threshold at + # 0.5, so integer, float and bool columns behave identically), + # and count NaNs rather than let them pass unremarked. values = values.astype(float) undefined = np.isnan(values) n_undefined += int(undefined.sum()) - flagged = np.where(undefined, False, values > 0.5) + flagged = undefined | (values > 0.5) masked |= flagged if n_undefined: # NaN means the masking stage recorded no verdict for this object. - # Treat it as un-masked (keep the object) so an incomplete mask - # column cannot silently delete sky, but say so loudly: a nonzero - # count here means the input product is defective. + # Treat it as masked, as the config-driven cuts (``kind: equal, + # value: False``) do, and say so: a nonzero count means the input + # product is defective. warnings.warn( f"{n_undefined} NaN value(s) in mask column(s) {columns};" - + " treated as not masked. The mask columns of a complete" + + " treated as masked. The mask columns of a complete" + " ShapePipe product hold only 0 and 1.", RuntimeWarning, stacklevel=2, diff --git a/src/sp_validation/tests/test_campaign_readers.py b/src/sp_validation/tests/test_campaign_readers.py index bcfee833..2d01d988 100644 --- a/src/sp_validation/tests/test_campaign_readers.py +++ b/src/sp_validation/tests/test_campaign_readers.py @@ -10,6 +10,7 @@ from sp_validation import catalog, galaxy from sp_validation.catalog_builders import JointCat +from sp_validation.masks import Mask GAL_DTYPE = np.dtype( [ @@ -378,14 +379,12 @@ def test_float_and_int_mask_columns(self): err_msg=f"mask column dtype {dtype}", ) - def test_nan_mask_value_is_not_masked_and_warns(self): - """Test that a NaN mask value keeps the object, loudly. + def test_nan_mask_value_is_masked_and_warns(self): + """Test that a NaN mask value drops the object, loudly. - A bare astype(bool) reads NaN as True, i.e. masked, so an - incomplete mask column would silently delete sky. NaN means the - masking stage recorded no verdict, so keep the object and warn: - a nonzero count means the input product is defective. The real - smk-g7 catalogue carries no NaNs, so this is defensive. + NaN means the masking stage recorded no verdict. mask_cut drops the + object, as the config-driven cut ``kind: equal, value: False`` does, + and warns: a nonzero count means the input product is defective. """ dat = np.zeros(3, dtype=[(col, "f8") for col in galaxy.DEFAULT_MASK_COLUMNS]) dat["MASK_n4"][0] = np.nan @@ -394,9 +393,11 @@ def test_nan_mask_value_is_not_masked_and_warns(self): with self.assertWarns(RuntimeWarning) as ctx: keep = galaxy.mask_cut(dat) - # row 0 NaN -> kept, row 1 masked, row 2 clean -> kept - npt.assert_array_equal(keep, np.array([True, False, True])) + # row 0 NaN -> masked, row 1 masked, row 2 clean -> kept + npt.assert_array_equal(keep, np.array([False, False, True])) self.assertIn("NaN", str(ctx.warning)) + config_keep = Mask("MASK_n4", "stars", kind="equal", value=False, dat=dat)._mask + npt.assert_array_equal(config_keep, keep) def test_no_warning_when_no_nan(self): dat = np.zeros(2, dtype=[(col, "f8") for col in galaxy.DEFAULT_MASK_COLUMNS]) From ea8820f44aa25bf78d733a66a04a821734a3cc6f Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 06:33:53 +0200 Subject: [PATCH 32/39] Comments describe the present; calibration configs' relative input_path Drop wording that goes stale ("being fixed upstream", "unconfirmed for the Aug-2026 products", "legacy" layouts and files) in favour of what is true now, and ApplyHspMasks.write_hdf5_header's documented but nonexistent campaigns parameter. test_configured_paths_exist_on_candide resolved a calibration config's relative params.input_path against config/calibration; such configs run as config_mask.yaml in their run directory, where the relative path names a file, so the guard skips those. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LsAQRxNg6yDsUX4y2SWxTJ --- CLAUDE.md | 2 +- config/calibration/mask_v2.0.yaml | 8 +++---- scripts/calibration/params.py | 4 ++-- scripts/combine_results.py | 4 ++-- scripts/compute_m_bias_image_sims.py | 3 ++- src/sp_validation/catalog.py | 2 +- src/sp_validation/catalog_builders.py | 2 -- src/sp_validation/galaxy.py | 3 +-- .../tests/test_campaign_readers.py | 22 +++++++++---------- .../tests/test_config_paths_exist.py | 9 ++++++++ 10 files changed, 33 insertions(+), 26 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index db1d1f02..b7269cff 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -71,7 +71,7 @@ Main configuration in `scripts/calibration/params.py` with parameters: - `campaign`: Campaign name (the ShapePipe tile list); names the input products - `data_dir`: Input data directory - `galaxy_cat_path`: Galaxy catalogue path (.fits/.hdf5) -- `star_cat_path`: Star catalogue path (.hdf5, or legacy .fits) +- `star_cat_path`: Star catalogue path (.hdf5, or a v1 .fits) ### Key Dependencies - astropy, numpy, scipy for core calculations diff --git a/config/calibration/mask_v2.0.yaml b/config/calibration/mask_v2.0.yaml index 22a5444e..8c138b37 100644 --- a/config/calibration/mask_v2.0.yaml +++ b/config/calibration/mask_v2.0.yaml @@ -6,8 +6,8 @@ # # Reason bits of the ShapePipe r-band default bitmask: # -# MASK_n1, MASK_n2 star halos (which is faint and which is bright is -# unconfirmed for the Aug-2026 products) +# MASK_n1, MASK_n2 star halos (which is faint and which is bright is not +# documented) # MASK_n4 stars # MASK_n8 manual galaxy mask # MASK_n64 undocumented reason bit @@ -54,8 +54,8 @@ dat: label: "stars" kind: equal value: False - # n1/n2 are the two star-halo bits; the faint/bright assignment is - # unconfirmed for the Aug-2026 products. + # n1/n2 are the two star-halo bits; the faint/bright assignment is not + # documented. - col_name: MASK_n1 label: "star halos (n1)" kind: equal diff --git a/scripts/calibration/params.py b/scripts/calibration/params.py index 558fcc67..3b5da373 100644 --- a/scripts/calibration/params.py +++ b/scripts/calibration/params.py @@ -56,8 +56,8 @@ ### Star and PSF catalog name; optional, set to `None` if not required star_cat_path = f"{data_dir}/full_starcat_{campaign}.hdf5" -# HDU number of star and PSF catalogue; only used for the legacy FITS star -# catalogue (a path ending in .fits), ignored for the v2 HDF5 product +# HDU number of star and PSF catalogue; only used for a FITS star catalogue +# (a path ending in .fits), ignored for the HDF5 product hdu_star_cat = 1 ### External mask; optional, set to `None` if not required diff --git a/scripts/combine_results.py b/scripts/combine_results.py index 8244c41c..4c4f9510 100755 --- a/scripts/combine_results.py +++ b/scripts/combine_results.py @@ -219,8 +219,8 @@ def print_all( def get_area(fname): """Return the unmasked area in deg^2 read from an area.txt file. - Accepts both the v2 wording ("campaign") and the legacy one ("patch"), - so results computed before the campaign rename can still be combined. + Accepts both the "campaign" and the "patch" wording of the area line, + so area files of either wording combine. Raises rather than returning a placeholder: a wrong area silently rescales every density. """ diff --git a/scripts/compute_m_bias_image_sims.py b/scripts/compute_m_bias_image_sims.py index eb0a76a8..dcee2919 100644 --- a/scripts/compute_m_bias_image_sims.py +++ b/scripts/compute_m_bias_image_sims.py @@ -60,7 +60,8 @@ def get_n_tiles(grids_dir, num): """Detect number of tiles from final_cat HDF5 files. Layout-agnostic: uses ``sp_validation.catalog.find_dataset_group``, so it - works on both the legacy nested and the flat per-tile HDF5 layouts. + works on both the nested ``patches//`` and the flat + per-tile HDF5 layouts. """ import h5py diff --git a/src/sp_validation/catalog.py b/src/sp_validation/catalog.py index 127b1555..4415d876 100644 --- a/src/sp_validation/catalog.py +++ b/src/sp_validation/catalog.py @@ -786,7 +786,7 @@ def find_dataset_group(hdf5_file): This makes the reader independent of how deeply the products nest that group: it walks down as long as the current node holds exactly one sub-group, and stops as soon as the members are datasets. It therefore - reads both the legacy ``patches//`` layout (the + reads both the nested ``patches//`` layout (the "patches" key is a ShapePipe-side compatibility shim, not a concept) and a flat ``tiles/`` or ``exposures/`` layout. diff --git a/src/sp_validation/catalog_builders.py b/src/sp_validation/catalog_builders.py index 065b7dff..8efbdfaf 100644 --- a/src/sp_validation/catalog_builders.py +++ b/src/sp_validation/catalog_builders.py @@ -982,8 +982,6 @@ def write_hdf5_header(self, hd5file): ---------- hd5file : h5py.File input HDF5 file - campaigns : list, optional - input campaign names, list of str, default is ``None`` """ super().write_hdf5_header(hd5file) diff --git a/src/sp_validation/galaxy.py b/src/sp_validation/galaxy.py index fbe688bd..766d1d34 100644 --- a/src/sp_validation/galaxy.py +++ b/src/sp_validation/galaxy.py @@ -35,8 +35,7 @@ #: columns, presented under these names by ``sp_validation.grammar``. #: #: Reason bits making up the r-band default bitmask: n1/n2 star halos -#: (which of the two is faint and which bright is unconfirmed for the -#: Aug-2026 products), n4 stars, n8 manual galaxy mask, n64 (an +#: (which of the two is faint and which bright is not documented), n4 stars, n8 manual galaxy mask, n64 (an #: undocumented reason bit), n1024 MaxiMask. #: #: Per-band coverage flags: n16 (u), n32 (g), n128 (i), n256 (z). There is diff --git a/src/sp_validation/tests/test_campaign_readers.py b/src/sp_validation/tests/test_campaign_readers.py index 2d01d988..405aa010 100644 --- a/src/sp_validation/tests/test_campaign_readers.py +++ b/src/sp_validation/tests/test_campaign_readers.py @@ -36,9 +36,9 @@ def make_galaxy_data(n_obj, offset=0): def write_campaign(path, layout, tiles): - """Write a campaign hdf5 file in the legacy or flat layout.""" + """Write a campaign hdf5 file in the nested or flat layout.""" with h5py.File(path, "w") as f: - if layout == "legacy": + if layout == "nested": group = f.create_group("patches").create_group("CAMPAIGN") else: group = f.create_group("tiles") @@ -65,9 +65,9 @@ def _expected(self): """Expected concatenation: datasets in sorted-key order.""" return np.concatenate([self._tiles[key] for key in sorted(self._tiles)]) - def test_legacy_layout(self): + def test_nested_layout(self): path = self._dir / "final_cat_CAMPAIGN.hdf5" - write_campaign(path, "legacy", self._tiles) + write_campaign(path, "nested", self._tiles) dat = catalog.read_campaign_catalogue(path, verbose=False) @@ -84,13 +84,13 @@ def test_flat_layout(self): npt.assert_array_equal(dat["RA"], self._expected()["RA"]) def test_layouts_agree(self): - legacy = self._dir / "legacy.hdf5" + nested = self._dir / "nested.hdf5" flat = self._dir / "flat.hdf5" - write_campaign(legacy, "legacy", self._tiles) + write_campaign(nested, "nested", self._tiles) write_campaign(flat, "flat", self._tiles) npt.assert_array_equal( - catalog.read_campaign_catalogue(legacy, verbose=False), + catalog.read_campaign_catalogue(nested, verbose=False), catalog.read_campaign_catalogue(flat, verbose=False), ) @@ -200,7 +200,7 @@ def test_dtype_promoted_across_tiles(self): def test_iter_campaign_tiles(self): """The streaming reader yields one restricted tile at a time.""" path = self._dir / "final_cat_CAMPAIGN.hdf5" - write_campaign(path, "legacy", self._tiles) + write_campaign(path, "nested", self._tiles) tiles = list( catalog.iter_campaign_tiles(str(path), param_list=["RA"], verbose=False) @@ -362,8 +362,8 @@ def test_catalogue_without_mask_columns_raises(self): def test_float_and_int_mask_columns(self): """Test that float and int mask columns cut like bool ones. - ShapePipe's writer emits the real MASK_n* columns as float64 - {0.0, 1.0} rather than bool (being fixed upstream), so the cut + ShapePipe's writer can emit the MASK_n* columns as float64 + {0.0, 1.0} rather than bool, so the cut must decide on the value, not on the dtype. """ for dtype in ("f8", "i4"): @@ -417,7 +417,7 @@ def setUp(self): self._dir = Path(self._tmp.name) self._paths = [] - for name, layout, n_obj in (("W3", "legacy", 3), ("SGC", "flat", 2)): + for name, layout, n_obj in (("W3", "nested", 3), ("SGC", "flat", 2)): path = self._dir / f"final_cat_{name}.hdf5" write_campaign(path, layout, {"000.000": make_galaxy_data(n_obj)}) self._paths.append(str(path)) diff --git a/src/sp_validation/tests/test_config_paths_exist.py b/src/sp_validation/tests/test_config_paths_exist.py index bc5fa662..4f3ea04b 100644 --- a/src/sp_validation/tests/test_config_paths_exist.py +++ b/src/sp_validation/tests/test_config_paths_exist.py @@ -187,8 +187,17 @@ def _candidate_paths() -> list[tuple[Path, str, Path]]: candidates = [] for config_path in _config_files(): iterator = _iter_ini_paths if config_path.suffix == ".ini" else _iter_yaml_paths + calibration = config_path.parent == root / "config/calibration" for source, key, value, base_dir in iterator(config_path): expanded = Path(value).expanduser() + if ( + calibration + and key == "params.input_path" + and not expanded.is_absolute() + ): + # A calibration config runs as config_mask.yaml in its run + # directory; a relative input_path names a file there. + continue if expanded.is_absolute(): resolved = expanded elif base_dir is not None: From 330ac8d418439bc16934d66db6a8970bea43ac3e Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 06:37:01 +0200 Subject: [PATCH 33/39] tests: calibration path reproduces a v1.4.6.3 release window (candide, slow) Runs scripts/calibration/calibrate_comprehensive_cat.py with mask_v1.X.6.yaml on rows 200M..201M of the v1.4.c comprehensive HDF5 and checks the released cut catalogue's rows from that window, matched by (RA, Dec) and bracketed by rows from outside it: same objects, same order, and identical per-object columns, including the no-shear reconvolved-PSF size against the release's NGMIX_Tpsf_NOSHEAR. Globally calibrated columns (e1, e2, w_des, leakage-corrected) are not compared. Skipped where the release products are absent. Passes in ~30 s on n09 (131,526 objects); before the no-shear reconvolved-PSF correction it fails, selecting 134,936. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LsAQRxNg6yDsUX4y2SWxTJ --- .../tests/test_release_regression.py | 149 ++++++++++++++++++ 1 file changed, 149 insertions(+) create mode 100644 src/sp_validation/tests/test_release_regression.py diff --git a/src/sp_validation/tests/test_release_regression.py b/src/sp_validation/tests/test_release_regression.py new file mode 100644 index 00000000..7fb03437 --- /dev/null +++ b/src/sp_validation/tests/test_release_regression.py @@ -0,0 +1,149 @@ +"""Candide-local regression of the calibration path against the v1.4.6.3 release. + +Runs ``scripts/calibration/calibrate_comprehensive_cat.py`` with the release's +mask config (``config/calibration/mask_v1.X.6.yaml``) on a row window of the +v1.4.c comprehensive HDF5, and checks that it selects exactly the objects the +released cut catalogue holds from that window, in the same order, with the same +per-object values. The release preserves input order, so the window's objects +form one contiguous run of released rows, found by (RA, Dec). + +Globally calibrated columns (``e1``, ``e2``, ``w_des``, the leakage-corrected +ellipticities) depend on the whole-survey response and weights, not on the +window, and are not compared. Everything the release derives object by object +is, including the no-shear reconvolved-PSF size, which the release took from +``NGMIX_Tpsf_1P`` (``sp_validation.grammar``). + +Skipped where the release products are absent. +""" + +import os +import runpy +from pathlib import Path + +import h5py +import numpy as np +import pytest +import yaml +from astropy.io import fits + +WL = Path("/n17data/UNIONS/WL/v1.4.x") +COMPREHENSIVE = WL / "unions_shapepipe_comprehensive_struc_2024_v1.4.c.hdf5" +RELEASE = WL / "v1.4.6.3" / "unions_shapepipe_cut_struc_2024_v1.4.6.3.fits" +REPO = Path(__file__).resolve().parents[3] +SCRIPT = REPO / "scripts" / "calibration" / "calibrate_comprehensive_cat.py" +CONFIG = REPO / "config" / "calibration" / "mask_v1.X.6.yaml" + +START, N_ROWS = 200_000_000, 1_000_000 + +#: Per-object output columns and their release counterparts. +PER_OBJECT = { + name: name + for name in ( + "RA", + "Dec", + "mag", + "snr", + "e1_uncal", + "e2_uncal", + "w_iv", + "FLUX_RADIUS", + "FWHM_IMAGE", + "FWHM_WORLD", + "MAGERR_AUTO", + "MAG_WIN", + "MAGERR_WIN", + "FLUX_AUTO", + "FLUXERR_AUTO", + "FLUX_APER", + "FLUXERR_APER", + "NGMIX_T_NOSHEAR", + "fwhm_PSF", + "R_g11", + "R_g12", + "R_g21", + "R_g22", + "e1_PSF", + "e2_PSF", + ) +} | {"NGMIX_T_PSF_RECONV_NOSHEAR": "NGMIX_Tpsf_NOSHEAR"} + +pytestmark = [ + pytest.mark.slow, + pytest.mark.skipif( + not (COMPREHENSIVE.exists() and RELEASE.exists()), + reason="v1.4.c comprehensive catalogue or v1.4.6.3 release absent", + ), +] + + +def _window_file(path): + """Write rows START:START+N_ROWS of the comprehensive HDF5 to ``path``. + + Returns the window's (RA, Dec) as complex keys. + """ + with h5py.File(COMPREHENSIVE, "r") as src, h5py.File(path, "w") as dst: + for name in ("data", "data_ext"): + window = src[name][START : START + N_ROWS] + dst.create_dataset(name, data=window) + if name == "data": + keys = window["RA"] + 1j * window["Dec"] + return keys + + +def _run_calibration(run_dir): + """Run the calibration script in ``run_dir``. + + Returns its cut catalogue and the input window's (RA, Dec) keys. + """ + comprehensive = run_dir / "unions_shapepipe_comprehensive_window.hdf5" + window_keys = _window_file(comprehensive) + config = yaml.safe_load(CONFIG.read_text()) + config["params"]["input_path"] = str(comprehensive) + (run_dir / "config_mask.yaml").write_text(yaml.safe_dump(config)) + + cwd = os.getcwd() + os.chdir(run_dir) + try: + runpy.run_path(str(SCRIPT), run_name="__main__") + except SystemExit as exit: + assert exit.code in (0, None), f"calibration script exited with {exit.code}" + finally: + os.chdir(cwd) + return fits.getdata(run_dir / "unions_shapepipe_cut_window.fits", 1), window_keys + + +def _released_region(ra, dec): + """Return the released rows from our first object on, with a margin.""" + release = fits.getdata(RELEASE, 1, memmap=True) + first = np.flatnonzero( + (np.asarray(release["RA"]) == ra[0]) & (np.asarray(release["Dec"]) == dec[0]) + ) + assert len(first) == 1, "window's first object not found once in the release" + start = int(first[0]) + margin = 1000 + lo, hi = max(start - margin, 0), start + len(ra) + margin + return release[lo:hi] + + +def test_calibration_reproduces_the_release_window(tmp_path): + ours, window_keys = _run_calibration(tmp_path) + assert len(ours) > 0 + ra, dec = np.asarray(ours["RA"]), np.asarray(ours["Dec"]) + + # The released rows that come from the window, bracketed on both sides + # by released rows that do not, so none is missed. + region = _released_region(ra, dec) + region_keys = np.asarray(region["RA"]) + 1j * np.asarray(region["Dec"]) + in_window = np.isin(region_keys, window_keys) + assert not in_window[0] and not in_window[-1], "margin too small" + released = region[in_window] + + # Selection: exactly our objects, in our order. + np.testing.assert_array_equal( + np.asarray(released["RA"]) + 1j * np.asarray(released["Dec"]), ra + 1j * dec + ) + + for name, released_name in PER_OBJECT.items(): + np.testing.assert_array_equal( + np.asarray(ours[name]), np.asarray(released[released_name]), err_msg=name + ) From 9f86152617f105374f4d5a500fc34518012436ec Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Tue, 29 Sep 2026 16:25:00 +0200 Subject: [PATCH 34/39] Drop the SExtractor-only columns DR6 catalogue mode lacks ShapePipe v2 catalogues in DR6 catalogue mode no longer carry MAG_WIN, MAGERR_WIN, SNR_WIN, FLUX_AUTO, FLUXERR_AUTO, FLUX_APER, FLUXERR_APER, FWHM_IMAGE, FWHM_WORLD (shapepipe #924). Stop passing them through in extract_info (params.add_cols) and calibrate_comprehensive_cat, and drop the SExtractor SNR_WIN curve from the galaxy SNR histogram. The v1.4.6.3 release regression compares the remaining per-object columns. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01BeQNTBcjTqu6TPktxysovQ --- scripts/calibration/calibrate_comprehensive_cat.py | 8 -------- scripts/calibration/extract_info.py | 11 ++--------- scripts/calibration/params.py | 8 -------- src/sp_validation/tests/test_release_regression.py | 8 -------- 4 files changed, 2 insertions(+), 33 deletions(-) diff --git a/scripts/calibration/calibrate_comprehensive_cat.py b/scripts/calibration/calibrate_comprehensive_cat.py index 4912b963..7456d61f 100644 --- a/scripts/calibration/calibrate_comprehensive_cat.py +++ b/scripts/calibration/calibrate_comprehensive_cat.py @@ -178,15 +178,7 @@ add_cols = [ "w_iv", "FLUX_RADIUS", - "FWHM_IMAGE", - "FWHM_WORLD", "MAGERR_AUTO", - "MAG_WIN", - "MAGERR_WIN", - "FLUX_AUTO", - "FLUXERR_AUTO", - "FLUX_APER", - "FLUXERR_APER", "NGMIX_T_NOSHEAR", "NGMIX_T_PSF_RECONV_NOSHEAR", "fwhm_PSF", diff --git a/scripts/calibration/extract_info.py b/scripts/calibration/extract_info.py index d9de957e..2db5c19a 100644 --- a/scripts/calibration/extract_info.py +++ b/scripts/calibration/extract_info.py @@ -570,20 +570,13 @@ n_bin = 500 x_cut = gal_snr_min -labels = [] if shape == "ngmix": # Do not apply `mask_ns`, so use all galaxies - xs = [ - dd["NGMIX_FLUX_NOSHEAR"][m_gal] / dd["NGMIX_FLUX_ERR_NOSHEAR"][m_gal], - dd["SNR_WIN"][m_gal], - ] - labels.append(["$F/\\sigma(F)$"]) - + xs = [dd["NGMIX_FLUX_NOSHEAR"][m_gal] / dd["NGMIX_FLUX_ERR_NOSHEAR"][m_gal]] + labels = ["$F/\\sigma(F)$"] else: raise ValueError(f"Unknown shape measurement method {shape}") -labels.append("SExtractor SNR") - title = "Galaxies" out_name = f"hist_SNR_{shape}.pdf" diff --git a/scripts/calibration/params.py b/scripts/calibration/params.py index 3b5da373..64168555 100644 --- a/scripts/calibration/params.py +++ b/scripts/calibration/params.py @@ -109,15 +109,7 @@ ### Additional output columns add_cols = [ "FLUX_RADIUS", - "FWHM_IMAGE", - "FWHM_WORLD", "MAGERR_AUTO", - "MAG_WIN", - "MAGERR_WIN", - "FLUX_AUTO", - "FLUXERR_AUTO", - "FLUX_APER", - "FLUXERR_APER", "NGMIX_T_NOSHEAR", "NGMIX_T_PSF_RECONV_NOSHEAR", ] diff --git a/src/sp_validation/tests/test_release_regression.py b/src/sp_validation/tests/test_release_regression.py index 7fb03437..2e5f13c0 100644 --- a/src/sp_validation/tests/test_release_regression.py +++ b/src/sp_validation/tests/test_release_regression.py @@ -47,15 +47,7 @@ "e2_uncal", "w_iv", "FLUX_RADIUS", - "FWHM_IMAGE", - "FWHM_WORLD", "MAGERR_AUTO", - "MAG_WIN", - "MAGERR_WIN", - "FLUX_AUTO", - "FLUXERR_AUTO", - "FLUX_APER", - "FLUXERR_APER", "NGMIX_T_NOSHEAR", "fwhm_PSF", "R_g11", From a619c902f8ad3a23fe2d54d4bab02fb91e3ec771 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Wed, 30 Sep 2026 03:28:04 +0200 Subject: [PATCH 35/39] Describe halo bit identities and guard independent mask cuts --- config/calibration/mask_v2.0.yaml | 15 +++++++------- docs/ngmix_psf_column_migration.md | 5 +++-- src/sp_validation/galaxy.py | 5 ++--- src/sp_validation/tests/test_grammar.py | 26 +++++++++++++++++++++++++ 4 files changed, 38 insertions(+), 13 deletions(-) diff --git a/config/calibration/mask_v2.0.yaml b/config/calibration/mask_v2.0.yaml index 1f0bdda1..e32a647a 100644 --- a/config/calibration/mask_v2.0.yaml +++ b/config/calibration/mask_v2.0.yaml @@ -6,9 +6,9 @@ # # Reason bits of the ShapePipe r-band default bitmask: # -# MASK_n1, MASK_n2 star halos (which is faint and which is bright is not -# documented) -# MASK_n4 stars +# MASK_n1 faint star halos (bit 0; value 1) +# MASK_n2 bright star halos (bit 1; value 2) +# MASK_n4 star bodies (bit 2; value 4) # MASK_n8 manual galaxy mask # MASK_n64 r-band coverage # MASK_n1024 MaxiMask @@ -52,17 +52,16 @@ dat: # ShapePipe masks (boolean, True = masked) - col_name: MASK_n4 - label: "stars" + label: "star bodies" kind: equal value: False - # n1/n2 are the two star-halo bits; the faint/bright assignment is not - # documented. + # Star-halo bits follow the UNIONS mask producer schema. - col_name: MASK_n1 - label: "star halos (n1)" + label: "faint star halos" kind: equal value: False - col_name: MASK_n2 - label: "star halos (n2)" + label: "bright star halos" kind: equal value: False - col_name: MASK_n8 diff --git a/docs/ngmix_psf_column_migration.md b/docs/ngmix_psf_column_migration.md index 5a2e7c5b..973cb50c 100644 --- a/docs/ngmix_psf_column_migration.md +++ b/docs/ngmix_psf_column_migration.md @@ -29,8 +29,9 @@ releases, while the labels remain documented here and in the config: | bit | old `data_ext` name | meaning | |---|---|---| -| 1, 2 | `1_Faint_star_halos`, `2_Bright_star_halos` | star halos | -| 4 | `4_Stars` | star mask | +| 1 (bit 0) | `1_Faint_star_halos` | faint star halos (`MASK_n1`) | +| 2 (bit 1) | `2_Bright_star_halos` | bright star halos (`MASK_n2`) | +| 4 (bit 2) | `4_Stars` | star body mask (`MASK_n4`) | | 8 | `8_Manual` | manual mask (large galaxies) | | 16–256 | `16_u`, `32_g`, `64_r`, `128_i`, `256_z` | per-band coverage; `256_z` is the HSC z-band | | 512 | `512_Tile_RA_DEC_cut` | outside the tile's unique region | diff --git a/src/sp_validation/galaxy.py b/src/sp_validation/galaxy.py index 0c9d4611..6ccd7861 100644 --- a/src/sp_validation/galaxy.py +++ b/src/sp_validation/galaxy.py @@ -34,9 +34,8 @@ #: not write n512). Post-processed v1 comprehensive catalogues carry the same #: columns, presented under these names by ``sp_validation.grammar``. #: -#: The r-band default bitmask uses n1/n2 star halos, n4 stars, n8 manual -#: galaxy mask, n64 r-band coverage, and n1024 MaxiMask. Which of n1/n2 -#: is faint or bright is unconfirmed for the Aug-2026 products. +#: The r-band default bitmask uses n1 faint star halos, n2 bright star halos, +#: n4 star bodies, n8 manual galaxy mask, n64 r-band coverage, and n1024 MaxiMask. #: #: Per-band coverage flags: n16 (u), n32 (g), n64 (r), n128 (i), n256 #: (HSC z). n2048 is ``True`` where Pan-STARRS z-band (z2) coverage is absent. diff --git a/src/sp_validation/tests/test_grammar.py b/src/sp_validation/tests/test_grammar.py index 189842fe..e7fb6c92 100644 --- a/src/sp_validation/tests/test_grammar.py +++ b/src/sp_validation/tests/test_grammar.py @@ -336,6 +336,32 @@ def test_data_ext_mask_flags_rename_without_a_generation(): np.testing.assert_array_equal(joined["HSM_T_PSF"], v2["HSM_T_PSF"]) +def test_halo_masks_keep_faint_and_bright_selections_distinct(): + """Producer halo names retain their bit identity through adaptation and cuts.""" + from sp_validation.galaxy import mask_cut + + # Fixed source names and distinct values avoid deriving the fixture from + # MASK_LABELS, which would hide a reversed faint/bright mapping. + source = _structured_n( + { + "1_Faint_star_halos": np.array([False, True, False, True]), + "2_Bright_star_halos": np.array([False, False, True, True]), + } + ) + view = adapt(source) + np.testing.assert_array_equal(view["MASK_n1"], [False, True, False, True]) + np.testing.assert_array_equal(view["MASK_n2"], [False, False, True, True]) + np.testing.assert_array_equal( + mask_cut(view, ["MASK_n1"]), [True, False, True, False] + ) + np.testing.assert_array_equal( + mask_cut(view, ["MASK_n2"]), [True, True, False, False] + ) + np.testing.assert_array_equal( + mask_cut(view, ["MASK_n1", "MASK_n2"]), [True, False, False, False] + ) + + def test_new_mask_names_pass_through_and_both_names_conflict(): old, new = _data_ext_mask_table() assert adapt(new) is new From 2b565093762cd8fbaa0432b6b5750d2dc0a2e929 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Thu, 1 Oct 2026 03:22:04 +0200 Subject: [PATCH 36/39] Name mask columns MASK_{bit}_{label} The external-mask column for bit b is MASK_{b}_{label}, labels from grammar.MASK_LABELS (grammar.mask_column), so the column carries both the bit identity and its meaning. ApplyHspMasks writes these names; every mask config, script and the galaxy defaults cut on them (galaxy derives its tuples from mask_column). The adapter presents the v1 data_ext spelling {b}_{label} and the pre-release ShapePipe v2 spelling MASK_n{b} under the canonical name; neither marks a generation. Tests cover both spellings mapping to the same column and the same cuts, including the fixed-name faint/bright halo check. Co-Authored-By: Claude Opus 5.5 --- .../calibration/mask_from_compr_v1.X.11.yaml | 8 +- .../calibration/mask_from_compr_v1.X.6.yaml | 8 +- config/calibration/mask_v1.X.10.yaml | 8 +- config/calibration/mask_v1.X.11.yaml | 8 +- config/calibration/mask_v1.X.2.yaml | 12 +-- config/calibration/mask_v1.X.3.yaml | 8 +- config/calibration/mask_v1.X.4.yaml | 12 +-- config/calibration/mask_v1.X.5.yaml | 8 +- config/calibration/mask_v1.X.6.yaml | 8 +- config/calibration/mask_v1.X.6_ppv1.yaml | 8 +- config/calibration/mask_v1.X.7.yaml | 10 +- config/calibration/mask_v1.X.8.yaml | 12 +-- config/calibration/mask_v1.X.9.yaml | 8 +- .../mask_v1.X.9_im_sim.overlay.yaml | 10 +- config/calibration/mask_v2.0.yaml | 44 ++++----- docs/ngmix_psf_column_migration.md | 40 ++++---- papers/catalog/hist_mag.py | 14 +-- scripts/calibration/params.py | 34 +++---- .../examples/demo_calibrate_minimal_cat.py | 6 +- .../demo_comprehensive_to_minimal_cat.py | 12 +-- scripts/masking.py | 22 ++--- src/sp_validation/catalog_builders.py | 4 +- src/sp_validation/galaxy.py | 40 +++----- src/sp_validation/grammar.py | 32 ++++--- src/sp_validation/masks.py | 2 +- src/sp_validation/plots.py | 14 ++- .../tests/test_campaign_readers.py | 42 ++++---- src/sp_validation/tests/test_grammar.py | 96 ++++++++++++++----- .../tests/test_grammar_calibration.py | 13 +-- src/sp_validation/tests/test_masks.py | 6 +- workflow/image_sims/params_im_sim.py | 2 +- 31 files changed, 296 insertions(+), 255 deletions(-) diff --git a/config/calibration/mask_from_compr_v1.X.11.yaml b/config/calibration/mask_from_compr_v1.X.11.yaml index 6d19cfc5..d76d7fb3 100644 --- a/config/calibration/mask_from_compr_v1.X.11.yaml +++ b/config/calibration/mask_from_compr_v1.X.11.yaml @@ -28,25 +28,25 @@ dat: # HDF5's data_ext dataset) # Stars - - col_name: MASK_n4 + - col_name: MASK_4_Stars label: "Stars" kind: equal value: False # Manual mask - - col_name: MASK_n8 + - col_name: MASK_8_Manual label: "manual mask" kind: equal value: False # r-band footprint - - col_name: MASK_n64 + - col_name: MASK_64_r label: "r-band imaging" kind: equal value: False # Maximask - - col_name: MASK_n1024 + - col_name: MASK_1024_Maximask label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_from_compr_v1.X.6.yaml b/config/calibration/mask_from_compr_v1.X.6.yaml index e090141e..a71f9974 100644 --- a/config/calibration/mask_from_compr_v1.X.6.yaml +++ b/config/calibration/mask_from_compr_v1.X.6.yaml @@ -27,25 +27,25 @@ dat: # HDF5's data_ext dataset) # Stars - - col_name: MASK_n4 + - col_name: MASK_4_Stars label: "Stars" kind: equal value: False # Manual mask - - col_name: MASK_n8 + - col_name: MASK_8_Manual label: "manual mask" kind: equal value: False # r-band footprint - - col_name: MASK_n64 + - col_name: MASK_64_r label: "r-band imaging" kind: equal value: False # Maximask - - col_name: MASK_n1024 + - col_name: MASK_1024_Maximask label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.10.yaml b/config/calibration/mask_v1.X.10.yaml index 037bc2d9..2678a49b 100644 --- a/config/calibration/mask_v1.X.10.yaml +++ b/config/calibration/mask_v1.X.10.yaml @@ -61,25 +61,25 @@ dat: # HDF5's data_ext dataset) # Stars - - col_name: MASK_n4 + - col_name: MASK_4_Stars label: "Stars" kind: equal value: False # Manual mask - - col_name: MASK_n8 + - col_name: MASK_8_Manual label: "manual mask" kind: equal value: False # r-band footprint - - col_name: MASK_n64 + - col_name: MASK_64_r label: "r-band imaging" kind: equal value: False # Maximask - - col_name: MASK_n1024 + - col_name: MASK_1024_Maximask label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.11.yaml b/config/calibration/mask_v1.X.11.yaml index df81d5bd..b87aa267 100644 --- a/config/calibration/mask_v1.X.11.yaml +++ b/config/calibration/mask_v1.X.11.yaml @@ -61,25 +61,25 @@ dat: # HDF5's data_ext dataset) # Stars - - col_name: MASK_n4 + - col_name: MASK_4_Stars label: "Stars" kind: equal value: False # Manual mask - - col_name: MASK_n8 + - col_name: MASK_8_Manual label: "manual mask" kind: equal value: False # r-band footprint - - col_name: MASK_n64 + - col_name: MASK_64_r label: "r-band imaging" kind: equal value: False # Maximask - - col_name: MASK_n1024 + - col_name: MASK_1024_Maximask label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.2.yaml b/config/calibration/mask_v1.X.2.yaml index 4ef4c22f..e7ad7758 100644 --- a/config/calibration/mask_v1.X.2.yaml +++ b/config/calibration/mask_v1.X.2.yaml @@ -60,37 +60,37 @@ dat: # Healsparse mask bits and post-processing columns (the comprehensive # HDF5's data_ext dataset) # Faint star halos - - col_name: MASK_n1 + - col_name: MASK_1_Faint_star_halos label: "Faint star halos" kind: equal value: False # Bright star halos - - col_name: MASK_n2 + - col_name: MASK_2_Bright_star_halos label: "Bright star halos" kind: equal value: False # Stars - - col_name: MASK_n4 + - col_name: MASK_4_Stars label: "Stars" kind: equal value: False # Manual mask - - col_name: MASK_n8 + - col_name: MASK_8_Manual label: "manual mask" kind: equal value: False # r-band footprint - - col_name: MASK_n64 + - col_name: MASK_64_r label: "r-band imaging" kind: equal value: False # Maximask - - col_name: MASK_n1024 + - col_name: MASK_1024_Maximask label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.3.yaml b/config/calibration/mask_v1.X.3.yaml index 2d12a458..3c14c5fd 100644 --- a/config/calibration/mask_v1.X.3.yaml +++ b/config/calibration/mask_v1.X.3.yaml @@ -61,25 +61,25 @@ dat: # HDF5's data_ext dataset) # Stars - - col_name: MASK_n4 + - col_name: MASK_4_Stars label: "Stars" kind: equal value: False # Manual mask - - col_name: MASK_n8 + - col_name: MASK_8_Manual label: "manual mask" kind: equal value: False # r-band footprint - - col_name: MASK_n64 + - col_name: MASK_64_r label: "r-band imaging" kind: equal value: False # Maximask - - col_name: MASK_n1024 + - col_name: MASK_1024_Maximask label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.4.yaml b/config/calibration/mask_v1.X.4.yaml index 93779a44..37b67325 100644 --- a/config/calibration/mask_v1.X.4.yaml +++ b/config/calibration/mask_v1.X.4.yaml @@ -60,37 +60,37 @@ dat: # Healsparse mask bits and post-processing columns (the comprehensive # HDF5's data_ext dataset) # Faint star halos - - col_name: MASK_n1 + - col_name: MASK_1_Faint_star_halos label: "Faint star halos" kind: equal value: False # Bright star halos - - col_name: MASK_n2 + - col_name: MASK_2_Bright_star_halos label: "Bright star halos" kind: equal value: False # Stars - - col_name: MASK_n4 + - col_name: MASK_4_Stars label: "Stars" kind: equal value: False # Manual mask - - col_name: MASK_n8 + - col_name: MASK_8_Manual label: "manual mask" kind: equal value: False # r-band footprint - - col_name: MASK_n64 + - col_name: MASK_64_r label: "r-band imaging" kind: equal value: False # Maximask - - col_name: MASK_n1024 + - col_name: MASK_1024_Maximask label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.5.yaml b/config/calibration/mask_v1.X.5.yaml index c80c5f04..bf717380 100644 --- a/config/calibration/mask_v1.X.5.yaml +++ b/config/calibration/mask_v1.X.5.yaml @@ -61,25 +61,25 @@ dat: # HDF5's data_ext dataset) # Stars - - col_name: MASK_n4 + - col_name: MASK_4_Stars label: "Stars" kind: equal value: False # Manual mask - - col_name: MASK_n8 + - col_name: MASK_8_Manual label: "manual mask" kind: equal value: False # r-band footprint - - col_name: MASK_n64 + - col_name: MASK_64_r label: "r-band imaging" kind: equal value: False # Maximask - - col_name: MASK_n1024 + - col_name: MASK_1024_Maximask label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.6.yaml b/config/calibration/mask_v1.X.6.yaml index 3eb33b95..fc021a2c 100644 --- a/config/calibration/mask_v1.X.6.yaml +++ b/config/calibration/mask_v1.X.6.yaml @@ -61,25 +61,25 @@ dat: # HDF5's data_ext dataset) # Stars - - col_name: MASK_n4 + - col_name: MASK_4_Stars label: "Stars" kind: equal value: False # Manual mask - - col_name: MASK_n8 + - col_name: MASK_8_Manual label: "manual mask" kind: equal value: False # r-band footprint - - col_name: MASK_n64 + - col_name: MASK_64_r label: "r-band imaging" kind: equal value: False # Maximask - - col_name: MASK_n1024 + - col_name: MASK_1024_Maximask label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.6_ppv1.yaml b/config/calibration/mask_v1.X.6_ppv1.yaml index 639dde4a..bcaffb15 100644 --- a/config/calibration/mask_v1.X.6_ppv1.yaml +++ b/config/calibration/mask_v1.X.6_ppv1.yaml @@ -88,25 +88,25 @@ dat: # HDF5's data_ext dataset) # Stars - - col_name: MASK_n4 + - col_name: MASK_4_Stars label: "Stars" kind: equal value: False # Manual mask - - col_name: MASK_n8 + - col_name: MASK_8_Manual label: "manual mask" kind: equal value: False # r-band footprint - - col_name: MASK_n64 + - col_name: MASK_64_r label: "r-band imaging" kind: equal value: False # Maximask - - col_name: MASK_n1024 + - col_name: MASK_1024_Maximask label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.7.yaml b/config/calibration/mask_v1.X.7.yaml index b92ddf58..585481d9 100644 --- a/config/calibration/mask_v1.X.7.yaml +++ b/config/calibration/mask_v1.X.7.yaml @@ -61,30 +61,30 @@ dat: # HDF5's data_ext dataset) # Bright star halos - - col_name: MASK_n2 + - col_name: MASK_2_Bright_star_halos label: "Bright star halos" kind: equal value: False # Stars - - col_name: MASK_n4 + - col_name: MASK_4_Stars label: "Stars" kind: equal value: False # Manual mask - - col_name: MASK_n8 + - col_name: MASK_8_Manual label: "manual mask" kind: equal value: False # r-band footprint - - col_name: MASK_n64 + - col_name: MASK_64_r label: "r-band imaging" kind: equal value: False # Maximask - - col_name: MASK_n1024 + - col_name: MASK_1024_Maximask label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.8.yaml b/config/calibration/mask_v1.X.8.yaml index 3f5e599d..5dafbb72 100644 --- a/config/calibration/mask_v1.X.8.yaml +++ b/config/calibration/mask_v1.X.8.yaml @@ -61,36 +61,36 @@ dat: # HDF5's data_ext dataset) # Faint star halos - - col_name: MASK_n1 + - col_name: MASK_1_Faint_star_halos label: "Faint star halos" kind: equal value: False # Bright star halos - - col_name: MASK_n2 + - col_name: MASK_2_Bright_star_halos label: "Bright star halos" kind: equal value: False # Stars - - col_name: MASK_n4 + - col_name: MASK_4_Stars label: "Stars" kind: equal value: False # Manual mask - - col_name: MASK_n8 + - col_name: MASK_8_Manual label: "manual mask" kind: equal value: False # r-band footprint - - col_name: MASK_n64 + - col_name: MASK_64_r label: "r-band imaging" kind: equal value: False # Maximask - - col_name: MASK_n1024 + - col_name: MASK_1024_Maximask label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.9.yaml b/config/calibration/mask_v1.X.9.yaml index 0baef956..906578df 100644 --- a/config/calibration/mask_v1.X.9.yaml +++ b/config/calibration/mask_v1.X.9.yaml @@ -61,25 +61,25 @@ dat: # HDF5's data_ext dataset) # Stars - - col_name: MASK_n4 + - col_name: MASK_4_Stars label: "Stars" kind: equal value: False # Manual mask - - col_name: MASK_n8 + - col_name: MASK_8_Manual label: "manual mask" kind: equal value: False # r-band footprint - - col_name: MASK_n64 + - col_name: MASK_64_r label: "r-band imaging" kind: equal value: False # Maximask - - col_name: MASK_n1024 + - col_name: MASK_1024_Maximask label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v1.X.9_im_sim.overlay.yaml b/config/calibration/mask_v1.X.9_im_sim.overlay.yaml index e334d0c5..9c15edab 100644 --- a/config/calibration/mask_v1.X.9_im_sim.overlay.yaml +++ b/config/calibration/mask_v1.X.9_im_sim.overlay.yaml @@ -48,7 +48,7 @@ ops: # --- mask bits (post-processing / coverage masks) ---------------------- - why: >- No coverage masks on sims: the healsparse mask-bit cuts (stars, manual - mask, n64, Maximask) are survey post-processing with no analogue in the + mask, r coverage, Maximask) are survey post-processing with no analogue in the simulated tiles. drop: |2 @@ -56,25 +56,25 @@ ops: # HDF5's data_ext dataset) # Stars - - col_name: MASK_n4 + - col_name: MASK_4_Stars label: "Stars" kind: equal value: False # Manual mask - - col_name: MASK_n8 + - col_name: MASK_8_Manual label: "manual mask" kind: equal value: False # r-band footprint - - col_name: MASK_n64 + - col_name: MASK_64_r label: "r-band imaging" kind: equal value: False # Maximask - - col_name: MASK_n1024 + - col_name: MASK_1024_Maximask label: "maximask" kind: equal value: False diff --git a/config/calibration/mask_v2.0.yaml b/config/calibration/mask_v2.0.yaml index e32a647a..6310f66e 100644 --- a/config/calibration/mask_v2.0.yaml +++ b/config/calibration/mask_v2.0.yaml @@ -1,28 +1,28 @@ # Config file for masking and calibration, ShapePipe v2 catalogues. # # ShapePipe v2 replaces the single IMAFLAGS_ISO column with per-reason boolean -# mask columns MASK_n, where True means MASKED. They are therefore cut -# with `kind: equal, value: False` (keep the un-masked objects): +# mask columns MASK__