From 09edd5e71bb740f9216be255b8ee1d6e640de388 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sun, 30 Aug 2026 20:25:38 -0400 Subject: [PATCH 01/40] docs(astra): record the pipeline's scientific decisions in astra.yaml MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ShapePipe's scientific choices — detection thresholds, masking geometry, star selection, PSF model, ngmix priors and seeding, flag semantics, completeness floors — live in code and committed configs with their reasoning nowhere, or spread across PRs, papers and comments. astra.yaml gathers them: 50 decisions across eight sub-analyses, each with its rationale, the alternatives that were rejected and why, and a greppable anchor back to the code or config that implements it. universes/committed.yaml pins the option this branch selects for every one. The record is ASTRA (astra-tools; `uvx astra-tools@0.2.17 guide`), applied here at codebase level rather than to a single analysis. Conventions are stated in the file's header: anchors as `path::symbol` / `path#SECTION.KEY` and never line numbers, [HARDCODED] for a scientific value with no config exposure, [LINT] for a place where the record and the code — or the code and itself — disagree, [PENDING #NNN] for state not yet on develop. Authoring it surfaced nine such lints, two of which #873 fixes, and mapped ten places where the published Guinot+22 / Farrens+22 descriptions have drifted from the code since publication; 16 decisions carry verbatim paper quotes as prior insights. CLAUDE.md gains the standing instruction: a scientific change is not finished until the record is, amended in the same PR. The membership test is whether a different defensible choice would change which objects enter the shear catalogue, or the numbers attached to them. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01Y2muA2sRojbxRNxU2SKQeP --- CLAUDE.md | 42 +- astra.yaml | 1837 ++++++++++++++++++++++++++++++++++++++ universes/committed.yaml | 70 ++ 3 files changed, 1948 insertions(+), 1 deletion(-) create mode 100644 astra.yaml create mode 100644 universes/committed.yaml diff --git a/CLAUDE.md b/CLAUDE.md index 01deff1fc..56579f03c 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -118,4 +118,44 @@ keep in their own stores outside it. A `.felt/` directory (a markdown "fiber" no store used with the `felt` CLI) is **not tracked here**: it's gitignored, and where it exists it's a machine-local symlink into a private, separately git-synced store, so a fresh clone won't have one. Record durable decisions in the PR, issue, or docs -where the change lives. +where the change lives — and *scientific* decisions in `astra.yaml`, below. + +## Scientific decisions live in `astra.yaml` + +`astra.yaml` at the repo root is the pipeline's decision record: every +consequential scientific choice embedded in the code and the committed configs, +each with its rationale, the alternatives that were considered and why they were +rejected, and an anchor back to the code or config that implements it. +`universes/committed.yaml` pins the option this branch's configuration +selects for every decision. The format +is ASTRA; `uvx astra-tools@0.2.17 guide` is the briefing and +`uvx astra-tools@0.2.17 spec` the field reference. + +**A scientific change is not finished until the record is.** When a change moves +what the pipeline measures, amend `astra.yaml` in the same PR — add the decision +if it is new, or edit its rationale, options and anchors if it moved — pin the +selected option in `universes/committed.yaml`, and say so in the PR description. +Purely technical changes (refactors, performance, packaging, I/O) leave it alone, +except where they move a value the record carries: the completeness floors in +`workflow/scripts/completeness.py` are orchestration code holding a scientific +decision. + +The membership test is whether *a different defensible choice would change which +objects enter the shear catalogue, or the numbers attached to them.* Detection +threshold and deblending contrast, masking geometry, star-selection cuts, PSF +model degree, ngmix priors and seeding, flag semantics, completeness floors — in. +Manifest sentinels, chunk sizes, allocation strategy, directory layout — out; +those live in the PR and the PRD. + +The file's own header states the conventions it follows. In short: every +rationale ends with a greppable `Anchor: path::symbol; path#SECTION.KEY` +sentence whose refs never cite line numbers; `[HARDCODED]` marks a scientific value +with no config exposure; `[LINT]` marks a place where the record and the code, or +the code and itself, disagree. Validate before committing: + +```bash +uvx astra-tools@0.2.17 validate +``` + +The record was authored against this branch's workflow configs; entries marked +`[PENDING #NNN]` describe state that has not yet reached `develop`. diff --git a/astra.yaml b/astra.yaml new file mode 100644 index 000000000..70b25b554 --- /dev/null +++ b/astra.yaml @@ -0,0 +1,1837 @@ +# ASTRA record for ShapePipe: the scientific decisions embedded in the code and +# the committed configs, with their reasoning and the alternatives that were +# rejected. It is the place scientific decisions are written down — see the +# "Scientific decisions" section of CLAUDE.md for when and how to amend it. +# +# The record describes the pipeline as orchestrated by workflow/Snakefile +# (PRD CosmoStat/shapepipe#848, PR #852). Conventions: +# +# * Decisions anchor to code, not recipes. Every rationale ends with one +# sentence "Anchor: ; ; ..." in a strict, greppable grammar. +# Each ref is a path relative to the shapepipe repo root, in one of three +# forms: CODE `path::symbol`, CONFIG `path#SECTION.KEY` (or `path#KEY` for +# sectionless .sex/.psfex/.ww/.param files), FILE `path` for a whole file +# or package. No line numbers — they rot; a line-level fact names its +# enclosing symbol. The analysis-ASTRA rule "never hardcode; reference via +# {decisions.x}" cannot hold in a codebase — the committed configs ARE the +# values. [HARDCODED] marks a scientific value living in code with no +# config exposure: the silent defaults the record exists to surface. +# * The default universe IS the committed configuration (universes/committed). +# Alternatives are excluded-with-reasons or genuinely open forks. +# * Sub-analyses follow the pipeline's methodological units — masking, +# detection, preparation, star selection + PSF, shape measurement, PSF +# diagnostics, survey geometry, catalogue assembly — not its ~20 Snakemake +# rules. Cross-cutting decisions stay top-level. A prior_insight repeated +# inside a sub-analysis carries a `_local` suffix: ids are scoped, and the +# duplicate keeps the sub-analysis readable on its own. +# * Outputs are representative product FAMILIES (one final_cat per tile), +# not enumerable artifacts; no recipes — the executor is the Snakemake +# workflow. +# * [LINT] marks places where this record and the code already disagree, or +# where the code disagrees with itself — found while authoring this file. +# * [PENDING #NNN] marks state that is live on feat/snakemake-orchestration +# — and therefore in smk-g4, the 34-tile validation campaign run under +# this branch — but not yet merged to develop. The record follows the +# branch and names the open PR. +# * A `path#KEY` anchor names the key's position in the file, not its +# activation: where the decision is "this is deliberately off", the key +# it points at may be commented out (e.g. final_cat.param#SPREAD_CLASS). + +version: "0.0.14" +name: ShapePipe scientific decisions +description: >- + Codebase-level decision record for the ShapePipe weak-lensing pipeline + (UNIONS/CFIS). Membership test: "a different defensible choice would change + which objects enter the shear catalogue, or the numbers attached to them." + Workflow mechanics that reproduce identical numbers (manifest sentinels, + clean-cascade cut, directory() outputs, allocation strategy, chunking under + position seeding) are deliberately absent; they live in the PRD and code. +tags: [shapepipe, weak-lensing, unions, codebase-record] +container: shapepipe-develop-runtime.sif + +inputs: + - id: tile_images + type: data + source: CADC-staged CFIS/UNIONS r-band tile stacks + exposure triplets (workflow/config.yaml) + description: >- + Pre-staged P3 tiles and single-exposure image/weight/flag triplets on + /project; get_images runs with RETRIEVE=symlink against this store. + - id: gsc_star_catalogue + type: data + source: GSC 2.3 (Vizier I/305/out) cone queries — scripts/python/create_star_cat.py + description: >- + Reference star catalogue driving bright-star masking. Catalogue choice, + query geometry, and magnitude handling are decisions in the masking + sub-analysis. + +outputs: + - id: final_cat + type: data + format: fits + description: >- + Per-tile shear catalogue family, the terminal science product (one per + campaign tile; make_cat_runner). Column selection and failure sentinels + are decisions in catalogue_assembly. + inputs: [tile_images] + decisions: [per_unit_count_floor, postage_stamp_size, photometric_zeropoint] + +decisions: + + # ── cross-cutting ──────────────────────────────────────────────────────── + + per_unit_count_floor: + label: Per-unit completeness policy under partial failure + rationale: >- + A 40-CCD stage where some CCDs legitimately produce nothing (sparse CCD, + setools rejects everything) cannot be all-or-nothing. The field's + converged answer (DES PSF blacklist, Rubin quantum registry) is per-unit + outcome records gated on a quality floor: record the attrition, fail + loud only below the floor, continue the survey. The floor VALUES are the + scientific content — how much silent per-CCD attrition can enter the + catalogue. The COMPLETENESS table holds them (exp_split + expect=121/floor=41, exp_mask expect=40/floor=1, psfex expect=80/floor=2, + psfex_interp floor=0 warn-only). Related leak the floor does not cover: + merge_sep_cats warns-and-skips a missing ngmix chunk, silently shrinking + a tile's shape catalogue below the floor's radar; and make_cat's own 10% + size-shortfall guard is commented out (see + catalogue_assembly.shape_catalogue_shortfall_guard). + Anchor: workflow/scripts/completeness.py::COMPLETENESS; + src/shapepipe/modules/merge_sep_cats_package/merge_sep_cats.py::MergeSep.process. + default: count_floor + options: + count_floor: + label: Count-floor table (expect/floor per runner; fail below floor) + insights: [des_psf_blacklist, guinot22_star_floor_22] + all_or_nothing: + label: Every expected sub-product required + excluded: true + excluded_reason: >- + Legitimately-absent CCDs would fail whole exposures and poison their + downstream cone; Snakemake has no optional-output primitive; field + precedent is tolerated, recorded attrition. + no_floor: + label: Accept whatever is produced, no gate + excluded: true + excluded_reason: >- + Silent attrition — a stage producing 2 of 40 CCDs would flow into + the catalogue unremarked. + + postage_stamp_size: + label: Postage-stamp size, 51 px everywhere + rationale: >- + One number pins three coupled apertures: the SExtractor vignet cut + around each detection (VIGNET(51,51) in default_noimaflags.param / + default.param, VIGNET_SIZE=51 in the dormant external-catalogue path, + example/cfis/config_tile_Uc.ini), the vignetmaker + stamps that feed ngmix (STAMP_SIZE=51 in config_tile_PiViVi.ini, both + runs; nearest-pixel centring, no sub-pixel interpolation in + VignetMaker._get_stamp), and the PSFEx model stamp (PSF_SIZE 51,51 in + default.psfex). The stamp IS the pixel data ngmix fits: it bounds + measurable galaxy size and truncates the wings of large galaxies. + Rationale for 51 not recorded in code. + Anchor: workflow/config/cfis/default_noimaflags.param#VIGNET; + example/cfis/config_tile_Uc.ini#READ_EXT_SEXCAT_RUNNER.VIGNET_SIZE; + workflow/config/cfis/config_tile_PiViVi.ini#VIGNETMAKER_RUNNER_RUN_1.STAMP_SIZE; + workflow/config/cfis/default.psfex#PSF_SIZE; + src/shapepipe/modules/vignetmaker_package/vignetmaker.py::VignetMaker._get_stamp. + default: px_51 + options: + px_51: + label: 51x51 px (~9.5 arcsec at 0.187"/px) + larger_adaptive: + label: Larger or size-adaptive stamps + excluded: true + excluded_reason: >- + Not wired; would need coupled changes in three places (a change in + any one alone desynchronises galaxy stamp, PSF stamp, and vignet). + + photometric_zeropoint: + label: Magnitude zero-point convention, fixed 30.0 on tiles + rationale: >- + Tiles use a hard-coded MAG_ZEROPOINT 30.0 for every tile + (default_tile.sex; ZP_FROM_HEADER=False in config_tile_Sx.ini), and + ngmix repeats it (MAG_ZP=30.0 in config_tile_Ng_template.ini). + Exposures instead read the per-image header zero-point + (ZP_FROM_HEADER=True, ZP_KEY=PHOTZP in config_exp_psfex.ini). The tile + convention leans on MegaPipe's calibrated stacks; the star-selection + magnitude window (18-22) and mask magnitude limits inherit whichever + convention their stage uses. SExtractorCaller.get_zero_point is the + header-reading path, unused on tiles. + Anchor: workflow/config/cfis/default_tile.sex#MAG_ZEROPOINT; + workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER; + workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.MAG_ZP; + workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.ZP_KEY; + src/shapepipe/modules/sextractor_package/sextractor_script.py::SExtractorCaller.get_zero_point. + default: fixed_30_tiles_header_exposures + options: + fixed_30_tiles_header_exposures: + label: Tiles fixed 30.0; exposures from header PHOTZP + header_everywhere: + label: Per-image header zero-points on tiles too + excluded: true + excluded_reason: >- + MegaPipe stacks are calibrated to ZP 30 by construction; per-tile + header reads add a failure path for no expected numerical change. + (If that claim is wrong, this is a real fork — verify.) + + baseline_validation_criterion: + label: Validation criterion against the v2.0 bash baseline + rationale: >- + Because shape_measurement.ngmix_seed_mode deliberately changes noise + streams, P1 validation against v2.0 is statistical parity + (population-level agreement), not bit parity. Everything upstream of + ngmix (through PSFEx) validated bit-exactly (P0: 4/4 PASS). This + defines the evidence standard for "the same pipeline" — surfaced to the + collaboration as open Q5 in PRD #848. + [PENDING #873] Run-to-run determinism, which is a different property + from parity with v2.0, is now complete. With the setools star split + seeded (star_selection_psf.psf_train_validation_split) the last unseeded + draw in the science chain is gone: two runs of this code over the same + inputs now produce the same PSF star sample, the same PSF models and + the same shapes, which they did not before. That also settles a tension + this record carried — the bit-parity claim above sat next to an + unseeded star split that could not have been bit-reproducible, and the + P0 exposure-stage comparison did see PSF-validation CCD attrition + differ between the two sides. Statistical rather than bit parity is + therefore demanded only against the v2.0 baseline, not between runs of + the current pipeline. + Anchor: workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.SEED_FROM_POSITION; + src/shapepipe/modules/ngmix_package/ngmix.py::position_seed; + src/shapepipe/modules/setools_package/setools.py::SETools._make_rand_split. + default: statistical_parity + options: + statistical_parity: + label: Population-level agreement in shear observables + bit_parity: + label: Bit-identical catalogues + excluded: true + excluded_reason: >- + Impossible by construction once the seed mode changed; requiring it + would freeze the chunk-dependent v2.0 RNG forever. + +prior_insights: + des_psf_blacklist: + claim: >- + DES enters a CCD's PSF model into a blacklist rather than failing the + exposure - in Y3, any CCD with fewer than 25 stars surviving outlier + rejection is blacklisted and excluded downstream (~2% of data removed), + and processing proceeds. + created_at: "2026-07-16T00:00:00Z" + evidence: + - id: ev_jarvis_y3 + doi: "10.48550/arXiv.2011.03409" + quote: + exact: "we enter it into a" + suffix: " \u201cblacklist\u201d and exclude this CCD" + location: { page: 10 } + guinot22_star_floor_22: + claim: >- + The published ShapePipe/UNIONS analysis applies a per-CCD quality floor + rather than failing whole exposures: a CCD with fewer than 22 selected + stars is discarded for PSF estimation and contributes no epoch to the + shape measurement, while processing continues. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_star_floor + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'The dashed line represents the cut at 22 stars/CCD below which the CCD is discarded for the PSF estimation.' + location: { page: 4 } + +findings: + orchestration_parity: + claim: >- + The Snakemake orchestration reproduces the bash baseline bit-exactly + through PSFEx (P0 validation, 4/4 PASS on the 186/187 quad). + Read it with two caveats. It is a statement about the pre-#873 code: + both #873 changes move products (a different realised star split, a + stricter science-path star gate), so re-establishing parity would mean + regenerating the baseline under the current branch. And the parity is + bit-exact in the products compared, not everywhere: the P0 + exposure-stage comparison did see PSF-validation CCD attrition differ + between the two sides, which the then-unseeded star split explains + (see baseline_validation_criterion). + created_at: "2026-08-19T00:00:00Z" + evidence: + - id: ev_final_cat + artifact: final_cat + record_authoring_found_defects: + claim: >- + Nine places in the code disagree with themselves or with their + documentation, each carried as a [LINT] mark: the 22-vs-20 + STAR_THRESH mismatch between PSF validation and science interpolation; + the unseeded train/validation rand_split (setools.py:664 — the star + sample entering the PSF model is irreproducible run-to-run); additive + (non-bitwise) mask-plane combination, safe today only because the + committed flag values are disjoint; the dead MESSIER_PIXEL_SCALE config + key; final_cat.param requesting IMAFLAGS_ISO that the merged catalogue + never receives; the centroid_source default disagreement (runner "wcs" + vs module "hsm", latent for direct callers); setools logging a FWHM cut + (mode +- 0.1 px in arcsec) half the applied one (mode +- 0.2 px), and + mixing pixel scales 0.187/0.186 within one file; TILE_LIST + overlap-flagging documented but never implemented; the mccd_plots + module docstring advertising rho statistics that live downstream now. + Status: the first two are fixed. CosmoStat/shapepipe#873 seeds the + rand_split and raises the science-path STAR_THRESH to 22, and commit + 90782098 mirrors that threshold into the workflow's own committed + config fork. #873 is OPEN against develop; both fixes are live on + feat/snakemake-orchestration only, and the 34-tile smk-g4 campaign is + the first run under them. The other seven stand, including the + mccd_plots docstring that still advertises rho statistics the package + no longer computes. + created_at: "2026-08-29T00:00:00Z" + derived: true + evidence: + - id: ev_final_cat_defects + artifact: final_cat + code_paper_divergence: + claim: >- + The two ShapePipe papers state roughly 17 of this record's 50 decisions + (now carried as prior_insights with verbatim quotes), have drifted from + the code on 10 of them since publication, and are silent on the rest. + Among them: DETECT_MINAREA 10 -> 5; DEBLEND_MINCONT 0.001 -> + 0.0005 on tiles; tile background AUTO -> MANUAL 0; in-line spread-model + star/galaxy classification -> disabled and deferred downstream; HSM + moment initialisation -> WCS centroids and prior-based guesses; GSC 2.2 + via cdsclient -> GSC 2.3 via astroquery; PSF acceptance 22 stars/CCD + published for the science path vs 20 committed there — closed since by + #873 + 90782098, which put the science path on 22, so nine of the ten + drifts remain open on the orchestration branch. + created_at: "2026-08-29T00:00:00Z" + derived: true + evidence: + - id: ev_final_cat_divergence + artifact: final_cat + +analyses: + + # ═════════════════════════════════════════════════════════════════════════ + masking: + description: >- + Which pixels are excluded before anything is measured. Modules: + src/shapepipe/modules/mask_package/mask.py (halo/spike/DSO/border + builders, WeightWatcher driver), scripts/python/create_star_cat.py + (star-catalogue fetch), configs config_exp_Ma.ini + + config_onthefly.mask / config_tile_onthefly.mask + mask_default/. + [LINT] MESSIER_PIXEL_SCALE is set in config_tile_onthefly.mask but + never read — mask_dso takes pixel scale from the WCS. [LINT] + _build_final_mask combines mask planes by ADDITION (mask.py:1141+), + not bitwise OR; the committed flag values (2/4/16/32/128) are disjoint + so no live collision exists, but any future duplicate value corrupts + the flag semantics silently. (FLAG_OUTFLAGS 2 in default.ww is inert: + no input flag image is passed to WeightWatcher — mask.py:1047-1074.) + inputs: + - id: ccd_images + type: data + source: split per-CCD exposure images + weights + CFIS flag maps (exp_split family) + - id: star_catalogue + type: data + source: GSC 2.3 per-exposure catalogues (exp_star_cat cache) + outputs: + - id: exposure_mask + type: data + format: fits + description: Per-CCD pipeline flag maps (run_sp_exp_Ma family). + decisions: + [star_catalogue_query, star_magnitude_definition, + bright_star_mask_geometry, deep_sky_object_masking, + border_mask_width, pixel_threshold_flags, external_flag_usage] + decisions: + star_catalogue_query: + label: Reference star catalogue and query for bright-star masking + rationale: >- + GSC 2.3 (Vizier I/305/out), columns GSC2.3/RAJ2000/DEJ2000/Fmag/ + jmag/Vmag/Nmag/Class, cone radius covering the full CCD mosaic, no + magnitude cut at query time. GSC 2.2 rejected in a code comment + beside the catalogue ID ("does not have Fmag"). + Provenance hazard: the star-cat cache is not keyed by script + version — a semantic change to this query reruns the rule but takes + the skip-if-exists branch; clear the cache by hand for the change + to reach the data (workflow/config.yaml star_cats comment). + Query geometry: search radius = half the image diagonal about the + field centre (Mask._get_image_radius); source precedence: with + CDSCLIENT_PATH set in the .mask configs the online-query branch + wins unless an external star cat is passed (USE_EXT_STAR=True in + config_exp_Ma.ini routes the exp_star_cat cache in). + Published description (Farrens+22 p.2): cdsclient downloads GSC 2.2 + (with cdsclient 3.84 pinned in its Table A.1); current code: GSC 2.3 + queried through astroquery — two things drifted, the catalogue + version (Fmag is needed for the magnitude cut) and the query + transport, since cdsclient is never invoked yet survives as a + required-but-unused CDSCLIENT_PATH still set to the stale + findgsc2.2 in config_tile_onthefly.mask. + Anchor: scripts/python/create_star_cat.py::CDS_CAT_ID; + src/shapepipe/modules/mask_package/mask.py::Mask._CDS_cat_ID; + src/shapepipe/modules/mask_package/mask.py::Mask._cds_keys; + workflow/config.yaml. + default: gsc_23_vizier + options: + gsc_23_vizier: + label: GSC 2.3 cone queries, all bands, no query-time mag cut + insights: [farrens22_star_cat_on_disk] + gaia: + label: Gaia-based star catalogue + excluded: true + excluded_reason: >- + Not wired. Deeper and better photometry; switching changes mask + geometry and hence the selection function — a real DR-level fork. + star_magnitude_definition: + label: Per-star magnitude for mask scaling + rationale: >- + mag = unweighted mean of the finite GSC bands among F, j, V, N; + stars with no finite band are logged and not masked; only Class==0 + objects masked. Comment records why not a naive mean: NaN bands + would NaN-poison the mag < mag_limit test, leaving exactly the + bright stars with incomplete photometry unmasked. + Anchor: src/shapepipe/modules/mask_package/mask.py::Mask._create_mask. + default: mean_finite_bands + options: + mean_finite_bands: { label: Mean of finite F/j/V/N; Class==0 only } + single_band: + label: Single-band (Fmag) magnitude + excluded: true + excluded_reason: Drops stars with missing Fmag from masking entirely. + bright_star_mask_geometry: + label: Halo + diffraction-spike mask geometry and magnitude scaling + rationale: >- + DS9 polygon templates scaled linearly with magnitude about a pivot: + halo HALO_MAG_LIM=13, HALO_SCALE_FACTOR=0.05, HALO_MAG_PIVOT=13.8 + (halo_mask.reg, ~270 px); spike SPIKE_MAG_LIM=18, + SPIKE_SCALE_FACTOR=0.3, SPIKE_MAG_PIVOT=13.8 + (MEGAPRIME_star_i_13.8.reg); scaling = 1 - factor*(mag-pivot), + floored at 0.1 by Mask._scaling_min. + Identical in exposure and tile configs. Template filename encodes + provenance (MegaPrime i-band mag-13.8 star); numeric rationale not + recorded. Note the 5-mag gap: stars in 13-18 get spikes but no halo. + Anchor: workflow/config/cfis/config_onthefly.mask#HALO_PARAMETERS.HALO_MAG_LIM; + workflow/config/cfis/config_onthefly.mask#SPIKE_PARAMETERS.SPIKE_MAG_LIM; + workflow/config/cfis/mask_default/halo_mask.reg; + workflow/config/cfis/mask_default/MEGAPRIME_star_i_13.8.reg; + src/shapepipe/modules/mask_package/mask.py::Mask._create_mask; + src/shapepipe/modules/mask_package/mask.py::Mask._scaling_min. + default: megaprime_polygon_linear_scaling + options: + megaprime_polygon_linear_scaling: + label: Fixed MegaPrime templates, linear mag scaling, floor 0.1 + radial_profile_fit: + label: Per-star radial-profile-driven mask size + excluded: true + excluded_reason: Not wired; the survey precedent is template-based. + deep_sky_object_masking: + label: Messier + NGC objects masked as circles, no enlargement + rationale: >- + Circles of radius max(size_X, size_Y), MESSIER_SIZE_PLUS=0, + NGC_SIZE_PLUS=0 (function default is 0.1 — the 0 is a choice); + flags 16/32. A comment records the overlap-test fix (corner-only + test missed small interior objects). + Anchor: workflow/config/cfis/config_onthefly.mask#MESSIER_PARAMETERS.MESSIER_SIZE_PLUS; + workflow/config/cfis/config_tile_onthefly.mask#NGC_PARAMETERS.NGC_SIZE_PLUS; + src/shapepipe/modules/mask_package/mask.py::Mask.mask_dso. + default: circles_no_padding + options: + circles_no_padding: + label: "size_plus = 0: mask exactly the catalogued extent" + insights: [farrens22_messier_mask] + padded_circles: + label: size_plus > 0 (code default 0.1) + excluded: true + excluded_reason: >- + Rationale for dropping the padding not recorded; flagged as a + question rather than an endorsed exclusion. + border_mask_width: + label: CCD border mask, 50 px on exposures, none on tiles + rationale: >- + Exposures BORDER_WIDTH=50 (flag 4); tiles BORDER_MAKE=False. + Mask.mask_border's own default is 100 — the committed 50 is a + choice, unrecorded. Trims CCD edges where PSF and astrometry + degrade; changes the effective footprint. + Anchor: workflow/config/cfis/config_onthefly.mask#BORDER_PARAMETERS.BORDER_WIDTH; + workflow/config/cfis/config_tile_onthefly.mask#BORDER_PARAMETERS.BORDER_MAKE; + src/shapepipe/modules/mask_package/mask.py::Mask.mask_border. + default: px50_exposures_only + options: + px50_exposures_only: + label: 50 px exposure borders; tiles unmasked + insights: [farrens22_border_mask] + px100: + label: 100 px (module default) + excluded: true + excluded_reason: Halves usable edge area for no recorded gain. + pixel_threshold_flags: + label: WeightWatcher weight/flag thresholds into mask bits + rationale: >- + WEIGHT_MIN 0, WEIGHT_MAX 1000, WEIGHT_OUTFLAGS 1; FLAG_MASKS 0x01, + FLAG_OUTFLAGS 2; POLY_OUTWEIGHTS 0. Zero-weight and externally + flagged pixels excluded on these thresholds. Values are stock, not + derived from the CFIS weight distribution; rationale not recorded. + The FLAG_* keys are inert in the committed invocation — no flag + image is passed to WeightWatcher by Mask._exec_WW. + Anchor: workflow/config/cfis/mask_default/default.ww#WEIGHT_MIN; + workflow/config/cfis/mask_default/default.ww#FLAG_MASKS; + src/shapepipe/modules/mask_package/mask.py::Mask._exec_WW. + default: stock_ww_thresholds + options: + stock_ww_thresholds: + label: Stock WeightWatcher thresholds + insights: [farrens22_weightwatcher] + external_flag_usage: + label: CFIS external flag maps folded into exposure masks + rationale: >- + USE_EXT_FLAG=True on exposures (imports CADC-provided bad-pixel / + cosmic-ray / trail flags); EF_MAKE=False on tiles. The external + plane enters via Mask._build_final_mask's path_external_flag branch. + Anchor: workflow/config/cfis/config_exp_Ma.ini#MASK_RUNNER.USE_EXT_FLAG; + workflow/config/cfis/config_tile_onthefly.mask#EXTERNAL_FLAG.EF_MAKE; + src/shapepipe/modules/mask_package/mask.py::Mask._build_final_mask. + default: exposures_only + options: + exposures_only: { label: "External flags on exposures, not tiles" } + ignore_external: + label: Pipeline-generated masks only + excluded: true + excluded_reason: Discards upstream knowledge of bad pixels. + prior_insights: + farrens22_star_cat_on_disk: + claim: >- + The ShapePipe release paper documents an on-disk star catalogue, in + GSC format, as a supported substitute for the online query, + motivated by compute nodes without internet access. + created_at: "2022-06-01T00:00:00Z" + evidence: + - id: ev_farrens22_star_cat_disk + doi: "10.48550/arXiv.2206.14689" + quote: + exact: 'Alternatively, a star catalogue available on disk (with the same format as the GSC) can also be used' + location: { page: 2 } + farrens22_messier_mask: + claim: >- + Messier objects are named in the published masking procedure as one + of the object classes ShapePipe masks. + created_at: "2022-06-01T00:00:00Z" + evidence: + - id: ev_farrens22_messier + doi: "10.48550/arXiv.2206.14689" + quote: + exact: 'Messier objects, and border regions.' + location: { page: 2 } + farrens22_border_mask: + claim: >- + CCD border regions are named in the published masking procedure as + one of the regions ShapePipe masks. + created_at: "2022-06-01T00:00:00Z" + evidence: + - id: ev_farrens22_border + doi: "10.48550/arXiv.2206.14689" + quote: + exact: 'Messier objects, and border regions.' + location: { page: 2 } + farrens22_weightwatcher: + claim: >- + The published pipeline generates the mask image itself with + WeightWatcher (Marmo & Bertin 2008), fixing the tool but none of its + threshold values. + created_at: "2022-06-01T00:00:00Z" + evidence: + - id: ev_farrens22_ww + doi: "10.48550/arXiv.2206.14689" + location: { page: 2 } + + # ═════════════════════════════════════════════════════════════════════════ + detection: + description: >- + Object detection on r-band tiles (single-image mode) and exposures (for + star finding). Module: src/shapepipe/modules/sextractor_package/ + sextractor_script.py (config assembly, ZP/background overrides, + post-processing that assigns per-epoch CCD membership). Configs: + config_tile_Sx.ini + default_tile.sex + default.conv + + default_noimaflags.param (tiles); default_exp.sex (exposures — same + thresholds, but DEBLEND_MINCONT 0.001 vs tile 0.0005 and BACK_TYPE AUTO + vs tile MANUAL 0, both deliberate and unexplained divergences). + [LINT] final_cat.param (consumed by the post-proc merge_final_cat, + not by make_cat) requests IMAFLAGS_ISO, but the tile chain never + produces it (FLAG_IMAGE=False, default_noimaflags.param); the + exposure-side IMAFLAGS_ISO stays exposure-side (merge_starcat.py:807 + only). The merged catalogue never receives the column. + inputs: + - id: tile_stack + type: data + source: MegaPipe r-band tile stack + weight (uncompressed, merged headers) + outputs: + - id: tile_sexcat + type: data + format: fits + description: Per-tile SExtractor LDAC catalogue with per-epoch CCD membership. + decisions: + [detection_threshold_policy, deblending_policy, background_model, + weighting_and_interpolation, detection_source_mode, + epoch_membership_ccd_bounds, photometry_parameters, + cleaning_and_neighbour_masking] + decisions: + photometry_parameters: + label: Photometric aperture definitions — Kron parameters, apertures, half-light fraction + rationale: >- + PHOT_AUTOPARAMS 2.5,3.5 (Kron factor / minimum radius), + PHOT_APERTURES 5 px, PHOT_FLUXFRAC 0.5, BACKPHOTO_TYPE GLOBAL — + identical in both .sex files. MAG_AUTO is the axis of the + star-selection magnitude box AND the catalogue magnitude; FLUX_AUTO + is PSFEx's photometric normalisation (default.psfex PHOTFLUX_KEY). + A different Kron factor shifts magnitudes systematically, moving + which stars build the PSF model and every magnitude-based + downstream cut. Rationale not recorded (stock values). Anchor: + workflow/config/cfis/default_tile.sex#PHOT_AUTOPARAMS; + workflow/config/cfis/default_exp.sex#PHOT_AUTOPARAMS; + workflow/config/cfis/default.psfex#PHOTFLUX_KEY. + default: kron_25_35 + options: + kron_25_35: { label: "Kron 2.5/3.5, aperture 5 px, FLUXFRAC 0.5, global background" } + cleaning_and_neighbour_masking: + label: Spurious-detection cleaning and neighbour-pixel correction + rationale: >- + CLEAN Y with CLEAN_PARAM 1.0 deletes detections consistent with + being wings of a brighter neighbour — a post-deblend change to the + object list; MASK_TYPE CORRECT replaces neighbour pixels during + photometry (vs BLANK/NONE), changing fluxes and windowed moments + of blends. Identical in both .sex files; rationale not recorded. + Anchor: workflow/config/cfis/default_tile.sex#CLEAN; + workflow/config/cfis/default_tile.sex#MASK_TYPE; + workflow/config/cfis/default_exp.sex#CLEAN. + default: clean_1_correct + options: + clean_1_correct: { label: "CLEAN 1.0 + MASK_TYPE CORRECT" } + detection_threshold_policy: + label: Detection significance, minimum area, matched filter + rationale: >- + DETECT_THRESH 1.5 sigma RELATIVE, ANALYSIS_THRESH 1.5, + DETECT_MINAREA 5, FILTER default.conv (3x3 pyramid kernel, "all + ground, FWHM = 2 pixels" — vs CFIS seeing ~0.65 arcsec = 3.5 px at + 0.187"/px, so the filter is not matched to the survey PSF). + Sets the faint end of the source sample. Rationale not recorded + (stock EB 2017 header). + Published description (Guinot+22 p.5, Table 2): DETECT_MINAREA 10; + current code: 5, in default_exp.sex as well as default_tile.sex — + the small-object end has been loosened since publication on both the + star-detection and tile-detection passes, while DETECT_THRESH 1.5 + RELATIVE and the 3x3 FWHM=2 px kernel still match. + Anchor: workflow/config/cfis/default_tile.sex#DETECT_THRESH; + workflow/config/cfis/default_tile.sex#DETECT_MINAREA; + workflow/config/cfis/default.conv. + default: thresh_1p5_minarea5_fwhm2px_filter + options: + thresh_1p5_minarea5_fwhm2px_filter: + label: 1.5 sigma, minarea 5, FWHM=2px kernel + seeing_matched_filter: + label: Kernel matched to CFIS seeing (~3.5 px) + excluded: true + excluded_reason: >- + Not wired; would change depth and the faint-end selection + function — a real fork, excluded only as not-the-committed-path. + deblending_policy: + label: Deblending sub-thresholds and contrast + rationale: >- + DEBLEND_NTHRESH 32, DEBLEND_MINCONT 0.0005 on tiles (2x more + aggressive splitting than the exposure 0.001 and 10x more than the + SExtractor default 0.005 — divergences not documented), CLEAN Y + PARAM 1.0. Controls object count, centroids, and blend + contamination in shapes. + Published description (Guinot+22 p.5, Table 2, galaxy detection on + the stacked tiles): DEBLEND_MINCONT 0.001; current code: 0.0005 on + tiles, with only the exposure side still carrying 0.001 — the + divergence lands on precisely the configuration the paper documents. + NTHRESH 32 matches. + Anchor: workflow/config/cfis/default_tile.sex#DEBLEND_MINCONT; + workflow/config/cfis/default_exp.sex#DEBLEND_MINCONT. + default: mincont_5em4_tiles + options: + mincont_5em4_tiles: { label: "MINCONT 0.0005 tiles / 0.001 exposures" } + background_model: + label: Tile background fixed to zero, not estimated + rationale: >- + BACK_TYPE MANUAL, BACK_VALUE 0.0, BKG_FROM_HEADER=False on tiles — + trusts MegaPipe stack background removal; exposures use BACK_TYPE + AUTO (64/3 mesh). Residual sky offsets propagate into thresholds, + fluxes, completeness. Divergence deliberate, unexplained. + Published description (Guinot+22 p.5): Table 2's caption asserts all + non-tabulated SExtractor parameters keep their defaults, i.e. + BACK_TYPE AUTO, and the paper never mentions the background choice + at all; current code: BACK_TYPE MANUAL with BACK_VALUE 0.0 on tiles, + identically in workflow/ and example/ — the standing tile + configuration, not a one-off, diverging silently from the published + parametrisation. + Anchor: workflow/config/cfis/default_tile.sex#BACK_TYPE; + workflow/config/cfis/default_exp.sex#BACK_TYPE; + workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.BKG_FROM_HEADER; + src/shapepipe/modules/sextractor_package/sextractor_script.py::SExtractorCaller.get_background. + default: manual_zero_tiles_auto_exposures + options: + manual_zero_tiles_auto_exposures: + label: Tiles trust the stack (0.0); exposures estimate + auto_everywhere: + label: SExtractor AUTO background on tiles too + excluded: true + excluded_reason: >- + Double-subtracts if MegaPipe already removed it; if MegaPipe + residuals are nonzero this exclusion is wrong — verify. + weighting_and_interpolation: + label: Weight-map usage and zero-weight pixel interpolation + rationale: >- + Two settings depart from stock SExtractor: WEIGHT_TYPE MAP_WEIGHT + (default NONE) and INTERP_TYPE ALL (default NONE — SExtractor + invents flux across zero-weight pixels). The accompanying + RESCALE_WEIGHTS Y, WEIGHT_GAIN Y, MASK_TYPE CORRECT and + INTERP_MAXXLAG/INTERP_MAXYLAG 16 are the SExtractor defaults, so + they are settings the configs restate rather than choices. The + variance policy sets effective per-pixel SNR and thus the detection + set; INTERP_TYPE ALL alters pixel data feeding measurements. + Rationale not recorded. + Published description (Guinot+22 p.5): Table 2's "all other + parameters are kept to their default values" silently covers both + non-default settings; current code: MAP_WEIGHT + INTERP_TYPE ALL in + default_tile.sex and default_exp.sex alike — the paper gives no hint + that the weight map or the zero-weight interpolation is in play. + Anchor: workflow/config/cfis/default_tile.sex#WEIGHT_TYPE; + workflow/config/cfis/default_tile.sex#INTERP_TYPE; + src/shapepipe/modules/sextractor_package/sextractor_script.py::SExtractorCaller.set_input_files. + default: map_weight_interp_all + options: + map_weight_interp_all: { label: MAP_WEIGHT + INTERP ALL + MASK CORRECT } + no_interpolation: + label: INTERP_TYPE NONE + excluded: true + excluded_reason: >- + Changes photometry near masks; the committed choice is itself + unjustified in code — flagged as a question, not an endorsement. + detection_source_mode: + label: Single-image detection on the r-band tile + rationale: >- + DETECTION_IMAGE=False, FLAG_IMAGE=False at detection, + param file default_noimaflags.param. No dual-image mode, no + detection coadd, no flag propagation at detection time. The + sx_nomask variant is the committed chain because it matches the + validated bash baseline. STATUS: DELIBERATELY UNDECIDED (Cail, + 2026-08-29) — whether DR6 detects masked or unmasked is punted to + the planned masking-unification rework (not yet tracked in an issue); + the default records baseline + parity, not a settled methodological choice. The masked variant is + one config + one rule + a tile-side star-cat analogue away. + Anchor: workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.DETECTION_IMAGE; + workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE; + workflow/config/cfis/default_noimaflags.param; + workflow/rules/tile.smk. + default: sx_nomask_single_image + options: + sx_nomask_single_image: + label: Unmasked single-image r-band detection + insights: [guinot22_stacked_detection] + sx_masked: + label: Detection on the masked tile + dual_image_coadd: + label: Dual-image mode with a detection coadd + excluded: true + excluded_reason: No detection coadd exists in UNIONS r-band processing. + epoch_membership_ccd_bounds: + label: Which exposure CCDs an object belongs to (N_EPOCH) + rationale: >- + CCD_SIZE = 33,2080,1,4612 with strict inequalities — the 33-px left + trim silently discards a CCD strip from epoch membership; WCS + inversion failures skip the CCD ("no epoch recorded"), changing + N_EPOCH. Sets how many exposures contribute to each galaxy's + multi-epoch fit. Rationale beyond "number of pixels in a CCD" not + recorded. + Anchor: workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.CCD_SIZE; + src/shapepipe/modules/sextractor_package/sextractor_script.py::make_post_process; + src/shapepipe/modules/sextractor_package/sextractor_script.py::ccd_candidate_mask. + default: trimmed_bounds_33_2080 + options: + trimmed_bounds_33_2080: { label: "x in (33,2080), y in (1,4612), strict" } + full_ccd: + label: Full 1-2048 x-range, inclusive bounds + excluded: true + excluded_reason: >- + The trim presumably excludes a bad edge region, but nothing in + code says so — flagged as a question. + prior_insights: + guinot22_stacked_detection: + claim: >- + Source extraction in the published analysis is performed on the + stacked tile images, for signal-to-noise and because artefacts are + suppressed relative to single exposures. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_stacked_detection + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'We do the extraction on stacked images which provide a better signal-to-noise ratio, and most artifacts have a reduced amplitude with respect to single exposures' + location: { page: 5 } + + # ═════════════════════════════════════════════════════════════════════════ + preparation: + description: >- + How pixels, WCS, and epoch membership are prepared before anything is + measured. Modules: + src/shapepipe/modules/split_exp_package/split_exp.py, + merge_headers_package/merge_headers.py, + find_exposures_package/find_exposures.py, + vignetmaker_package/vignetmaker.py. + inputs: + - id: exposure_files + type: data + source: delivered CFIS exposure triplets (image/weight/flag MEF) + tile stacks + outputs: + - id: epoch_stamps + type: data + format: fits + description: Per-object multi-epoch vignets + per-CCD WCS log feeding ngmix. + decisions: + [astrometric_solution_source, ccd_split_extent, + epoch_provenance_from_tile_history, object_position_columns, + stamp_positioning_and_padding, epoch_flag_source] + decisions: + astrometric_solution_source: + label: Astrometry taken verbatim from delivered per-CCD headers + rationale: >- + split_exp builds WCS(h) from each raw CCD header at split time, + pickles it, and merge_headers writes the lot into + log_exp_headers.sqlite; every downstream world<->pixel transform + (stamp positioning, epoch membership, position seeding) uses that + stored solution. No re-derivation, no astrometric refinement — the + survey's delivered astrometry IS the pipeline's astrometry. + Alternative (a joint astrometric re-fit a la DES/Rubin) would move + every stamp centre and every position seed. Anchor: + src/shapepipe/modules/split_exp_package/split_exp.py::SplitExposures.create_hdus; + src/shapepipe/modules/merge_headers_package/merge_headers.py::merge_headers; + src/shapepipe/modules/vignetmaker_package/vignetmaker.py::VignetMaker._get_stamp_me. + default: delivered_headers + options: + delivered_headers: + label: "WCS(header) verbatim, stored at split time" + insights: [guinot22_gaia_astrometry] + astrometric_refit: + label: Joint astrometric re-solution + excluded: true + excluded_reason: Not wired; CFIS delivered astrometry is trusted. + ccd_split_extent: + label: All 40 MegaCam HDUs split and carried as candidate epochs + rationale: >- + N_HDU=40 with a hard check (any other HDU count raises) — every + CCD including the ear CCDs 36-39 is a candidate epoch wherever the + WCS lands it. The MegaCamFlip special-casing of 36/37 shows the + ears flow through shape measurement. Alternative: exclude ear CCDs + (different optical path/orientation history). Anchor: + workflow/config/cfis/config_exp_Sp.ini#SPLIT_EXP_RUNNER.N_HDU; + src/shapepipe/modules/split_exp_package/split_exp.py::SplitExposures.create_hdus. + default: all_40_hdus + options: + all_40_hdus: + label: "40 HDUs, hard-fail on any other count" + insights: [guinot22_forty_chips] + epoch_provenance_from_tile_history: + label: Epoch sets parsed from tile FITS HISTORY cards + rationale: >- + A tile's contributing exposures are recovered by parsing column 3 + of each HISTORY line, stripping prefix "p", deduplicating — the + coadd's own provenance record is trusted as the epoch list. The + LSB s-prefix rename is present but commented out. A mis-parse + changes N_EPOCH and which exposures are fit. Anchor: + workflow/config/cfis/config_tile_Fe.ini#FIND_EXPOSURES_RUNNER.COLNUM; + src/shapepipe/modules/find_exposures_package/find_exposures.py::FindExposures.get_exposure_list. + default: history_parse + options: + history_parse: { label: "HISTORY column 3, prefix p, dedup" } + object_position_columns: + label: Windowed centroids (XWIN/YWIN) define every position + rationale: >- + PSF interpolation sites, tile stamp centres, multi-epoch stamp + centres, and the catalogue sky position all use SExtractor's + windowed centroid — XWIN_WORLD/YWIN_WORLD on the tile side (SPHE), + XWIN_IMAGE/YWIN_IMAGE exposure-side (PIX). Windowed vs isophotal + vs model centroids differ systematically for blends and asymmetric + galaxies, and the centroid definition feeds the position seed. + Anchor: workflow/config/cfis/config_tile_PiViVi.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS; + workflow/config/cfis/config_tile_PiViVi.ini#VIGNETMAKER_RUNNER_RUN_2.POSITION_PARAMS; + workflow/config/cfis/config_exp_psfex.ini#POSITION_PARAMS. + default: xwin_windowed + options: + xwin_windowed: { label: Windowed centroids everywhere } + stamp_positioning_and_padding: + label: Nearest-pixel stamp centring; edge stamps zero-padded + rationale: >- + Multi-epoch stamps are placed by round-tripping the tile world + position through the stored per-CCD WCS, then rounding to the + nearest pixel (no sub-pixel interpolation — the residual sub-pixel + offset is absorbed by the fit's centroid prior, cen sigma = 1 + pixel). Objects whose stamp overruns a CCD or tile edge are KEPT, + out-of-image pixels zero-filled (sf_tools FetchStamps + pad_mode='constant'); no boundary rejection exists — zero-padded + pixels enter the fit as data with whatever weight the padded + weight stamp carries. Anchor: + src/shapepipe/modules/vignetmaker_package/vignetmaker.py::VignetMaker._get_stamp; + src/shapepipe/modules/vignetmaker_package/vignetmaker.py::VignetMaker._get_stamp_me. + default: round_and_zero_pad + options: + round_and_zero_pad: { label: "Nearest-pixel + zero padding, no edge rejection" } + epoch_flag_source: + label: Per-epoch flag stamps come from RAW CFIS flags, not the pipeline mask + rationale: >- + The multi-epoch vignet run reads its flag stamps from + split_exp_runner output — the delivered instrumental flags — + while mask_runner's pipeline_flag (halos, spikes, DSOs, borders) + feeds only the exposure-side star finding + (config_exp_psfex.ini FILE_PATTERN pipeline_flag). Combined with + unmasked tile detection (detection.detection_source_mode), the + consequence is stark: THE BRIGHT-STAR MASKS CURRENTLY AFFECT ONLY + PSF-STAR SELECTION — neither the galaxy sample (no tile mask, no + IMAFLAGS cut possible) nor the pixels ngmix fits (raw flags only) + see them. Whether that is intended belongs to the + planned masking-unification rework, whose object-level half is + sp_validation's IMAFLAGS_ISO cut on a column this chain never + produces; this is the pixel-level half. Anchor: + workflow/config/cfis/config_tile_PiViVi.ini#VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_EXP_RUNNERS; + workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.FILE_PATTERN; + workflow/config/cfis/config_exp_Ma.ini#MASK_RUNNER.PREFIX. + default: raw_flags + options: + raw_flags: { label: split_exp raw flags gate epoch pixels } + pipeline_flags: + label: pipeline_flag (incl. bright-star masks) gates epoch pixels + prior_insights: + guinot22_gaia_astrometry: + claim: >- + The astrometric solution the analysis relies on is the upstream + MegaPipe/Gaia DR2 calibration, accurate to within 20 mas; no + astrometric re-fit inside the pipeline is described. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_astrometry + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'An astrometric calibration within 20 mas was achieved using the Gaia DR2 observations' + location: { page: 2 } + guinot22_forty_chips: + claim: >- + Star selection and PSF modelling are carried out independently on + each of the 40 MegaCam chips, with no chip excluded. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_forty_chips + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'is performed independently on each of the 40 chips that constitute the MegaCAM' + location: { page: 3 } + + # ═════════════════════════════════════════════════════════════════════════ + star_selection_psf: + description: >- + Which objects constrain the PSF, and the PSF model itself. Modules: + src/shapepipe/pipeline/str_handler.py (_mode — the iterative + histogram-zoom FWHM mode estimator centring the star box; median + fallback below N=20), src/shapepipe/modules/setools_package/setools.py + (_make_rand_split), src/shapepipe/modules/psfex_interp_package/ + psfex_interp.py (acceptance gates, HSM shapes). Configs: + star_selection.setools, default.psfex, config_exp_psfex.ini. + [LINT] star_stat logs the FWHM cut as mode +- 0.1*0.187 while the mask + applies mode +- 0.2 px — the run's own log misstates the selection. + [LINT] pixel scale appears as 0.187 (load-bearing) and 0.186 + (plot-only) in the same setools file. + inputs: + - id: exposure_sexcat + type: data + source: per-CCD exposure SExtractor catalogues (default_exp.sex run) + outputs: + - id: psf_model + type: data + format: psf + description: >- + Per-CCD PSFEx models + interpolated PSFs at object positions + (run_sp_exp_SxSePsfPi family), with HSM shape diagnostics. + decisions: + [star_selection_box, psf_train_validation_split, + psfex_candidate_vetting, psf_modelling_software, + psf_model_complexity, psf_acceptance_thresholds] + decisions: + star_selection_box: + label: Stellar-locus selection — mag window + FWHM window around the mode + rationale: >- + 18 < MAG_AUTO < 22, |FWHM - mode| <= 0.2 px, FLAGS==0, + IMAFLAGS_ISO==0; the mode is computed on a preselection + (MAG_AUTO<21, 0.3-1.5 arcsec at 0.187"/px) via the iterative + histogram-zoom estimator (str_handler.py::_mode, eps=0.001; median + fallback for N<20, -1 for N=0 — small-N behaviour changes selection + on sparse CCDs). PSFEx's automatic FWHM-range selection is off + (SAMPLE_AUTOSELECT N) and bad-pixel filtering is off, but PSFEx's + compiled-in fixed sample cuts still apply on top of this box — + see psfex_candidate_vetting. + Anchor: workflow/config/cfis/star_selection.setools#MASK:star_selection.MAG_AUTO; + workflow/config/cfis/star_selection.setools#MASK:preselect.MAG_AUTO; + workflow/config/cfis/default.psfex#SAMPLE_AUTOSELECT; + src/shapepipe/pipeline/str_handler.py::_mode. + default: mode_centred_box + options: + mode_centred_box: + label: FWHM-mode-centred box, +-0.2 px, mag 18-22, setools-only vetting + insights: [guinot22_star_box] + size_mag_locus_fit: + label: Fitted size-magnitude stellar locus + excluded: true + excluded_reason: Not wired; the mode-box is the validated v2.0 selection. + psfex_autoselect: + label: PSFEx SAMPLE_AUTOSELECT vetting on top + excluded: true + excluded_reason: >- + Deliberately disabled so selection lives in one place; rationale + not recorded in code. + psf_train_validation_split: + label: Random 80/20 star split — model fit vs held-out validation + rationale: >- + RAND_SPLIT ratio 20: star_split_ratio_80 fits the PSFEx model + (config_exp_psfex.ini FILE_PATTERN, and the tile multi-epoch + interpolation ME_DOT_PSF_PATTERN in config_tile_PiViVi.ini); + star_split_ratio_20 is the independent PSF-residual diagnostic + (PSFEX_INTERP MODE=VALIDATION). Trades model precision (fewer + training stars per CCD, interacting with the STAR_THRESH gate) + against an independent residual test. + [PENDING #873] The split is DETERMINISTIC: _make_rand_split takes + np.random.RandomState(seed).permutation(cat_size), the seed being + the digits of the unit's file number mod 2^32 — a pure function of + the input catalogue, fixed per CCD and independent of processing + order, the same philosophy as shape_measurement.ngmix_seed_mode's + SEED_FROM_POSITION. Before this the split drew from unseeded + np.random.randint, so the star sample entering the PSF model — and + therefore every shape downstream of it — differed between + identical runs; it was the one unseeded draw the position-seed work + left uncovered. One-off cost: the realised 80/20 membership changes + once (it is one further draw, now frozen), so PSF models and shapes + shift by that draw relative to every earlier product. + Anchor: workflow/config/cfis/star_selection.setools#RAND_SPLIT:star_split.RATIO; + src/shapepipe/modules/setools_package/setools.py::SETools._make_rand_split; + workflow/config/cfis/config_exp_psfex.ini#PSFEX_RUNNER.FILE_PATTERN; + workflow/config/cfis/config_tile_PiViVi.ini#PSFEX_INTERP_RUNNER.ME_DOT_PSF_PATTERN. + default: split_80_20_seeded + options: + split_80_20_seeded: + label: 80% train / 20% validation, seeded from the file number + insights: [guinot22_star_split] + split_80_20_unseeded: + label: Same split, unseeded np.random (pre-#873) + excluded: true + excluded_reason: >- + Retired by #873: it made the PSF star sample — and every shape + downstream of it — irreproducible run-to-run, the single + remaining unseeded draw in the science chain. Kept on the + record because every UNIONS product built before the smk-g4 + campaign was produced under it. + no_holdout: + label: 100% of stars in the model, no held-out diagnostic + excluded: true + excluded_reason: Loses the independent rho-statistic input. + psfex_candidate_vetting: + label: PSFEx-side candidate vetting — built-in defaults, unpinned + rationale: >- + default.psfex sets only SAMPLE_AUTOSELECT N; SAMPLE_MINSN, + SAMPLE_MAXELLIP, SAMPLE_FWHMRANGE, SAMPLE_VARIABILITY are absent, + so PSFEx's compiled-in defaults apply silently (MINSN 20, + MAXELLIP 0.3, FWHMRANGE 2-10 px, VARIABILITY 0.2) — a second star + selection nobody's config records, and one that changes if the + PSFEx binary version changes. BADPIXEL_FILTER N + PSF_RECENTER N: + star vignets with flagged/sentinel pixels are accepted unfiltered + and candidates are not recentred (CENTER_KEYS XWIN). The setools + box is therefore not the whole selection. [HARDCODED] (in the + PSFEx binary). + Anchor: workflow/config/cfis/default.psfex#SAMPLE_AUTOSELECT; + workflow/config/cfis/default.psfex#BADPIXEL_FILTER; + workflow/config/cfis/default.psfex#PSF_RECENTER. + default: builtin_defaults + options: + builtin_defaults: + label: "Compiled-in MINSN 20 / MAXELLIP 0.3 / FWHMRANGE 2-10, no bad-pixel filter" + insights: [guinot22_psfex_preselection_off] + pinned_explicit: + label: Write the SAMPLE_* values explicitly into default.psfex + psf_modelling_software: + label: PSF modelling software — PSFEx per-CCD vs MCCD focal-plane + rationale: >- + The committed chain fits PSFEx independently per CCD. MCCD + (Liaudat+2021) is a maintained in-tree alternative: a focal-plane + model fit across all 40 CCDs at once with a hybrid local+global + decomposition (src/shapepipe/modules/mccd_package/ + six + mccd_*_runner.py; knobs in example/cfis/config_MCCD.ini — + N_COMP_LOC=8, D_COMP_GLOB=8, LOC_MODEL=hybrid, MIN_N_STARS=20, + RMSE_THRESH=1.25). Unwired in workflow/config/cfis/ (needs the + MCCD config adapted, and config_exp_mccd.ini carries a stale + hardcoded PSF_MODEL_DIR path). The image-simulation path + substitutes PSF modelling entirely: fake_psf_runner injects the + true input PSF from a SKiLLS dictionary in psfex_interp's output + format. + Anchor: src/shapepipe/modules/mccd_package; + src/shapepipe/modules/fake_psf_package; + example/cfis/config_MCCD.ini#INSTANCE.N_COMP_LOC; + example/cfis/config_MCCD.ini#INPUTS.MIN_N_STARS; + example/cfis/config_exp_mccd.ini. + default: psfex + options: + psfex: + label: PSFEx, independent per-CCD models + insights: [guinot22_psfex_software, farrens22_two_psf_methods] + mccd_focal_plane: + label: MCCD hybrid local+global focal-plane model + true_input_psf: + label: fake_psf injection of the simulation's true PSF + excluded: true + excluded_reason: >- + Only meaningful on simulated images where the true PSF exists; + not a data-analysis option. + psf_model_complexity: + label: PSFEx model — pixel basis, degree-2 spatial polynomial per CCD + rationale: >- + BASIS_TYPE PIXEL, BASIS_NUMBER 20, PSF_SIZE 51,51, PSF_SAMPLING 1, + PSFVAR_DEGREES 2 in XWIN,YWIN per CCD (MEF_TYPE INDEPENDENT, + STABILITY_TYPE EXPOSURE), PSF_RECENTER N. Model flexibility sets + the PSF-leakage/overfitting balance — the dominant additive + systematic in cosmic shear. Values are the stock EB 2017 header; + rationale not recorded in code. + Anchor: workflow/config/cfis/default.psfex#BASIS_TYPE; + workflow/config/cfis/default.psfex#PSFVAR_DEGREES; + workflow/config/cfis/default.psfex#PSF_SIZE. + default: pixel_basis_deg2_per_ccd + options: + pixel_basis_deg2_per_ccd: + label: "PIXEL basis, degree 2, per-CCD" + insights: [guinot22_psf_no_oversampling] + deg3: + label: Degree-3 spatial variation + excluded: true + excluded_reason: >- + More flexibility per CCD needs more stars per CCD than the + count-floor world guarantees; not validated. + psf_acceptance_thresholds: + label: Per-CCD PSF-model quality gate (min stars, max chi2) + rationale: >- + A CCD whose model has ACCEPTED < STAR_THRESH or CHI2 > 2 is not + interpolated — its galaxies drop from the shear catalogue: direct + footprint selection, the in-code analogue of the DES blacklist. + [PENDING #873] Both passes now gate at 22 stars: the VALIDATION-mode + exposure config always did (config_exp_psfex.ini), and #873 raised + the MULTI-EPOCH science path 20 -> 22 in example/cfis + (config_tile_PiViVi_canfar_{sx,uc}.ini), with commit 90782098 + mirroring it into workflow/config/cfis/config_tile_PiViVi.ini — the + committed config fork this workflow actually reads (#848 D2). + Provenance of the retired 20, which is what makes this a fix rather + than a preference: commit fdc86553 (Kilbinger, 2020-06-30, "Forgot + to update new star number threshold for 80% of stars") deliberately + bumped 20 -> 22 to account for the 80/20 split, but only in the + validation config; the tile config kept the pre-split 20, so for + five years the SCIENCE path gated on the value that 2020 fix meant + to retire. (20 is also the psfex_interp function default, so the + stale-value reading rested on the commit provenance rather than on + the config alone.) + Published description (Guinot+22 p.4, Fig. 3): 22 stars/CCD, applied + to exactly the CCDs feeding multi-epoch shape measurement — the + number now agrees. Two gaps remain: an undocumented CHI2_THRESH=2 in + both configs, and the mechanism — interpsfex tests the PSFEx header + ACCEPTED/CHI2 at interpolation time and drops that epoch for objects + on the CCD, rather than excluding the CCD from PSF modelling as the + paper describes. + Anchor: src/shapepipe/modules/psfex_interp_package/psfex_interp.py::PSFExInterpolator.interpsfex; + workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH; + workflow/config/cfis/config_tile_PiViVi.ini#PSFEX_INTERP_RUNNER.STAR_THRESH; + example/cfis/config_tile_PiViVi_canfar_sx.ini#PSFEX_INTERP_RUNNER.STAR_THRESH; + example/cfis/config_tile_PiViVi_canfar_uc.ini#PSFEX_INTERP_RUNNER.STAR_THRESH. + default: stars22_chi2_2 + options: + stars22_chi2_2: + label: ">= 22 stars on both passes, chi2 <= 2" + insights: [des_psf_blacklist_local, guinot22_star_floor_22_local] + stars20_chi2_2: + label: ">= 20 stars on the science path, 22 in validation (pre-#873)" + excluded: true + excluded_reason: >- + Retired by #873 + 90782098. It was never a chosen value: it is + the pre-split threshold fdc86553 raised to 22 in 2020 for the + validation config and forgot on the science path, leaving the + science gate below both the published floor (Guinot+22 Fig. 3) + and the pipeline's own intent. Every UNIONS product built before + the smk-g4 campaign carries it. + des_25: + label: DES Y3 threshold (25 stars) + excluded: true + excluded_reason: >- + Not adopted; CFIS CCDs are smaller than DECam's — the right + number is survey-specific. + prior_insights: + des_psf_blacklist_local: + claim: >- + DES Y3 blacklists any CCD with fewer than 25 stars surviving + outlier rejection in the PSF fit. + created_at: "2026-07-16T00:00:00Z" + evidence: + - id: ev_jarvis_y3_local + doi: "10.48550/arXiv.2011.03409" + quote: + exact: "fewer than 25 stars survived the outlier rejection" + location: { page: 10 } + guinot22_star_floor_22_local: + claim: >- + The published ShapePipe/UNIONS analysis discards a CCD from the PSF + estimation when fewer than 22 stars are selected on it — the floor + the science-path PSF-interpolation gate now applies. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_star_floor_local + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'The dashed line represents the cut at 22 stars/CCD below which the CCD is discarded for the PSF estimation.' + location: { page: 4 } + guinot22_star_box: + claim: >- + The published star selection keeps objects whose FWHM lies within + 0.04 arcsec of the mode of a size preselection, restricted to the + magnitude range 18 < r < 22. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_star_box + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'From this pre-selection we keep objects for which the FWHM is within 0.04 arcsec of the mode. In addition to these size cuts, we only use star candidates in the magnitude range 18 < r < 22.' + location: { page: 4 } + guinot22_star_split: + claim: >- + The published analysis randomly splits the star sample in two, 80% + building the PSF model and 20% held out for the validation tests. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_star_split + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'To be able to perform these tests properly, our star sample has been randomly divided in two:' + location: { page: 7 } + guinot22_psfex_preselection_off: + claim: >- + PSFEx's internal pre-selection is deliberately disabled so that the + pipeline's own star selection is the only one, the paper describing + the model as fit on the entire star sample. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_preselection_off + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'Since we carry out our own star selection (see Sect. 4.1), we disable the internal PSFEx pre-selection, and the PSF is thus obtained using the entire star sample.' + location: { page: 4 } + guinot22_psfex_software: + claim: >- + The published UNIONS shear catalogue uses PSFEx for PSF modelling, + with MCCD named as upcoming rather than current work. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_psfex_software + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'We make use of the PSFEx software package' + location: { page: 4 } + farrens22_two_psf_methods: + claim: >- + ShapePipe ships two PSF modelling methods, PSFEx and MCCD, either or + both of which may be run and used for the galaxy shape measurement. + created_at: "2022-06-01T00:00:00Z" + evidence: + - id: ev_farrens22_two_psf + doi: "10.48550/arXiv.2206.14689" + quote: + exact: 'ShapePipe allows either or both methods to be run and subsequently used for the galaxy shape measurement.' + location: { page: 3 } + guinot22_psf_no_oversampling: + claim: >- + The PSFEx parametrisation is tabulated in the paper (pixel basis, + degree-2 spatial variation), with the deliberate choice not to + over-sample the PSF models. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_psf_complexity + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'The PSFEx parameters we used are presented in Table 1. We have chosen not to over-sample the PSF models.' + location: { page: 5 } + + # ═════════════════════════════════════════════════════════════════════════ + shape_measurement: + description: >- + Galaxy shape estimation: ngmix single-Gaussian fits with metacalibration. + Modules: src/shapepipe/modules/ngmix_package/ngmix.py (priors, metacal + setup, epoch handling, postage-stamp prep), ngmix_runner.py (config + exposure). Config: config_tile_Ng_template.ini. Most values here are + HARDCODED — scientific choices living in code with no config exposure; + this sub-analysis is where the silent-default risk concentrates. + [LINT] centroid_source default disagrees between the runner ("wcs", + production; always passed explicitly, ngmix_runner.py:170) and every + module-level signature ("hsm") — unreachable in the pipeline path, but + direct callers (tests, notebooks) silently get the other choice. + [LINT] pixel scale is 0.186 here (config PIXEL_SCALE) vs 0.187 in the + setools/masking configs — and star_selection.setools itself mixes + 0.187 (cuts) with 0.186 (SCATTER stat, :75). + inputs: + - id: vignets + type: data + source: 51x51 galaxy/weight/background-RMS vignets + interpolated PSFs (vignetmaker, psfex_interp) + outputs: + - id: ngmix_cat + type: data + format: fits + description: Per-tile metacal shear catalogue chunks (ngmix_runner family). + decisions: + [ngmix_seed_mode, galaxy_model, fit_priors, metacal_scheme, + centroid_source, epoch_quality_and_weighting, noise_model, + psf_epoch_loss_policy, megacam_ccd_flip] + decisions: + ngmix_seed_mode: + label: ngmix per-object RNG seeding + rationale: >- + Production historically seeded one RandomState from the tile ID, + consumed in object order — results depended on chunk boundaries. + The position seed (3-arcsec sky boxes + CCD offsets, zig-zag fold + + Cantor pairing mod 2^32; ngmix.py::position_seed) makes every + stream a function of sky position: chunk-invariant, + bit-reproducible, and metacal fixnoise counter-noise cancels across + image-simulation branches (ngmix#796). Consequence: chunking is + demoted to a pure throughput knob (reverting this decision + re-promotes it). Cost: noise streams change vs v2.0 — see + top-level baseline_validation_criterion. + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::position_seed; + workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.SEED_FROM_POSITION; + src/shapepipe/modules/ngmix_runner.py::ngmix_runner. + default: position_seed + options: + position_seed: + label: Per-object seed from (ra, dec, ccd), 3-arcsec boxes + tile_seed: + label: Tile-wide RandomState (v2.0) + excluded: true + excluded_reason: >- + Chunk-dependent; retired outright (SEED_FROM_POSITION=False now + raises — ngmix_runner.py:110-116). + galaxy_model: + label: Galaxy and PSF model — single Gaussian [HARDCODED] + rationale: >- + ngmix.fitting.Fitter(model='gauss') for both galaxy and PSF + (ngmix.py::make_runners); guessers TPSFFluxAndPriorGuesser / + TFluxGuesser with T=0.25 and catalogue-flux guess, Runner ntry=5, + PSFRunner ntry=2 — with a non-convex likelihood, guess and retries + decide which objects converge (failed fits are NaN-filled with + flags, not raised). Rationale not recorded. Under metacal, model + bias largely cancels in the response, which is the standard defense + of 'gauss'; not stated in code. + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::make_runners. + default: gauss + options: + gauss: + label: "Single Gaussian, T guess 0.25, ntry 5/2" + insights: [guinot22_gaussian_model] + exp_or_bdf: + label: exp / bdf galaxy models + excluded: true + excluded_reason: >- + Slower, and metacal makes the gain marginal; not validated on + CFIS. + fit_priors: + label: ngmix joint prior — GPriorBA(0.4), cen sigma = pixel scale, flat T/F [HARDCODED] + rationale: >- + Ellipticity GPriorBA sigma=0.4; centroid CenPrior sigma = one pixel + scale (0.186 arcsec, config PIXEL_SCALE — the coupling + sigma=pixel_scale is itself the hardcoded choice); flat T in + [-1, 1e3], flat F in [-100, 1e9] with negative support (bounds + decide which noisy fits survive vs rail). get_prior takes T/F range + arguments but no caller passes them. Prior width drives noise bias; + rationale not recorded. + Published description (Guinot+22 p.7): centroid sigma = pixel scale + ~0.187 arcsec, flat F in [-1e4, 1e9], flat half-light radius r50 in + [-10, 1e6] arcsec, ellipticity prior from Bernstein & Armstrong + (2014); current code: PIXEL_SCALE 0.186, flat F in [-100, 1e9], and + a flat prior on ngmix's second-moment size T in [-1, 1e3] rather + than on r50 — the prior families agree, the flux bound and pixel + scale have drifted, and the size prior is a different + parameterisation rather than a changed number. + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::get_prior; + workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.PIXEL_SCALE. + default: gpriorba04_flat + options: + gpriorba04_flat: { label: "GPriorBA 0.4 + flat T/F with negative support" } + nonneg_informative: + label: Non-negative or informative T/F priors + excluded: true + excluded_reason: >- + Truncating negative support biases the noshear ensemble mean; + metacal wants symmetric noise response. + metacal_scheme: + label: Metacalibration — 5 types, step 0.01, fitgauss reconv, fixnoise [HARDCODED] + rationale: >- + types [noshear,1p,1m,2p,2m], step 0.01, psf='fitgauss' (runner + default; moves the metacal response directly — alternatives gauss/ + dilate/azgauss listed in the docstring), fixnoise=True, + use_noise_image=True, MetacalBootstrapper(ignore_failed_psf=True) + (changes which epochs enter the fit). No *_psf sheared types, so no + mcal_R_psf PSF-response term in the catalogue. fixnoise rationale + appears only in the position_seed docstring (counter-noise + cancellation). + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal. + default: five_types_step001_fitgauss + options: + five_types_step001_fitgauss: + label: "noshear+1p/1m/2p/2m, step 0.01, fitgauss, fixnoise" + insights: [guinot22_metacal_five_images] + with_psf_response: + label: Add sheared-PSF types for R_psf + excluded: true + excluded_reason: >- + Not wired; leakage is instead diagnosed via PSF_ORIG columns + + rho statistics downstream. + centroid_source: + label: Jacobian origin from WCS astrometry, not HSM moments + rationale: >- + Production runner default "wcs"; hsm is "legacy... noisy for stars + and flagged as incorrect by Fabian — see #767" (runner comment; a + rare recorded rationale). Moves the centroid-prior centre per + object. The runner reads an optional CENTROID_SOURCE config option + that no committed CFIS config sets. [LINT] module-level default is + still "hsm" — see this sub-analysis's description. + Published description (Guinot+22 p.7): HSM adaptive moments, run on + each sheared version, supplied the whole initial guess vector + (centroid, r50, flux) for the least-squares fit; current code: that + initialisation is gone — guesses come from ngmix's + TPSFFluxAndPriorGuesser with fixed T=0.25 and a catalogue flux, and + the only surviving HSM role is the optional stamp re-centering that + sets the Jacobian origin. So the drift is wider than a swapped + centroid source. (The paper's other HSM use, PSF/star shape + diagnostics, is unaffected.) + Anchor: src/shapepipe/modules/ngmix_runner.py::ngmix_runner; + src/shapepipe/modules/ngmix_package/ngmix.py::make_ngmix_observation. + default: wcs + options: + wcs: { label: WCS-projected catalogue position } + hsm: + label: HSM adaptive-moment centroid + excluded: true + excluded_reason: Noisy for stars; flagged incorrect (shapepipe#767). + epoch_quality_and_weighting: + label: Epoch admission, masking cut, and multi-epoch combination [HARDCODED] + rationale: >- + An epoch is dropped if >1/3 of its stamp is masked + (prepare_postage_stamps; the comment says "objects", the code drops + epochs — an object with zero surviving epochs drops out); failed + PSF fits drop epochs (flags != 0); fluxes rescaled by header FSCALE + (gal*Fscale, weight/Fscale^2); the diagnostic PSF is averaged over + epochs weighted by obs.weight.sum(). Joint multi-epoch fit over + survivors. Rationale for 1/3 and for the weight choice not + recorded. + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_postage_stamps; + src/shapepipe/modules/ngmix_package/ngmix.py::rescale_epoch_fluxes; + src/shapepipe/modules/ngmix_package/ngmix.py::_average_psf_fits. + default: third_masked_cut + options: + third_masked_cut: { label: "Drop epoch if >1/3 masked; FSCALE rescale; weight-sum PSF average" } + noise_model: + label: Per-pixel inverse variance from background-RMS vignets + rationale: >- + BKG_RMS_VIGNET_PATH set in the CFIS template: weight = + 1/bkg_rms^2 per pixel (all-or-nothing; missing file errors); + fallback scalar 1/sigma_mad^2. Masked pixels filled with Gaussian + noise at sig_noise. A scalar sigma "mis-reports errors and erodes + the inverse-variance advantage whenever the RMS map actually + varies" (recorded rationale, fixnoise bookkeeping). PSF observation + gets a flat weight from PSF_NOISE=1e-5 — hardcoded module constant, + validated 1e-4..1e-6 on the digital twin (#749/#774 comment); + without it the g-prior swamps the PSF likelihood. Per-epoch + background subtraction BKG_SUB=True (off only for sims). + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_ngmix_weights; + src/shapepipe/modules/ngmix_package/ngmix.py::PSF_NOISE; + src/shapepipe/modules/ngmix_package/ngmix.py::background_subtract; + workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.BKG_RMS_VIGNET_PATH. + default: rms_vignet_weights + options: + rms_vignet_weights: { label: Per-pixel RMS-map weights + PSF_NOISE 1e-5 } + scalar_sigma_mad: + label: Scalar sigma_mad per epoch + excluded: true + excluded_reason: Mis-reports errors where the RMS map varies (recorded). + psf_epoch_loss_policy: + label: Object-level policy when CCDs fail PSF interpolation + rationale: >- + When k of ~40 CCDs fail the PSF acceptance gate (~5-6% attrition, + per-exposure clustered, matches the bash baseline — but measured + with the science gate at 20 stars, so [PENDING #873] at 22 it can + only rise, and smk-g4 is the first campaign to re-measure it), the + pipeline applies NO further quality gate: tiles complete, each + object records NGMIX_N_EPOCH, and sp report surfaces per-tile + epoch loss. + Object-level protection is delegated entirely to the validation + stage's epoch-count cut (sp_validation's galaxy selection cuts on N_EPOCH >= 1). Rationale (Cail, + 2026-08-29, PRD walk): epoch loss is a per-object depth effect + already recorded in the catalogue; gating at pipeline level would + fail whole tiles for a versionable catalogue decision. PRD #848's + open-questions section was removed accordingly. Per-tile epoch loss + is surfaced by the run report; NGMIX_N_EPOCH is the per-object + record. + Anchor: workflow/scripts/run_report.py; + workflow/config/cfis/final_cat.param#NGMIX_N_EPOCH. + default: record_and_delegate + options: + record_and_delegate: + label: Record NGMIX_N_EPOCH, report attrition, no pipeline gate + pipeline_epoch_floor: + label: Fail tiles below a minimum surviving-epoch fraction + excluded: true + excluded_reason: >- + Fails whole tiles for what is a versionable per-object + catalogue decision; the depth effect is already recorded. + megacam_ccd_flip: + label: 180-degree tile-vignet rotation for MegaCam CCDs <18 and 36-37 [HARDCODED] + rationale: >- + "MegaPipe has CCDs that are upside down" (docstring) — the tile + vignet is rotated to register with epoch stamps; a wrong flip + mis-registers the tile mask against the epoch, changing flagged + pixels and the 1/3-masked cut. Carries its own recorded caveat: + "will give incorrect results when used with THELI ccds. Fix this." + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::Ngmix.MegaCamFlip. + default: megapipe_flip + options: + megapipe_flip: { label: Flip CCDs <18 and 36/37 (MegaPipe orientation) } + prior_insights: + guinot22_gaussian_model: + claim: >- + The published shape measurement models galaxies with a single + Gaussian profile, arguing the resulting model bias is small and + largely absorbed by metacalibration. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_gaussian + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'Despite being very simple, the model bias (Kacprzak et al.' + suffix: ' 2014) is small.' + location: { page: 7 } + guinot22_metacal_five_images: + claim: >- + Metacalibration in the published analysis creates four sheared + images for calibration plus one for measurement, with a shear step + of 0.01 and a 90-degree-rotated noise image to cancel noise + correlations. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_metacal + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'This method creates four images used for the calibration, and one for the measurement.' + location: { page: 6 } + + # ═════════════════════════════════════════════════════════════════════════ + psf_diagnostics: + description: >- + The PSF-fidelity diagnostic chain: merge the held-out (20%) validation + stars into one catalogue, bin PSF shapes and residuals over the focal + plane. DORMANT in the committed snakemake chain — no rule runs it. + Modules: src/shapepipe/modules/merge_starcat_runner.py (+ per-model + merge classes in merge_starcat.py), mccd_plots_runner.py (serves both + PSF models despite its name). Boundary note: the module docstring + (mccd_package/__init__.py:157) still advertises rho-statistics plots, + but no rho/treecorr code remains in shapepipe — rho/tau statistics + moved downstream to sp_validation (rho_tau.py via + shear_psf_leakage.RhoStat/TauStat; treecorr min_sep/max_sep/nbins, + jackknife patch numbers hardcoded per survey with a "TODO to yaml"). + The diagnostic decision chain thus crosses the repo boundary into + sp_validation. [LINT] the module docstring still advertises rho + statistics this package no longer computes. + inputs: + - id: validation_star_cats + type: data + source: per-CCD star_split_ratio_20 catalogues with PSF/star HSM shapes (psfex_interp VALIDATION mode) + outputs: + - id: merged_star_catalogue + type: data + format: fits + description: >- + One full_starcat over the run — the input rho/tau statistics and + leakage diagnostics consume downstream. + decisions: [starcat_merge_source, meanshape_binning] + decisions: + starcat_merge_source: + label: Which PSF model's validation output feeds the merged star catalogue + rationale: >- + merge_starcat_runner dispatches on PSF_MODEL in {psfex, mccd, + setools} to per-model merge classes (different HDU conventions: + mccd HDU 1, psfex/setools HDU 2). Follows star_selection_psf. + psf_modelling_software; recorded separately because the merge can + also consume raw setools output (pre-model diagnostics). + Anchor: src/shapepipe/modules/merge_starcat_runner.py::merge_starcat_runner; + src/shapepipe/modules/merge_starcat_package/merge_starcat.py. + default: psfex + options: + psfex: + label: PSFEx validation catalogues (HDU 2) + insights: [guinot22_psfex_for_v1] + mccd: { label: MCCD validation catalogues (HDU 1) } + setools: { label: Raw setools star catalogues } + meanshape_binning: + label: Focal-plane mean-shape binning and outlier handling + rationale: >- + PSF ellipticity/size and residuals binned per CCD over the focal + plane: X_GRID=5, Y_GRID=10 bins per CCD, colour scales MAX_E=0.05, + MAX_DE=0.005, REMOVE_OUTLIERS=False (example/cfis config; no + committed workflow config exists). Grid resolution sets which + spatial PSF-residual structure is visible; outlier removal changes + what the diagnostic hides. + Published description (Guinot+22 p.8): focal-plane residual maps + averaged in ~20 arcsec cells per CCD; current code: no committed + workflow config picks a grid at all — example/cfis carries both the + 5x10 grid recorded here (~77x86 arcsec) and, in + config_valjoint_Pl_mccd.ini, a 20x40 grid (~19x22 arcsec) that + reproduces the paper. This is therefore an undetermined knob with + two committed precedents rather than a value that drifted. The + uniform REMOVE_OUTLIERS=False is a genuine paper-silence gap. + Anchor: example/cfis/config_MsPl_psfex.ini#MCCD_PLOTS_RUNNER.X_GRID; + example/cfis/config_MsPl_psfex.ini#MCCD_PLOTS_RUNNER.REMOVE_OUTLIERS; + src/shapepipe/modules/mccd_plots_runner.py; + src/shapepipe/modules/mccd_package/mccd_plot_utilities.py::plot_meanshapes. + default: grid_5x10 + options: + grid_5x10: { label: "5x10 per CCD, outliers kept" } + prior_insights: + guinot22_psfex_for_v1: + claim: >- + PSFEx is the PSF model behind the published UNIONS v1 catalogue, so + the merged validation star catalogue and its diagnostics are fed by + PSFEx output. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_psfex_v1 + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'We make use of the PSFEx software package' + location: { page: 4 } + + # ═════════════════════════════════════════════════════════════════════════ + survey_geometry: + description: >- + Effective survey area and mask geometry for two-point estimators. + DORMANT — no committed workflow rule. Module: + src/shapepipe/modules/random_cat_package/random_cat.py (+ runner): + uniform randoms over each tile rejected against the pipeline mask, + yielding effective area (overlap- and mask-corrected) and optionally + the mask itself as a HEALPix map (save_as_healpix). This is the + in-repo ancestor of the planned healsparse external-mask rework — whichever way that rework lands, this + sub-analysis is where its geometry decisions belong. Bitrot risk: + healpy is imported but absent from pyproject.toml dependencies. + inputs: + - id: tile_masks + type: data + source: per-tile pipeline flag maps + final catalogues + outputs: + - id: random_catalogue + type: data + format: fits + description: Per-tile random catalogue + effective area (+ optional HEALPix mask). + decisions: [random_sampling, healpix_mask_export] + decisions: + random_sampling: + label: Random-point density for area estimation + rationale: >- + N_RANDOM=50000 with DENSITY=True (per square degree; False = total + per tile) in the example config; no committed workflow value. + Sampling density sets the Monte Carlo noise floor on effective + area, which propagates to two-point normalisation. + Anchor: example/cfis/config_Rc.ini#RANDOM_CAT_RUNNER.N_RANDOM; + example/cfis/config_Rc.ini#RANDOM_CAT_RUNNER.DENSITY; + src/shapepipe/modules/random_cat_runner.py::random_cat_runner. + default: per_sqdeg_50k + options: + per_sqdeg_50k: { label: "50000 per sq deg" } + healpix_mask_export: + label: HEALPix export resolution for the pipeline mask + rationale: >- + SAVE_MASK_AS_HEALPIX=True, HEALPIX_OUT_NSIDE=1024 (~3.4 arcmin + pixels) in the example config — coarser than the arcsecond-scale + mask features it rasterises; the resolution choice decides what the + exported mask can represent. Supersession candidate under + masking-unification (healsparse). + Anchor: example/cfis/config_Rc.ini#RANDOM_CAT_RUNNER.SAVE_MASK_AS_HEALPIX; + example/cfis/config_Rc.ini#RANDOM_CAT_RUNNER.HEALPIX_OUT_NSIDE; + src/shapepipe/modules/random_cat_package/random_cat.py::RandomCat.save_as_healpix. + default: nside_1024 + options: + nside_1024: { label: nside 1024 } + + # ═════════════════════════════════════════════════════════════════════════ + catalogue_assembly: + description: >- + Final per-tile catalogue: merging shape chunks, attaching photometry + and PSF diagnostics, classification, sentinels. Modules: + src/shapepipe/modules/make_cat_package/make_cat.py (+ runner), + merge_sep_cats.py, vignetmaker_package (stamps, see top-level + postage_stamp_size), find_exposures_package (epoch list from tile + HISTORY cards). Configs: config_tile_Mc.ini, config_tile_PiViVi.ini, + final_cat.param. + inputs: + - id: ngmix_chunks + type: data + source: per-tile ngmix catalogue chunks + tile sexcat + PSF diagnostics + outputs: + - id: tile_final_cat + type: data + format: fits + description: The assembled per-tile science catalogue (final_cat family). + decisions: + [star_galaxy_classification, tile_overlap_handling, + column_selection, failure_sentinels, postproc_run_provenance, + shape_catalogue_shortfall_guard] + decisions: + star_galaxy_classification: + label: Star/galaxy separation — deferred out of the pipeline + rationale: >- + Production sets SM_DO_CLASSIFICATION=False (config_tile_Mc.ini) and + wires no spread-model input: SPREAD_MODEL/SPREADERR_MODEL are + sentinel 99, no SPREAD_CLASS column, and the SPREAD_* entries in + final_cat.param are commented out. The dormant machinery + (make_cat.py::save_sm_data) classifies on class = sm + 2*sm_err + with star |class|<0.003, galaxy class>0.01 — thresholds hardcoded + in the function signature. Reactivation is a 4-line config diff + (run spread_model_runner after psfex_interp+vignetmaker, add its + output to make_cat inputs, flip the switch — the exact diff + between example/cfis/config_make_cat_psfex.ini and _nosm.ini; the + defunct tile wiring config_tile_PiViSmVi.ini is the reference). + Separation therefore happens entirely downstream (sp_validation); + the pipeline ships everything. Rationale for deferring not + recorded. + Published description (Guinot+22 p.5-6): galaxies are selected + inside the pipeline with the spread model, at s + 2*sigma_s > + 0.0003 together with s > 0 and 20 < MAG_AUTO < 26; current code: + classification is disabled entirely and separation deferred to + sp_validation, with the dormant make_cat thresholds putting the + like-for-like galaxy boundary at 0.01, some thirty times the + published cut (0.003 is its separate star-side bound). Even + reactivated, the code implements only the spread-model test — the + paper's companion cuts have no in-pipeline counterpart. + Anchor: workflow/config/cfis/config_tile_Mc.ini#MAKE_CAT_RUNNER.SM_DO_CLASSIFICATION; + src/shapepipe/modules/make_cat_package/make_cat.py::save_sm_data; + workflow/config/cfis/final_cat.param#SPREAD_CLASS; + example/cfis/config_make_cat_psfex_nosm.ini; + example/cfis/defunct/config_tile_PiViSmVi.ini. + default: deferred_downstream + options: + deferred_downstream: + label: No in-pipeline classification; catalogue ships all objects + spread_model_inline: + label: spread_model classification in make_cat (0.003/0.01) + excluded: true + excluded_reason: >- + Machinery present but unwired in production; reactivating it + changes which objects downstream sees as galaxies. + tile_overlap_handling: + label: Tile-overlap duplicates — neither removed nor flagged + rationale: >- + Adjacent tiles overlap; objects in the overlap are measured in + both. make_cat attaches only TILE_ID (parsed from the sexcat + filename); no unique-object rule, no overlap flag. [LINT] the + documented config key TILE_LIST ("used to flag objects in areas of + overlap between tiles", in the make_cat package docstring) is + implemented nowhere — grep across src/ and workflow/ finds only + the docstring. VERIFIED downstream: sp_validation dedups at + classification time (galaxy.py::classification_galaxy_overlap_ra_dec + RA/Dec cuts to non-overlapping tile areas, and the WCS-based + mask_overlap variant; applied as cut_overlap in + classification_galaxy_base) — so this is today's division of labour, + and the pipeline's contract is "ship duplicates, TILE_ID is the + handle"; flagging overlaps in the catalogue stays an open option. The dead TILE_LIST docstring remains a lint. + Anchor: src/shapepipe/modules/make_cat_package/make_cat.py::save_sextractor_data; + src/shapepipe/modules/make_cat_package/__init__.py. + default: no_dedup_in_pipeline + options: + no_dedup_in_pipeline: + label: Ship duplicates; TILE_ID is the only handle + overlap_flagging: + label: Implement the documented TILE_LIST overlap flag + nearest_tile_centre: + label: Keep each object only in its nearest tile + excluded: true + excluded_reason: >- + Requires cross-tile coordination at assembly time, which the + per-tile DAG deliberately avoids; dedup belongs downstream if + anywhere. + column_selection: + label: Which columns survive into the science catalogue + rationale: >- + final_cat.param: positions XWIN/YWIN_WORLD, TILE_ID, flags + (FLAGS, IMAFLAGS_ISO, NGMIX_MCAL_FLAGS), PSF ellipticity from + PSF_ORIG only, all five metacal branches of G1/G2/T/FLUX/FLAGS, + but shear errors only for NOSHEAR (sheared-branch error columns + commented out — downstream response-weighted estimators cannot + propagate per-branch errors), SExtractor photometry (MAG_AUTO, + FLUX_APER, FLUX_RADIUS, SNR_WIN, FWHM_*), N_EPOCH/NGMIX_N_EPOCH, + NGMIX_MOM_FAIL. Doesn't change membership, but determines which + numbers exist for downstream cuts and calibration. Note + final_cat.param is read by scripts/python/create_final_cat.py in + post-processing, outside the per-tile DAG. [LINT] see detection: + IMAFLAGS_ISO is requested but never reaches the merged catalogue. + Anchor: workflow/config/cfis/final_cat.param; + scripts/python/create_final_cat.py. + default: committed_param_list + options: + committed_param_list: { label: The committed final_cat.param set } + failure_sentinels: + label: Objects without shape measurements kept, with sentinel values [HARDCODED] + rationale: >- + Unmatched objects (no ngmix row) stay in the catalogue with + sentinels: sizes/fluxes/flags 0, error fluxes/mags -1, + ellipticities -10, T_ERR 1e30. The sentinel choice defines what a + downstream cut must exclude — a naive G1 > -1 cut silently changes + the sample. Flag-0-for-failure is the sharpest hazard: a failed + object's NGMIX_MCAL_FLAGS reads as success. Rationale not recorded. + Anchor: src/shapepipe/modules/make_cat_package/make_cat.py::SaveCatalogue._save_ngmix_data. + default: sentinel_values + options: + sentinel_values: { label: "Keep with sentinels (flags 0, e -10, T_ERR 1e30)" } + drop_unmatched: + label: Drop objects without shapes + excluded: true + excluded_reason: >- + Loses the photometry-only population and hides attrition from + the completeness accounting. + postproc_run_provenance: + label: Post-proc run selection — newest mtime wins, merged patches never refresh + rationale: >- + create_final_cat picks each tile's make_cat run by newest + directory mtime (skipping runs without an output FITS), and the + merged patch catalogue is incremental: a tile already present is + never refreshed — a reprocessed tile reaches the patch only via + an explicit single-ID remove+add. mtime is filesystem state, not + provenance: a touched old run can outrank a newer one. Outside + the per-tile DAG (scripts/, not workflow/). Anchor: + scripts/python/create_final_cat.py::process. + default: newest_mtime_incremental + options: + newest_mtime_incremental: { label: "Newest mtime, incremental merge, manual refresh" } + shape_catalogue_shortfall_guard: + label: 10% shape-shortfall guard — logged, not enforced [HARDCODED] + rationale: >- + If the merged shape catalogue covers <10% of the detection + catalogue, make_cat logs an error but the enforcement (return + + raise) is commented out in both the module and its runner: a tile + whose shapes are 90% missing from a processing error is written and + looks normal. The comment distinguishes the two causes (measurement + failure = ok; premature merge = error) but not why enforcement is + off. Interacts with top-level per_unit_count_floor, which floors + tile_ngmix at 1 file and cannot see intra-file attrition. + Anchor: src/shapepipe/modules/make_cat_package/make_cat.py::SaveCatalogue._save_ngmix_data; + src/shapepipe/modules/make_cat_runner.py::make_cat_runner. + default: log_only + options: + log_only: { label: "Log the shortfall, write the tile anyway" } + enforce_10pct: + label: Fail the tile below 10% coverage + excluded: true + excluded_reason: >- + Was the coded intent, then disabled — reason unrecorded; + flagged as a question, not an endorsement. + +# ── Dormant science paths, surveyed but not yet recorded as sub-analyses ──── +# Candidates for future passes (each carries real scientific knobs): +# * External photometry match — match_external_package (TOLERANCE=0.3 +# arcsec against UNIONS ugriz; the external catalogue path is hardcoded +# to an IAP/candide location). +# * Image-simulation validation wiring — example/cfis_image_sims/: +# same chain over SKiLLS images with fake_psf substitution; bash-shaped, +# not yet ported to snakemake. diff --git a/universes/committed.yaml b/universes/committed.yaml new file mode 100644 index 000000000..5530226cf --- /dev/null +++ b/universes/committed.yaml @@ -0,0 +1,70 @@ +id: committed +description: The committed configuration on feat/snakemake-orchestration. +decisions: + per_unit_count_floor: count_floor + postage_stamp_size: px_51 + photometric_zeropoint: fixed_30_tiles_header_exposures + baseline_validation_criterion: statistical_parity +analyses: + masking: + decisions: + star_catalogue_query: gsc_23_vizier + star_magnitude_definition: mean_finite_bands + bright_star_mask_geometry: megaprime_polygon_linear_scaling + deep_sky_object_masking: circles_no_padding + border_mask_width: px50_exposures_only + pixel_threshold_flags: stock_ww_thresholds + external_flag_usage: exposures_only + detection: + decisions: + detection_threshold_policy: thresh_1p5_minarea5_fwhm2px_filter + deblending_policy: mincont_5em4_tiles + background_model: manual_zero_tiles_auto_exposures + weighting_and_interpolation: map_weight_interp_all + detection_source_mode: sx_nomask_single_image + epoch_membership_ccd_bounds: trimmed_bounds_33_2080 + photometry_parameters: kron_25_35 + cleaning_and_neighbour_masking: clean_1_correct + preparation: + decisions: + astrometric_solution_source: delivered_headers + ccd_split_extent: all_40_hdus + epoch_provenance_from_tile_history: history_parse + object_position_columns: xwin_windowed + stamp_positioning_and_padding: round_and_zero_pad + epoch_flag_source: raw_flags + star_selection_psf: + decisions: + star_selection_box: mode_centred_box + psfex_candidate_vetting: builtin_defaults + psf_modelling_software: psfex + psf_train_validation_split: split_80_20_seeded + psf_model_complexity: pixel_basis_deg2_per_ccd + psf_acceptance_thresholds: stars22_chi2_2 + shape_measurement: + decisions: + ngmix_seed_mode: position_seed + galaxy_model: gauss + fit_priors: gpriorba04_flat + metacal_scheme: five_types_step001_fitgauss + centroid_source: wcs + epoch_quality_and_weighting: third_masked_cut + noise_model: rms_vignet_weights + psf_epoch_loss_policy: record_and_delegate + megacam_ccd_flip: megapipe_flip + psf_diagnostics: + decisions: + starcat_merge_source: psfex + meanshape_binning: grid_5x10 + survey_geometry: + decisions: + random_sampling: per_sqdeg_50k + healpix_mask_export: nside_1024 + catalogue_assembly: + decisions: + star_galaxy_classification: deferred_downstream + tile_overlap_handling: no_dedup_in_pipeline + column_selection: committed_param_list + failure_sentinels: sentinel_values + postproc_run_provenance: newest_mtime_incremental + shape_catalogue_shortfall_guard: log_only From 206b48d6c4818578ed5ecfff4e730cfcf80efcc2 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 02:34:43 +0200 Subject: [PATCH 02/40] test(astra): validate decision anchors and universe pins --- tests/helpers/astra_record.py | 284 +++++++++++++++++++++++++++++++ tests/unit/test_astra_anchors.py | 50 ++++++ 2 files changed, 334 insertions(+) create mode 100644 tests/helpers/astra_record.py create mode 100644 tests/unit/test_astra_anchors.py diff --git a/tests/helpers/astra_record.py b/tests/helpers/astra_record.py new file mode 100644 index 000000000..2b8f70ed9 --- /dev/null +++ b/tests/helpers/astra_record.py @@ -0,0 +1,284 @@ +"""Reusable parsing and resolution helpers for ShapePipe's ASTRA record.""" + +import ast +import configparser +from dataclasses import dataclass +from pathlib import Path +import re + +import yaml + + +@dataclass(frozen=True) +class Anchor: + """An anchor sentence found in a YAML value.""" + + location: str + references: tuple[str, ...] + error: str | None = None + + +def load_yaml(path): + """Load YAML with PyYAML's safe loader.""" + + return yaml.safe_load(Path(path).read_text(encoding="utf-8")) + + +def _walk(value, location=""): + if isinstance(value, dict): + for key, child in value.items(): + path = f"{location}.{key}" if location else str(key) + yield from _walk(child, path) + elif isinstance(value, list): + for index, child in enumerate(value): + yield from _walk(child, f"{location}[{index}]") + else: + yield location, value + + +def _rationales(document): + for location, value in _walk(document): + if location.endswith(".rationale"): + yield location, value + + +def extract_anchors(document): + """Parse all ``Anchor:`` sentences and check every rationale has one.""" + + anchors = [] + for location, value in _walk(document): + if not isinstance(value, str) or "Anchor:" not in value: + continue + tail = value.split("Anchor:", 1)[1].strip() + error = None + refs = () + if value.count("Anchor:") != 1: + count = value.count("Anchor:") + error = f"expected one Anchor: marker, found {count}" + elif not tail.endswith("."): + error = "anchor sentence must end with a period" + else: + refs = tuple(part.strip() for part in tail[:-1].split(";")) + if not refs or any(not ref for ref in refs): + error = "anchor sentence contains an empty ref" + anchors.append(Anchor(location, refs, error)) + + for location, value in _rationales(document): + if ( + not isinstance(value, str) + or value.count("Anchor:") != 1 + or not value.rstrip().endswith(".") + ): + anchors.append( + Anchor( + location, + (), + "rationale must end with exactly one Anchor: sentence", + ) + ) + return anchors + + +def _parse_reference(reference): + if "::" in reference: + path, symbol = reference.split("::", 1) + return "code", path, symbol + if "#" in reference: + path, key = reference.split("#", 1) + return "config", path, key + return "path", reference, "" + + +def resolve_anchor(root, reference): + """Return ``None`` if a reference resolves, otherwise a diagnostic.""" + + kind, relative, selector = _parse_reference(reference) + path = Path(relative) + if path.is_absolute() or ".." in path.parts: + return "path must be relative to the repository root" + target = Path(root) / path + if not target.exists(): + return "path does not exist" + if kind == "path": + return None + if not target.is_file(): + return "code/config refs must name a file" + + try: + text = target.read_text(encoding="utf-8") + except (OSError, UnicodeError) as error: + return f"cannot read file: {error}" + + if kind == "code": + if target.suffix != ".py": + return "code-symbol refs must name a .py file" + try: + tree = ast.parse(text, filename=str(target)) + except SyntaxError as error: + return f"cannot parse Python file: {error}" + if not _has_symbol(tree, selector): + return f"no def/class/assignment target named {selector!r}" + return None + + suffix = target.suffix.lower() + if suffix == ".ini": + return _ini_key(text, selector) + if suffix == ".setools": + return _setools_key(text, selector) + if suffix in {".sex", ".psfex", ".ww", ".param", ".conf"}: + key = selector.rsplit(".", 1)[-1] + pattern = re.compile(rf"^\s*(?:#\s*)?{re.escape(key)}(?=$|\s|=|\()") + if any(pattern.search(line) for line in text.splitlines()): + return None + return f"no line starts with key {key!r} (commented keys are allowed)" + return f"unsupported config-key file type {suffix or '(no extension)'}" + + +def _ini_key(text, selector): + if "." not in selector: + return "INI config ref needs SECTION.KEY" + section, key = selector.rsplit(".", 1) + parser = configparser.ConfigParser( + interpolation=None, strict=False, allow_no_value=True + ) + try: + parser.read_string(text) + except configparser.Error as error: + return f"cannot parse INI file: {error}" + if not parser.has_section(section): + return f"INI section {section!r} is missing" + if not parser.has_option(section, key): + return f"INI key {key!r} is missing from section {section!r}" + return None + + +def _setools_key(text, selector): + if "." not in selector: + return "SETools config ref needs SECTION.KEY" + section, key = selector.rsplit(".", 1) + pattern = re.compile(rf"^\s*(?:#\s*)?{re.escape(key)}(?=$|\s|=|<|>)") + active = False + for line in text.splitlines(): + stripped = line.strip() + if stripped.startswith("[") and stripped.endswith("]"): + active = stripped[1:-1].strip() == section + elif active and pattern.search(line): + return None + return f"SETools key {key!r} is missing from section {section!r}" + + +def _target_names(target): + if isinstance(target, ast.Name): + return [target.id] + if isinstance(target, ast.Attribute): + return [target.attr] + if isinstance(target, (ast.Tuple, ast.List)): + return [name for item in target.elts for name in _target_names(item)] + if isinstance(target, ast.Starred): + return _target_names(target.value) + return [] + + +def _bindings(scope): + """Collect declarations and assignment targets in one lexical scope.""" + + result = {} + + def visit(node): + if isinstance( + node, + (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef), + ): + result[node.name] = node + return + if isinstance(node, ast.Lambda): + return + if isinstance(node, ast.Assign): + targets = node.targets + elif isinstance(node, (ast.AnnAssign, ast.AugAssign, ast.NamedExpr)): + targets = [node.target] + elif isinstance(node, (ast.For, ast.AsyncFor)): + targets = [node.target] + elif isinstance(node, (ast.With, ast.AsyncWith)): + targets = [item.optional_vars for item in node.items] + else: + targets = [] + for target in targets: + if target is not None: + result.update(dict.fromkeys(_target_names(target), node)) + if isinstance(node, ast.ExceptHandler) and node.name: + result[node.name] = node + for child in ast.iter_child_nodes(node): + visit(child) + + for statement in scope.body: + visit(statement) + return result + + +def _has_symbol(tree, symbol): + scope = tree + parts = symbol.split(".") + for index, part in enumerate(parts): + declaration = _bindings(scope).get(part) + if declaration is None: + return False + if index == len(parts) - 1: + return True + if not isinstance( + declaration, + (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef), + ): + return False + scope = declaration + return False + + +def universe_errors(record, universe): + """Check scoped decision IDs and options against the ASTRA record.""" + + record_decisions = _decisions(record) + pinned = _decisions(universe) + errors = [] + for location in sorted(pinned.keys() - record_decisions.keys()): + errors.append( + f"{location}: universe decision is absent from astra.yaml" + ) + for location in sorted(record_decisions.keys() - pinned.keys()): + errors.append( + f"{location}: astra.yaml decision is not pinned in the universe" + ) + for location in sorted(record_decisions.keys() & pinned.keys()): + definition = record_decisions[location] + options = ( + definition.get("options", {}) + if isinstance(definition, dict) + else {} + ) + if not isinstance(options, dict) or pinned[location] not in options: + errors.append( + f"{location}: pinned option {pinned[location]!r} is not in " + "ASTRA options" + ) + return errors + + +def _decisions(document, location=""): + scope = document if isinstance(document, dict) else {} + result = {} + for decision_id, definition in (scope.get("decisions") or {}).items(): + key = ( + f"{location}.decisions.{decision_id}" + if location + else f"decisions.{decision_id}" + ) + result[key] = definition + for analysis_id, analysis in (scope.get("analyses") or {}).items(): + child = ( + f"{location}.analyses.{analysis_id}" + if location + else f"analyses.{analysis_id}" + ) + if isinstance(analysis, dict): + result.update(_decisions(analysis, child)) + return result diff --git a/tests/unit/test_astra_anchors.py b/tests/unit/test_astra_anchors.py new file mode 100644 index 000000000..b4d95c2d1 --- /dev/null +++ b/tests/unit/test_astra_anchors.py @@ -0,0 +1,50 @@ +"""Keep ASTRA decision anchors and the committed universe resolvable.""" + +from pathlib import Path + +from tests.helpers.astra_record import ( + extract_anchors, + load_yaml, + resolve_anchor, + universe_errors, +) + + +REPO_ROOT = Path(__file__).resolve().parents[2] + + +def test_every_astra_anchor_resolves(): + record = load_yaml(REPO_ROOT / "astra.yaml") + anchors = extract_anchors(record) + errors = [] + + assert anchors, "astra.yaml contains no Anchor: sentences" + for anchor in anchors: + if anchor.error: + errors.append(f"{anchor.location}: {anchor.error}") + continue + for reference in anchor.references: + problem = resolve_anchor(REPO_ROOT, reference) + if problem: + errors.append( + f"{anchor.location}: {reference}: {problem}" + ) + + message = ( + "Unresolved ASTRA anchors or rationales:\n - " + + "\n - ".join(errors) + ) + assert not errors, message + + +def test_committed_universe_matches_astra_decisions(): + record = load_yaml(REPO_ROOT / "astra.yaml") + universe = load_yaml(REPO_ROOT / "universes" / "committed.yaml") + + errors = universe_errors(record, universe) + + message = ( + "ASTRA / committed universe mismatch:\n - " + + "\n - ".join(errors) + ) + assert not errors, message From 9f9ec5ea182bb24a991f82365b19ce82f8a1bef4 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 02:54:27 +0200 Subject: [PATCH 03/40] test(astra): resolve Snakemake rule anchors; JSON report mode Co-Authored-By: Claude Fable 5.1 --- tests/helpers/astra_record.py | 102 +++++++++++++++++++++++++++++++ tests/unit/test_astra_anchors.py | 25 ++++++++ 2 files changed, 127 insertions(+) diff --git a/tests/helpers/astra_record.py b/tests/helpers/astra_record.py index 2b8f70ed9..1a7123c7d 100644 --- a/tests/helpers/astra_record.py +++ b/tests/helpers/astra_record.py @@ -1,10 +1,14 @@ """Reusable parsing and resolution helpers for ShapePipe's ASTRA record.""" +import argparse import ast import configparser from dataclasses import dataclass +import json from pathlib import Path import re +import subprocess +import sys import yaml @@ -89,6 +93,26 @@ def _parse_reference(reference): return "path", reference, "" +_SNAKEMAKE_SUFFIXES = {".smk"} +_SNAKEFILE_NAMES = {"Snakefile"} + + +def _is_snakemake_file(target): + return target.suffix in _SNAKEMAKE_SUFFIXES or target.name in _SNAKEFILE_NAMES + + +def _snakemake_symbol(text, symbol): + rule_pattern = re.compile( + rf"^\s*(?:rule|checkpoint)\s+{re.escape(symbol)}\s*:", re.MULTILINE + ) + if rule_pattern.search(text): + return None + def_pattern = re.compile(rf"^\s*def\s+{re.escape(symbol)}\(", re.MULTILINE) + if def_pattern.search(text): + return None + return f"no rule/checkpoint/def named {symbol!r}" + + def resolve_anchor(root, reference): """Return ``None`` if a reference resolves, otherwise a diagnostic.""" @@ -110,6 +134,8 @@ def resolve_anchor(root, reference): return f"cannot read file: {error}" if kind == "code": + if _is_snakemake_file(target): + return _snakemake_symbol(text, selector) if target.suffix != ".py": return "code-symbol refs must name a .py file" try: @@ -282,3 +308,79 @@ def _decisions(document, location=""): if isinstance(analysis, dict): result.update(_decisions(analysis, child)) return result + + +def _git_sha(root): + try: + return subprocess.run( + ["git", "rev-parse", "HEAD"], + cwd=root, + capture_output=True, + check=True, + text=True, + ).stdout.strip() + except (OSError, subprocess.CalledProcessError): + return None + + +def build_report(root): + """Resolve every anchor and universe pin under ``root`` into a report dict.""" + + root = Path(root) + astra_yaml = root / "astra.yaml" + record = load_yaml(astra_yaml) + anchors = extract_anchors(record) + + unresolved = [] + for anchor in anchors: + if anchor.error: + unresolved.append( + {"location": anchor.location, "ref": None, "problem": anchor.error} + ) + continue + for reference in anchor.references: + problem = resolve_anchor(root, reference) + if problem: + unresolved.append( + { + "location": anchor.location, + "ref": reference, + "problem": problem, + } + ) + + universe = load_yaml(root / "universes" / "committed.yaml") + errors = universe_errors(record, universe) + + return { + "astra_yaml": str(astra_yaml), + "git_sha": _git_sha(root), + "anchors_total": len(anchors), + "unresolved": unresolved, + "universe_errors": errors, + "ok": not unresolved and not errors, + } + + +def main(argv=None): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--report", required=True, help="path to write the JSON report to" + ) + parser.add_argument( + "--root", + default=Path(__file__).resolve().parents[2], + help="repository root (default: repo root inferred from this file)", + ) + args = parser.parse_args(argv) + + report = build_report(args.root) + report_path = Path(args.report) + report_path.parent.mkdir(parents=True, exist_ok=True) + report_path.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + + return 0 if report["ok"] else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/unit/test_astra_anchors.py b/tests/unit/test_astra_anchors.py index b4d95c2d1..38b334f12 100644 --- a/tests/unit/test_astra_anchors.py +++ b/tests/unit/test_astra_anchors.py @@ -13,6 +13,31 @@ REPO_ROOT = Path(__file__).resolve().parents[2] +def test_snakemake_rule_and_function_anchors_resolve(tmp_path): + rules_dir = tmp_path / "workflow" / "rules" + rules_dir.mkdir(parents=True) + (rules_dir / "example.smk").write_text( + "def tile_local(tile):\n return tile\n\n\n" + "rule tile_detect:\n input: 'a'\n output: 'b'\n", + encoding="utf-8", + ) + + assert resolve_anchor(tmp_path, "workflow/rules/example.smk::tile_detect") is None + assert resolve_anchor(tmp_path, "workflow/rules/example.smk::tile_local") is None + + problem = resolve_anchor(tmp_path, "workflow/rules/example.smk::no_such_rule") + assert problem == "no rule/checkpoint/def named 'no_such_rule'" + + +def test_snakefile_rule_anchor_resolves(tmp_path): + (tmp_path / "workflow").mkdir() + (tmp_path / "workflow" / "Snakefile").write_text( + "checkpoint plan:\n input: 'a'\n", encoding="utf-8" + ) + + assert resolve_anchor(tmp_path, "workflow/Snakefile::plan") is None + + def test_every_astra_anchor_resolves(): record = load_yaml(REPO_ROOT / "astra.yaml") anchors = extract_anchors(record) From 5368d9df7e981caabc2369665fd18f60c52c6d96 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 03:04:34 +0200 Subject: [PATCH 04/40] docs(astra): rewrite the decision record against develop - masking: describe healsparse queries (mask_query MASK_EXT on exposures, make_cat MASK_ on tiles) and the instrument flag image as the only pixel mask, replacing the deleted in-house mask generation - detection: tiles follow the MegaPipe (Gwyn) SExtractor parameters (#896); option ids no longer encode the retired values - shape_measurement: import defect_fill, blend_handling and epoch_masked_fraction_cut from the digital twin with their literature insights; defaults are what the committed code selects - prune to the membership test: drop psf_diagnostics, survey_geometry, the workflow-policy decisions and the root findings; split compound decisions; reserve excluded for considered-and-rejected; strip chronology - re-point anchors to the current configs; the anchor test passes Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014bvNTrAmZxcfb1ee83ApPK --- astra.yaml | 2397 ++++++++++++++++++-------------------- universes/committed.yaml | 54 +- 2 files changed, 1187 insertions(+), 1264 deletions(-) diff --git a/astra.yaml b/astra.yaml index 70b25b554..1b27db982 100644 --- a/astra.yaml +++ b/astra.yaml @@ -1,162 +1,129 @@ -# ASTRA record for ShapePipe: the scientific decisions embedded in the code and -# the committed configs, with their reasoning and the alternatives that were -# rejected. It is the place scientific decisions are written down — see the -# "Scientific decisions" section of CLAUDE.md for when and how to amend it. +# ASTRA record of ShapePipe's scientific decisions: the choices embedded in +# the code and the committed workflow configs (workflow/config/cfis/), why +# they stand, and the alternatives. CLAUDE.md says when to amend it; the +# anchor test tests/unit/test_astra_anchors.py keeps it resolvable. # -# The record describes the pipeline as orchestrated by workflow/Snakefile -# (PRD CosmoStat/shapepipe#848, PR #852). Conventions: -# -# * Decisions anchor to code, not recipes. Every rationale ends with one -# sentence "Anchor: ; ; ..." in a strict, greppable grammar. -# Each ref is a path relative to the shapepipe repo root, in one of three -# forms: CODE `path::symbol`, CONFIG `path#SECTION.KEY` (or `path#KEY` for -# sectionless .sex/.psfex/.ww/.param files), FILE `path` for a whole file -# or package. No line numbers — they rot; a line-level fact names its -# enclosing symbol. The analysis-ASTRA rule "never hardcode; reference via -# {decisions.x}" cannot hold in a codebase — the committed configs ARE the -# values. [HARDCODED] marks a scientific value living in code with no -# config exposure: the silent defaults the record exists to surface. -# * The default universe IS the committed configuration (universes/committed). -# Alternatives are excluded-with-reasons or genuinely open forks. -# * Sub-analyses follow the pipeline's methodological units — masking, -# detection, preparation, star selection + PSF, shape measurement, PSF -# diagnostics, survey geometry, catalogue assembly — not its ~20 Snakemake -# rules. Cross-cutting decisions stay top-level. A prior_insight repeated -# inside a sub-analysis carries a `_local` suffix: ids are scoped, and the -# duplicate keeps the sub-analysis readable on its own. -# * Outputs are representative product FAMILIES (one final_cat per tile), -# not enumerable artifacts; no recipes — the executor is the Snakemake -# workflow. -# * [LINT] marks places where this record and the code already disagree, or -# where the code disagrees with itself — found while authoring this file. -# * [PENDING #NNN] marks state that is live on feat/snakemake-orchestration -# — and therefore in smk-g4, the 34-tile validation campaign run under -# this branch — but not yet merged to develop. The record follows the -# branch and names the open PR. -# * A `path#KEY` anchor names the key's position in the file, not its -# activation: where the decision is "this is deliberately off", the key -# it points at may be commented out (e.g. final_cat.param#SPREAD_CLASS). +# Conventions: +# * Every rationale ends with one sentence "Anchor: ; ." Each ref +# is a path relative to the repo root: CODE `path::Symbol` (a def, class +# or assignment target, dotted for nesting), CONFIG `path#SECTION.KEY` +# (`path#KEY` for sectionless .sex/.psfex/.param files, where a +# commented-out key still resolves), or FILE `path`. No line numbers. +# Numeric values are stated once, in the rationale, next to the anchor +# that holds them. +# * A decision's `default` is the option the committed code and configs +# select; universes/committed.yaml pins it. +# * `excluded: true` means considered and rejected. An option the code does +# not implement says so in its description and is not excluded. +# * [HARDCODED] in a rationale marks a committed choice fixed in code with +# no config key to change it. +# * [LINT] in a rationale marks a place where the code disagrees with itself +# or with its own documentation. +# * A prior insight repeated inside a sub-analysis carries a `_local` +# suffix, because insight ids are scoped. version: "0.0.14" name: ShapePipe scientific decisions description: >- Codebase-level decision record for the ShapePipe weak-lensing pipeline - (UNIONS/CFIS). Membership test: "a different defensible choice would change - which objects enter the shear catalogue, or the numbers attached to them." - Workflow mechanics that reproduce identical numbers (manifest sentinels, - clean-cascade cut, directory() outputs, allocation strategy, chunking under - position seeding) are deliberately absent; they live in the PRD and code. + (UNIONS/CFIS) as orchestrated by workflow/Snakefile. Membership test: a + different defensible choice would change which objects enter the shear + catalogue, or the numbers attached to them. tags: [shapepipe, weak-lensing, unions, codebase-record] -container: shapepipe-develop-runtime.sif +container: ghcr.io/cosmostat/shapepipe:develop inputs: - id: tile_images type: data - source: CADC-staged CFIS/UNIONS r-band tile stacks + exposure triplets (workflow/config.yaml) + source: CFIS/UNIONS r-band MegaPipe tile stacks and single-exposure image/weight/flag triplets (workflow/config.yaml) description: >- - Pre-staged P3 tiles and single-exposure image/weight/flag triplets on - /project; get_images runs with RETRIEVE=symlink against this store. - - id: gsc_star_catalogue + Tiles and the exposures that built them; each exposure carries its + instrument flag image. + - id: healsparse_masks type: data - source: GSC 2.3 (Vizier I/305/out) cone queries — scripts/python/create_star_cat.py + source: UNIONS healsparse mask products (e.g. mask_ugriz_nside131072_n4.hsp) description: >- - Reference star catalogue driving bright-star masking. Catalogue choice, - query geometry, and magnitude handling are decisions in the masking - sub-analysis. + Sky-fixed mask maps, built outside ShapePipe and queried at object + positions; how they are used is the masking sub-analysis. outputs: - id: final_cat type: data format: fits description: >- - Per-tile shear catalogue family, the terminal science product (one per - campaign tile; make_cat_runner). Column selection and failure sentinels - are decisions in catalogue_assembly. - inputs: [tile_images] - decisions: [per_unit_count_floor, postage_stamp_size, photometric_zeropoint] + Per-tile shear catalogue family, the terminal science product + (make_cat_runner). + inputs: [tile_images, healsparse_masks] + decisions: [per_unit_completeness, postage_stamp_size, photometric_zeropoint] decisions: # ── cross-cutting ──────────────────────────────────────────────────────── - per_unit_count_floor: - label: Per-unit completeness policy under partial failure + per_unit_completeness: + label: Per-unit completeness gate rationale: >- - A 40-CCD stage where some CCDs legitimately produce nothing (sparse CCD, - setools rejects everything) cannot be all-or-nothing. The field's - converged answer (DES PSF blacklist, Rubin quantum registry) is per-unit - outcome records gated on a quality floor: record the attrition, fail - loud only below the floor, continue the survey. The floor VALUES are the - scientific content — how much silent per-CCD attrition can enter the - catalogue. The COMPLETENESS table holds them (exp_split - expect=121/floor=41, exp_mask expect=40/floor=1, psfex expect=80/floor=2, - psfex_interp floor=0 warn-only). Related leak the floor does not cover: - merge_sep_cats warns-and-skips a missing ngmix chunk, silently shrinking - a tile's shape catalogue below the floor's radar; and make_cat's own 10% - size-shortfall guard is commented out (see - catalogue_assembly.shape_catalogue_shortfall_guard). - Anchor: workflow/scripts/completeness.py::COMPLETENESS; - src/shapepipe/modules/merge_sep_cats_package/merge_sep_cats.py::MergeSep.process. - default: count_floor + Every rule checks its products against a nominal per-runner count: a + runner below its count fails the unit (an exposure or a tile), so a + partial unit never enters the catalogue; the missing unit's objects do + not appear at all. The one tolerated shortfall is the exposure-side + psfex_interp VALIDATION output, where a CCD whose model fails the + acceptance gate (star_selection_psf.psf_acceptance_thresholds) produces + nothing and the unit only warns; the MCCD chain, never run in a + campaign, warns on every runner. Science-path PSF rejection does not go + through this table: psfex_interp drops the epoch per object inside the + tile run. + Anchor: workflow/scripts/completeness.py::COMPLETENESS. + default: exact_counts options: - count_floor: - label: Count-floor table (expect/floor per runner; fail below floor) + exact_counts: + label: Nominal count per runner; psfex_interp validation shortfall warns insights: [des_psf_blacklist, guinot22_star_floor_22] - all_or_nothing: - label: Every expected sub-product required + count_floor: + label: Tolerate recorded attrition down to a per-runner floor excluded: true excluded_reason: >- - Legitimately-absent CCDs would fail whole exposures and poison their - downstream cone; Snakemake has no optional-output primitive; field - precedent is tolerated, recorded attrition. - no_floor: - label: Accept whatever is produced, no gate + The floors had no basis: across a 127-exposure, 64-tile campaign + every non-warning runner produced exactly its nominal count, so a + floor below it only admits failed units unremarked. + no_gate: + label: Accept whatever is produced excluded: true excluded_reason: >- - Silent attrition — a stage producing 2 of 40 CCDs would flow into - the catalogue unremarked. + A stage producing 2 of 40 CCDs would flow into the catalogue + unremarked. postage_stamp_size: - label: Postage-stamp size, 51 px everywhere + label: Postage-stamp size shared by vignets, ngmix stamps and PSF models rationale: >- - One number pins three coupled apertures: the SExtractor vignet cut - around each detection (VIGNET(51,51) in default_noimaflags.param / - default.param, VIGNET_SIZE=51 in the dormant external-catalogue path, - example/cfis/config_tile_Uc.ini), the vignetmaker - stamps that feed ngmix (STAMP_SIZE=51 in config_tile_PiViVi.ini, both - runs; nearest-pixel centring, no sub-pixel interpolation in - VignetMaker._get_stamp), and the PSFEx model stamp (PSF_SIZE 51,51 in - default.psfex). The stamp IS the pixel data ngmix fits: it bounds - measurable galaxy size and truncates the wings of large galaxies. - Rationale for 51 not recorded in code. + One number, 51 px (about 9.5 arcsec), pins three coupled apertures: + the SExtractor VIGNET(51,51) around each detection, the vignetmaker + stamps that feed ngmix (STAMP_SIZE in both vignetmaker runs), and the + PSFEx model stamp (PSF_SIZE 51,51). The stamp is the pixel data ngmix + fits: it bounds the measurable galaxy size and truncates the wings of + large galaxies. No rationale for 51 is recorded. Anchor: workflow/config/cfis/default_noimaflags.param#VIGNET; - example/cfis/config_tile_Uc.ini#READ_EXT_SEXCAT_RUNNER.VIGNET_SIZE; - workflow/config/cfis/config_tile_PiViVi.ini#VIGNETMAKER_RUNNER_RUN_1.STAMP_SIZE; - workflow/config/cfis/default.psfex#PSF_SIZE; - src/shapepipe/modules/vignetmaker_package/vignetmaker.py::VignetMaker._get_stamp. + workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.STAMP_SIZE; + workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.STAMP_SIZE; + workflow/config/cfis/default.psfex#PSF_SIZE. default: px_51 options: px_51: - label: 51x51 px (~9.5 arcsec at 0.187"/px) + label: 51x51 px everywhere larger_adaptive: label: Larger or size-adaptive stamps - excluded: true - excluded_reason: >- - Not wired; would need coupled changes in three places (a change in - any one alone desynchronises galaxy stamp, PSF stamp, and vignet). + description: >- + Not implemented; needs the vignet, stamp and PSF sizes changed + together, since changing one alone desynchronises them. photometric_zeropoint: - label: Magnitude zero-point convention, fixed 30.0 on tiles + label: Magnitude zero-point convention rationale: >- - Tiles use a hard-coded MAG_ZEROPOINT 30.0 for every tile - (default_tile.sex; ZP_FROM_HEADER=False in config_tile_Sx.ini), and - ngmix repeats it (MAG_ZP=30.0 in config_tile_Ng_template.ini). - Exposures instead read the per-image header zero-point - (ZP_FROM_HEADER=True, ZP_KEY=PHOTZP in config_exp_psfex.ini). The tile - convention leans on MegaPipe's calibrated stacks; the star-selection - magnitude window (18-22) and mask magnitude limits inherit whichever - convention their stage uses. SExtractorCaller.get_zero_point is the - header-reading path, unused on tiles. + Tiles use a fixed MAG_ZEROPOINT 30.0 (ZP_FROM_HEADER=False), and ngmix + repeats it in MAG_ZP; exposures read the per-image header PHOTZP. The + tile convention relies on MegaPipe stacks being calibrated to 30 by + construction. Magnitude cuts (the star-selection window, downstream + galaxy cuts) inherit whichever convention their stage uses. Anchor: workflow/config/cfis/default_tile.sex#MAG_ZEROPOINT; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER; workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.MAG_ZP; @@ -170,44 +137,9 @@ decisions: label: Per-image header zero-points on tiles too excluded: true excluded_reason: >- - MegaPipe stacks are calibrated to ZP 30 by construction; per-tile - header reads add a failure path for no expected numerical change. - (If that claim is wrong, this is a real fork — verify.) - - baseline_validation_criterion: - label: Validation criterion against the v2.0 bash baseline - rationale: >- - Because shape_measurement.ngmix_seed_mode deliberately changes noise - streams, P1 validation against v2.0 is statistical parity - (population-level agreement), not bit parity. Everything upstream of - ngmix (through PSFEx) validated bit-exactly (P0: 4/4 PASS). This - defines the evidence standard for "the same pipeline" — surfaced to the - collaboration as open Q5 in PRD #848. - [PENDING #873] Run-to-run determinism, which is a different property - from parity with v2.0, is now complete. With the setools star split - seeded (star_selection_psf.psf_train_validation_split) the last unseeded - draw in the science chain is gone: two runs of this code over the same - inputs now produce the same PSF star sample, the same PSF models and - the same shapes, which they did not before. That also settles a tension - this record carried — the bit-parity claim above sat next to an - unseeded star split that could not have been bit-reproducible, and the - P0 exposure-stage comparison did see PSF-validation CCD attrition - differ between the two sides. Statistical rather than bit parity is - therefore demanded only against the v2.0 baseline, not between runs of - the current pipeline. - Anchor: workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.SEED_FROM_POSITION; - src/shapepipe/modules/ngmix_package/ngmix.py::position_seed; - src/shapepipe/modules/setools_package/setools.py::SETools._make_rand_split. - default: statistical_parity - options: - statistical_parity: - label: Population-level agreement in shear observables - bit_parity: - label: Bit-identical catalogues - excluded: true - excluded_reason: >- - Impossible by construction once the seed mode changed; requiring it - would freeze the chunk-dependent v2.0 RNG forever. + MegaPipe stacks are calibrated to 30, so a header read adds a failure + path for no numerical change; it becomes a real fork only if tiles + ever carry a different PHOTZP. prior_insights: des_psf_blacklist: @@ -222,7 +154,7 @@ prior_insights: doi: "10.48550/arXiv.2011.03409" quote: exact: "we enter it into a" - suffix: " \u201cblacklist\u201d and exclude this CCD" + suffix: " “blacklist” and exclude this CCD" location: { page: 10 } guinot22_star_floor_22: claim: >- @@ -238,320 +170,147 @@ prior_insights: exact: 'The dashed line represents the cut at 22 stars/CCD below which the CCD is discarded for the PSF estimation.' location: { page: 4 } -findings: - orchestration_parity: - claim: >- - The Snakemake orchestration reproduces the bash baseline bit-exactly - through PSFEx (P0 validation, 4/4 PASS on the 186/187 quad). - Read it with two caveats. It is a statement about the pre-#873 code: - both #873 changes move products (a different realised star split, a - stricter science-path star gate), so re-establishing parity would mean - regenerating the baseline under the current branch. And the parity is - bit-exact in the products compared, not everywhere: the P0 - exposure-stage comparison did see PSF-validation CCD attrition differ - between the two sides, which the then-unseeded star split explains - (see baseline_validation_criterion). - created_at: "2026-08-19T00:00:00Z" - evidence: - - id: ev_final_cat - artifact: final_cat - record_authoring_found_defects: - claim: >- - Nine places in the code disagree with themselves or with their - documentation, each carried as a [LINT] mark: the 22-vs-20 - STAR_THRESH mismatch between PSF validation and science interpolation; - the unseeded train/validation rand_split (setools.py:664 — the star - sample entering the PSF model is irreproducible run-to-run); additive - (non-bitwise) mask-plane combination, safe today only because the - committed flag values are disjoint; the dead MESSIER_PIXEL_SCALE config - key; final_cat.param requesting IMAFLAGS_ISO that the merged catalogue - never receives; the centroid_source default disagreement (runner "wcs" - vs module "hsm", latent for direct callers); setools logging a FWHM cut - (mode +- 0.1 px in arcsec) half the applied one (mode +- 0.2 px), and - mixing pixel scales 0.187/0.186 within one file; TILE_LIST - overlap-flagging documented but never implemented; the mccd_plots - module docstring advertising rho statistics that live downstream now. - Status: the first two are fixed. CosmoStat/shapepipe#873 seeds the - rand_split and raises the science-path STAR_THRESH to 22, and commit - 90782098 mirrors that threshold into the workflow's own committed - config fork. #873 is OPEN against develop; both fixes are live on - feat/snakemake-orchestration only, and the 34-tile smk-g4 campaign is - the first run under them. The other seven stand, including the - mccd_plots docstring that still advertises rho statistics the package - no longer computes. - created_at: "2026-08-29T00:00:00Z" - derived: true - evidence: - - id: ev_final_cat_defects - artifact: final_cat - code_paper_divergence: - claim: >- - The two ShapePipe papers state roughly 17 of this record's 50 decisions - (now carried as prior_insights with verbatim quotes), have drifted from - the code on 10 of them since publication, and are silent on the rest. - Among them: DETECT_MINAREA 10 -> 5; DEBLEND_MINCONT 0.001 -> - 0.0005 on tiles; tile background AUTO -> MANUAL 0; in-line spread-model - star/galaxy classification -> disabled and deferred downstream; HSM - moment initialisation -> WCS centroids and prior-based guesses; GSC 2.2 - via cdsclient -> GSC 2.3 via astroquery; PSF acceptance 22 stars/CCD - published for the science path vs 20 committed there — closed since by - #873 + 90782098, which put the science path on 22, so nine of the ten - drifts remain open on the orchestration branch. - created_at: "2026-08-29T00:00:00Z" - derived: true - evidence: - - id: ev_final_cat_divergence - artifact: final_cat - analyses: # ═════════════════════════════════════════════════════════════════════════ masking: description: >- - Which pixels are excluded before anything is measured. Modules: - src/shapepipe/modules/mask_package/mask.py (halo/spike/DSO/border - builders, WeightWatcher driver), scripts/python/create_star_cat.py - (star-catalogue fetch), configs config_exp_Ma.ini + - config_onthefly.mask / config_tile_onthefly.mask + mask_default/. - [LINT] MESSIER_PIXEL_SCALE is set in config_tile_onthefly.mask but - never read — mask_dso takes pixel scale from the WCS. [LINT] - _build_final_mask combines mask planes by ADDITION (mask.py:1141+), - not bitwise OR; the committed flag values (2/4/16/32/128) are disjoint - so no live collision exists, but any future duplicate value corrupts - the flag semantics silently. (FLAG_OUTFLAGS 2 in default.ww is inert: - no input flag image is passed to WeightWatcher — mask.py:1047-1074.) + Which masks reach the measurement, and where. ShapePipe generates no + masks. The instrument flag image delivered with each exposure is the + only mask that reaches pixels. Sky-fixed masks (star halos and bodies, + manual regions, missing bands) are healsparse maps built outside + ShapePipe; their geometry is decided there. Inside ShapePipe they are + only queried at object positions into catalogue columns, and no stage + cuts on those columns. inputs: - - id: ccd_images + - id: exposure_flags type: data - source: split per-CCD exposure images + weights + CFIS flag maps (exp_split family) - - id: star_catalogue + source: per-CCD instrument flag images split from each exposure (exp_split family) + - id: sky_masks type: data - source: GSC 2.3 per-exposure catalogues (exp_star_cat cache) + source: UNIONS healsparse mask maps outputs: - - id: exposure_mask + - id: masked_measurement_inputs type: data format: fits - description: Per-CCD pipeline flag maps (run_sp_exp_Ma family). - decisions: - [star_catalogue_query, star_magnitude_definition, - bright_star_mask_geometry, deep_sky_object_masking, - border_mask_width, pixel_threshold_flags, external_flag_usage] + description: >- + Exposure detection catalogues carrying IMAFLAGS_ISO (and MASK_EXT when + maps are configured), the flag stamps ngmix reads, and the final + catalogue's optional MASK_ columns. + inputs: [exposure_flags, sky_masks] + decisions: [pixel_mask_source, psf_star_mask_veto, sky_mask_application] decisions: - star_catalogue_query: - label: Reference star catalogue and query for bright-star masking + pixel_mask_source: + label: Only the instrument flag image reaches pixels rationale: >- - GSC 2.3 (Vizier I/305/out), columns GSC2.3/RAJ2000/DEJ2000/Fmag/ - jmag/Vmag/Nmag/Class, cone radius covering the full CCD mosaic, no - magnitude cut at query time. GSC 2.2 rejected in a code comment - beside the catalogue ID ("does not have Fmag"). - Provenance hazard: the star-cat cache is not keyed by script - version — a semantic change to this query reruns the rule but takes - the skip-if-exists branch; clear the cache by hand for the change - to reach the data (workflow/config.yaml star_cats comment). - Query geometry: search radius = half the image diagonal about the - field centre (Mask._get_image_radius); source precedence: with - CDSCLIENT_PATH set in the .mask configs the online-query branch - wins unless an external star cat is passed (USE_EXT_STAR=True in - config_exp_Ma.ini routes the exp_star_cat cache in). - Published description (Farrens+22 p.2): cdsclient downloads GSC 2.2 - (with cdsclient 3.84 pinned in its Table A.1); current code: GSC 2.3 - queried through astroquery — two things drifted, the catalogue - version (Fmag is needed for the magnitude cut) and the query - transport, since cdsclient is never invoked yet survives as a - required-but-unused CDSCLIENT_PATH still set to the stale - findgsc2.2 in config_tile_onthefly.mask. - Anchor: scripts/python/create_star_cat.py::CDS_CAT_ID; - src/shapepipe/modules/mask_package/mask.py::Mask._CDS_cat_ID; - src/shapepipe/modules/mask_package/mask.py::Mask._cds_keys; - workflow/config.yaml. - default: gsc_23_vizier + On exposures SExtractor reads the split flag image (FLAG_IMAGE=True), + producing IMAFLAGS_ISO, which the PSF star selection requires to be + zero. The multi-epoch vignet run cuts flag stamps from the same split + flag image, and ngmix gives weight 0 to every flagged pixel and drops + epochs that are mostly flagged (shape_measurement.defect_fill, + shape_measurement.epoch_masked_fraction_cut). Tiles have no flag + image, so tile detection runs unflagged + (detection.detection_source_mode). Sky-fixed masks never touch + pixels: an object inside a star halo is measured from the same + pixels as one outside it. + Anchor: workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE; + workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_PATTERN; + src/shapepipe/modules/ngmix_package/ngmix.py::prepare_ngmix_weights. + default: instrument_flags_only options: - gsc_23_vizier: - label: GSC 2.3 cone queries, all bands, no query-time mag cut - insights: [farrens22_star_cat_on_disk] - gaia: - label: Gaia-based star catalogue + instrument_flags_only: + label: Instrument flags gate pixels; sky masks stay at catalogue level + rasterised_sky_masks: + label: Rasterise healsparse masks into the pixel flags + insights: [farrens22_pipeline_masks] excluded: true excluded_reason: >- - Not wired. Deeper and better photometry; switching changes mask - geometry and hence the selection function — a real DR-level fork. - star_magnitude_definition: - label: Per-star magnitude for mask scaling + Sky-fixed masks say where an object sits, not that its pixels are + corrupted, so what to do about them is an analysis decision. + Rasterising them would bake one mask version into every shape; + catalogue columns leave the choice downstream. + psf_star_mask_veto: + label: PSF-star candidates rejected on instrument flags only rationale: >- - mag = unweighted mean of the finite GSC bands among F, j, V, N; - stars with no finite band are logged and not masked; only Class==0 - objects masked. Comment records why not a naive mean: NaN bands - would NaN-poison the mag < mag_limit test, leaving exactly the - bright stars with incomplete photometry unmasked. - Anchor: src/shapepipe/modules/mask_package/mask.py::Mask._create_mask. - default: mean_finite_bands + The star selection cuts IMAFLAGS_ISO == 0 and nothing else from the + masks. mask_query sits in the exposure module chain between + SExtractor and setools: when MASK_PATHS names maps it writes MASK_EXT + (0 clean, nonzero flagged; off-coverage counts as clean) onto each + CCD's catalogue. MASK_PATHS ships commented out, so the committed + module passes the catalogue through with no MASK_EXT column. The + intended map is the UNIONS star-body product (bit 2); halo bits 0 + and 1 are excluded because halos say nothing about whether a star is + a good PSF sample. Imposing the veto is one line per mask block in + star_selection.setools (MASK_EXT == 0). [LINT] the + workflow/rules/exposure.smk docstring names the column FLAG_EXT; the + code writes MASK_EXT. + Anchor: workflow/config/cfis/config_exp_psfex.ini; + workflow/config/cfis/star_selection.setools#MASK:star_selection.IMAFLAGS_ISO; + src/shapepipe/modules/mask_query_runner.py::mask_query_runner; + src/shapepipe/utilities/mask_query.py::flag_positions. + default: instrument_flags_only options: - mean_finite_bands: { label: Mean of finite F/j/V/N; Class==0 only } - single_band: - label: Single-band (Fmag) magnitude - excluded: true - excluded_reason: Drops stars with missing Fmag from masking entirely. - bright_star_mask_geometry: - label: Halo + diffraction-spike mask geometry and magnitude scaling - rationale: >- - DS9 polygon templates scaled linearly with magnitude about a pivot: - halo HALO_MAG_LIM=13, HALO_SCALE_FACTOR=0.05, HALO_MAG_PIVOT=13.8 - (halo_mask.reg, ~270 px); spike SPIKE_MAG_LIM=18, - SPIKE_SCALE_FACTOR=0.3, SPIKE_MAG_PIVOT=13.8 - (MEGAPRIME_star_i_13.8.reg); scaling = 1 - factor*(mag-pivot), - floored at 0.1 by Mask._scaling_min. - Identical in exposure and tile configs. Template filename encodes - provenance (MegaPrime i-band mag-13.8 star); numeric rationale not - recorded. Note the 5-mag gap: stars in 13-18 get spikes but no halo. - Anchor: workflow/config/cfis/config_onthefly.mask#HALO_PARAMETERS.HALO_MAG_LIM; - workflow/config/cfis/config_onthefly.mask#SPIKE_PARAMETERS.SPIKE_MAG_LIM; - workflow/config/cfis/mask_default/halo_mask.reg; - workflow/config/cfis/mask_default/MEGAPRIME_star_i_13.8.reg; - src/shapepipe/modules/mask_package/mask.py::Mask._create_mask; - src/shapepipe/modules/mask_package/mask.py::Mask._scaling_min. - default: megaprime_polygon_linear_scaling - options: - megaprime_polygon_linear_scaling: - label: Fixed MegaPrime templates, linear mag scaling, floor 0.1 - radial_profile_fit: - label: Per-star radial-profile-driven mask size - excluded: true - excluded_reason: Not wired; the survey precedent is template-based. - deep_sky_object_masking: - label: Messier + NGC objects masked as circles, no enlargement - rationale: >- - Circles of radius max(size_X, size_Y), MESSIER_SIZE_PLUS=0, - NGC_SIZE_PLUS=0 (function default is 0.1 — the 0 is a choice); - flags 16/32. A comment records the overlap-test fix (corner-only - test missed small interior objects). - Anchor: workflow/config/cfis/config_onthefly.mask#MESSIER_PARAMETERS.MESSIER_SIZE_PLUS; - workflow/config/cfis/config_tile_onthefly.mask#NGC_PARAMETERS.NGC_SIZE_PLUS; - src/shapepipe/modules/mask_package/mask.py::Mask.mask_dso. - default: circles_no_padding - options: - circles_no_padding: - label: "size_plus = 0: mask exactly the catalogued extent" - insights: [farrens22_messier_mask] - padded_circles: - label: size_plus > 0 (code default 0.1) + instrument_flags_only: + label: IMAFLAGS_ISO == 0; no sky map queried + star_body_veto: + label: Also reject candidates on the star-body map (MASK_EXT == 0) + description: >- + Set MASK_PATHS to the star-body map and add MASK_EXT == 0 beside + each IMAFLAGS_ISO cut. Querying without the cut records MASK_EXT + and changes no star. + star_body_and_halo_veto: + label: Also reject candidates inside star halos excluded: true excluded_reason: >- - Rationale for dropping the padding not recorded; flagged as a - question rather than an endorsed exclusion. - border_mask_width: - label: CCD border mask, 50 px on exposures, none on tiles - rationale: >- - Exposures BORDER_WIDTH=50 (flag 4); tiles BORDER_MAKE=False. - Mask.mask_border's own default is 100 — the committed 50 is a - choice, unrecorded. Trims CCD edges where PSF and astrometry - degrade; changes the effective footprint. - Anchor: workflow/config/cfis/config_onthefly.mask#BORDER_PARAMETERS.BORDER_WIDTH; - workflow/config/cfis/config_tile_onthefly.mask#BORDER_PARAMETERS.BORDER_MAKE; - src/shapepipe/modules/mask_package/mask.py::Mask.mask_border. - default: px50_exposures_only - options: - px50_exposures_only: - label: 50 px exposure borders; tiles unmasked - insights: [farrens22_border_mask] - px100: - label: 100 px (module default) - excluded: true - excluded_reason: Halves usable edge area for no recorded gain. - pixel_threshold_flags: - label: WeightWatcher weight/flag thresholds into mask bits - rationale: >- - WEIGHT_MIN 0, WEIGHT_MAX 1000, WEIGHT_OUTFLAGS 1; FLAG_MASKS 0x01, - FLAG_OUTFLAGS 2; POLY_OUTWEIGHTS 0. Zero-weight and externally - flagged pixels excluded on these thresholds. Values are stock, not - derived from the CFIS weight distribution; rationale not recorded. - The FLAG_* keys are inert in the committed invocation — no flag - image is passed to WeightWatcher by Mask._exec_WW. - Anchor: workflow/config/cfis/mask_default/default.ww#WEIGHT_MIN; - workflow/config/cfis/mask_default/default.ww#FLAG_MASKS; - src/shapepipe/modules/mask_package/mask.py::Mask._exec_WW. - default: stock_ww_thresholds - options: - stock_ww_thresholds: - label: Stock WeightWatcher thresholds - insights: [farrens22_weightwatcher] - external_flag_usage: - label: CFIS external flag maps folded into exposure masks + Halos flag objects for the final catalogue; a star inside another + star's halo is not thereby a bad PSF sample. + sky_mask_application: + label: Object-level sky masking deferred downstream rationale: >- - USE_EXT_FLAG=True on exposures (imports CADC-provided bad-pixel / - cosmic-ray / trail flags); EF_MAKE=False on tiles. The external - plane enters via Mask._build_final_mask's path_external_flag branch. - Anchor: workflow/config/cfis/config_exp_Ma.ini#MASK_RUNNER.USE_EXT_FLAG; - workflow/config/cfis/config_tile_onthefly.mask#EXTERNAL_FLAG.EF_MAKE; - src/shapepipe/modules/mask_package/mask.py::Mask._build_final_mask. - default: exposures_only + The final catalogue ships every detected object. When MASK_EXT_PATHS + lists band:path pairs, make_cat queries each healsparse map at the + object's windowed position and writes one MASK_ column holding + the map value verbatim; an object off a map's coverage gets that + map's sentinel (False for boolean maps, which reads as unmasked; + typically -1 for integer maps). The committed make_cat config sets no + MASK_EXT_PATHS, so no mask column is written and every mask cut + happens downstream against the maps themselves. + Anchor: src/shapepipe/modules/make_cat_runner.py::make_cat_runner; + src/shapepipe/modules/make_cat_package/make_cat.py::save_mask_ext_data; + src/shapepipe/utilities/mask_query.py::query_map. + default: deferred_downstream options: - exposures_only: { label: "External flags on exposures, not tiles" } - ignore_external: - label: Pipeline-generated masks only + deferred_downstream: + label: No mask columns; all objects shipped + catalogue_columns: + label: Per-band MASK_ columns from MASK_EXT_PATHS, no cut + pipeline_cut: + label: Drop masked objects inside the pipeline excluded: true - excluded_reason: Discards upstream knowledge of bad pixels. + excluded_reason: >- + Location flags are analysis decisions; a pipeline cut would fix + one mask version into the catalogue. prior_insights: - farrens22_star_cat_on_disk: + farrens22_pipeline_masks: claim: >- - The ShapePipe release paper documents an on-disk star catalogue, in - GSC format, as a supported substitute for the online query, - motivated by compute nodes without internet access. + The published ShapePipe pipeline generated its own masks, including + Messier objects and CCD borders, and applied them to the images; the + current pipeline generates none. created_at: "2022-06-01T00:00:00Z" evidence: - - id: ev_farrens22_star_cat_disk - doi: "10.48550/arXiv.2206.14689" - quote: - exact: 'Alternatively, a star catalogue available on disk (with the same format as the GSC) can also be used' - location: { page: 2 } - farrens22_messier_mask: - claim: >- - Messier objects are named in the published masking procedure as one - of the object classes ShapePipe masks. - created_at: "2022-06-01T00:00:00Z" - evidence: - - id: ev_farrens22_messier + - id: ev_farrens22_masks doi: "10.48550/arXiv.2206.14689" quote: exact: 'Messier objects, and border regions.' location: { page: 2 } - farrens22_border_mask: - claim: >- - CCD border regions are named in the published masking procedure as - one of the regions ShapePipe masks. - created_at: "2022-06-01T00:00:00Z" - evidence: - - id: ev_farrens22_border - doi: "10.48550/arXiv.2206.14689" - quote: - exact: 'Messier objects, and border regions.' - location: { page: 2 } - farrens22_weightwatcher: - claim: >- - The published pipeline generates the mask image itself with - WeightWatcher (Marmo & Bertin 2008), fixing the tool but none of its - threshold values. - created_at: "2022-06-01T00:00:00Z" - evidence: - - id: ev_farrens22_ww - doi: "10.48550/arXiv.2206.14689" - location: { page: 2 } # ═════════════════════════════════════════════════════════════════════════ detection: description: >- - Object detection on r-band tiles (single-image mode) and exposures (for - star finding). Module: src/shapepipe/modules/sextractor_package/ - sextractor_script.py (config assembly, ZP/background overrides, - post-processing that assigns per-epoch CCD membership). Configs: - config_tile_Sx.ini + default_tile.sex + default.conv + - default_noimaflags.param (tiles); default_exp.sex (exposures — same - thresholds, but DEBLEND_MINCONT 0.001 vs tile 0.0005 and BACK_TYPE AUTO - vs tile MANUAL 0, both deliberate and unexplained divergences). - [LINT] final_cat.param (consumed by the post-proc merge_final_cat, - not by make_cat) requests IMAFLAGS_ISO, but the tile chain never - produces it (FLAG_IMAGE=False, default_noimaflags.param); the - exposure-side IMAFLAGS_ISO stays exposure-side (merge_starcat.py:807 - only). The merged catalogue never receives the column. + Object detection with SExtractor on r-band tiles (the galaxy sample) and + on single-exposure CCDs (PSF-star candidates). Tiles follow the MegaPipe + parameters of Gwyn's UNIONS tile catalogue; exposures keep ShapePipe's + stock values. inputs: - id: tile_stack type: data @@ -561,194 +320,204 @@ analyses: type: data format: fits description: Per-tile SExtractor LDAC catalogue with per-epoch CCD membership. + inputs: [tile_stack] decisions: [detection_threshold_policy, deblending_policy, background_model, - weighting_and_interpolation, detection_source_mode, + weight_map_usage, zero_weight_interpolation, detection_source_mode, epoch_membership_ccd_bounds, photometry_parameters, - cleaning_and_neighbour_masking] + spurious_detection_cleaning, blend_photometry_mask_type] decisions: - photometry_parameters: - label: Photometric aperture definitions — Kron parameters, apertures, half-light fraction - rationale: >- - PHOT_AUTOPARAMS 2.5,3.5 (Kron factor / minimum radius), - PHOT_APERTURES 5 px, PHOT_FLUXFRAC 0.5, BACKPHOTO_TYPE GLOBAL — - identical in both .sex files. MAG_AUTO is the axis of the - star-selection magnitude box AND the catalogue magnitude; FLUX_AUTO - is PSFEx's photometric normalisation (default.psfex PHOTFLUX_KEY). - A different Kron factor shifts magnitudes systematically, moving - which stars build the PSF model and every magnitude-based - downstream cut. Rationale not recorded (stock values). Anchor: - workflow/config/cfis/default_tile.sex#PHOT_AUTOPARAMS; - workflow/config/cfis/default_exp.sex#PHOT_AUTOPARAMS; - workflow/config/cfis/default.psfex#PHOTFLUX_KEY. - default: kron_25_35 - options: - kron_25_35: { label: "Kron 2.5/3.5, aperture 5 px, FLUXFRAC 0.5, global background" } - cleaning_and_neighbour_masking: - label: Spurious-detection cleaning and neighbour-pixel correction - rationale: >- - CLEAN Y with CLEAN_PARAM 1.0 deletes detections consistent with - being wings of a brighter neighbour — a post-deblend change to the - object list; MASK_TYPE CORRECT replaces neighbour pixels during - photometry (vs BLANK/NONE), changing fluxes and windowed moments - of blends. Identical in both .sex files; rationale not recorded. - Anchor: workflow/config/cfis/default_tile.sex#CLEAN; - workflow/config/cfis/default_tile.sex#MASK_TYPE; - workflow/config/cfis/default_exp.sex#CLEAN. - default: clean_1_correct - options: - clean_1_correct: { label: "CLEAN 1.0 + MASK_TYPE CORRECT" } detection_threshold_policy: label: Detection significance, minimum area, matched filter rationale: >- - DETECT_THRESH 1.5 sigma RELATIVE, ANALYSIS_THRESH 1.5, - DETECT_MINAREA 5, FILTER default.conv (3x3 pyramid kernel, "all - ground, FWHM = 2 pixels" — vs CFIS seeing ~0.65 arcsec = 3.5 px at - 0.187"/px, so the filter is not matched to the survey PSF). - Sets the faint end of the source sample. Rationale not recorded - (stock EB 2017 header). - Published description (Guinot+22 p.5, Table 2): DETECT_MINAREA 10; - current code: 5, in default_exp.sex as well as default_tile.sex — - the small-object end has been loosened since publication on both the - star-detection and tile-detection passes, while DETECT_THRESH 1.5 - RELATIVE and the 3x3 FWHM=2 px kernel still match. + Tiles: DETECT_THRESH and ANALYSIS_THRESH 1.0 sigma, DETECT_MINAREA + 3, filtered with a 7x7 Gaussian of FWHM 3 px (gauss_3.0_7x7.conv, + close to the ~3.5 px CFIS seeing). These are the parameters of the + MegaPipe tile catalogue, so ShapePipe's galaxy sample matches the + catalogue UNIONS adopts. Exposures: 1.5 sigma, minarea 5, the 3x3 + FWHM 2 px kernel (default.conv); they only feed star selection. + Guinot+22 Table 2 lists minarea 10 at 1.5 sigma with the FWHM 2 px + kernel, so the tiles differ from the paper on all three. Anchor: workflow/config/cfis/default_tile.sex#DETECT_THRESH; workflow/config/cfis/default_tile.sex#DETECT_MINAREA; - workflow/config/cfis/default.conv. - default: thresh_1p5_minarea5_fwhm2px_filter + workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE; + workflow/config/cfis/gauss_3.0_7x7.conv; + workflow/config/cfis/default_exp.sex#DETECT_THRESH. + default: megapipe_tiles options: - thresh_1p5_minarea5_fwhm2px_filter: - label: 1.5 sigma, minarea 5, FWHM=2px kernel - seeing_matched_filter: - label: Kernel matched to CFIS seeing (~3.5 px) + megapipe_tiles: + label: MegaPipe values on tiles; stock values on exposures + stock_tiles: + label: Stock ShapePipe values on tiles (1.5 sigma, minarea 5, FWHM 2 px) excluded: true excluded_reason: >- - Not wired; would change depth and the faint-end selection - function — a real fork, excluded only as not-the-committed-path. + Triggers spuriously on about 10% of grid-placed Sersic galaxies + in image simulations, and does not match the MegaPipe tile + catalogue. deblending_policy: - label: Deblending sub-thresholds and contrast + label: Deblending contrast rationale: >- - DEBLEND_NTHRESH 32, DEBLEND_MINCONT 0.0005 on tiles (2x more - aggressive splitting than the exposure 0.001 and 10x more than the - SExtractor default 0.005 — divergences not documented), CLEAN Y - PARAM 1.0. Controls object count, centroids, and blend - contamination in shapes. - Published description (Guinot+22 p.5, Table 2, galaxy detection on - the stacked tiles): DEBLEND_MINCONT 0.001; current code: 0.0005 on - tiles, with only the exposure side still carrying 0.001 — the - divergence lands on precisely the configuration the paper documents. - NTHRESH 32 matches. + DEBLEND_NTHRESH 32 on both passes; DEBLEND_MINCONT 0.002 on tiles + (the MegaPipe value) and 0.001 on exposures. Contrast sets object + count, centroids, and blend contamination in shapes. Guinot+22 Table + 2 lists 0.001 for tile detection. Anchor: workflow/config/cfis/default_tile.sex#DEBLEND_MINCONT; workflow/config/cfis/default_exp.sex#DEBLEND_MINCONT. - default: mincont_5em4_tiles + default: megapipe_tiles options: - mincont_5em4_tiles: { label: "MINCONT 0.0005 tiles / 0.001 exposures" } + megapipe_tiles: + label: MINCONT 0.002 tiles / 0.001 exposures + mincont_5em4_tiles: + label: MINCONT 0.0005 on tiles + excluded: true + excluded_reason: >- + Part of the stock tile parameter set rejected in favour of the + MegaPipe values (see detection_threshold_policy). background_model: - label: Tile background fixed to zero, not estimated + label: Background estimation and photometric background rationale: >- - BACK_TYPE MANUAL, BACK_VALUE 0.0, BKG_FROM_HEADER=False on tiles — - trusts MegaPipe stack background removal; exposures use BACK_TYPE - AUTO (64/3 mesh). Residual sky offsets propagate into thresholds, - fluxes, completeness. Divergence deliberate, unexplained. - Published description (Guinot+22 p.5): Table 2's caption asserts all - non-tabulated SExtractor parameters keep their defaults, i.e. - BACK_TYPE AUTO, and the paper never mentions the background choice - at all; current code: BACK_TYPE MANUAL with BACK_VALUE 0.0 on tiles, - identically in workflow/ and example/ — the standing tile - configuration, not a one-off, diverging silently from the published - parametrisation. - Anchor: workflow/config/cfis/default_tile.sex#BACK_TYPE; - workflow/config/cfis/default_exp.sex#BACK_TYPE; + Both passes estimate the background with SExtractor AUTO. Tiles use + the MegaPipe mesh BACK_SIZE 512 with BACK_FILTERSIZE 9 and a LOCAL + photometric background (annulus BACKPHOTO_THICK 30); exposures use + mesh 64, filter 3 and a GLOBAL photometric background. The exposure + BACKGROUND and BACKGROUND_RMS maps are also what ngmix subtracts from + each epoch and weights its pixels by + (shape_measurement.galaxy_pixel_weights). The header background path + is off (BKG_FROM_HEADER=False). Residual sky offsets propagate into + thresholds, fluxes, completeness and shapes. + Anchor: workflow/config/cfis/default_tile.sex#BACK_SIZE; + workflow/config/cfis/default_tile.sex#BACKPHOTO_TYPE; + workflow/config/cfis/default_exp.sex#BACK_SIZE; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.BKG_FROM_HEADER; src/shapepipe/modules/sextractor_package/sextractor_script.py::SExtractorCaller.get_background. - default: manual_zero_tiles_auto_exposures + default: auto_megapipe_tiles options: - manual_zero_tiles_auto_exposures: - label: Tiles trust the stack (0.0); exposures estimate - auto_everywhere: - label: SExtractor AUTO background on tiles too + auto_megapipe_tiles: + label: AUTO everywhere; MegaPipe mesh and LOCAL photometry on tiles + manual_zero_tiles: + label: Tile background fixed to 0, trusting the stack subtraction excluded: true excluded_reason: >- - Double-subtracts if MegaPipe already removed it; if MegaPipe - residuals are nonzero this exclusion is wrong — verify. - weighting_and_interpolation: - label: Weight-map usage and zero-weight pixel interpolation + Part of the stock tile parameter set rejected in favour of the + MegaPipe values (see detection_threshold_policy). + weight_map_usage: + label: Weight map as inverse variance for detection rationale: >- - Two settings depart from stock SExtractor: WEIGHT_TYPE MAP_WEIGHT - (default NONE) and INTERP_TYPE ALL (default NONE — SExtractor - invents flux across zero-weight pixels). The accompanying - RESCALE_WEIGHTS Y, WEIGHT_GAIN Y, MASK_TYPE CORRECT and - INTERP_MAXXLAG/INTERP_MAXYLAG 16 are the SExtractor defaults, so - they are settings the configs restate rather than choices. The - variance policy sets effective per-pixel SNR and thus the detection - set; INTERP_TYPE ALL alters pixel data feeding measurements. - Rationale not recorded. - Published description (Guinot+22 p.5): Table 2's "all other - parameters are kept to their default values" silently covers both - non-default settings; current code: MAP_WEIGHT + INTERP_TYPE ALL in - default_tile.sex and default_exp.sex alike — the paper gives no hint - that the weight map or the zero-weight interpolation is in play. + WEIGHT_TYPE MAP_WEIGHT on both passes (SExtractor default NONE): the + per-pixel variance sets the effective SNR and so the detection set. + RESCALE_WEIGHTS and WEIGHT_GAIN are SExtractor defaults. Guinot+22 + says all non-tabulated parameters keep their defaults, which would + mean no weight map. Anchor: workflow/config/cfis/default_tile.sex#WEIGHT_TYPE; - workflow/config/cfis/default_tile.sex#INTERP_TYPE; + workflow/config/cfis/default_exp.sex#WEIGHT_TYPE; src/shapepipe/modules/sextractor_package/sextractor_script.py::SExtractorCaller.set_input_files. - default: map_weight_interp_all + default: map_weight + options: + map_weight: + label: MAP_WEIGHT + no_weight: + label: No weight map (SExtractor default) + zero_weight_interpolation: + label: Interpolation across zero-weight pixels + rationale: >- + INTERP_TYPE ALL on both passes (SExtractor default NONE), with + INTERP_MAXXLAG/INTERP_MAXYLAG 16: SExtractor invents flux across + zero-weight pixels, which changes detections and photometry near + masked regions. No rationale is recorded. + Anchor: workflow/config/cfis/default_tile.sex#INTERP_TYPE; + workflow/config/cfis/default_exp.sex#INTERP_TYPE. + default: interp_all options: - map_weight_interp_all: { label: MAP_WEIGHT + INTERP ALL + MASK CORRECT } + interp_all: + label: INTERP_TYPE ALL no_interpolation: label: INTERP_TYPE NONE - excluded: true - excluded_reason: >- - Changes photometry near masks; the committed choice is itself - unjustified in code — flagged as a question, not an endorsement. + spurious_detection_cleaning: + label: Cleaning of spurious detections + rationale: >- + CLEAN Y with CLEAN_PARAM 1.0 on both passes deletes detections + consistent with being wings of a brighter neighbour, a post-deblend + change to the object list. Stock value; no rationale recorded. + Anchor: workflow/config/cfis/default_tile.sex#CLEAN_PARAM; + workflow/config/cfis/default_exp.sex#CLEAN_PARAM. + default: clean_1 + options: + clean_1: + label: CLEAN Y, CLEAN_PARAM 1.0 + blend_photometry_mask_type: + label: Neighbour pixels in blend photometry + rationale: >- + MASK_TYPE CORRECT on both passes replaces pixels belonging to a + neighbour by their mirror across the object centre during + photometry, changing fluxes and windowed moments of blends. Stock + value; no rationale recorded. + Anchor: workflow/config/cfis/default_tile.sex#MASK_TYPE; + workflow/config/cfis/default_exp.sex#MASK_TYPE. + default: correct + options: + correct: + label: MASK_TYPE CORRECT + blank: + label: MASK_TYPE BLANK + photometry_parameters: + label: Kron and aperture photometry definitions + rationale: >- + PHOT_AUTOPARAMS 2.5,3.5 (Kron factor, minimum radius), PHOT_APERTURES + 5 px and PHOT_FLUXFRAC 0.5 on both passes. MAG_AUTO is the axis of + the star-selection magnitude window and the catalogue magnitude; + FLUX_AUTO is PSFEx's photometric normalisation. A different Kron + factor shifts magnitudes and so every magnitude-based cut. Stock + values; no rationale recorded. + Anchor: workflow/config/cfis/default_tile.sex#PHOT_AUTOPARAMS; + workflow/config/cfis/default_exp.sex#PHOT_AUTOPARAMS; + workflow/config/cfis/default.psfex#PHOTFLUX_KEY. + default: kron_25_35 + options: + kron_25_35: + label: Kron 2.5/3.5, aperture 5 px, FLUXFRAC 0.5 detection_source_mode: - label: Single-image detection on the r-band tile + label: Single-image, unflagged detection on the r-band tile rationale: >- - DETECTION_IMAGE=False, FLAG_IMAGE=False at detection, - param file default_noimaflags.param. No dual-image mode, no - detection coadd, no flag propagation at detection time. The - sx_nomask variant is the committed chain because it matches the - validated bash baseline. STATUS: DELIBERATELY UNDECIDED (Cail, - 2026-08-29) — whether DR6 detects masked or unmasked is punted to - the planned masking-unification rework (not yet tracked in an issue); - the default records baseline - parity, not a settled methodological choice. The masked variant is - one config + one rule + a tile-side star-cat analogue away. + DETECTION_IMAGE=False and FLAG_IMAGE=False with the + default_noimaflags.param column list: tiles have no instrument flag + image and no detection coadd exists, so detection sees every tile + pixel and the tile catalogue carries no IMAFLAGS_ISO. [LINT] + final_cat.param, read by the post-processing merge, requests + IMAFLAGS_ISO, which the tile chain never produces. Anchor: workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.DETECTION_IMAGE; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE; workflow/config/cfis/default_noimaflags.param; - workflow/rules/tile.smk. + workflow/config/cfis/final_cat.param#IMAFLAGS_ISO. default: sx_nomask_single_image options: sx_nomask_single_image: - label: Unmasked single-image r-band detection + label: Unflagged single-image r-band detection insights: [guinot22_stacked_detection] - sx_masked: - label: Detection on the masked tile + masked_tile: + label: Detection on a tile masked by rasterised sky masks + description: >- + Not implemented; no tile pixel mask exists (see + masking.pixel_mask_source). dual_image_coadd: label: Dual-image mode with a detection coadd - excluded: true - excluded_reason: No detection coadd exists in UNIONS r-band processing. + description: Not implemented; no UNIONS detection coadd exists. epoch_membership_ccd_bounds: label: Which exposure CCDs an object belongs to (N_EPOCH) rationale: >- - CCD_SIZE = 33,2080,1,4612 with strict inequalities — the 33-px left - trim silently discards a CCD strip from epoch membership; WCS - inversion failures skip the CCD ("no epoch recorded"), changing - N_EPOCH. Sets how many exposures contribute to each galaxy's - multi-epoch fit. Rationale beyond "number of pixels in a CCD" not - recorded. + CCD_SIZE = 33,2080,1,4612 with strict inequalities: the 33-px left + trim removes a CCD strip from epoch membership, and a WCS inversion + failure skips the CCD, lowering N_EPOCH. This sets how many exposures + enter each galaxy's multi-epoch fit. The trim is unexplained beyond + "number of pixels in a CCD". Anchor: workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.CCD_SIZE; src/shapepipe/modules/sextractor_package/sextractor_script.py::make_post_process; src/shapepipe/modules/sextractor_package/sextractor_script.py::ccd_candidate_mask. default: trimmed_bounds_33_2080 options: - trimmed_bounds_33_2080: { label: "x in (33,2080), y in (1,4612), strict" } + trimmed_bounds_33_2080: + label: "x in (33,2080), y in (1,4612), strict" full_ccd: label: Full 1-2048 x-range, inclusive bounds - excluded: true - excluded_reason: >- - The trim presumably excludes a bad edge region, but nothing in - code says so — flagged as a question. prior_insights: guinot22_stacked_detection: claim: >- @@ -766,12 +535,8 @@ analyses: # ═════════════════════════════════════════════════════════════════════════ preparation: description: >- - How pixels, WCS, and epoch membership are prepared before anything is - measured. Modules: - src/shapepipe/modules/split_exp_package/split_exp.py, - merge_headers_package/merge_headers.py, - find_exposures_package/find_exposures.py, - vignetmaker_package/vignetmaker.py. + How exposures, astrometry, epoch lists and stamps are prepared before + anything is measured. inputs: - id: exposure_files type: data @@ -779,120 +544,95 @@ analyses: outputs: - id: epoch_stamps type: data - format: fits + format: sqlite description: Per-object multi-epoch vignets + per-CCD WCS log feeding ngmix. + inputs: [exposure_files] decisions: [astrometric_solution_source, ccd_split_extent, epoch_provenance_from_tile_history, object_position_columns, - stamp_positioning_and_padding, epoch_flag_source] + stamp_positioning_and_padding] decisions: astrometric_solution_source: label: Astrometry taken verbatim from delivered per-CCD headers rationale: >- - split_exp builds WCS(h) from each raw CCD header at split time, - pickles it, and merge_headers writes the lot into - log_exp_headers.sqlite; every downstream world<->pixel transform - (stamp positioning, epoch membership, position seeding) uses that - stored solution. No re-derivation, no astrometric refinement — the - survey's delivered astrometry IS the pipeline's astrometry. - Alternative (a joint astrometric re-fit a la DES/Rubin) would move - every stamp centre and every position seed. Anchor: - src/shapepipe/modules/split_exp_package/split_exp.py::SplitExposures.create_hdus; - src/shapepipe/modules/merge_headers_package/merge_headers.py::merge_headers; - src/shapepipe/modules/vignetmaker_package/vignetmaker.py::VignetMaker._get_stamp_me. + split_exp builds WCS(header) from each raw CCD header, and + merge_headers stores the lot; every downstream world-to-pixel + transform (stamp placement, epoch membership, position seeding) uses + that solution. [HARDCODED] no re-derivation or astrometric refinement + exists. A joint re-fit would move every stamp centre and position + seed. + Anchor: src/shapepipe/modules/split_exp_package/split_exp.py::SplitExposures.create_hdus; + src/shapepipe/modules/merge_headers_package/merge_headers.py::merge_headers. default: delivered_headers options: delivered_headers: - label: "WCS(header) verbatim, stored at split time" + label: WCS(header) verbatim, stored at split time insights: [guinot22_gaia_astrometry] astrometric_refit: label: Joint astrometric re-solution - excluded: true - excluded_reason: Not wired; CFIS delivered astrometry is trusted. + description: Not implemented. ccd_split_extent: label: All 40 MegaCam HDUs split and carried as candidate epochs rationale: >- - N_HDU=40 with a hard check (any other HDU count raises) — every - CCD including the ear CCDs 36-39 is a candidate epoch wherever the - WCS lands it. The MegaCamFlip special-casing of 36/37 shows the - ears flow through shape measurement. Alternative: exclude ear CCDs - (different optical path/orientation history). Anchor: - workflow/config/cfis/config_exp_Sp.ini#SPLIT_EXP_RUNNER.N_HDU; + N_HDU=40, and any other HDU count raises: every CCD including the + ear CCDs 36-39 is a candidate epoch wherever the WCS lands it. + Excluding the ear CCDs (a different optical path) is the alternative. + Anchor: workflow/config/cfis/config_exp_Sp.ini#SPLIT_EXP_RUNNER.N_HDU; src/shapepipe/modules/split_exp_package/split_exp.py::SplitExposures.create_hdus. default: all_40_hdus options: all_40_hdus: - label: "40 HDUs, hard-fail on any other count" + label: 40 HDUs, hard-fail on any other count insights: [guinot22_forty_chips] + exclude_ear_ccds: + label: Drop CCDs 36-39 as epochs + description: Not implemented. epoch_provenance_from_tile_history: label: Epoch sets parsed from tile FITS HISTORY cards rationale: >- - A tile's contributing exposures are recovered by parsing column 3 - of each HISTORY line, stripping prefix "p", deduplicating — the - coadd's own provenance record is trusted as the epoch list. The - LSB s-prefix rename is present but commented out. A mis-parse - changes N_EPOCH and which exposures are fit. Anchor: - workflow/config/cfis/config_tile_Fe.ini#FIND_EXPOSURES_RUNNER.COLNUM; + A tile's contributing exposures are column 3 (COLNUM) of each HISTORY + line, with prefix p stripped and duplicates removed: the coadd's own + provenance is trusted as the epoch list. A mis-parse changes N_EPOCH + and which exposures are fit. + Anchor: workflow/config/cfis/config_tile_Fe.ini#FIND_EXPOSURES_RUNNER.COLNUM; src/shapepipe/modules/find_exposures_package/find_exposures.py::FindExposures.get_exposure_list. default: history_parse options: - history_parse: { label: "HISTORY column 3, prefix p, dedup" } + history_parse: + label: HISTORY column 3, prefix p, deduplicated object_position_columns: label: Windowed centroids (XWIN/YWIN) define every position rationale: >- - PSF interpolation sites, tile stamp centres, multi-epoch stamp - centres, and the catalogue sky position all use SExtractor's - windowed centroid — XWIN_WORLD/YWIN_WORLD on the tile side (SPHE), - XWIN_IMAGE/YWIN_IMAGE exposure-side (PIX). Windowed vs isophotal - vs model centroids differ systematically for blends and asymmetric - galaxies, and the centroid definition feeds the position seed. - Anchor: workflow/config/cfis/config_tile_PiViVi.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS; - workflow/config/cfis/config_tile_PiViVi.ini#VIGNETMAKER_RUNNER_RUN_2.POSITION_PARAMS; - workflow/config/cfis/config_exp_psfex.ini#POSITION_PARAMS. + PSF interpolation sites, tile and multi-epoch stamp centres, and the + catalogue position all use SExtractor's windowed centroid: + XWIN_WORLD/YWIN_WORLD on the tile side, XWIN_IMAGE/YWIN_IMAGE on + exposures. Windowed, isophotal and model centroids differ + systematically for blends and asymmetric galaxies, and the centroid + feeds the position seed and the centroid prior. + Anchor: workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS; + workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.POSITION_PARAMS; + workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS. default: xwin_windowed options: - xwin_windowed: { label: Windowed centroids everywhere } + xwin_windowed: + label: Windowed centroids everywhere stamp_positioning_and_padding: - label: Nearest-pixel stamp centring; edge stamps zero-padded + label: Nearest-pixel stamp extraction with zero padding rationale: >- - Multi-epoch stamps are placed by round-tripping the tile world - position through the stored per-CCD WCS, then rounding to the - nearest pixel (no sub-pixel interpolation — the residual sub-pixel - offset is absorbed by the fit's centroid prior, cen sigma = 1 - pixel). Objects whose stamp overruns a CCD or tile edge are KEPT, - out-of-image pixels zero-filled (sf_tools FetchStamps - pad_mode='constant'); no boundary rejection exists — zero-padded - pixels enter the fit as data with whatever weight the padded - weight stamp carries. Anchor: - src/shapepipe/modules/vignetmaker_package/vignetmaker.py::VignetMaker._get_stamp; + [HARDCODED] stamps are cut around the pixel nearest the object's + position, with no sub-pixel interpolation; the sub-pixel remainder is + stored as the stamp's OFFSET, which ngmix uses as the Jacobian origin + (shape_measurement.centroid_source), so extraction and centroid + prior share one rounding. Multi-epoch stamps take the position from + the tile world coordinate through the stored per-CCD WCS. Objects + whose stamp overruns an image edge are kept, with out-of-image pixels + zero-filled; there is no boundary rejection. + Anchor: src/shapepipe/modules/vignetmaker_package/vignetmaker.py::get_stamps; src/shapepipe/modules/vignetmaker_package/vignetmaker.py::VignetMaker._get_stamp_me. default: round_and_zero_pad options: - round_and_zero_pad: { label: "Nearest-pixel + zero padding, no edge rejection" } - epoch_flag_source: - label: Per-epoch flag stamps come from RAW CFIS flags, not the pipeline mask - rationale: >- - The multi-epoch vignet run reads its flag stamps from - split_exp_runner output — the delivered instrumental flags — - while mask_runner's pipeline_flag (halos, spikes, DSOs, borders) - feeds only the exposure-side star finding - (config_exp_psfex.ini FILE_PATTERN pipeline_flag). Combined with - unmasked tile detection (detection.detection_source_mode), the - consequence is stark: THE BRIGHT-STAR MASKS CURRENTLY AFFECT ONLY - PSF-STAR SELECTION — neither the galaxy sample (no tile mask, no - IMAFLAGS cut possible) nor the pixels ngmix fits (raw flags only) - see them. Whether that is intended belongs to the - planned masking-unification rework, whose object-level half is - sp_validation's IMAFLAGS_ISO cut on a column this chain never - produces; this is the pixel-level half. Anchor: - workflow/config/cfis/config_tile_PiViVi.ini#VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_EXP_RUNNERS; - workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.FILE_PATTERN; - workflow/config/cfis/config_exp_Ma.ini#MASK_RUNNER.PREFIX. - default: raw_flags - options: - raw_flags: { label: split_exp raw flags gate epoch pixels } - pipeline_flags: - label: pipeline_flag (incl. bright-star masks) gates epoch pixels + round_and_zero_pad: + label: Nearest-pixel extraction, zero padding, no edge rejection prior_insights: guinot22_gaia_astrometry: claim: >- @@ -921,17 +661,8 @@ analyses: # ═════════════════════════════════════════════════════════════════════════ star_selection_psf: description: >- - Which objects constrain the PSF, and the PSF model itself. Modules: - src/shapepipe/pipeline/str_handler.py (_mode — the iterative - histogram-zoom FWHM mode estimator centring the star box; median - fallback below N=20), src/shapepipe/modules/setools_package/setools.py - (_make_rand_split), src/shapepipe/modules/psfex_interp_package/ - psfex_interp.py (acceptance gates, HSM shapes). Configs: - star_selection.setools, default.psfex, config_exp_psfex.ini. - [LINT] star_stat logs the FWHM cut as mode +- 0.1*0.187 while the mask - applies mode +- 0.2 px — the run's own log misstates the selection. - [LINT] pixel scale appears as 0.187 (load-bearing) and 0.186 - (plot-only) in the same setools file. + Which objects constrain the PSF, the PSF model itself, and which CCD + models are good enough to use. inputs: - id: exposure_sexcat type: data @@ -941,131 +672,109 @@ analyses: type: data format: psf description: >- - Per-CCD PSFEx models + interpolated PSFs at object positions - (run_sp_exp_SxSePsfPi family), with HSM shape diagnostics. + Per-CCD PSF models and the PSFs interpolated at object positions, + with HSM shape diagnostics. + inputs: [exposure_sexcat] decisions: [star_selection_box, psf_train_validation_split, psfex_candidate_vetting, psf_modelling_software, psf_model_complexity, psf_acceptance_thresholds] decisions: star_selection_box: - label: Stellar-locus selection — mag window + FWHM window around the mode + label: Stellar-locus selection, magnitude window and FWHM window around the mode rationale: >- - 18 < MAG_AUTO < 22, |FWHM - mode| <= 0.2 px, FLAGS==0, - IMAFLAGS_ISO==0; the mode is computed on a preselection - (MAG_AUTO<21, 0.3-1.5 arcsec at 0.187"/px) via the iterative - histogram-zoom estimator (str_handler.py::_mode, eps=0.001; median - fallback for N<20, -1 for N=0 — small-N behaviour changes selection - on sparse CCDs). PSFEx's automatic FWHM-range selection is off - (SAMPLE_AUTOSELECT N) and bad-pixel filtering is off, but PSFEx's - compiled-in fixed sample cuts still apply on top of this box — - see psfex_candidate_vetting. + 18 < MAG_AUTO < 22, |FWHM - mode| <= 0.2 px, FLAGS == 0 and + IMAFLAGS_ISO == 0. The mode is computed on a preselection (MAG_AUTO < + 21, FWHM 0.3-1.5 arcsec at 0.187 arcsec/px) by an iterative + histogram-zoom estimator that falls back to the median below 20 + objects, so small-N behaviour changes selection on sparse CCDs. + PSFEx's own selection is off (SAMPLE_AUTOSELECT N), but its + compiled-in cuts still apply (psfex_candidate_vetting). [LINT] the + file's statistics log the FWHM cut as mode +- 0.1 px and its plot + uses 0.186 arcsec/px, while the applied cut is +- 0.2 px at 0.187. Anchor: workflow/config/cfis/star_selection.setools#MASK:star_selection.MAG_AUTO; - workflow/config/cfis/star_selection.setools#MASK:preselect.MAG_AUTO; + workflow/config/cfis/star_selection.setools#MASK:preselect.FWHM_IMAGE; workflow/config/cfis/default.psfex#SAMPLE_AUTOSELECT; - src/shapepipe/pipeline/str_handler.py::_mode. + src/shapepipe/pipeline/str_handler.py::StrInterpreter._mode. default: mode_centred_box options: mode_centred_box: - label: FWHM-mode-centred box, +-0.2 px, mag 18-22, setools-only vetting + label: FWHM-mode-centred box, +-0.2 px, mag 18-22 insights: [guinot22_star_box] size_mag_locus_fit: label: Fitted size-magnitude stellar locus - excluded: true - excluded_reason: Not wired; the mode-box is the validated v2.0 selection. + description: Not implemented. psfex_autoselect: label: PSFEx SAMPLE_AUTOSELECT vetting on top + insights: [guinot22_psfex_preselection_off] excluded: true excluded_reason: >- - Deliberately disabled so selection lives in one place; rationale - not recorded in code. + Disabled so that the pipeline's own star selection is the only + one. psf_train_validation_split: - label: Random 80/20 star split — model fit vs held-out validation + label: Seeded 80/20 star split, model fit vs held-out validation rationale: >- - RAND_SPLIT ratio 20: star_split_ratio_80 fits the PSFEx model - (config_exp_psfex.ini FILE_PATTERN, and the tile multi-epoch - interpolation ME_DOT_PSF_PATTERN in config_tile_PiViVi.ini); - star_split_ratio_20 is the independent PSF-residual diagnostic - (PSFEX_INTERP MODE=VALIDATION). Trades model precision (fewer - training stars per CCD, interacting with the STAR_THRESH gate) - against an independent residual test. - [PENDING #873] The split is DETERMINISTIC: _make_rand_split takes - np.random.RandomState(seed).permutation(cat_size), the seed being - the digits of the unit's file number mod 2^32 — a pure function of - the input catalogue, fixed per CCD and independent of processing - order, the same philosophy as shape_measurement.ngmix_seed_mode's - SEED_FROM_POSITION. Before this the split drew from unseeded - np.random.randint, so the star sample entering the PSF model — and - therefore every shape downstream of it — differed between - identical runs; it was the one unseeded draw the position-seed work - left uncovered. One-off cost: the realised 80/20 membership changes - once (it is one further draw, now frozen), so PSF models and shapes - shift by that draw relative to every earlier product. + RAND_SPLIT RATIO 20: the 80% sample fits the PSFEx model and feeds the + tile multi-epoch interpolation (ME_DOT_PSF_PATTERN); the 20% sample + is the independent residual diagnostic (psfex_interp VALIDATION + mode). The split trades training stars per CCD, which interacts with + the acceptance gate, against an independent residual test. It is + deterministic: a permutation seeded from the unit's file number, so + the PSF star sample is a pure function of the input catalogue. Anchor: workflow/config/cfis/star_selection.setools#RAND_SPLIT:star_split.RATIO; src/shapepipe/modules/setools_package/setools.py::SETools._make_rand_split; workflow/config/cfis/config_exp_psfex.ini#PSFEX_RUNNER.FILE_PATTERN; - workflow/config/cfis/config_tile_PiViVi.ini#PSFEX_INTERP_RUNNER.ME_DOT_PSF_PATTERN. + workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.ME_DOT_PSF_PATTERN. default: split_80_20_seeded options: split_80_20_seeded: label: 80% train / 20% validation, seeded from the file number insights: [guinot22_star_split] split_80_20_unseeded: - label: Same split, unseeded np.random (pre-#873) + label: Same split from an unseeded random draw excluded: true excluded_reason: >- - Retired by #873: it made the PSF star sample — and every shape - downstream of it — irreproducible run-to-run, the single - remaining unseeded draw in the science chain. Kept on the - record because every UNIONS product built before the smk-g4 - campaign was produced under it. + Makes the PSF star sample, and every shape downstream of it, + irreproducible run-to-run. no_holdout: - label: 100% of stars in the model, no held-out diagnostic + label: All stars in the model, no held-out diagnostic excluded: true - excluded_reason: Loses the independent rho-statistic input. + excluded_reason: Loses the independent residual and rho-statistic input. psfex_candidate_vetting: - label: PSFEx-side candidate vetting — built-in defaults, unpinned + label: PSFEx built-in candidate cuts, unpinned rationale: >- - default.psfex sets only SAMPLE_AUTOSELECT N; SAMPLE_MINSN, - SAMPLE_MAXELLIP, SAMPLE_FWHMRANGE, SAMPLE_VARIABILITY are absent, - so PSFEx's compiled-in defaults apply silently (MINSN 20, - MAXELLIP 0.3, FWHMRANGE 2-10 px, VARIABILITY 0.2) — a second star - selection nobody's config records, and one that changes if the - PSFEx binary version changes. BADPIXEL_FILTER N + PSF_RECENTER N: - star vignets with flagged/sentinel pixels are accepted unfiltered - and candidates are not recentred (CENTER_KEYS XWIN). The setools - box is therefore not the whole selection. [HARDCODED] (in the - PSFEx binary). + default.psfex sets SAMPLE_AUTOSELECT N but omits SAMPLE_MINSN, + SAMPLE_MAXELLIP, SAMPLE_FWHMRANGE and SAMPLE_VARIABILITY, so + [HARDCODED] PSFEx's compiled-in defaults apply (MINSN 20, MAXELLIP + 0.3, FWHMRANGE 2-10 px, VARIABILITY 0.2): a second star selection no + config records, which changes with the PSFEx version. + BADPIXEL_FILTER N and PSF_RECENTER N accept flagged star vignets + unfiltered and do not recentre candidates. Anchor: workflow/config/cfis/default.psfex#SAMPLE_AUTOSELECT; workflow/config/cfis/default.psfex#BADPIXEL_FILTER; workflow/config/cfis/default.psfex#PSF_RECENTER. default: builtin_defaults options: builtin_defaults: - label: "Compiled-in MINSN 20 / MAXELLIP 0.3 / FWHMRANGE 2-10, no bad-pixel filter" - insights: [guinot22_psfex_preselection_off] + label: Compiled-in SAMPLE_* defaults, no bad-pixel filter pinned_explicit: label: Write the SAMPLE_* values explicitly into default.psfex psf_modelling_software: - label: PSF modelling software — PSFEx per-CCD vs MCCD focal-plane + label: PSF model, PSFEx per CCD or MCCD over the focal plane rationale: >- - The committed chain fits PSFEx independently per CCD. MCCD - (Liaudat+2021) is a maintained in-tree alternative: a focal-plane - model fit across all 40 CCDs at once with a hybrid local+global - decomposition (src/shapepipe/modules/mccd_package/ + six - mccd_*_runner.py; knobs in example/cfis/config_MCCD.ini — - N_COMP_LOC=8, D_COMP_GLOB=8, LOC_MODEL=hybrid, MIN_N_STARS=20, - RMSE_THRESH=1.25). Unwired in workflow/config/cfis/ (needs the - MCCD config adapted, and config_exp_mccd.ini carries a stale - hardcoded PSF_MODEL_DIR path). The image-simulation path - substitutes PSF modelling entirely: fake_psf_runner injects the - true input PSF from a SKiLLS dictionary in psfex_interp's output - format. - Anchor: src/shapepipe/modules/mccd_package; - src/shapepipe/modules/fake_psf_package; - example/cfis/config_MCCD.ini#INSTANCE.N_COMP_LOC; - example/cfis/config_MCCD.ini#INPUTS.MIN_N_STARS; - example/cfis/config_exp_mccd.ini. + workflow/config.yaml psf_model selects the exposure and tile config + pair; the committed value is psfex, which fits each CCD + independently. MCCD (Liaudat+2021) fits all 40 CCDs at once with a + hybrid local+global model (N_COMP_LOC 8, D_COMP_GLOB 8, MIN_N_STARS + 20, RMSE_THRESH 1.25); the completeness table treats its counts as + warnings because no campaign has run it. [LINT] config_exp_mccd.ini + still reads pipeline_flag images from mask_runner, which no longer + exists, so the MCCD exposure chain cannot run as committed. + Anchor: workflow/config.yaml; + workflow/config/cfis/config_MCCD.ini#INSTANCE.N_COMP_LOC; + workflow/config/cfis/config_MCCD.ini#INPUTS.MIN_N_STARS; + workflow/config/cfis/config_exp_mccd.ini#SEXTRACTOR_RUNNER.INPUT_MODULE; + src/shapepipe/modules/mccd_package. default: psfex options: psfex: @@ -1073,89 +782,57 @@ analyses: insights: [guinot22_psfex_software, farrens22_two_psf_methods] mccd_focal_plane: label: MCCD hybrid local+global focal-plane model - true_input_psf: - label: fake_psf injection of the simulation's true PSF - excluded: true - excluded_reason: >- - Only meaningful on simulated images where the true PSF exists; - not a data-analysis option. + insights: [farrens22_two_psf_methods] psf_model_complexity: - label: PSFEx model — pixel basis, degree-2 spatial polynomial per CCD + label: PSFEx pixel basis with degree-2 spatial variation per CCD rationale: >- - BASIS_TYPE PIXEL, BASIS_NUMBER 20, PSF_SIZE 51,51, PSF_SAMPLING 1, - PSFVAR_DEGREES 2 in XWIN,YWIN per CCD (MEF_TYPE INDEPENDENT, - STABILITY_TYPE EXPOSURE), PSF_RECENTER N. Model flexibility sets - the PSF-leakage/overfitting balance — the dominant additive - systematic in cosmic shear. Values are the stock EB 2017 header; - rationale not recorded in code. + BASIS_TYPE PIXEL, BASIS_NUMBER 20, PSF_SAMPLING 1, PSFVAR_DEGREES 2 + in XWIN/YWIN per CCD. Model flexibility sets the balance between PSF + leakage and overfitting, the dominant additive systematic in cosmic + shear. Stock values; no rationale recorded. Anchor: workflow/config/cfis/default.psfex#BASIS_TYPE; - workflow/config/cfis/default.psfex#PSFVAR_DEGREES; - workflow/config/cfis/default.psfex#PSF_SIZE. + workflow/config/cfis/default.psfex#BASIS_NUMBER; + workflow/config/cfis/default.psfex#PSFVAR_DEGREES. default: pixel_basis_deg2_per_ccd options: pixel_basis_deg2_per_ccd: - label: "PIXEL basis, degree 2, per-CCD" + label: PIXEL basis, degree 2, per CCD insights: [guinot22_psf_no_oversampling] deg3: label: Degree-3 spatial variation - excluded: true - excluded_reason: >- - More flexibility per CCD needs more stars per CCD than the - count-floor world guarantees; not validated. + description: Needs more stars per CCD than the acceptance gate guarantees. psf_acceptance_thresholds: - label: Per-CCD PSF-model quality gate (min stars, max chi2) + label: Per-CCD PSF-model quality gate rationale: >- - A CCD whose model has ACCEPTED < STAR_THRESH or CHI2 > 2 is not - interpolated — its galaxies drop from the shear catalogue: direct - footprint selection, the in-code analogue of the DES blacklist. - [PENDING #873] Both passes now gate at 22 stars: the VALIDATION-mode - exposure config always did (config_exp_psfex.ini), and #873 raised - the MULTI-EPOCH science path 20 -> 22 in example/cfis - (config_tile_PiViVi_canfar_{sx,uc}.ini), with commit 90782098 - mirroring it into workflow/config/cfis/config_tile_PiViVi.ini — the - committed config fork this workflow actually reads (#848 D2). - Provenance of the retired 20, which is what makes this a fix rather - than a preference: commit fdc86553 (Kilbinger, 2020-06-30, "Forgot - to update new star number threshold for 80% of stars") deliberately - bumped 20 -> 22 to account for the 80/20 split, but only in the - validation config; the tile config kept the pre-split 20, so for - five years the SCIENCE path gated on the value that 2020 fix meant - to retire. (20 is also the psfex_interp function default, so the - stale-value reading rested on the commit provenance rather than on - the config alone.) - Published description (Guinot+22 p.4, Fig. 3): 22 stars/CCD, applied - to exactly the CCDs feeding multi-epoch shape measurement — the - number now agrees. Two gaps remain: an undocumented CHI2_THRESH=2 in - both configs, and the mechanism — interpsfex tests the PSFEx header - ACCEPTED/CHI2 at interpolation time and drops that epoch for objects - on the CCD, rather than excluding the CCD from PSF modelling as the - paper describes. + A CCD whose model has ACCEPTED < STAR_THRESH = 22 or CHI2 > + CHI2_THRESH = 2 is not interpolated, on both the validation and the + multi-epoch pass; 22 applies the published floor to the 80% training + sample. In the science path the CCD's epoch is dropped for every + object on it; an object left with no epoch has no shape. There is no + minimum-epoch floor in the pipeline: NGMIX_N_EPOCH records what + survived and epoch-count cuts happen downstream. Guinot+22 describes + excluding the CCD from PSF modelling rather than gating at + interpolation, and does not state the chi2 cut. Anchor: src/shapepipe/modules/psfex_interp_package/psfex_interp.py::PSFExInterpolator.interpsfex; workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH; - workflow/config/cfis/config_tile_PiViVi.ini#PSFEX_INTERP_RUNNER.STAR_THRESH; - example/cfis/config_tile_PiViVi_canfar_sx.ini#PSFEX_INTERP_RUNNER.STAR_THRESH; - example/cfis/config_tile_PiViVi_canfar_uc.ini#PSFEX_INTERP_RUNNER.STAR_THRESH. + workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH; + workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.CHI2_THRESH. default: stars22_chi2_2 options: stars22_chi2_2: - label: ">= 22 stars on both passes, chi2 <= 2" + label: ">= 22 stars and chi2 <= 2 on both passes" insights: [des_psf_blacklist_local, guinot22_star_floor_22_local] - stars20_chi2_2: - label: ">= 20 stars on the science path, 22 in validation (pre-#873)" + stars20_science_path: + label: ">= 20 stars on the science path" excluded: true excluded_reason: >- - Retired by #873 + 90782098. It was never a chosen value: it is - the pre-split threshold fdc86553 raised to 22 in 2020 for the - validation config and forgot on the science path, leaving the - science gate below both the published floor (Guinot+22 Fig. 3) - and the pipeline's own intent. Every UNIONS product built before - the smk-g4 campaign carries it. + 20 is the pre-split floor; with 80% of stars in the model it gates + below both the published floor and the validation pass. des_25: label: DES Y3 threshold (25 stars) excluded: true excluded_reason: >- - Not adopted; CFIS CCDs are smaller than DECam's — the right - number is survey-specific. + CFIS CCDs are smaller than DECam's; the floor is survey-specific. prior_insights: des_psf_blacklist_local: claim: >- @@ -1171,8 +848,7 @@ analyses: guinot22_star_floor_22_local: claim: >- The published ShapePipe/UNIONS analysis discards a CCD from the PSF - estimation when fewer than 22 stars are selected on it — the floor - the science-path PSF-interpolation gate now applies. + estimation when fewer than 22 stars are selected on it. created_at: "2022-04-01T00:00:00Z" evidence: - id: ev_guinot22_star_floor_local @@ -1253,244 +929,431 @@ analyses: # ═════════════════════════════════════════════════════════════════════════ shape_measurement: description: >- - Galaxy shape estimation: ngmix single-Gaussian fits with metacalibration. - Modules: src/shapepipe/modules/ngmix_package/ngmix.py (priors, metacal - setup, epoch handling, postage-stamp prep), ngmix_runner.py (config - exposure). Config: config_tile_Ng_template.ini. Most values here are - HARDCODED — scientific choices living in code with no config exposure; - this sub-analysis is where the silent-default risk concentrates. - [LINT] centroid_source default disagrees between the runner ("wcs", - production; always passed explicitly, ngmix_runner.py:170) and every - module-level signature ("hsm") — unreachable in the pipeline path, but - direct callers (tests, notebooks) silently get the other choice. - [LINT] pixel scale is 0.186 here (config PIXEL_SCALE) vs 0.187 in the - setools/masking configs — and star_selection.setools itself mixes - 0.187 (cuts) with 0.186 (SCATTER stat, :75). + Galaxy shape estimation: joint multi-epoch ngmix Gaussian fits with + metacalibration. Most choices here are fixed in code; this is where the + silent-default risk concentrates. inputs: - id: vignets type: data - source: 51x51 galaxy/weight/background-RMS vignets + interpolated PSFs (vignetmaker, psfex_interp) + source: multi-epoch galaxy, weight, flag, background and background-RMS vignets + interpolated PSFs outputs: - id: ngmix_cat type: data format: fits description: Per-tile metacal shear catalogue chunks (ngmix_runner family). + inputs: [vignets] decisions: - [ngmix_seed_mode, galaxy_model, fit_priors, metacal_scheme, - centroid_source, epoch_quality_and_weighting, noise_model, - psf_epoch_loss_policy, megacam_ccd_flip] + [ngmix_seed_mode, galaxy_model, fit_initialisation, fit_priors, + metacal_scheme, centroid_source, epoch_flux_rescaling, + psf_epoch_averaging, galaxy_pixel_weights, psf_likelihood_noise, + megacam_ccd_flip, defect_fill, blend_handling, + epoch_masked_fraction_cut] decisions: ngmix_seed_mode: - label: ngmix per-object RNG seeding + label: Per-object RNG seeded from sky position rationale: >- - Production historically seeded one RandomState from the tile ID, - consumed in object order — results depended on chunk boundaries. - The position seed (3-arcsec sky boxes + CCD offsets, zig-zag fold + - Cantor pairing mod 2^32; ngmix.py::position_seed) makes every - stream a function of sky position: chunk-invariant, - bit-reproducible, and metacal fixnoise counter-noise cancels across - image-simulation branches (ngmix#796). Consequence: chunking is - demoted to a pure throughput knob (reverting this decision - re-promotes it). Cost: noise streams change vs v2.0 — see - top-level baseline_validation_criterion. + Each object's RNG (noise realisations, guesses, priors) is seeded + from its position: [HARDCODED] 3-arcsec sky boxes offset by the + first epoch's CCD number, folded and Cantor-paired mod 2^32. Every + stream is a function of sky position, so results do not depend on + how a tile is chunked, and metacal's fixnoise counter-noise cancels + across image-simulation branches that share an object's box. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::position_seed; - workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.SEED_FROM_POSITION; - src/shapepipe/modules/ngmix_runner.py::ngmix_runner. + src/shapepipe/modules/ngmix_package/ngmix.py::Ngmix.process. default: position_seed options: position_seed: label: Per-object seed from (ra, dec, ccd), 3-arcsec boxes tile_seed: - label: Tile-wide RandomState (v2.0) + label: One tile-wide RandomState consumed in object order excluded: true excluded_reason: >- - Chunk-dependent; retired outright (SEED_FROM_POSITION=False now - raises — ngmix_runner.py:110-116). + Makes every noise stream depend on chunk boundaries and object + order. galaxy_model: - label: Galaxy and PSF model — single Gaussian [HARDCODED] + label: Single-Gaussian galaxy and PSF models rationale: >- - ngmix.fitting.Fitter(model='gauss') for both galaxy and PSF - (ngmix.py::make_runners); guessers TPSFFluxAndPriorGuesser / - TFluxGuesser with T=0.25 and catalogue-flux guess, Runner ntry=5, - PSFRunner ntry=2 — with a non-convex likelihood, guess and retries - decide which objects converge (failed fits are NaN-filled with - flags, not raised). Rationale not recorded. Under metacal, model - bias largely cancels in the response, which is the standard defense - of 'gauss'; not stated in code. + [HARDCODED] ngmix Fitter(model='gauss') for both the galaxy and the + PSF. Under metacalibration, model bias largely cancels in the + response, which is the standard defence of the Gaussian; the code + does not state it. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::make_runners. default: gauss options: gauss: - label: "Single Gaussian, T guess 0.25, ntry 5/2" + label: Single Gaussian insights: [guinot22_gaussian_model] exp_or_bdf: label: exp / bdf galaxy models - excluded: true - excluded_reason: >- - Slower, and metacal makes the gain marginal; not validated on - CFIS. + description: Not exposed; make_runners fixes the model. + fit_initialisation: + label: Fit guesses and retries + rationale: >- + [HARDCODED] TPSFFluxAndPriorGuesser (galaxy) and TFluxGuesser (PSF) + start from T = 0.25 and the catalogue flux; the galaxy runner retries + 5 times, the PSF runner twice. With a non-convex likelihood the guess + and retries decide which objects converge; a failed fit is flagged, + not raised. Guinot+22 initialised the whole guess vector from HSM + adaptive moments on each sheared image; the code no longer does. + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::make_runners. + default: prior_guess_t025_ntry5_2 + options: + prior_guess_t025_ntry5_2: + label: T guess 0.25 + catalogue flux, ntry 5 / 2 + hsm_initialisation: + label: Guesses from HSM adaptive moments (Guinot+22) + description: Not implemented. fit_priors: - label: ngmix joint prior — GPriorBA(0.4), cen sigma = pixel scale, flat T/F [HARDCODED] + label: ngmix joint prior rationale: >- - Ellipticity GPriorBA sigma=0.4; centroid CenPrior sigma = one pixel - scale (0.186 arcsec, config PIXEL_SCALE — the coupling - sigma=pixel_scale is itself the hardcoded choice); flat T in - [-1, 1e3], flat F in [-100, 1e9] with negative support (bounds - decide which noisy fits survive vs rail). get_prior takes T/F range - arguments but no caller passes them. Prior width drives noise bias; - rationale not recorded. - Published description (Guinot+22 p.7): centroid sigma = pixel scale - ~0.187 arcsec, flat F in [-1e4, 1e9], flat half-light radius r50 in - [-10, 1e6] arcsec, ellipticity prior from Bernstein & Armstrong - (2014); current code: PIXEL_SCALE 0.186, flat F in [-100, 1e9], and - a flat prior on ngmix's second-moment size T in [-1, 1e3] rather - than on r50 — the prior families agree, the flux bound and pixel - scale have drifted, and the size prior is a different - parameterisation rather than a changed number. + [HARDCODED] ellipticity GPriorBA with sigma 0.4; flat T in [-1, 1e3] + and flat F in [-100, 1e9], with negative support (the bounds decide + which noisy fits survive and which rail); a centroid prior of width + one pixel scale, PIXEL_SCALE 0.186 arcsec (derived from the WCS when + the key is absent). Prior width drives noise bias; no rationale + recorded. Guinot+22 states a flat F in [-1e4, 1e9] and a flat r50 + prior rather than T. [LINT] the epoch stamps are exposure pixels, and + star selection uses 0.187 arcsec/px. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::get_prior; workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.PIXEL_SCALE. default: gpriorba04_flat options: - gpriorba04_flat: { label: "GPriorBA 0.4 + flat T/F with negative support" } + gpriorba04_flat: + label: GPriorBA 0.4 + flat T/F with negative support nonneg_informative: label: Non-negative or informative T/F priors excluded: true excluded_reason: >- Truncating negative support biases the noshear ensemble mean; - metacal wants symmetric noise response. + metacal wants a symmetric noise response. metacal_scheme: - label: Metacalibration — 5 types, step 0.01, fitgauss reconv, fixnoise [HARDCODED] + label: Metacalibration protocol rationale: >- - types [noshear,1p,1m,2p,2m], step 0.01, psf='fitgauss' (runner - default; moves the metacal response directly — alternatives gauss/ - dilate/azgauss listed in the docstring), fixnoise=True, - use_noise_image=True, MetacalBootstrapper(ignore_failed_psf=True) - (changes which epochs enter the fit). No *_psf sheared types, so no - mcal_R_psf PSF-response term in the catalogue. fixnoise rationale - appears only in the position_seed docstring (counter-noise - cancellation). - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal. + [HARDCODED] types noshear, 1p, 1m, 2p, 2m with step 0.01, fixnoise + with the noise image, and ignore_failed_psf (an epoch whose PSF fit + fails is dropped). The reconvolution kernel is METACAL_PSF, default + fitgauss, not set in the committed config; it moves the response + directly. No sheared-PSF types run, so the catalogue has no PSF + response term. + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal; + src/shapepipe/modules/ngmix_runner.py::ngmix_runner. default: five_types_step001_fitgauss options: five_types_step001_fitgauss: - label: "noshear+1p/1m/2p/2m, step 0.01, fitgauss, fixnoise" + label: noshear+1p/1m/2p/2m, step 0.01, fitgauss, fixnoise insights: [guinot22_metacal_five_images] with_psf_response: label: Add sheared-PSF types for R_psf - excluded: true - excluded_reason: >- - Not wired; leakage is instead diagnosed via PSF_ORIG columns + - rho statistics downstream. + description: Not implemented. centroid_source: - label: Jacobian origin from WCS astrometry, not HSM moments + label: Jacobian origin at the coadd centroid rationale: >- - Production runner default "wcs"; hsm is "legacy... noisy for stars - and flagged as incorrect by Fabian — see #767" (runner comment; a - rare recorded rationale). Moves the centroid-prior centre per - object. The runner reads an optional CENTROID_SOURCE config option - that no committed CFIS config sets. [LINT] module-level default is - still "hsm" — see this sub-analysis's description. - Published description (Guinot+22 p.7): HSM adaptive moments, run on - each sheared version, supplied the whole initial guess vector - (centroid, r50, flux) for the least-squares fit; current code: that - initialisation is gone — guesses come from ngmix's - TPSFFluxAndPriorGuesser with fixed T=0.25 and a catalogue flux, and - the only surviving HSM role is the optional stamp re-centering that - sets the Jacobian origin. So the drift is wider than a swapped - centroid source. (The paper's other HSM use, PSF/star shape - diagnostics, is unaffected.) + The Jacobian origin, where the centroid prior centres, is the + sub-pixel offset the stamp extractor stored when it cut the stamp + ("wcs", the default at every level; the committed config sets no + CENTROID_SOURCE). One projection and one rounding serve both + extraction and prior, so they cannot disagree near a rounding tie. + "hsm" re-centres on adaptive moments measured from the stamp. Anchor: src/shapepipe/modules/ngmix_runner.py::ngmix_runner; src/shapepipe/modules/ngmix_package/ngmix.py::make_ngmix_observation. default: wcs options: - wcs: { label: WCS-projected catalogue position } + wcs: + label: Coadd-centroid offset from the stamp extractor hsm: label: HSM adaptive-moment centroid excluded: true - excluded_reason: Noisy for stars; flagged incorrect (shapepipe#767). - epoch_quality_and_weighting: - label: Epoch admission, masking cut, and multi-epoch combination [HARDCODED] + excluded_reason: >- + Noisy, notably for stars, and follows the light rather than the + astrometry that placed the stamp. + epoch_flux_rescaling: + label: Epochs put on a common flux scale by header FSCALE rationale: >- - An epoch is dropped if >1/3 of its stamp is masked - (prepare_postage_stamps; the comment says "objects", the code drops - epochs — an object with zero surviving epochs drops out); failed - PSF fits drop epochs (flags != 0); fluxes rescaled by header FSCALE - (gal*Fscale, weight/Fscale^2); the diagnostic PSF is averaged over - epochs weighted by obs.weight.sum(). Joint multi-epoch fit over - survivors. Rationale for 1/3 and for the weight choice not - recorded. - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_postage_stamps; - src/shapepipe/modules/ngmix_package/ngmix.py::rescale_epoch_fluxes; - src/shapepipe/modules/ngmix_package/ngmix.py::_average_psf_fits. - default: third_masked_cut + [HARDCODED] each epoch's image is multiplied by its header FSCALE and + its weight divided by FSCALE squared (background RMS scaled with the + image) before the joint fit, so the epochs share the tile's + zero-point. + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::rescale_epoch_fluxes. + default: fscale options: - third_masked_cut: { label: "Drop epoch if >1/3 masked; FSCALE rescale; weight-sum PSF average" } - noise_model: + fscale: + label: Rescale by header FSCALE + psf_epoch_averaging: + label: Catalogue PSF quantities averaged over epochs by galaxy weight + rationale: >- + [HARDCODED] the PSF shape and size written to the catalogue (the + original image PSF and the metacal reconvolution kernel) are averages + over epochs weighted by the summed galaxy inverse variance of each + epoch; epochs whose PSF fit failed are left out. These columns feed + PSF-leakage estimates downstream. + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::_average_psf_fits; + src/shapepipe/modules/ngmix_package/ngmix.py::average_original_psf. + default: galaxy_weight_sum + options: + galaxy_weight_sum: + label: Weighted by summed galaxy inverse variance + galaxy_pixel_weights: label: Per-pixel inverse variance from background-RMS vignets rationale: >- - BKG_RMS_VIGNET_PATH set in the CFIS template: weight = - 1/bkg_rms^2 per pixel (all-or-nothing; missing file errors); - fallback scalar 1/sigma_mad^2. Masked pixels filled with Gaussian - noise at sig_noise. A scalar sigma "mis-reports errors and erodes - the inverse-variance advantage whenever the RMS map actually - varies" (recorded rationale, fixnoise bookkeeping). PSF observation - gets a flat weight from PSF_NOISE=1e-5 — hardcoded module constant, - validated 1e-4..1e-6 on the digital twin (#749/#774 comment); - without it the g-prior swamps the PSF likelihood. Per-epoch - background subtraction BKG_SUB=True (off only for sims). + With BKG_RMS_VIGNET_PATH set, each pixel's weight is 1/rms^2 from the + SExtractor background-RMS map (all or nothing; a missing file + raises), and the noise realisations use the same per-pixel RMS; the + fallback is a scalar 1/sigma_mad^2. A scalar sigma mis-reports errors + wherever the RMS varies. Each epoch is background-subtracted with the + SExtractor background vignet (BKG_SUB, on unless the key is set + False). Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_ngmix_weights; - src/shapepipe/modules/ngmix_package/ngmix.py::PSF_NOISE; src/shapepipe/modules/ngmix_package/ngmix.py::background_subtract; workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.BKG_RMS_VIGNET_PATH. default: rms_vignet_weights options: - rms_vignet_weights: { label: Per-pixel RMS-map weights + PSF_NOISE 1e-5 } + rms_vignet_weights: + label: Per-pixel background-RMS weights scalar_sigma_mad: label: Scalar sigma_mad per epoch excluded: true - excluded_reason: Mis-reports errors where the RMS map varies (recorded). - psf_epoch_loss_policy: - label: Object-level policy when CCDs fail PSF interpolation + excluded_reason: Mis-reports errors wherever the RMS map varies. + psf_likelihood_noise: + label: Flat PSF-observation weight rationale: >- - When k of ~40 CCDs fail the PSF acceptance gate (~5-6% attrition, - per-exposure clustered, matches the bash baseline — but measured - with the science gate at 20 stars, so [PENDING #873] at 22 it can - only rise, and smk-g4 is the first campaign to re-measure it), the - pipeline applies NO further quality gate: tiles complete, each - object records NGMIX_N_EPOCH, and sp report surfaces per-tile - epoch loss. - Object-level protection is delegated entirely to the validation - stage's epoch-count cut (sp_validation's galaxy selection cuts on N_EPOCH >= 1). Rationale (Cail, - 2026-08-29, PRD walk): epoch loss is a per-object depth effect - already recorded in the catalogue; gating at pipeline level would - fail whole tiles for a versionable catalogue decision. PRD #848's - open-questions section was removed accordingly. Per-tile epoch loss - is surfaced by the run report; NGMIX_N_EPOCH is the per-object - record. - Anchor: workflow/scripts/run_report.py; - workflow/config/cfis/final_cat.param#NGMIX_N_EPOCH. - default: record_and_delegate + [HARDCODED] the PSF observation carries a flat weight 1/PSF_NOISE^2 + with PSF_NOISE 1e-5; without it the g-prior swamps the PSF + likelihood. The recovered PSF shape and size are flat across 1e-4 to + 1e-6 on the digital twin. + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::PSF_NOISE; + src/shapepipe/modules/ngmix_package/ngmix.py::make_ngmix_observation. + default: psf_noise_1em5 options: - record_and_delegate: - label: Record NGMIX_N_EPOCH, report attrition, no pipeline gate - pipeline_epoch_floor: - label: Fail tiles below a minimum surviving-epoch fraction - excluded: true - excluded_reason: >- - Fails whole tiles for what is a versionable per-object - catalogue decision; the depth effect is already recorded. + psf_noise_1em5: + label: PSF_NOISE 1e-5 megacam_ccd_flip: - label: 180-degree tile-vignet rotation for MegaCam CCDs <18 and 36-37 [HARDCODED] + label: 180-degree rotation of tile vignets for CCDs below 18 and 36-37 rationale: >- - "MegaPipe has CCDs that are upside down" (docstring) — the tile - vignet is rotated to register with epoch stamps; a wrong flip - mis-registers the tile mask against the epoch, changing flagged - pixels and the 1/3-masked cut. Carries its own recorded caveat: - "will give incorrect results when used with THELI ccds. Fix this." + [HARDCODED] "MegaPipe has CCDs that are upside down": the tile vignet + and segmentation stamp are rotated to register with the epoch stamp. + A wrong flip mis-registers the tile coverage flag against the epoch, + changing flagged pixels and the masked-fraction cut. The docstring + warns it gives incorrect results for THELI CCDs. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::Ngmix.MegaCamFlip. default: megapipe_flip options: - megapipe_flip: { label: Flip CCDs <18 and 36/37 (MegaPipe orientation) } + megapipe_flip: + label: Flip CCDs < 18 and 36/37 (MegaPipe orientation) + defect_fill: + label: Image content of flagged / zero-weight pixels before metacal + rationale: >- + prepare_ngmix_weights gives weight 0 to every pixel with a nonzero + instrument flag, zero exposure weight or invalid background RMS. What + the IMAGE holds in those pixels still matters, because metacal never + looks at weights: ngmix builds a galsim InterpolatedImage from the + whole observation image, deconvolves, shears and reconvolves it, and + copies the weight map through unchanged. Whatever sits in a + zero-weight pixel is therefore spread into the weighted pixels within + about a PSF width, with ringing at sharp features. DES's own + corrector (ngmixer) says why it fills: "it may be important for codes + that take moments or use FFTs". The fill is coupled to the neighbour + treatment through the single BLEND_HANDLING key, which the committed + config leaves at its default. Under noisefill, masked pixels get an + independent noise realisation at the per-pixel RMS. Under uberseg, + the fill is skipped, so raw bad columns, bleeds, cosmic rays and + bright-star light enter metacal. [LINT] the prepare_ngmix_weights + docstring says noisefill keeps the weight of filled pixels (the code + zeroes it), and the ngmix_runner comment says noisefill fills + neighbour pixels (it fills flagged pixels and leaves neighbours + untouched). No DES metacal pipeline passed raw defects through + metacal: Y1 dropped every epoch with a masked pixel + (max_zero_weight_frac 0.0), Y3 filled symmetrized defects with the + best-fit central model (both candidate Y3 configs do this; which one + was production is not recorded), and Y6 interpolated symmetrized + defects in the image and in every noise image. The residual cost of + any fill is anisotropy. A filled bad column that crosses the galaxy + removes or misplaces light along one detector axis. The + reconvolution spreads that into an additive e1-type term, coherent on + the sky because CFHT/MegaCam, like DECam, has a fixed sky orientation + (inferred from CFHT's equatorial mount, no derotator). Sheldon & Huff + 2017 saw a large additive e1 even with model fill, removed by a + 90-degree compensating mask. Every DES pipeline symmetrized the + defect mask, and ShapePipe does not. The fill should be set + independently of blend_handling; the recommended option is + symmetrized_noise. + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_ngmix_weights; + src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal; + src/shapepipe/modules/ngmix_runner.py::ngmix_runner. + default: noise + options: + noise: + label: Noise fill (BLEND_HANDLING = noisefill) + description: >- + Masked pixels are replaced by an independent noise realisation at + the per-pixel background RMS and keep weight 0. Consistent with + metacal's fixnoise noise image, which covers every pixel. Removes + defects, but leaves an unsymmetrized hole in the galaxy light + wherever a defect crosses the object, which can give an e1-type + additive term. + insights: [mask_metacal_acts_on_whole_stamp, mask_bad_column_symmetrize] + raw: + label: "No fill: raw defect values (BLEND_HANDLING = uberseg)" + description: >- + Masked pixels keep weight 0, but their raw values (bad columns, + saturation, bleeds, cosmic rays, bright-star light) stay in the + image that metacal deconvolves, shears and reconvolves. + excluded: true + excluded_reason: >- + Metacal acts on every pixel regardless of weight, so raw defects + leak into the weighted pixels. No published metacal pipeline does + this: DES dropped, model-filled or interpolated defects. The code + reaches this option only as a side effect of BLEND_HANDLING = + uberseg. + insights: [mask_metacal_acts_on_whole_stamp, mask_des_defect_practice] + symmetrized_noise: + label: 90-degree-symmetrized mask, then noise fill + description: >- + Not implemented. OR the defect mask with its 90-degree rotation + about the stamp centre, zero the weight on the union, and + noise-fill the union as noisefill does. This mirrors the DES + mask symmetrization (Y1/Y3 ngmixer symmetrize_weight; Y6 + symmetrize_masking) and cancels the leading column-aligned + additive term. It roughly doubles the masked area, so the epoch + cut must be applied after symmetrizing. ShapePipe stamps are + square, so rot90 is well defined. The noise image needs no change. + insights: [mask_bad_column_symmetrize, mask_des_defect_practice, mask_fixed_orientation] + interpolate: + label: Symmetrize, then interpolate image and noise image (DES Y6) + description: >- + Not implemented. Symmetrize as above, then fill the union by 2D + Clough-Tocher interpolation (scipy) of the image and, identically, + of the fixnoise noise image. This restores galaxy light across + narrow defects instead of leaving a hole. It is the DES Y6 and + Rubin metadetect practice. Poor for large holes (star masks), + which Y6 zeroes with apodized edges. + insights: [mask_interpolate_with_noise, mask_bad_column_symmetrize, mask_sharp_edges_ring] + model: + label: Symmetrize, then fill with the best-fit central model (DES Y3) + description: >- + Not implemented. Fill symmetrized defects with the PSF-convolved + best-fit model of the central object from a pre-metacal fit + (ngmix v1.3.9 replace_masked_pixels). The uberseg-only Y3 config + adds no noise (add_noise=False); the MOF-corrector config adds + it. This restores galaxy light; without noise it leaves + noise-free patches that the full-stamp fixnoise noise image does + not mirror. + insights: [mask_bad_column_symmetrize, mask_des_defect_practice] + blend_handling: + label: Neighbour treatment before metacal + rationale: >- + This decision covers how pixels shared with a neighbour are treated, + and only that; defect fill is the separate decision above. The two + are coupled through BLEND_HANDLING: noisefill (the default; the + committed config sets no key) leaves neighbours fully weighted and + untouched, while uberseg zeroes the weight of pixels nearer a + neighbour's coadd segmentation footprint than the target's + (DILATE_NEIGHBOUR, default 1) and leaves the image untouched. + Official uberseg is weight-only: esheldon/meds get_uberseg returns a + weight map and never modifies the image (a nearest-segment-pixel + Voronoi split). DES Y1's fiducial metacal ran on uberseg-weighted + stamps with the raw neighbour light still in the image. The last Y3 + config does the same, though an earlier Y3 config subtracted MOF + neighbours first. DES Y6 does not mask neighbours before the shear + step and uses uberseg only as the weight of the fit after metacal. + None of the DES or Rubin metacal/metadetect pipelines noise-fills the + neighbour side. Leaving neighbour light raw is consistent with + metacal: it is real sky, and the artificial shear shears it along + with the target, as the real shear does. The reconvolution spreads it + slightly further across the Voronoi boundary than the PSF already + had. The known residual is about +2% m for uberseg-only against MOF + subtraction in DES Y1 simulations. Separately, Sheldon et al. 2020 + find that the blending bias of per-stamp metacal is dominated by + shear-dependent detection, which no pixel treatment fixes and which + DES Y3 calibrated with simulations. Noise-filling the neighbour side + would instead cut the target's own light along an unsheared + boundary, a sharp edge that rings in the FFTs. The recommended + comparison arm is uberseg (weight-only), with defect_fill held equal + across arms. + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::uberseg_weight; + src/shapepipe/modules/ngmix_package/ngmix.py::prepare_ngmix_weights; + src/shapepipe/modules/ngmix_runner.py::ngmix_runner. + default: none + options: + none: + label: No neighbour treatment (BLEND_HANDLING = noisefill) + description: >- + Neighbour pixels keep their full weight and image values. The fit + sees all neighbour light in the stamp, which is the configuration + Jarvis et al. 2016 found biased toward neighbours (worse than + their segmentation-only mask). It is a candidate cause of the + FLAGS=2 B-modes investigated in #814. + insights: [mask_uberseg_neighbour_bias] + uberseg: + label: UberSeg, weight-only + description: >- + Zero the weight of pixels nearer a neighbour footprint than the + target, as DES Y1/Y3 did. The image is untouched there, so the + neighbour's light is sheared coherently with the target. Needs + the coadd segmentation stamp (SEG_VIGNET_PATH). DILATE_NEIGHBOUR=1 + absorbs the coadd-vs-epoch overlay offset, because ShapePipe + reuses one coadd seg stamp for every epoch where MEDS reprojects + it. + insights: [mask_uberseg_weight_only, mask_uberseg_neighbour_bias, mask_des_y1_uberseg_only, mask_blend_bias_detection] + uberseg_fill: + label: UberSeg plus noise fill of neighbour-side pixels + description: >- + Not implemented. Additionally replace the neighbour-side pixels + with noise. This removes neighbour light from metacal, but cuts + the target's own wings along an unsheared Voronoi boundary: a + sharp edge (FFT ringing) that does not respond to the artificial + shear as sky does. No published pipeline does this. Useful only as + a diagnostic arm; a smooth (apodized) taper would be the less + damaging variant. + insights: [mask_sharp_edges_ring, mask_blend_bias_detection] + mof_subtract: + label: Subtract neighbour models, then UberSeg (DES Y1 alternative) + description: >- + Not implemented. Subtract multi-object-fit models of the + neighbours from the stamp before metacal and keep uberseg + weights. This removed the ~2% uberseg-only bias in DES Y1 + simulations, although Sheldon et al. 2020 found similar blend + biases with and without MOF. + insights: [mask_des_y1_uberseg_only, mask_blend_bias_detection] + epoch_masked_fraction_cut: + label: Per-epoch masked-fraction cut + rationale: >- + [HARDCODED] an epoch whose stamp has more than 1/3 of its pixels + flagged (any nonzero flag bit, including the tile-coverage bit 2**10 + set where the tile vignet is off-image) is dropped from the + multi-epoch fit; an object with no surviving epoch has no shape. DES + was stricter. Y1 rejected any epoch with a masked or zero-weight + pixel, and any whose central 4-pixel region was masked. Y3 cut at 10% + of raw zero-weight pixels. Y6 dropped images more than 10% missing + and cut objects at mfrac < 0.1, which its simulations show avoids + calibration bias. Sheldon & Huff 2017 recommend dropping problematic + epochs when many are available. The cut interacts with defect_fill: + symmetrizing roughly doubles the masked fraction, so the cut should + be applied after symmetrizing. UNIONS has fewer epochs than DES, so + the cost in effective number density has to be measured, not + assumed. A central-region veto (drop the epoch if a defect lies + within a few pixels of the centre) is a cheap refinement. + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_postage_stamps. + default: one_third + options: + one_third: + label: 1/3 of the stamp flagged + description: >- + Drop an epoch only if more than 1/3 of the stamp pixels carry a + nonzero flag. + ten_percent: + label: 10% (DES Y3 / Y6) + description: >- + Not implemented. Drop an epoch if more than 10% of the + (symmetrized) stamp is masked, matching DES Y3 + max_zero_weight_frac and Y6 max_masked_fraction. + insights: [mask_multi_epoch_drop] + any_masked: + label: Any masked pixel (DES Y1) + description: >- + Not implemented. Drop an epoch if any stamp pixel is masked, so + that no fill is ever needed. DES could afford this with about 10 + epochs per band; UNIONS likely cannot. + insights: [mask_multi_epoch_drop, mask_des_defect_practice] prior_insights: guinot22_gaussian_model: claim: >- @@ -1518,153 +1381,313 @@ analyses: quote: exact: 'This method creates four images used for the calibration, and one for the measurement.' location: { page: 6 } - - # ═════════════════════════════════════════════════════════════════════════ - psf_diagnostics: - description: >- - The PSF-fidelity diagnostic chain: merge the held-out (20%) validation - stars into one catalogue, bin PSF shapes and residuals over the focal - plane. DORMANT in the committed snakemake chain — no rule runs it. - Modules: src/shapepipe/modules/merge_starcat_runner.py (+ per-model - merge classes in merge_starcat.py), mccd_plots_runner.py (serves both - PSF models despite its name). Boundary note: the module docstring - (mccd_package/__init__.py:157) still advertises rho-statistics plots, - but no rho/treecorr code remains in shapepipe — rho/tau statistics - moved downstream to sp_validation (rho_tau.py via - shear_psf_leakage.RhoStat/TauStat; treecorr min_sep/max_sep/nbins, - jackknife patch numbers hardcoded per survey with a "TODO to yaml"). - The diagnostic decision chain thus crosses the repo boundary into - sp_validation. [LINT] the module docstring still advertises rho - statistics this package no longer computes. - inputs: - - id: validation_star_cats - type: data - source: per-CCD star_split_ratio_20 catalogues with PSF/star HSM shapes (psfex_interp VALIDATION mode) - outputs: - - id: merged_star_catalogue - type: data - format: fits - description: >- - One full_starcat over the run — the input rho/tau statistics and - leakage diagnostics consume downstream. - decisions: [starcat_merge_source, meanshape_binning] - decisions: - starcat_merge_source: - label: Which PSF model's validation output feeds the merged star catalogue - rationale: >- - merge_starcat_runner dispatches on PSF_MODEL in {psfex, mccd, - setools} to per-model merge classes (different HDU conventions: - mccd HDU 1, psfex/setools HDU 2). Follows star_selection_psf. - psf_modelling_software; recorded separately because the merge can - also consume raw setools output (pre-model diagnostics). - Anchor: src/shapepipe/modules/merge_starcat_runner.py::merge_starcat_runner; - src/shapepipe/modules/merge_starcat_package/merge_starcat.py. - default: psfex - options: - psfex: - label: PSFEx validation catalogues (HDU 2) - insights: [guinot22_psfex_for_v1] - mccd: { label: MCCD validation catalogues (HDU 1) } - setools: { label: Raw setools star catalogues } - meanshape_binning: - label: Focal-plane mean-shape binning and outlier handling - rationale: >- - PSF ellipticity/size and residuals binned per CCD over the focal - plane: X_GRID=5, Y_GRID=10 bins per CCD, colour scales MAX_E=0.05, - MAX_DE=0.005, REMOVE_OUTLIERS=False (example/cfis config; no - committed workflow config exists). Grid resolution sets which - spatial PSF-residual structure is visible; outlier removal changes - what the diagnostic hides. - Published description (Guinot+22 p.8): focal-plane residual maps - averaged in ~20 arcsec cells per CCD; current code: no committed - workflow config picks a grid at all — example/cfis carries both the - 5x10 grid recorded here (~77x86 arcsec) and, in - config_valjoint_Pl_mccd.ini, a 20x40 grid (~19x22 arcsec) that - reproduces the paper. This is therefore an undetermined knob with - two committed precedents rather than a value that drifted. The - uniform REMOVE_OUTLIERS=False is a genuine paper-silence gap. - Anchor: example/cfis/config_MsPl_psfex.ini#MCCD_PLOTS_RUNNER.X_GRID; - example/cfis/config_MsPl_psfex.ini#MCCD_PLOTS_RUNNER.REMOVE_OUTLIERS; - src/shapepipe/modules/mccd_plots_runner.py; - src/shapepipe/modules/mccd_package/mccd_plot_utilities.py::plot_meanshapes. - default: grid_5x10 - options: - grid_5x10: { label: "5x10 per CCD, outliers kept" } - prior_insights: - guinot22_psfex_for_v1: + mask_uberseg_weight_only: + label: UberSeg is a weight-map operation (Jarvis 2016) claim: >- - PSFEx is the PSF model behind the published UNIONS v1 catalogue, so - the merged validation star catalogue and its diagnostics are fed by - PSFEx output. - created_at: "2022-04-01T00:00:00Z" + UberSeg, introduced for DES SV, zeroes the WEIGHT of pixels that + belong to another object's coadd segmentation footprint or lie + closer to another object than to the target; MEDS weights are also + zeroed wherever a mask flag is set. The image is not modified: in + the forward-model fits it was designed for, a zero-weight pixel + drops out of the likelihood exactly. esheldon/meds get_uberseg + matches, which returns a weight map. + created_at: "2026-09-26T00:00:00Z" + derived: true evidence: - - id: ev_guinot22_psfex_v1 - doi: "10.48550/arXiv.2204.04798" + - id: ev1 + doi: 10.48550/arXiv.1507.05603 + version: 3 quote: - exact: 'We make use of the PSFEx software package' - location: { page: 4 } - - # ═════════════════════════════════════════════════════════════════════════ - survey_geometry: - description: >- - Effective survey area and mask geometry for two-point estimators. - DORMANT — no committed workflow rule. Module: - src/shapepipe/modules/random_cat_package/random_cat.py (+ runner): - uniform randoms over each tile rejected against the pipeline mask, - yielding effective area (overlap- and mask-corrected) and optionally - the mask itself as a HEALPix map (save_as_healpix). This is the - in-repo ancestor of the planned healsparse external-mask rework — whichever way that rework lands, this - sub-analysis is where its geometry decisions belong. Bitrot risk: - healpy is imported but absent from pyproject.toml dependencies. - inputs: - - id: tile_masks - type: data - source: per-tile pipeline flag maps + final catalogues - outputs: - - id: random_catalogue - type: data - format: fits - description: Per-tile random catalogue + effective area (+ optional HEALPix mask). - decisions: [random_sampling, healpix_mask_export] - decisions: - random_sampling: - label: Random-point density for area estimation - rationale: >- - N_RANDOM=50000 with DENSITY=True (per square degree; False = total - per tile) in the example config; no committed workflow value. - Sampling density sets the Monte Carlo noise floor on effective - area, which propagates to two-point normalisation. - Anchor: example/cfis/config_Rc.ini#RANDOM_CAT_RUNNER.N_RANDOM; - example/cfis/config_Rc.ini#RANDOM_CAT_RUNNER.DENSITY; - src/shapepipe/modules/random_cat_runner.py::random_cat_runner. - default: per_sqdeg_50k - options: - per_sqdeg_50k: { label: "50000 per sq deg" } - healpix_mask_export: - label: HEALPix export resolution for the pipeline mask - rationale: >- - SAVE_MASK_AS_HEALPIX=True, HEALPIX_OUT_NSIDE=1024 (~3.4 arcmin - pixels) in the example config — coarser than the arcsecond-scale - mask features it rasterises; the resolution choice decides what the - exported mask can represent. Supersession candidate under - masking-unification (healsparse). - Anchor: example/cfis/config_Rc.ini#RANDOM_CAT_RUNNER.SAVE_MASK_AS_HEALPIX; - example/cfis/config_Rc.ini#RANDOM_CAT_RUNNER.HEALPIX_OUT_NSIDE; - src/shapepipe/modules/random_cat_package/random_cat.py::RandomCat.save_as_healpix. - default: nside_1024 - options: - nside_1024: { label: nside 1024 } + exact: We then set pixels in the weight map to zero if they were either associated with other objects in the segmentation map or were closer to any other object than to the object of interest. + - id: ev2 + doi: 10.48550/arXiv.1507.05603 + version: 3 + quote: + exact: are both of the model-fitting variety + mask_uberseg_neighbour_bias: + label: Neighbour light biases shapes toward neighbours (Jarvis 2016) + claim: >- + With a plain segmentation-map mask, light from a bright neighbour + just outside its footprint entered the fit and biased the shape + toward the neighbour; UberSeg made that bias undetectable in + end-to-end simulations. ShapePipe's default (no neighbour treatment) + masks less than even the plain segmentation map. + created_at: "2026-09-26T00:00:00Z" + derived: false + evidence: + - id: ev1 + doi: 10.48550/arXiv.1507.05603 + version: 3 + quote: + exact: when using ordinary segmentation maps we found a significant bias of the galaxy shape in the direction toward neighbors. + - id: ev2 + doi: 10.48550/arXiv.1507.05603 + version: 3 + quote: + exact: we no longer detected any systematic bias in the shape estimates due to unmasked flux from neighboring objects + mask_metacal_acts_on_whole_stamp: + label: Metacal transforms every stamp pixel, weights unseen + claim: >- + Metacal builds an interpolated image of the whole postage stamp, + deconvolves, shears and reconvolves it; its Fourier transforms + cannot accommodate missing data. So the image values of zero-weight + pixels feed the sheared images. The method papers leave masking + unaddressed. ngmix implements exactly this: an InterpolatedImage of + obs.image, with the weights copied through. + created_at: "2026-09-26T00:00:00Z" + derived: true + evidence: + - id: ev1 + doi: 10.48550/arXiv.1702.02600 + version: 1 + quote: + exact: For each galaxy and PSF postage stamp, we first create an + - id: ev2 + doi: 10.48550/arXiv.1702.02600 + version: 1 + quote: + exact: We have made no attempt to deal with the effects of masked pixels or blending + - id: ev3 + doi: 10.48550/arXiv.1702.02601 + version: 2 + quote: + exact: convolutions cannot accommodate missing data + mask_bad_column_symmetrize: + label: Filled bad columns give additive e1; symmetrize the mask + claim: >- + Sheldon & Huff 2017 found that filling bad columns (with the + best-fit model, not even noise) gave a large additive e1 bias and a + few-per-mille multiplicative bias. Both vanished when a compensating + column rotated by 90 degrees about the stamp centre was added. + Sheldon et al. 2020 recommend the same compensating mask, and DES Y6 + OR-s each bad-pixel mask with its 90-degree rotation, as previous + DES pipelines did, to cancel additive biases. + created_at: "2026-09-26T00:00:00Z" + derived: true + evidence: + - id: ev1 + doi: 10.48550/arXiv.1702.02601 + version: 2 + quote: + exact: as well as a multiplicative bias of a few parts in a thousand + - id: ev2 + doi: 10.48550/arXiv.1702.02601 + version: 2 + quote: + exact: at 90 degree rotation about the center of the image, restoring symmetry to the image + - id: ev3 + doi: 10.48550/arXiv.1911.02505 + version: 2 + quote: + exact: An additional compensating mask, rotated at right angles to the real mask, can be used to restore symmetry to the image + - id: ev4 + doi: 10.48550/arXiv.2501.05665 + version: 2 + quote: + exact: we rotate the bad pixel mask by 90 degrees and apply it via a logical + - id: ev5 + doi: 10.48550/arXiv.2501.05665 + version: 2 + quote: + exact: This process helps to cancel additive biases in the final shape measurement. + mask_des_defect_practice: + label: DES never passed raw defects through metacal + claim: >- + Every DES metacal generation removed defects from the image before + metacal: Y1 dropped any epoch with a masked pixel, Y3 filled + symmetrized defects with the best-fit central model (ngmix-y1-config + / ngmix-y3-config, see the defect_fill rationale), and Y6 + interpolated symmetrized defects following "previous DES shear + measurement pipelines". None left raw defect values in a zero-weight + pixel, which is what ShapePipe's uberseg path does today. + created_at: "2026-09-26T00:00:00Z" + derived: true + evidence: + - id: ev1 + doi: 10.48550/arXiv.2501.05665 + version: 2 + quote: + exact: Following previous DES shear measurement pipelines + - id: ev2 + doi: 10.48550/arXiv.2501.05665 + version: 2 + quote: + exact: Then, we apply a two-dimensional Clough-Tocher interpolation + mask_interpolate_with_noise: + label: Interpolate defects, and the noise image identically + claim: >- + The metadetect papers interpolate masked regions (bad columns, + cosmic rays, saturation) before the shear step, taking care not to + create a spurious shear, and pass the noise image used for metacal's + correlated-noise correction through the same interpolation. + created_at: "2026-09-26T00:00:00Z" + derived: true + evidence: + - id: ev1 + doi: 10.48550/arXiv.1911.02505 + version: 2 + quote: + exact: The FFT does not permit missing data, so the masked regions must be interpolated in some way. + - id: ev2 + doi: 10.48550/arXiv.1911.02505 + version: 2 + quote: + exact: so the noise field used for correcting correlated noise effects must also be propagated through the same coadding and interpolation + - id: ev3 + doi: 10.48550/arXiv.2303.03947 + version: 2 + quote: + exact: Before warping, we interpolated the simulated artifacts in each image such as cosmic rays, bad columns, and saturated pixels. + - id: ev4 + doi: 10.48550/arXiv.2303.03947 + version: 2 + quote: + exact: we must also run the noise image through the same procedures + mask_sharp_edges_ring: + label: Sharp mask edges ring through metacal; apodize large masks + claim: >- + Masks with sharp edges cause ringing in the metacal FFTs, so + bright-star regions are set to zero with an apodized (smoothly + tapered) edge, and the noise image gets the same masking. The same + reasoning argues against a hard noise fill cut along a neighbour's + Voronoi boundary. + created_at: "2026-09-26T00:00:00Z" + derived: true + evidence: + - id: ev1 + doi: 10.48550/arXiv.2303.03947 + version: 2 + quote: + exact: Masks with sharp features can cause ringing in the FFTs used by the + - id: ev2 + doi: 10.48550/arXiv.2501.05665 + version: 2 + quote: + exact: we further mask bright stars with apodization to avoid FFT artifacts during the deconvolution, shearing, and reconvolution processes + - id: ev3 + doi: 10.48550/arXiv.2501.05665 + version: 2 + quote: + exact: we also applied the same masking and apodization to the + mask_multi_epoch_drop: + label: Drop heavily masked epochs (10% in DES) + claim: >- + With many dithered epochs, problematic data can simply be dropped + (Sheldon & Huff 2017). DES Y6 drops input images more than 10% + missing and cuts objects at masked fraction mfrac < 0.1, which its + image simulations show is enough to avoid shear calibration bias. + ShapePipe's cut is 1/3. + created_at: "2026-09-26T00:00:00Z" + derived: true + evidence: + - id: ev1 + doi: 10.48550/arXiv.1702.02601 + version: 2 + quote: + exact: data deemed problematic can simply be left out of the fit + - id: ev2 + doi: 10.48550/arXiv.2501.05665 + version: 2 + quote: + exact: Images with a missing pixel fraction higher than 10 + - id: ev3 + doi: 10.48550/arXiv.2501.05665 + version: 2 + quote: + exact: is sufficient not to introduce shear calibration biases + mask_des_y1_uberseg_only: + label: 'DES Y1 metacal: uberseg-only fiducial, ~2% m residual' + claim: >- + DES Y1 metacal's fiducial catalogue handled neighbours with uberseg + only (raw neighbour light in the image). In dense deblending + simulations it carried m of about +2% (2.18 +/- 0.16 per cent at S/N + > 10), removed by subtracting MOF models of the neighbours. In data + the relative uberseg-vs-MOF m was 0.023 +/- 0.009. + created_at: "2026-09-26T00:00:00Z" + derived: false + evidence: + - id: ev1 + doi: 10.48550/arXiv.1708.01533 + version: 2 + quote: + exact: masks pixels close to neighbours rather than assigning a fraction of the light in each pixel to them + - id: ev2 + doi: 10.48550/arXiv.1708.01533 + version: 2 + quote: + exact: We detect no bias when subtracting the light from neighbours. + mask_blend_bias_detection: + label: Stamp-metacal blend bias is mostly shear-dependent detection + claim: >- + Sheldon et al. 2020 find that the few-percent blending bias of + per-stamp metacal comes from shear-dependent detection, not from + blended light itself: it is similar with and without neighbour + subtraction, and in metacal the space between objects is sheared + coherently. DES Y3 kept Y1's approach (Gatti et al. list no change + to neighbour or mask handling) and calibrated the resulting 2-3% + with image simulations. + created_at: "2026-09-26T00:00:00Z" + derived: true + evidence: + - id: ev1 + doi: 10.48550/arXiv.1911.02505 + version: 2 + quote: + exact: We show that this bias is not due to blending itself, but rather to shear-dependent object detection. + - id: ev2 + doi: 10.48550/arXiv.1911.02505 + version: 2 + quote: + exact: we see similar biases when no deblending is performed + - id: ev3 + doi: 10.48550/arXiv.1911.02505 + version: 2 + quote: + exact: the space between objects is sheared coherently + - id: ev4 + doi: 10.48550/arXiv.2011.03408 + version: 3 + quote: + exact: shape catalogue differs from DES Y1 in the following ways + - id: ev5 + doi: 10.48550/arXiv.2011.03408 + version: 3 + quote: + exact: which affects the shear estimates when objects are blended + mask_fixed_orientation: + label: Fixed-orientation cameras make mask anisotropy coherent + claim: >- + Masks have a preferred direction (columns, bleeds, spikes). On a + camera with a fixed sky orientation (DECam, and CFHT/MegaCam on its + equatorial mount), mask-induced shape errors therefore add up + coherently; with Rubin's camera rotation, unmasked trails averaged + away. DES Y3 has an unexplained mean e1 of 3.5e-4 that its + simulations, which include the real bad-pixel masks, do not + reproduce; whether masks cause it is not established. + created_at: "2026-09-26T00:00:00Z" + derived: true + evidence: + - id: ev1 + doi: 10.48550/arXiv.1708.01533 + version: 2 + quote: + exact: The DES focal plane does not rotate, so these effects always correspond to the same orientation in sky coordinates. + - id: ev2 + doi: 10.48550/arXiv.2303.03947 + version: 2 + quote: + exact: We find that these unmasked trails do not cause a shear bias, which we attribute to the camera rotations that randomize the direction of the trail on the sky. + - id: ev3 + doi: 10.48550/arXiv.2011.03408 + version: 3 + quote: + exact: Our shear catalogue is characterized by a non-null mean shear in one of the two components, whose origin is unknown. # ═════════════════════════════════════════════════════════════════════════ catalogue_assembly: description: >- - Final per-tile catalogue: merging shape chunks, attaching photometry - and PSF diagnostics, classification, sentinels. Modules: - src/shapepipe/modules/make_cat_package/make_cat.py (+ runner), - merge_sep_cats.py, vignetmaker_package (stamps, see top-level - postage_stamp_size), find_exposures_package (epoch list from tile - HISTORY cards). Configs: config_tile_Mc.ini, config_tile_PiViVi.ini, - final_cat.param. + The per-tile science catalogue: shape chunks merged and joined to the + detection catalogue and PSF quantities. inputs: - id: ngmix_chunks type: data @@ -1674,68 +1697,43 @@ analyses: type: data format: fits description: The assembled per-tile science catalogue (final_cat family). + inputs: [ngmix_chunks] decisions: [star_galaxy_classification, tile_overlap_handling, - column_selection, failure_sentinels, postproc_run_provenance, - shape_catalogue_shortfall_guard] + failure_sentinels] decisions: star_galaxy_classification: - label: Star/galaxy separation — deferred out of the pipeline + label: Star/galaxy separation deferred out of the pipeline rationale: >- - Production sets SM_DO_CLASSIFICATION=False (config_tile_Mc.ini) and - wires no spread-model input: SPREAD_MODEL/SPREADERR_MODEL are - sentinel 99, no SPREAD_CLASS column, and the SPREAD_* entries in - final_cat.param are commented out. The dormant machinery - (make_cat.py::save_sm_data) classifies on class = sm + 2*sm_err - with star |class|<0.003, galaxy class>0.01 — thresholds hardcoded - in the function signature. Reactivation is a 4-line config diff - (run spread_model_runner after psfex_interp+vignetmaker, add its - output to make_cat inputs, flip the switch — the exact diff - between example/cfis/config_make_cat_psfex.ini and _nosm.ini; the - defunct tile wiring config_tile_PiViSmVi.ini is the reference). - Separation therefore happens entirely downstream (sp_validation); - the pipeline ships everything. Rationale for deferring not - recorded. - Published description (Guinot+22 p.5-6): galaxies are selected - inside the pipeline with the spread model, at s + 2*sigma_s > - 0.0003 together with s > 0 and 20 < MAG_AUTO < 26; current code: - classification is disabled entirely and separation deferred to - sp_validation, with the dormant make_cat thresholds putting the - like-for-like galaxy boundary at 0.01, some thirty times the - published cut (0.003 is its separate star-side bound). Even - reactivated, the code implements only the spread-model test — the - paper's companion cuts have no in-pipeline counterpart. + SM_DO_CLASSIFICATION=False and no spread-model input is wired, so the + catalogue ships every object and separation happens downstream. The + dormant make_cat classifier uses class = sm + 2 sm_err with stars at + |class| < 0.003 and galaxies at class > 0.01, thresholds fixed in the + function signature. Guinot+22 selected galaxies in the pipeline at s + + 2 sigma_s > 0.0003 together with s > 0 and 20 < MAG_AUTO < 26; the + dormant code implements only the spread-model test, at a galaxy + boundary thirty times higher. Anchor: workflow/config/cfis/config_tile_Mc.ini#MAKE_CAT_RUNNER.SM_DO_CLASSIFICATION; src/shapepipe/modules/make_cat_package/make_cat.py::save_sm_data; - workflow/config/cfis/final_cat.param#SPREAD_CLASS; - example/cfis/config_make_cat_psfex_nosm.ini; - example/cfis/defunct/config_tile_PiViSmVi.ini. + workflow/config/cfis/final_cat.param#SPREAD_CLASS. default: deferred_downstream options: deferred_downstream: label: No in-pipeline classification; catalogue ships all objects spread_model_inline: - label: spread_model classification in make_cat (0.003/0.01) - excluded: true - excluded_reason: >- - Machinery present but unwired in production; reactivating it - changes which objects downstream sees as galaxies. + label: spread_model classification in make_cat (0.003 / 0.01) + description: >- + Not wired in the workflow; needs spread_model_runner after the + PSF interpolation and its output added to make_cat's inputs. tile_overlap_handling: - label: Tile-overlap duplicates — neither removed nor flagged + label: Tile-overlap duplicates neither removed nor flagged rationale: >- - Adjacent tiles overlap; objects in the overlap are measured in - both. make_cat attaches only TILE_ID (parsed from the sexcat - filename); no unique-object rule, no overlap flag. [LINT] the - documented config key TILE_LIST ("used to flag objects in areas of - overlap between tiles", in the make_cat package docstring) is - implemented nowhere — grep across src/ and workflow/ finds only - the docstring. VERIFIED downstream: sp_validation dedups at - classification time (galaxy.py::classification_galaxy_overlap_ra_dec - RA/Dec cuts to non-overlapping tile areas, and the WCS-based - mask_overlap variant; applied as cut_overlap in - classification_galaxy_base) — so this is today's division of labour, - and the pipeline's contract is "ship duplicates, TILE_ID is the - handle"; flagging overlaps in the catalogue stays an open option. The dead TILE_LIST docstring remains a lint. + Adjacent tiles overlap, and objects in the overlap are measured in + both. [HARDCODED] make_cat attaches only TILE_ID; there is no unique-object rule + and no overlap flag, and sp_validation removes duplicates by cutting + each tile to its non-overlapping area. [LINT] the make_cat package + docstring documents a TILE_LIST key that flags overlap objects; + nothing implements it. Anchor: src/shapepipe/modules/make_cat_package/make_cat.py::save_sextractor_data; src/shapepipe/modules/make_cat_package/__init__.py. default: no_dedup_in_pipeline @@ -1743,95 +1741,30 @@ analyses: no_dedup_in_pipeline: label: Ship duplicates; TILE_ID is the only handle overlap_flagging: - label: Implement the documented TILE_LIST overlap flag + label: Flag overlap objects in the catalogue + description: Not implemented. nearest_tile_centre: label: Keep each object only in its nearest tile excluded: true excluded_reason: >- Requires cross-tile coordination at assembly time, which the - per-tile DAG deliberately avoids; dedup belongs downstream if - anywhere. - column_selection: - label: Which columns survive into the science catalogue - rationale: >- - final_cat.param: positions XWIN/YWIN_WORLD, TILE_ID, flags - (FLAGS, IMAFLAGS_ISO, NGMIX_MCAL_FLAGS), PSF ellipticity from - PSF_ORIG only, all five metacal branches of G1/G2/T/FLUX/FLAGS, - but shear errors only for NOSHEAR (sheared-branch error columns - commented out — downstream response-weighted estimators cannot - propagate per-branch errors), SExtractor photometry (MAG_AUTO, - FLUX_APER, FLUX_RADIUS, SNR_WIN, FWHM_*), N_EPOCH/NGMIX_N_EPOCH, - NGMIX_MOM_FAIL. Doesn't change membership, but determines which - numbers exist for downstream cuts and calibration. Note - final_cat.param is read by scripts/python/create_final_cat.py in - post-processing, outside the per-tile DAG. [LINT] see detection: - IMAFLAGS_ISO is requested but never reaches the merged catalogue. - Anchor: workflow/config/cfis/final_cat.param; - scripts/python/create_final_cat.py. - default: committed_param_list - options: - committed_param_list: { label: The committed final_cat.param set } + per-tile DAG deliberately avoids. failure_sentinels: - label: Objects without shape measurements kept, with sentinel values [HARDCODED] + label: Objects without shape measurements kept, with sentinel values rationale: >- - Unmatched objects (no ngmix row) stay in the catalogue with - sentinels: sizes/fluxes/flags 0, error fluxes/mags -1, - ellipticities -10, T_ERR 1e30. The sentinel choice defines what a - downstream cut must exclude — a naive G1 > -1 cut silently changes - the sample. Flag-0-for-failure is the sharpest hazard: a failed - object's NGMIX_MCAL_FLAGS reads as success. Rationale not recorded. + [HARDCODED] detections with no ngmix row stay in the catalogue with + sentinels: sizes, fluxes, magnitudes and flags 0, flux and magnitude + errors -1, ellipticities -10, size errors 1e30. The sentinels define + what a downstream cut must exclude: a failed object's + NGMIX_MCAL_FLAGS reads 0, the success value, so a cut on flags alone + keeps it. Anchor: src/shapepipe/modules/make_cat_package/make_cat.py::SaveCatalogue._save_ngmix_data. default: sentinel_values options: - sentinel_values: { label: "Keep with sentinels (flags 0, e -10, T_ERR 1e30)" } + sentinel_values: + label: Keep with sentinels (flags 0, e -10, T_ERR 1e30) drop_unmatched: label: Drop objects without shapes excluded: true excluded_reason: >- - Loses the photometry-only population and hides attrition from - the completeness accounting. - postproc_run_provenance: - label: Post-proc run selection — newest mtime wins, merged patches never refresh - rationale: >- - create_final_cat picks each tile's make_cat run by newest - directory mtime (skipping runs without an output FITS), and the - merged patch catalogue is incremental: a tile already present is - never refreshed — a reprocessed tile reaches the patch only via - an explicit single-ID remove+add. mtime is filesystem state, not - provenance: a touched old run can outrank a newer one. Outside - the per-tile DAG (scripts/, not workflow/). Anchor: - scripts/python/create_final_cat.py::process. - default: newest_mtime_incremental - options: - newest_mtime_incremental: { label: "Newest mtime, incremental merge, manual refresh" } - shape_catalogue_shortfall_guard: - label: 10% shape-shortfall guard — logged, not enforced [HARDCODED] - rationale: >- - If the merged shape catalogue covers <10% of the detection - catalogue, make_cat logs an error but the enforcement (return + - raise) is commented out in both the module and its runner: a tile - whose shapes are 90% missing from a processing error is written and - looks normal. The comment distinguishes the two causes (measurement - failure = ok; premature merge = error) but not why enforcement is - off. Interacts with top-level per_unit_count_floor, which floors - tile_ngmix at 1 file and cannot see intra-file attrition. - Anchor: src/shapepipe/modules/make_cat_package/make_cat.py::SaveCatalogue._save_ngmix_data; - src/shapepipe/modules/make_cat_runner.py::make_cat_runner. - default: log_only - options: - log_only: { label: "Log the shortfall, write the tile anyway" } - enforce_10pct: - label: Fail the tile below 10% coverage - excluded: true - excluded_reason: >- - Was the coded intent, then disabled — reason unrecorded; - flagged as a question, not an endorsement. - -# ── Dormant science paths, surveyed but not yet recorded as sub-analyses ──── -# Candidates for future passes (each carries real scientific knobs): -# * External photometry match — match_external_package (TOLERANCE=0.3 -# arcsec against UNIONS ugriz; the external catalogue path is hardcoded -# to an IAP/candide location). -# * Image-simulation validation wiring — example/cfis_image_sims/: -# same chain over SKiLLS images with fake_psf substitution; bash-shaped, -# not yet ported to snakemake. + Loses the photometry-only population and hides attrition. diff --git a/universes/committed.yaml b/universes/committed.yaml index 5530226cf..b250eef95 100644 --- a/universes/committed.yaml +++ b/universes/committed.yaml @@ -1,30 +1,27 @@ id: committed -description: The committed configuration on feat/snakemake-orchestration. +description: The committed configuration on develop. decisions: - per_unit_count_floor: count_floor + per_unit_completeness: exact_counts postage_stamp_size: px_51 photometric_zeropoint: fixed_30_tiles_header_exposures - baseline_validation_criterion: statistical_parity analyses: masking: decisions: - star_catalogue_query: gsc_23_vizier - star_magnitude_definition: mean_finite_bands - bright_star_mask_geometry: megaprime_polygon_linear_scaling - deep_sky_object_masking: circles_no_padding - border_mask_width: px50_exposures_only - pixel_threshold_flags: stock_ww_thresholds - external_flag_usage: exposures_only + pixel_mask_source: instrument_flags_only + psf_star_mask_veto: instrument_flags_only + sky_mask_application: deferred_downstream detection: decisions: - detection_threshold_policy: thresh_1p5_minarea5_fwhm2px_filter - deblending_policy: mincont_5em4_tiles - background_model: manual_zero_tiles_auto_exposures - weighting_and_interpolation: map_weight_interp_all + detection_threshold_policy: megapipe_tiles + deblending_policy: megapipe_tiles + background_model: auto_megapipe_tiles + weight_map_usage: map_weight + zero_weight_interpolation: interp_all + spurious_detection_cleaning: clean_1 + blend_photometry_mask_type: correct + photometry_parameters: kron_25_35 detection_source_mode: sx_nomask_single_image epoch_membership_ccd_bounds: trimmed_bounds_33_2080 - photometry_parameters: kron_25_35 - cleaning_and_neighbour_masking: clean_1_correct preparation: decisions: astrometric_solution_source: delivered_headers @@ -32,39 +29,32 @@ analyses: epoch_provenance_from_tile_history: history_parse object_position_columns: xwin_windowed stamp_positioning_and_padding: round_and_zero_pad - epoch_flag_source: raw_flags star_selection_psf: decisions: star_selection_box: mode_centred_box + psf_train_validation_split: split_80_20_seeded psfex_candidate_vetting: builtin_defaults psf_modelling_software: psfex - psf_train_validation_split: split_80_20_seeded psf_model_complexity: pixel_basis_deg2_per_ccd psf_acceptance_thresholds: stars22_chi2_2 shape_measurement: decisions: ngmix_seed_mode: position_seed galaxy_model: gauss + fit_initialisation: prior_guess_t025_ntry5_2 fit_priors: gpriorba04_flat metacal_scheme: five_types_step001_fitgauss centroid_source: wcs - epoch_quality_and_weighting: third_masked_cut - noise_model: rms_vignet_weights - psf_epoch_loss_policy: record_and_delegate + epoch_flux_rescaling: fscale + psf_epoch_averaging: galaxy_weight_sum + galaxy_pixel_weights: rms_vignet_weights + psf_likelihood_noise: psf_noise_1em5 megacam_ccd_flip: megapipe_flip - psf_diagnostics: - decisions: - starcat_merge_source: psfex - meanshape_binning: grid_5x10 - survey_geometry: - decisions: - random_sampling: per_sqdeg_50k - healpix_mask_export: nside_1024 + defect_fill: noise + blend_handling: none + epoch_masked_fraction_cut: one_third catalogue_assembly: decisions: star_galaxy_classification: deferred_downstream tile_overlap_handling: no_dedup_in_pipeline - column_selection: committed_param_list failure_sentinels: sentinel_values - postproc_run_provenance: newest_mtime_incremental - shape_catalogue_shortfall_guard: log_only From a6eff5e6831a918141b4fafcb4c700d796e1727a Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 03:04:34 +0200 Subject: [PATCH 05/40] docs(claude): point the scientific-decisions section at the anchor test Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014bvNTrAmZxcfb1ee83ApPK --- CLAUDE.md | 47 ++++++++++++++++++----------------------------- 1 file changed, 18 insertions(+), 29 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index 56579f03c..b5140cac6 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -124,38 +124,27 @@ where the change lives — and *scientific* decisions in `astra.yaml`, below. `astra.yaml` at the repo root is the pipeline's decision record: every consequential scientific choice embedded in the code and the committed configs, -each with its rationale, the alternatives that were considered and why they were -rejected, and an anchor back to the code or config that implements it. -`universes/committed.yaml` pins the option this branch's configuration -selects for every decision. The format -is ASTRA; `uvx astra-tools@0.2.17 guide` is the briefing and -`uvx astra-tools@0.2.17 spec` the field reference. +with its rationale, its alternatives, and an anchor to the code or config that +implements it. `universes/committed.yaml` pins the option the committed +configuration selects for every decision. The format is ASTRA; +`uvx astra-tools@0.2.17 guide` is the briefing and `uvx astra-tools@0.2.17 spec` +the field reference. The file's header states its conventions (anchor grammar, +`[HARDCODED]`, `[LINT]`). + +Membership test: a different defensible choice would change which objects enter +the shear catalogue, or the numbers attached to them. Detection thresholds, +masking, star-selection cuts, PSF model degree, ngmix priors and seeding, flag +semantics, completeness gates: in. Workflow policy (manifests, chunking, +allocation, failure reporting, provenance) is out; it lives in the PR and in the +PRD, CosmoStat/shapepipe#848. **A scientific change is not finished until the record is.** When a change moves -what the pipeline measures, amend `astra.yaml` in the same PR — add the decision -if it is new, or edit its rationale, options and anchors if it moved — pin the -selected option in `universes/committed.yaml`, and say so in the PR description. -Purely technical changes (refactors, performance, packaging, I/O) leave it alone, -except where they move a value the record carries: the completeness floors in -`workflow/scripts/completeness.py` are orchestration code holding a scientific -decision. - -The membership test is whether *a different defensible choice would change which -objects enter the shear catalogue, or the numbers attached to them.* Detection -threshold and deblending contrast, masking geometry, star-selection cuts, PSF -model degree, ngmix priors and seeding, flag semantics, completeness floors — in. -Manifest sentinels, chunk sizes, allocation strategy, directory layout — out; -those live in the PR and the PRD. - -The file's own header states the conventions it follows. In short: every -rationale ends with a greppable `Anchor: path::symbol; path#SECTION.KEY` -sentence whose refs never cite line numbers; `[HARDCODED]` marks a scientific value -with no config exposure; `[LINT]` marks a place where the record and the code, or -the code and itself, disagree. Validate before committing: +what the pipeline measures, amend `astra.yaml` in the same PR (add the decision, +or edit its rationale, options and anchors), pin the selected option in +`universes/committed.yaml`, and say so in the PR description. The anchor test +`tests/unit/test_astra_anchors.py` runs in CI; a scientific change that breaks it +or leaves the record stale is unfinished. Before committing: ```bash uvx astra-tools@0.2.17 validate ``` - -The record was authored against this branch's workflow configs; entries marked -`[PENDING #NNN]` describe state that has not yet reached `develop`. From e3f6f4adbc838032b5bc2d81460c312670252a31 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 03:22:45 +0200 Subject: [PATCH 06/40] docs(astra): correct seven rationale claims against the code - epoch_provenance: names keep their trailing p; EXP_PREFIX is a no-op [LINT] - fit_initialisation: only the PSF guesser takes the catalogue flux; an exception in Ngmix.process drops the object with no row - star_galaxy_classification: thresholds come from SM_STAR_THRESH / SM_GAL_THRESH, which the committed config does not set - psf_train_validation_split: seeded from the unit's file number - stamp_positioning: an out-of-image stamp centre raises - object_position_columns: tile stamps are cut at XWIN_IMAGE (COORD=PIX) - mark the PSFEx built-in SAMPLE_* behaviour and the 33-px trim unverified - record the galaxy prior reused for PSF fits and the silent epoch drops before the 1/3 cut; carry stale completeness, exposure.smk, _mode and pixel-scale comments as [LINT] Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014bvNTrAmZxcfb1ee83ApPK --- astra.yaml | 290 ++++++++++++++++++++++++++++++++--------------------- 1 file changed, 173 insertions(+), 117 deletions(-) diff --git a/astra.yaml b/astra.yaml index 1b27db982..7ad57e61f 100644 --- a/astra.yaml +++ b/astra.yaml @@ -63,16 +63,21 @@ decisions: per_unit_completeness: label: Per-unit completeness gate rationale: >- - Every rule checks its products against a nominal per-runner count: a - runner below its count fails the unit (an exposure or a tile), so a - partial unit never enters the catalogue; the missing unit's objects do - not appear at all. The one tolerated shortfall is the exposure-side + Every rule that runs shapepipe_run checks its products against a + nominal per-runner count (tile_ngmix per chunk): a runner below + its count fails the unit (an exposure or a tile), so a partial + unit never enters the catalogue; the missing unit's objects do not + appear at all. The one tolerated shortfall is the exposure-side psfex_interp VALIDATION output, where a CCD whose model fails the - acceptance gate (star_selection_psf.psf_acceptance_thresholds) produces - nothing and the unit only warns; the MCCD chain, never run in a - campaign, warns on every runner. Science-path PSF rejection does not go - through this table: psfex_interp drops the epoch per object inside the - tile run. + acceptance gate (star_selection_psf.psf_acceptance_thresholds) + produces nothing and the unit only warns; the MCCD chain, never + run in a campaign, warns on every runner. Science-path PSF + rejection does not go through this table: psfex_interp drops the + epoch per object inside the tile run. [LINT] the completeness.py + docstring gives tile psfex_interp as its warn example, but that + runner is mandatory, and a workflow/rules/exposure.smk comment + says a floor's :warn tolerates setools rejecting a sparse CCD, but + setools has a mandatory count and no floor exists. Anchor: workflow/scripts/completeness.py::COMPLETENESS. default: exact_counts options: @@ -232,18 +237,20 @@ analyses: psf_star_mask_veto: label: PSF-star candidates rejected on instrument flags only rationale: >- - The star selection cuts IMAFLAGS_ISO == 0 and nothing else from the - masks. mask_query sits in the exposure module chain between - SExtractor and setools: when MASK_PATHS names maps it writes MASK_EXT - (0 clean, nonzero flagged; off-coverage counts as clean) onto each - CCD's catalogue. MASK_PATHS ships commented out, so the committed - module passes the catalogue through with no MASK_EXT column. The - intended map is the UNIONS star-body product (bit 2); halo bits 0 - and 1 are excluded because halos say nothing about whether a star is - a good PSF sample. Imposing the veto is one line per mask block in - star_selection.setools (MASK_EXT == 0). [LINT] the - workflow/rules/exposure.smk docstring names the column FLAG_EXT; the - code writes MASK_EXT. + The star selection cuts IMAFLAGS_ISO == 0 and nothing else + from the masks. mask_query sits in the exposure module + chain between SExtractor and setools: when MASK_PATHS + names maps it writes MASK_EXT (0 clean, nonzero flagged; + off-coverage counts as clean) onto each CCD's catalogue. + MASK_PATHS ships commented out, so the committed module + passes the catalogue through with no MASK_EXT column. The + intended map is the UNIONS star-body product (bit 2); halo + bits 0 and 1 are excluded because halos say nothing about + whether a star is a good PSF sample. Imposing the veto is + one line per mask block in star_selection.setools + (MASK_EXT == 0). [LINT] the workflow/rules/exposure.smk + docstring names the column FLAG_EXT and says setools cuts + on it; the code writes MASK_EXT and nothing cuts on it. Anchor: workflow/config/cfis/config_exp_psfex.ini; workflow/config/cfis/star_selection.setools#MASK:star_selection.IMAFLAGS_ISO; src/shapepipe/modules/mask_query_runner.py::mask_query_runner; @@ -504,11 +511,12 @@ analyses: epoch_membership_ccd_bounds: label: Which exposure CCDs an object belongs to (N_EPOCH) rationale: >- - CCD_SIZE = 33,2080,1,4612 with strict inequalities: the 33-px left - trim removes a CCD strip from epoch membership, and a WCS inversion - failure skips the CCD, lowering N_EPOCH. This sets how many exposures - enter each galaxy's multi-epoch fit. The trim is unexplained beyond - "number of pixels in a CCD". + CCD_SIZE = 33,2080,1,4612 with strict inequalities: + positions with x at or below 33 are outside the bounds, + and a WCS inversion failure skips the CCD, lowering + N_EPOCH. This sets how many exposures enter each galaxy's + multi-epoch fit. The x range spans 2048 px; whether the + excluded strip is prescan or science pixels is unverified. Anchor: workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.CCD_SIZE; src/shapepipe/modules/sextractor_package/sextractor_script.py::make_post_process; src/shapepipe/modules/sextractor_package/sextractor_script.py::ccd_candidate_mask. @@ -517,7 +525,7 @@ analyses: trimmed_bounds_33_2080: label: "x in (33,2080), y in (1,4612), strict" full_ccd: - label: Full 1-2048 x-range, inclusive bounds + label: Untrimmed x-range, inclusive bounds prior_insights: guinot22_stacked_detection: claim: >- @@ -590,11 +598,17 @@ analyses: epoch_provenance_from_tile_history: label: Epoch sets parsed from tile FITS HISTORY cards rationale: >- - A tile's contributing exposures are column 3 (COLNUM) of each HISTORY - line, with prefix p stripped and duplicates removed: the coadd's own - provenance is trusted as the epoch list. A mis-parse changes N_EPOCH - and which exposures are fit. + A tile's contributing exposures are the file names in + column 3 (COLNUM) of each HISTORY line, stripped of their + extension and deduplicated: the coadd's own provenance is + trusted as the epoch list. Names keep their trailing p + (2243881p); downstream code drops it when it needs the + bare exposure ID. A mis-parse changes N_EPOCH and which + exposures are fit. [LINT] EXP_PREFIX = p is passed to + removeprefix, which does nothing to names where p is a + suffix, so the key has no effect. Anchor: workflow/config/cfis/config_tile_Fe.ini#FIND_EXPOSURES_RUNNER.COLNUM; + workflow/config/cfis/config_tile_Fe.ini#FIND_EXPOSURES_RUNNER.EXP_PREFIX; src/shapepipe/modules/find_exposures_package/find_exposures.py::FindExposures.get_exposure_list. default: history_parse options: @@ -603,13 +617,18 @@ analyses: object_position_columns: label: Windowed centroids (XWIN/YWIN) define every position rationale: >- - PSF interpolation sites, tile and multi-epoch stamp centres, and the - catalogue position all use SExtractor's windowed centroid: - XWIN_WORLD/YWIN_WORLD on the tile side, XWIN_IMAGE/YWIN_IMAGE on - exposures. Windowed, isophotal and model centroids differ - systematically for blends and asymmetric galaxies, and the centroid - feeds the position seed and the centroid prior. - Anchor: workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS; + PSF interpolation sites, tile and multi-epoch stamp + centres, and the catalogue position all use SExtractor's + windowed centroid. Tile stamps are cut at + XWIN_IMAGE/YWIN_IMAGE in tile pixels (COORD = PIX); PSF + interpolation and multi-epoch stamps use + XWIN_WORLD/YWIN_WORLD; exposure-side PSF validation uses + XWIN_IMAGE/YWIN_IMAGE. Windowed, isophotal and model + centroids differ systematically for blends and asymmetric + galaxies, and the centroid feeds the position seed and the + centroid prior. + Anchor: workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.POSITION_PARAMS; + workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS; workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.POSITION_PARAMS; workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS. default: xwin_windowed @@ -619,14 +638,17 @@ analyses: stamp_positioning_and_padding: label: Nearest-pixel stamp extraction with zero padding rationale: >- - [HARDCODED] stamps are cut around the pixel nearest the object's - position, with no sub-pixel interpolation; the sub-pixel remainder is - stored as the stamp's OFFSET, which ngmix uses as the Jacobian origin - (shape_measurement.centroid_source), so extraction and centroid - prior share one rounding. Multi-epoch stamps take the position from - the tile world coordinate through the stored per-CCD WCS. Objects - whose stamp overruns an image edge are kept, with out-of-image pixels - zero-filled; there is no boundary rejection. + [HARDCODED] stamps are cut around the pixel nearest the + object's position, with no sub-pixel interpolation; the + sub-pixel remainder is stored as the stamp's OFFSET, which + ngmix uses as the Jacobian origin + (shape_measurement.centroid_source), so extraction and + centroid prior share one rounding. Multi-epoch stamps take + the position from the tile world coordinate through the + stored per-CCD WCS. Objects whose stamp overruns an image + edge are kept, with out-of-image pixels zero-filled. A + stamp centre that rounds outside the image raises, which + fails the vignet run for the whole tile. Anchor: src/shapepipe/modules/vignetmaker_package/vignetmaker.py::get_stamps; src/shapepipe/modules/vignetmaker_package/vignetmaker.py::VignetMaker._get_stamp_me. default: round_and_zero_pad @@ -683,17 +705,20 @@ analyses: star_selection_box: label: Stellar-locus selection, magnitude window and FWHM window around the mode rationale: >- - 18 < MAG_AUTO < 22, |FWHM - mode| <= 0.2 px, FLAGS == 0 and - IMAFLAGS_ISO == 0. The mode is computed on a preselection (MAG_AUTO < - 21, FWHM 0.3-1.5 arcsec at 0.187 arcsec/px) by an iterative - histogram-zoom estimator that falls back to the median below 20 - objects, so small-N behaviour changes selection on sparse CCDs. - PSFEx's own selection is off (SAMPLE_AUTOSELECT N), but its - compiled-in cuts still apply (psfex_candidate_vetting). [LINT] the - file's statistics log the FWHM cut as mode +- 0.1 px and its plot - uses 0.186 arcsec/px, while the applied cut is +- 0.2 px at 0.187. + 18 < MAG_AUTO < 22, |FWHM - mode| <= 0.2 px, FLAGS == 0 + and IMAFLAGS_ISO == 0. The mode is computed on a + preselection (MAG_AUTO < 21, FWHM 0.3-1.5 arcsec at 0.187 + arcsec/px) by an iterative histogram-zoom estimator that + falls back to the median below 20 objects, so small-N + behaviour changes selection on sparse CCDs. PSFEx's own + selection is off (SAMPLE_AUTOSELECT N); see + psfex_candidate_vetting for what PSFEx may still apply. + [LINT] the file's statistics log the FWHM cut as mode +- + 0.1 px and its plot uses 0.186 arcsec/px, while the + applied cut is +- 0.2 px at 0.187; the _mode docstring + puts the median fallback at 10 objects, the code at 20. Anchor: workflow/config/cfis/star_selection.setools#MASK:star_selection.MAG_AUTO; - workflow/config/cfis/star_selection.setools#MASK:preselect.FWHM_IMAGE; + workflow/config/cfis/star_selection.setools#MASK:preselect.MAG_AUTO; workflow/config/cfis/default.psfex#SAMPLE_AUTOSELECT; src/shapepipe/pipeline/str_handler.py::StrInterpreter._mode. default: mode_centred_box @@ -714,13 +739,15 @@ analyses: psf_train_validation_split: label: Seeded 80/20 star split, model fit vs held-out validation rationale: >- - RAND_SPLIT RATIO 20: the 80% sample fits the PSFEx model and feeds the - tile multi-epoch interpolation (ME_DOT_PSF_PATTERN); the 20% sample - is the independent residual diagnostic (psfex_interp VALIDATION - mode). The split trades training stars per CCD, which interacts with - the acceptance gate, against an independent residual test. It is - deterministic: a permutation seeded from the unit's file number, so - the PSF star sample is a pure function of the input catalogue. + RAND_SPLIT RATIO 20: the 80% sample fits the PSFEx model + and feeds the tile multi-epoch interpolation + (ME_DOT_PSF_PATTERN); the 20% sample is the independent + residual diagnostic (psfex_interp VALIDATION mode). The + split trades training stars per CCD, which interacts with + the acceptance gate, against an independent residual test. + It is deterministic: a permutation seeded from the digits + of the unit's file number, so a given CCD gets the same + split on every run. Anchor: workflow/config/cfis/star_selection.setools#RAND_SPLIT:star_split.RATIO; src/shapepipe/modules/setools_package/setools.py::SETools._make_rand_split; workflow/config/cfis/config_exp_psfex.ini#PSFEX_RUNNER.FILE_PATTERN; @@ -743,20 +770,21 @@ analyses: psfex_candidate_vetting: label: PSFEx built-in candidate cuts, unpinned rationale: >- - default.psfex sets SAMPLE_AUTOSELECT N but omits SAMPLE_MINSN, - SAMPLE_MAXELLIP, SAMPLE_FWHMRANGE and SAMPLE_VARIABILITY, so - [HARDCODED] PSFEx's compiled-in defaults apply (MINSN 20, MAXELLIP - 0.3, FWHMRANGE 2-10 px, VARIABILITY 0.2): a second star selection no - config records, which changes with the PSFEx version. - BADPIXEL_FILTER N and PSF_RECENTER N accept flagged star vignets - unfiltered and do not recentre candidates. + default.psfex sets SAMPLE_AUTOSELECT N but omits + SAMPLE_MINSN, SAMPLE_MAXELLIP, SAMPLE_FWHMRANGE and + SAMPLE_VARIABILITY, so [HARDCODED] whatever PSFEx compiles + in for them governs, changing with the PSFEx version; + which of these cuts still act with SAMPLE_AUTOSELECT N is + unverified. BADPIXEL_FILTER N and PSF_RECENTER N accept + flagged star vignets unfiltered and do not recentre + candidates. Anchor: workflow/config/cfis/default.psfex#SAMPLE_AUTOSELECT; workflow/config/cfis/default.psfex#BADPIXEL_FILTER; workflow/config/cfis/default.psfex#PSF_RECENTER. default: builtin_defaults options: builtin_defaults: - label: Compiled-in SAMPLE_* defaults, no bad-pixel filter + label: PSFEx built-in SAMPLE_* values, no bad-pixel filter pinned_explicit: label: Write the SAMPLE_* values explicitly into default.psfex psf_modelling_software: @@ -989,13 +1017,19 @@ analyses: fit_initialisation: label: Fit guesses and retries rationale: >- - [HARDCODED] TPSFFluxAndPriorGuesser (galaxy) and TFluxGuesser (PSF) - start from T = 0.25 and the catalogue flux; the galaxy runner retries - 5 times, the PSF runner twice. With a non-convex likelihood the guess - and retries decide which objects converge; a failed fit is flagged, - not raised. Guinot+22 initialised the whole guess vector from HSM - adaptive moments on each sheared image; the code no longer does. - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::make_runners. + [HARDCODED] the galaxy guesser (TPSFFluxAndPriorGuesser) + starts from T = 0.25 with a flux taken from a PSF-flux + fit; the PSF guesser (TFluxGuesser) starts from T = 0.25 + and the catalogue flux. The galaxy runner retries 5 times, + the PSF runner twice. With a non-convex likelihood the + guess and retries decide which objects converge. A failed + ngmix fit is flagged; any exception during an object's fit + drops it with no ngmix row, leaving it to the catalogue + sentinels (catalogue_assembly.failure_sentinels). + Guinot+22 initialised the whole guess vector from HSM + adaptive moments on each sheared image; the code does not. + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::make_runners; + src/shapepipe/modules/ngmix_package/ngmix.py::Ngmix.process. default: prior_guess_t025_ntry5_2 options: prior_guess_t025_ntry5_2: @@ -1006,15 +1040,21 @@ analyses: fit_priors: label: ngmix joint prior rationale: >- - [HARDCODED] ellipticity GPriorBA with sigma 0.4; flat T in [-1, 1e3] - and flat F in [-100, 1e9], with negative support (the bounds decide - which noisy fits survive and which rail); a centroid prior of width - one pixel scale, PIXEL_SCALE 0.186 arcsec (derived from the WCS when - the key is absent). Prior width drives noise bias; no rationale - recorded. Guinot+22 states a flat F in [-1e4, 1e9] and a flat r50 - prior rather than T. [LINT] the epoch stamps are exposure pixels, and - star selection uses 0.187 arcsec/px. + [HARDCODED] ellipticity GPriorBA with sigma 0.4; flat T in + [-1, 1e3] and flat F in [-100, 1e9], with negative support + (the bounds decide which noisy fits survive and which + rail); a centroid prior of width one pixel scale, + PIXEL_SCALE 0.186 arcsec (derived from the WCS when the + key is absent). The same joint prior also constrains the + PSF fits: the PSF fitter is built with the galaxy prior. + Prior width drives noise bias; no rationale recorded. + Guinot+22 states a flat F in [-1e4, 1e9] and a flat r50 + prior rather than T. [LINT] the epoch stamps are exposure + pixels, and star selection uses 0.187 arcsec/px; the + ngmix_runner comment says pixel scale also sets a noise + window, but get_noise is never called. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::get_prior; + src/shapepipe/modules/ngmix_package/ngmix.py::make_runners; workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.PIXEL_SCALE. default: gpriorba04_flat options: @@ -1317,21 +1357,28 @@ analyses: epoch_masked_fraction_cut: label: Per-epoch masked-fraction cut rationale: >- - [HARDCODED] an epoch whose stamp has more than 1/3 of its pixels - flagged (any nonzero flag bit, including the tile-coverage bit 2**10 - set where the tile vignet is off-image) is dropped from the - multi-epoch fit; an object with no surviving epoch has no shape. DES - was stricter. Y1 rejected any epoch with a masked or zero-weight - pixel, and any whose central 4-pixel region was masked. Y3 cut at 10% - of raw zero-weight pixels. Y6 dropped images more than 10% missing - and cut objects at mfrac < 0.1, which its simulations show avoids - calibration bias. Sheldon & Huff 2017 recommend dropping problematic - epochs when many are available. The cut interacts with defect_fill: - symmetrizing roughly doubles the masked fraction, so the cut should - be applied after symmetrizing. UNIONS has fewer epochs than DES, so - the cost in effective number density has to be measured, not - assumed. A central-region veto (drop the epoch if a defect lies - within a few pixels of the centre) is a cheap refinement. + [HARDCODED] an epoch whose stamp has more than 1/3 of its + pixels flagged (any nonzero flag bit, including the + tile-coverage bit 2**10 set where the tile vignet is + off-image) is dropped from the multi-epoch fit; an object + with no surviving epoch has no shape. The cut counts flag + pixels only, not zero-weight or invalid-RMS pixels. Before + it, an epoch is dropped silently if its galaxy stamp is + all zeros or its background-subtracted noise estimate + (sigma_mad) is not positive. DES was stricter. Y1 rejected + any epoch with a masked or zero-weight pixel, and any + whose central 4-pixel region was masked. Y3 cut at 10% of + raw zero-weight pixels. Y6 dropped images more than 10% + missing and cut objects at mfrac < 0.1, which its + simulations show avoids calibration bias. Sheldon & Huff + 2017 recommend dropping problematic epochs when many are + available. The cut interacts with defect_fill: + symmetrizing roughly doubles the masked fraction, so the + cut should be applied after symmetrizing. UNIONS has fewer + epochs than DES, so the cost in effective number density + has to be measured, not assumed. A central-region veto + (drop the epoch if a defect lies within a few pixels of + the centre) is a cheap refinement. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_postage_stamps. default: one_third options: @@ -1705,15 +1752,20 @@ analyses: star_galaxy_classification: label: Star/galaxy separation deferred out of the pipeline rationale: >- - SM_DO_CLASSIFICATION=False and no spread-model input is wired, so the - catalogue ships every object and separation happens downstream. The - dormant make_cat classifier uses class = sm + 2 sm_err with stars at - |class| < 0.003 and galaxies at class > 0.01, thresholds fixed in the - function signature. Guinot+22 selected galaxies in the pipeline at s - + 2 sigma_s > 0.0003 together with s > 0 and 20 < MAG_AUTO < 26; the - dormant code implements only the spread-model test, at a galaxy - boundary thirty times higher. + SM_DO_CLASSIFICATION=False and no spread-model input is + wired, so the catalogue ships every object and separation + happens downstream. The dormant make_cat classifier uses + class = sm + 2 sm_err with stars at |class| < + SM_STAR_THRESH and galaxies at class > SM_GAL_THRESH, read + from config when classification is on; the function + defaults are 0.003 and 0.01, and the committed config sets + neither key, so enabling classification alone raises. + Guinot+22 selected galaxies in the pipeline at s + 2 + sigma_s > 0.0003 together with s > 0 and 20 < MAG_AUTO < + 26; the dormant code implements only the spread-model + test. Anchor: workflow/config/cfis/config_tile_Mc.ini#MAKE_CAT_RUNNER.SM_DO_CLASSIFICATION; + src/shapepipe/modules/make_cat_runner.py::make_cat_runner; src/shapepipe/modules/make_cat_package/make_cat.py::save_sm_data; workflow/config/cfis/final_cat.param#SPREAD_CLASS. default: deferred_downstream @@ -1721,10 +1773,11 @@ analyses: deferred_downstream: label: No in-pipeline classification; catalogue ships all objects spread_model_inline: - label: spread_model classification in make_cat (0.003 / 0.01) + label: spread_model classification in make_cat description: >- Not wired in the workflow; needs spread_model_runner after the - PSF interpolation and its output added to make_cat's inputs. + PSF interpolation, its output added to make_cat's inputs, and + SM_STAR_THRESH / SM_GAL_THRESH set. tile_overlap_handling: label: Tile-overlap duplicates neither removed nor flagged rationale: >- @@ -1752,12 +1805,15 @@ analyses: failure_sentinels: label: Objects without shape measurements kept, with sentinel values rationale: >- - [HARDCODED] detections with no ngmix row stay in the catalogue with - sentinels: sizes, fluxes, magnitudes and flags 0, flux and magnitude - errors -1, ellipticities -10, size errors 1e30. The sentinels define - what a downstream cut must exclude: a failed object's - NGMIX_MCAL_FLAGS reads 0, the success value, so a cut on flags alone - keeps it. + [HARDCODED] detections with no ngmix row (no surviving + epoch, or an exception during the fit) stay in the + catalogue with sentinels: sizes, fluxes, magnitudes and + flags 0, flux and magnitude errors -1, ellipticities and + their errors -10, size errors 1e30, and NGMIX_N_EPOCH 0. + The sentinels define what a downstream cut must exclude: a + failed object's NGMIX_MCAL_FLAGS reads 0, the success + value, so a cut on flags alone keeps it; a cut on + NGMIX_N_EPOCH > 0 removes it. Anchor: src/shapepipe/modules/make_cat_package/make_cat.py::SaveCatalogue._save_ngmix_data. default: sentinel_values options: From 9fade3c5dad67046993aba089608397f3b01559b Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 03:27:25 +0200 Subject: [PATCH 07/40] docs(astra): rephrase unverifiable claims; add Guinot+22 insights; sentinels and completeness precision Follow-up to the correction pass: claims the repo cannot check are stated as what the config assumes; five Guinot+22 prior insights with page-verified quotes replace bare paper citations; failure_sentinels says an NGMIX_N_EPOCH > 0 cut removes failed objects; per_unit_completeness counts only rules that run shapepipe_run. Co-Authored-By: Claude Opus 5.5 --- astra.yaml | 232 +++++++++++++++++++++++++++++++++++++---------------- 1 file changed, 165 insertions(+), 67 deletions(-) diff --git a/astra.yaml b/astra.yaml index 7ad57e61f..e40d15b33 100644 --- a/astra.yaml +++ b/astra.yaml @@ -124,11 +124,12 @@ decisions: photometric_zeropoint: label: Magnitude zero-point convention rationale: >- - Tiles use a fixed MAG_ZEROPOINT 30.0 (ZP_FROM_HEADER=False), and ngmix - repeats it in MAG_ZP; exposures read the per-image header PHOTZP. The - tile convention relies on MegaPipe stacks being calibrated to 30 by - construction. Magnitude cuts (the star-selection window, downstream - galaxy cuts) inherit whichever convention their stage uses. + Tiles use a fixed MAG_ZEROPOINT 30.0 (ZP_FROM_HEADER=False), and + ngmix repeats it in MAG_ZP; exposures read the per-image header + PHOTZP. The fixed tile value assumes the MegaPipe stacks are + calibrated to 30; nothing in the repo checks it. Magnitude cuts + (the star-selection window, downstream galaxy cuts) inherit + whichever convention their stage uses. Anchor: workflow/config/cfis/default_tile.sex#MAG_ZEROPOINT; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER; workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.MAG_ZP; @@ -140,11 +141,9 @@ decisions: label: Tiles fixed 30.0; exposures from header PHOTZP header_everywhere: label: Per-image header zero-points on tiles too - excluded: true - excluded_reason: >- - MegaPipe stacks are calibrated to 30, so a header read adds a failure - path for no numerical change; it becomes a real fork only if tiles - ever carry a different PHOTZP. + description: >- + ZP_FROM_HEADER=True on tiles; changes magnitudes only where a tile's + header zero-point differs from 30. prior_insights: des_psf_blacklist: @@ -337,14 +336,17 @@ analyses: detection_threshold_policy: label: Detection significance, minimum area, matched filter rationale: >- - Tiles: DETECT_THRESH and ANALYSIS_THRESH 1.0 sigma, DETECT_MINAREA - 3, filtered with a 7x7 Gaussian of FWHM 3 px (gauss_3.0_7x7.conv, - close to the ~3.5 px CFIS seeing). These are the parameters of the - MegaPipe tile catalogue, so ShapePipe's galaxy sample matches the - catalogue UNIONS adopts. Exposures: 1.5 sigma, minarea 5, the 3x3 - FWHM 2 px kernel (default.conv); they only feed star selection. - Guinot+22 Table 2 lists minarea 10 at 1.5 sigma with the FWHM 2 px - kernel, so the tiles differ from the paper on all three. + Tiles: DETECT_THRESH and ANALYSIS_THRESH 1.0 sigma, + DETECT_MINAREA 3, filtered with a 7x7 Gaussian of FWHM 3 + px (gauss_3.0_7x7.conv), near the CFIS average seeing of + 0.65 arcsec (about 3.5 px at 0.187 arcsec/px; the .sex + files set SEEING_FWHM 0.6). These are the parameters of + the MegaPipe tile catalogue, so ShapePipe's galaxy sample + matches the catalogue UNIONS adopts. Exposures: 1.5 sigma, + minarea 5, the 3x3 FWHM 2 px kernel (default.conv); they + only feed star selection. Guinot+22 lists 1.5 sigma, + minarea 10 and the default 3x3 kernel, so the tiles differ + from the paper on all three. Anchor: workflow/config/cfis/default_tile.sex#DETECT_THRESH; workflow/config/cfis/default_tile.sex#DETECT_MINAREA; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE; @@ -354,8 +356,10 @@ analyses: options: megapipe_tiles: label: MegaPipe values on tiles; stock values on exposures + insights: [guinot22_cfis_seeing] stock_tiles: label: Stock ShapePipe values on tiles (1.5 sigma, minarea 5, FWHM 2 px) + insights: [guinot22_sextractor_params] excluded: true excluded_reason: >- Triggers spuriously on about 10% of grid-placed Sersic galaxies @@ -364,16 +368,17 @@ analyses: deblending_policy: label: Deblending contrast rationale: >- - DEBLEND_NTHRESH 32 on both passes; DEBLEND_MINCONT 0.002 on tiles - (the MegaPipe value) and 0.001 on exposures. Contrast sets object - count, centroids, and blend contamination in shapes. Guinot+22 Table - 2 lists 0.001 for tile detection. + DEBLEND_NTHRESH 32 on both passes; DEBLEND_MINCONT 0.002 + on tiles (the MegaPipe value) and 0.001 on exposures. + Contrast sets object count, centroids, and blend + contamination in shapes. Guinot+22 lists 0.001. Anchor: workflow/config/cfis/default_tile.sex#DEBLEND_MINCONT; workflow/config/cfis/default_exp.sex#DEBLEND_MINCONT. default: megapipe_tiles options: megapipe_tiles: label: MINCONT 0.002 tiles / 0.001 exposures + insights: [guinot22_sextractor_params] mincont_5em4_tiles: label: MINCONT 0.0005 on tiles excluded: true @@ -410,11 +415,11 @@ analyses: weight_map_usage: label: Weight map as inverse variance for detection rationale: >- - WEIGHT_TYPE MAP_WEIGHT on both passes (SExtractor default NONE): the - per-pixel variance sets the effective SNR and so the detection set. - RESCALE_WEIGHTS and WEIGHT_GAIN are SExtractor defaults. Guinot+22 - says all non-tabulated parameters keep their defaults, which would - mean no weight map. + WEIGHT_TYPE MAP_WEIGHT on both passes (SExtractor default + NONE): the per-pixel variance sets the effective SNR and + so the detection set. RESCALE_WEIGHTS and WEIGHT_GAIN are + SExtractor defaults. Guinot+22 keeps every non-tabulated + parameter at its default, which would mean no weight map. Anchor: workflow/config/cfis/default_tile.sex#WEIGHT_TYPE; workflow/config/cfis/default_exp.sex#WEIGHT_TYPE; src/shapepipe/modules/sextractor_package/sextractor_script.py::SExtractorCaller.set_input_files. @@ -424,6 +429,7 @@ analyses: label: MAP_WEIGHT no_weight: label: No weight map (SExtractor default) + insights: [guinot22_sextractor_params] zero_weight_interpolation: label: Interpolation across zero-weight pixels rationale: >- @@ -511,12 +517,13 @@ analyses: epoch_membership_ccd_bounds: label: Which exposure CCDs an object belongs to (N_EPOCH) rationale: >- - CCD_SIZE = 33,2080,1,4612 with strict inequalities: - positions with x at or below 33 are outside the bounds, - and a WCS inversion failure skips the CCD, lowering - N_EPOCH. This sets how many exposures enter each galaxy's - multi-epoch fit. The x range spans 2048 px; whether the - excluded strip is prescan or science pixels is unverified. + CCD_SIZE = 33,2080,1,4612 with strict inequalities bounds + each CCD's usable x range, and a WCS inversion failure + skips the CCD, lowering N_EPOCH. This sets how many + exposures enter each galaxy's multi-epoch fit. The x range + (33, 2080) spans exactly 2048 px and matches MegaCam's raw + DATASEC, so the excluded strip is likely prescan; that is + unverified here. Anchor: workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.CCD_SIZE; src/shapepipe/modules/sextractor_package/sextractor_script.py::make_post_process; src/shapepipe/modules/sextractor_package/sextractor_script.py::ccd_candidate_mask. @@ -524,9 +531,41 @@ analyses: options: trimmed_bounds_33_2080: label: "x in (33,2080), y in (1,4612), strict" - full_ccd: - label: Untrimmed x-range, inclusive bounds + inclusive_bounds: + label: Same bounds, inclusive prior_insights: + guinot22_sextractor_params: + claim: >- + The published tile detection uses DETECT_THRESH 1.5, DETECT_MINAREA + 10, the default 3x3 filter and DEBLEND_MINCONT 0.001, with every + other SExtractor parameter at its default. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_table2 + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'Table 2. SExtractor parametrisation. All other parameters are kept to their default values.' + location: { page: 5 } + - id: ev_guinot22_mincont + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'DEBLEND_MINCONT 0.001' + location: { page: 5 } + - id: ev_guinot22_minarea + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'DETECT_MINAREA 10' + location: { page: 5 } + guinot22_cfis_seeing: + claim: >- + CFIS r-band data have an average seeing of 0.65 arcsec. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_seeing + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'is taking r-band data with an average seeing of 0.65 arcsec' + location: { page: 1 } guinot22_stacked_detection: claim: >- Source extraction in the published analysis is performed on the @@ -582,9 +621,9 @@ analyses: ccd_split_extent: label: All 40 MegaCam HDUs split and carried as candidate epochs rationale: >- - N_HDU=40, and any other HDU count raises: every CCD including the - ear CCDs 36-39 is a candidate epoch wherever the WCS lands it. - Excluding the ear CCDs (a different optical path) is the alternative. + N_HDU=40, and any other HDU count raises: every one of the + 40 CCDs is a candidate epoch wherever the WCS lands it. + Excluding a subset of CCDs is the alternative. Anchor: workflow/config/cfis/config_exp_Sp.ini#SPLIT_EXP_RUNNER.N_HDU; src/shapepipe/modules/split_exp_package/split_exp.py::SplitExposures.create_hdus. default: all_40_hdus @@ -592,8 +631,8 @@ analyses: all_40_hdus: label: 40 HDUs, hard-fail on any other count insights: [guinot22_forty_chips] - exclude_ear_ccds: - label: Drop CCDs 36-39 as epochs + exclude_ccd_subset: + label: Exclude a subset of CCDs as epochs description: Not implemented. epoch_provenance_from_tile_history: label: Epoch sets parsed from tile FITS HISTORY cards @@ -717,8 +756,8 @@ analyses: 0.1 px and its plot uses 0.186 arcsec/px, while the applied cut is +- 0.2 px at 0.187; the _mode docstring puts the median fallback at 10 objects, the code at 20. - Anchor: workflow/config/cfis/star_selection.setools#MASK:star_selection.MAG_AUTO; - workflow/config/cfis/star_selection.setools#MASK:preselect.MAG_AUTO; + Anchor: workflow/config/cfis/star_selection.setools#MASK:star_selection.FLAGS; + workflow/config/cfis/star_selection.setools#MASK:preselect.FLAGS; workflow/config/cfis/default.psfex#SAMPLE_AUTOSELECT; src/shapepipe/pipeline/str_handler.py::StrInterpreter._mode. default: mode_centred_box @@ -790,16 +829,19 @@ analyses: psf_modelling_software: label: PSF model, PSFEx per CCD or MCCD over the focal plane rationale: >- - workflow/config.yaml psf_model selects the exposure and tile config - pair; the committed value is psfex, which fits each CCD - independently. MCCD (Liaudat+2021) fits all 40 CCDs at once with a - hybrid local+global model (N_COMP_LOC 8, D_COMP_GLOB 8, MIN_N_STARS - 20, RMSE_THRESH 1.25); the completeness table treats its counts as - warnings because no campaign has run it. [LINT] config_exp_mccd.ini - still reads pipeline_flag images from mask_runner, which no longer - exists, so the MCCD exposure chain cannot run as committed. + workflow/config.yaml psf_model selects the exposure and + tile config pair; the committed value is psfex, which fits + each CCD independently. MCCD (Liaudat+2021) fits one + hybrid local+global model over the focal plane + (FP_GEOMETRY CFIS; N_COMP_LOC 8, D_COMP_GLOB 8, + MIN_N_STARS 20, RMSE_THRESH 1.25); the completeness table + treats its counts as warnings because no campaign has run + it. [LINT] config_exp_mccd.ini still reads pipeline_flag + images from mask_runner, which no longer exists, so the + MCCD exposure chain cannot run as committed. Anchor: workflow/config.yaml; workflow/config/cfis/config_MCCD.ini#INSTANCE.N_COMP_LOC; + workflow/config/cfis/config_MCCD.ini#INSTANCE.FP_GEOMETRY; workflow/config/cfis/config_MCCD.ini#INPUTS.MIN_N_STARS; workflow/config/cfis/config_exp_mccd.ini#SEXTRACTOR_RUNNER.INPUT_MODULE; src/shapepipe/modules/mccd_package. @@ -832,15 +874,16 @@ analyses: psf_acceptance_thresholds: label: Per-CCD PSF-model quality gate rationale: >- - A CCD whose model has ACCEPTED < STAR_THRESH = 22 or CHI2 > - CHI2_THRESH = 2 is not interpolated, on both the validation and the - multi-epoch pass; 22 applies the published floor to the 80% training - sample. In the science path the CCD's epoch is dropped for every - object on it; an object left with no epoch has no shape. There is no - minimum-epoch floor in the pipeline: NGMIX_N_EPOCH records what - survived and epoch-count cuts happen downstream. Guinot+22 describes - excluding the CCD from PSF modelling rather than gating at - interpolation, and does not state the chi2 cut. + A CCD whose model has ACCEPTED < STAR_THRESH = 22 or CHI2 + > CHI2_THRESH = 2 is not interpolated, on both the + validation and the multi-epoch pass; 22 applies the + published floor to the 80% training sample. In the science + path the CCD's epoch is dropped for every object on it; an + object left with no epoch has no shape. There is no + minimum-epoch floor in the pipeline: NGMIX_N_EPOCH records + what survived and epoch-count cuts happen downstream. + Guinot+22 describes discarding the CCD from PSF estimation + rather than gating at interpolation. Anchor: src/shapepipe/modules/psfex_interp_package/psfex_interp.py::PSFExInterpolator.interpsfex; workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH; workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH; @@ -1036,6 +1079,7 @@ analyses: label: T guess 0.25 + catalogue flux, ntry 5 / 2 hsm_initialisation: label: Guesses from HSM adaptive moments (Guinot+22) + insights: [guinot22_hsm_initialisation] description: Not implemented. fit_priors: label: ngmix joint prior @@ -1048,8 +1092,8 @@ analyses: key is absent). The same joint prior also constrains the PSF fits: the PSF fitter is built with the galaxy prior. Prior width drives noise bias; no rationale recorded. - Guinot+22 states a flat F in [-1e4, 1e9] and a flat r50 - prior rather than T. [LINT] the epoch stamps are exposure + Guinot+22 states a flat F in [-1e4, 1e9] and a flat prior + on the half-light radius r50 rather than on T. [LINT] the epoch stamps are exposure pixels, and star selection uses 0.187 arcsec/px; the ngmix_runner comment says pixel scale also sets a noise window, but get_noise is never called. @@ -1060,6 +1104,7 @@ analyses: options: gpriorba04_flat: label: GPriorBA 0.4 + flat T/F with negative support + insights: [guinot22_ngmix_priors] nonneg_informative: label: Non-negative or informative T/F priors excluded: true @@ -1402,6 +1447,39 @@ analyses: epochs per band; UNIONS likely cannot. insights: [mask_multi_epoch_drop, mask_des_defect_practice] prior_insights: + guinot22_ngmix_priors: + claim: >- + The published ngmix fit uses a Gaussian centroid prior of width one + pixel scale, a flat prior on the half-light radius r50 in [-10, 1e6] + arcsec and a flat flux prior in [-1e4, 1e9]. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_cen_prior + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'Gaussian distribution centred on 0 with σ = pixel scale' + location: { page: 7 } + - id: ev_guinot22_r50_prior + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'r50: flat distribution in' + location: { page: 7 } + - id: ev_guinot22_flux_prior + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'F: flat distribution in' + location: { page: 7 } + guinot22_hsm_initialisation: + claim: >- + The published fit is initialised from adaptive moments measured on + each sheared version of the object. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_hsm_init + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'We thus run first an adaptive moments algorithm to initialise the least-square operation.' + location: { page: 7 } guinot22_gaussian_model: claim: >- The published shape measurement models galaxies with a single @@ -1774,6 +1852,7 @@ analyses: label: No in-pipeline classification; catalogue ships all objects spread_model_inline: label: spread_model classification in make_cat + insights: [guinot22_spread_model_cut] description: >- Not wired in the workflow; needs spread_model_runner after the PSF interpolation, its output added to make_cat's inputs, and @@ -1781,12 +1860,13 @@ analyses: tile_overlap_handling: label: Tile-overlap duplicates neither removed nor flagged rationale: >- - Adjacent tiles overlap, and objects in the overlap are measured in - both. [HARDCODED] make_cat attaches only TILE_ID; there is no unique-object rule - and no overlap flag, and sp_validation removes duplicates by cutting - each tile to its non-overlapping area. [LINT] the make_cat package - docstring documents a TILE_LIST key that flags overlap objects; - nothing implements it. + Adjacent tiles overlap, and objects in the overlap are + measured in both. [HARDCODED] make_cat attaches only + TILE_ID; there is no unique-object rule and no overlap + flag, so duplicates are left to downstream selection. + [LINT] the make_cat package docstring documents a + TILE_LIST key that flags overlap objects; nothing + implements it. Anchor: src/shapepipe/modules/make_cat_package/make_cat.py::save_sextractor_data; src/shapepipe/modules/make_cat_package/__init__.py. default: no_dedup_in_pipeline @@ -1824,3 +1904,21 @@ analyses: excluded: true excluded_reason: >- Loses the photometry-only population and hides attrition. + prior_insights: + guinot22_spread_model_cut: + claim: >- + The published analysis selects galaxies inside the pipeline with the + spread model, at s + 2 sigma_s > 0.0003 together with s > 0 and + 20 < MAG_AUTO < 26. + created_at: "2022-04-01T00:00:00Z" + evidence: + - id: ev_guinot22_spread_model + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 'We then use the spread model to make the selection presented below:' + location: { page: 5 } + - id: ev_guinot22_spread_model_cut + doi: "10.48550/arXiv.2204.04798" + quote: + exact: 's + 2σs > 0.0003' + location: { page: 5 } From 64327ad2084764b6891f8aac49658418e8154c4e Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 03:28:29 +0200 Subject: [PATCH 08/40] test(contracts): validate @sc contracts against the decision record tests/helpers/contracts.py parses @sc/@cc contracts with sc-list's line grammar from Python docstrings, Snakemake comment blocks and CONTRACTS files under src/, workflow/ and scripts/. test_contracts.py fails on malformed lines, missing or duplicate ids, tag lines hidden in .py comments (invisible to sc-list), and decision: metas naming no decision in astra.yaml. A report-only test prints decisions no contract cites and contracts off the record's anchored symbols. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014bvNTrAmZxcfb1ee83ApPK --- tests/helpers/contracts.py | 276 +++++++++++++++++++++++++++++++++++ tests/unit/test_contracts.py | 177 ++++++++++++++++++++++ 2 files changed, 453 insertions(+) create mode 100644 tests/helpers/contracts.py create mode 100644 tests/unit/test_contracts.py diff --git a/tests/helpers/contracts.py b/tests/helpers/contracts.py new file mode 100644 index 000000000..2957d1ea4 --- /dev/null +++ b/tests/helpers/contracts.py @@ -0,0 +1,276 @@ +"""Parse and validate the repository's scientific contracts. + +A contract is a tagged block colocated with the code it governs: a tag line +(``@sc`` or ``@cc``, an optional bracketed ``key:value`` meta list, then a +stable id) followed by prose up to the next blank line. The line grammar is +the loom ``sc-list`` reference parser's, regex for regex, so a contract that +parses here parses there. + +Where contracts live: + +* Python: in the docstring of the module, class or function they govern. + ``sc-list`` reads docstrings only, so a tag line in a ``#`` comment of a + ``.py`` file is reported as an error rather than silently ignored. +* Snakemake (``.smk``, ``Snakefile``): in ``#`` comment blocks. +* ``CONTRACTS`` files: anywhere in the file; they govern their directory. + +A ``decision:`` meta names a decision in ``astra.yaml``: a top-level +decision by its bare id, a sub-analysis decision as ``.``. +""" + +from dataclasses import dataclass, field +import ast +import io +from pathlib import Path +import re +import tokenize +import warnings + +from tests.helpers.astra_record import extract_anchors + +TAG = re.compile(r"^\s*@(sc|cc)\b.*$") +VALID = re.compile(r"^\s*@(sc|cc)(?:\s+\[([^\]]*)\])?\s+([\w][\w.-]*)\s*$") +META_PAIR = re.compile(r"[\w.-]+:[^,\s]+") +MISSING_ID = re.compile(r"\s*@(sc|cc)(?:\s+\[[^\]]*\])?\s*") + +SCAN_ROOTS = ("src", "workflow", "scripts") +SKIP = {".git", ".venv", "venv", "__pycache__", "node_modules", ".felt"} + + +@dataclass +class Contract: + """One parsed contract.""" + + tag: str + id: str + meta: dict = field(default_factory=dict) + prose: str = "" + path: str = "" + line: int = 0 + scope: str = "" + + +def parse_block(text, path, offset=0, scope=""): + """Parse every contract in ``text``; return ``(contracts, errors)``.""" + + lines = text.splitlines() + found, errors = [], [] + for index, line in enumerate(lines): + if not TAG.match(line): + continue + where = f"{path}:{index + 1 + offset}" + match = VALID.match(line) + if not match: + detail = ( + "missing contract id" + if MISSING_ID.fullmatch(line) + else "malformed contract line" + ) + errors.append(f"{where}: {detail}: {line.strip()}") + continue + tag, meta_text, ident = match.groups() + meta = {} + if meta_text is not None: + pairs = [part.strip() for part in meta_text.split(",")] + if any(not META_PAIR.fullmatch(part) for part in pairs): + errors.append(f"{where}: malformed contract metadata: {meta_text}") + continue + meta = dict(part.split(":", 1) for part in pairs) + prose = [] + for body in lines[index + 1:]: + if not body.strip(): + break + prose.append(body.strip()) + found.append( + Contract(tag, ident, meta, " ".join(prose), str(path), + index + 1 + offset, scope) + ) + return found, errors + + +def _declarations(tree): + parents = { + child: parent + for parent in ast.walk(tree) + for child in ast.iter_child_nodes(parent) + } + yield tree, "module" + for node in ast.walk(tree): + if not isinstance( + node, (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef) + ): + continue + names, parent = [node.name], parents.get(node) + while parent is not None and not isinstance(parent, ast.Module): + if isinstance( + parent, (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef) + ): + names.append(parent.name) + parent = parents.get(parent) + yield node, ".".join(reversed(names)) + + +def python_contracts(path, source): + """Contracts in a Python file's docstrings; tags in comments are errors.""" + + try: + with warnings.catch_warnings(): + warnings.simplefilter("ignore", SyntaxWarning) + tree = ast.parse(source, filename=str(path)) + except SyntaxError as error: + return [], [f"{path}: cannot parse: {error}"] + found, errors = [], [] + for node, scope in _declarations(tree): + doc = ast.get_docstring(node, clean=False) + if not doc: + continue + first = 1 if node is tree else node.body[0].lineno + records, issues = parse_block(doc, path, first - 1, scope) + found.extend(records) + errors.extend(issues) + try: + tokens = tokenize.generate_tokens(io.StringIO(source).readline) + for token in tokens: + if token.type == tokenize.COMMENT and TAG.match( + token.string.lstrip("#") + ): + errors.append( + f"{path}:{token.start[0]}: contract in a comment; " + "move it into the governing docstring" + ) + except (tokenize.TokenError, SyntaxError): + pass + return found, errors + + +def snakemake_contracts(path, source): + """Contracts in a Snakemake file's ``#`` comment blocks.""" + + stripped = [] + for line in source.splitlines(): + text = line.strip() + stripped.append(text[1:] if text.startswith("#") else "") + return parse_block("\n".join(stripped), path, 0, "file") + + +def _is_snakemake(path): + return path.suffix == ".smk" or path.name == "Snakefile" + + +def contract_files(root, scan_roots=SCAN_ROOTS): + """Yield every file under ``scan_roots`` that can carry contracts.""" + + root = Path(root) + for top in scan_roots: + base = root / top + if not base.is_dir(): + continue + for path in sorted(base.rglob("*")): + if any(part in SKIP for part in path.parts) or not path.is_file(): + continue + if ( + path.name == "CONTRACTS" + or path.suffix == ".py" + or _is_snakemake(path) + ): + yield path + + +def collect(root, scan_roots=SCAN_ROOTS): + """Parse every contract under ``root``; return ``(contracts, errors)``. + + Errors cover malformed tag lines, malformed meta, missing ids and ids + used more than once. Paths are relative to ``root``. + """ + + root = Path(root) + contracts, errors = [], [] + for path in contract_files(root, scan_roots): + relative = path.relative_to(root) + text = path.read_text(encoding="utf-8") + if path.name == "CONTRACTS": + found, issues = parse_block( + text, relative, 0, str(relative.parent) + ) + elif _is_snakemake(path): + found, issues = snakemake_contracts(relative, text) + else: + found, issues = python_contracts(relative, text) + contracts.extend(found) + errors.extend(issues) + seen = {} + for contract in contracts: + seen.setdefault(contract.id, []).append(contract) + for ident, items in seen.items(): + if len(items) > 1: + where = ", ".join(f"{c.path}:{c.line}" for c in items) + errors.append(f"duplicate contract id {ident}: {where}") + return contracts, errors + + +def decision_ids(record, prefix=""): + """Scoped decision ids: bare at top level, dotted inside sub-analyses.""" + + ids = set() + for decision in (record.get("decisions") or {}): + ids.add(f"{prefix}{decision}") + for name, analysis in (record.get("analyses") or {}).items(): + if isinstance(analysis, dict): + ids |= decision_ids(analysis, f"{prefix}{name}.") + return ids + + +def decision_errors(contracts, record): + """Contracts whose ``decision:`` meta names no decision in the record.""" + + known = decision_ids(record) + return [ + f"{c.path}:{c.line}: contract {c.id} cites unknown decision " + f"{c.meta['decision']!r}" + for c in contracts + if "decision" in c.meta and c.meta["decision"] not in known + ] + + +def anchored_refs(record): + """``(code_symbols, anchored_paths)`` named by the record's anchors. + + ``code_symbols`` holds ``(path, Symbol)`` pairs from ``path::Symbol`` + refs; ``anchored_paths`` holds every path any ref names. + """ + + symbols, paths = set(), set() + for anchor in extract_anchors(record): + for reference in anchor.references: + if "::" in reference: + path, symbol = reference.split("::", 1) + symbols.add((path, symbol)) + else: + path = reference.split("#", 1)[0] + paths.add(path) + return symbols, paths + + +def coverage_report(contracts, record): + """Report-only gaps between the contracts and the record. + + Returns ``(uncovered, unanchored)``: decision ids no contract cites, and + ``@sc`` contracts whose declaration is not an anchored symbol. A + module-docstring contract counts as anchored when an anchor names its + file or a symbol in it. + """ + + cited = {c.meta["decision"] for c in contracts if "decision" in c.meta} + uncovered = sorted(decision_ids(record) - cited) + symbols, paths = anchored_refs(record) + unanchored = [] + for contract in contracts: + if contract.tag != "sc": + continue + if contract.scope == "module": + if contract.path in paths: + continue + elif (contract.path, contract.scope) in symbols: + continue + unanchored.append(contract) + return uncovered, unanchored diff --git a/tests/unit/test_contracts.py b/tests/unit/test_contracts.py new file mode 100644 index 000000000..53f108fe2 --- /dev/null +++ b/tests/unit/test_contracts.py @@ -0,0 +1,177 @@ +"""Keep the @sc contracts well-formed and tied to the ASTRA decision record.""" + +from functools import cache +from pathlib import Path +import textwrap + +from tests.helpers.astra_record import load_yaml +from tests.helpers.contracts import ( + collect, + coverage_report, + decision_errors, + decision_ids, +) + +REPO_ROOT = Path(__file__).resolve().parents[2] + +RECORD = { + "decisions": {"top_choice": {}}, + "analyses": {"stage": {"decisions": {"inner_choice": {}}}}, +} + + +@cache +def _repository(): + record = load_yaml(REPO_ROOT / "astra.yaml") + contracts, errors = collect(REPO_ROOT) + return record, contracts, errors + + +def _write(root, relative, text): + path = root / relative + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(textwrap.dedent(text), encoding="utf-8") + + +def test_parser_reads_docstrings_contracts_files_and_snakemake(tmp_path): + _write(tmp_path, "src/pkg/mod.py", ''' + """Module.""" + + + class Thing: + def method(self): + """Do it. + + @sc [decision:stage.inner_choice,label:convention] method-rule + The method keeps its promise. + Second prose line. + + Returns + ------- + None + """ + ''') + _write(tmp_path, "src/pkg/CONTRACTS", """ + @cc boundary-rule + forbid: pkg.a.* -> pkg.b.* + """) + _write(tmp_path, "workflow/rules/x.smk", """ + # @sc [decision:top_choice] rule-contract + # The rule keeps its promise. + + rule x: + output: "a" + """) + + contracts, errors = collect(tmp_path) + + assert errors == [] + by_id = {c.id: c for c in contracts} + assert set(by_id) == {"method-rule", "boundary-rule", "rule-contract"} + method = by_id["method-rule"] + assert method.scope == "Thing.method" + assert method.meta == { + "decision": "stage.inner_choice", + "label": "convention", + } + assert method.prose == "The method keeps its promise. Second prose line." + assert method.line == 9 + assert by_id["boundary-rule"].scope == "src/pkg" + assert decision_errors(contracts, RECORD) == [] + + +def test_malformed_meta_and_missing_id_are_errors(tmp_path): + _write(tmp_path, "src/mod.py", ''' + def f(): + """F. + + @sc [decision stage.inner_choice] bad-meta + Prose. + + @sc [label:x] + Prose. + + @sc two words + Prose. + """ + ''') + + contracts, errors = collect(tmp_path) + + assert contracts == [] + assert len(errors) == 3 + assert "malformed contract metadata" in errors[0] + assert "missing contract id" in errors[1] + assert "malformed contract line" in errors[2] + + +def test_duplicate_ids_and_comment_contracts_are_errors(tmp_path): + _write(tmp_path, "src/a.py", ''' + def f(): + """F. + + @sc same-id + Prose. + """ + ''') + _write(tmp_path, "scripts/b.py", ''' + def g(): + """G. + + @sc same-id + Prose. + """ + # @sc hidden-id + return None + ''') + + _, errors = collect(tmp_path) + + assert any("duplicate contract id same-id" in e for e in errors) + assert any("contract in a comment" in e for e in errors) + + +def test_unknown_decision_is_an_error(tmp_path): + _write(tmp_path, "src/mod.py", ''' + def f(): + """F. + + @sc [decision:inner_choice] undotted + Sub-analysis decisions need their analysis prefix. + + @sc [decision:no_such_choice] unknown + Prose. + """ + ''') + + contracts, errors = collect(tmp_path) + + assert errors == [] + problems = decision_errors(contracts, RECORD) + assert len(problems) == 2 + assert "'inner_choice'" in problems[0] + assert "'no_such_choice'" in problems[1] + assert decision_ids(RECORD) == {"top_choice", "stage.inner_choice"} + + +def test_repository_contracts_are_valid_and_cite_real_decisions(): + record, contracts, errors = _repository() + errors = errors + decision_errors(contracts, record) + + message = "Contract problems:\n - " + "\n - ".join(errors) + assert not errors, message + + +def test_contract_coverage_report(): + """Report-only: print record decisions and contracts that lack a partner.""" + + record, contracts, _ = _repository() + uncovered, unanchored = coverage_report(contracts, record) + + print(f"\n{len(contracts)} contracts; " + f"{len(uncovered)} decisions cited by no contract:") + for decision in uncovered: + print(f" {decision}") + print(f"{len(unanchored)} @sc contracts off the record's anchors:") + for contract in unanchored: + print(f" {contract.id} at {contract.path}::{contract.scope}") From 1abd1cd0a7d79135e4328ec9bfeac63d0611a0c5 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 03:31:29 +0200 Subject: [PATCH 09/40] docs(sc): first scientific contracts at the record's anchors Sixteen @sc contracts in the docstrings of declarations astra.yaml anchors, each citing its decision: star-selection mode and split seeding, SExtractor weight wiring and epoch membership bounds, CCD splitting and WCS source, epoch provenance, stamp rounding, the PSF acceptance gate, catalogue classification scope, never-fit sentinels, mask-column and mask-flag semantics, and per-unit completeness. Where the record carries a [LINT] at the declaration, the contract states the intended behaviour and names the lint. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014bvNTrAmZxcfb1ee83ApPK --- .../find_exposures_package/find_exposures.py | 8 +++++++ .../modules/make_cat_package/make_cat.py | 14 ++++++++++++ src/shapepipe/modules/make_cat_runner.py | 10 ++++++++- src/shapepipe/modules/mask_query_runner.py | 12 +++++++++- .../psfex_interp_package/psfex_interp.py | 6 +++++ .../modules/setools_package/setools.py | 8 +++++++ .../sextractor_package/sextractor_script.py | 22 +++++++++++++++++++ .../modules/split_exp_package/split_exp.py | 13 +++++++++++ .../vignetmaker_package/vignetmaker.py | 7 ++++++ src/shapepipe/pipeline/str_handler.py | 7 ++++++ src/shapepipe/utilities/mask_query.py | 6 +++++ workflow/scripts/completeness.py | 8 +++++++ 12 files changed, 119 insertions(+), 2 deletions(-) diff --git a/src/shapepipe/modules/find_exposures_package/find_exposures.py b/src/shapepipe/modules/find_exposures_package/find_exposures.py index eab72801e..2670e15fe 100644 --- a/src/shapepipe/modules/find_exposures_package/find_exposures.py +++ b/src/shapepipe/modules/find_exposures_package/find_exposures.py @@ -61,6 +61,14 @@ def get_exposure_list(self): Return list of exposure file used for the tile in process, from tiles FITS header. + @sc [decision:preparation.epoch_provenance_from_tile_history,label:convention] epochs-from-tile-history + The epoch list is the deduplicated file names in HISTORY column COLNUM + with the extension stripped and the trailing ``p`` kept; a tile whose + header cannot be read must fail, never yield an empty or partial list. + EXP_PREFIX is meant to strip a name prefix; removeprefix does nothing + to CFIS names, where ``p`` is a suffix, a [LINT] the decision record + carries. + Returns ------- list diff --git a/src/shapepipe/modules/make_cat_package/make_cat.py b/src/shapepipe/modules/make_cat_package/make_cat.py index 062ac5719..eafba894a 100644 --- a/src/shapepipe/modules/make_cat_package/make_cat.py +++ b/src/shapepipe/modules/make_cat_package/make_cat.py @@ -250,6 +250,12 @@ def save_mask_ext_data(final_cat_file, band_paths, w_log): The lookup itself is ``shapepipe.utilities.mask_query.query_map``, shared with the ``mask_query`` module: one primitive, two consumers. + @sc [decision:masking.sky_mask_application,label:scope] mask-columns-verbatim + Each ``MASK_`` holds the map value at the object's windowed position + verbatim, off-coverage sentinel included; nothing here thresholds, + interprets or removes an object. Without MASK_EXT_PATHS no column is + written and every mask cut happens downstream. + Parameters ---------- final_cat_file : file_io.FITSCatalogue @@ -412,6 +418,14 @@ def _save_ngmix_data(self, ngmix_cat_path, moments=False): moments : bool, optional If True, write the parallel ``NGMIXm_*`` (moments-branch) columns. + @sc [decision:catalogue_assembly.failure_sentinels,label:coupling] never-fit-sentinels-out-of-range + A detection with no ngmix row keeps NGMIX_N_EPOCH 0 and shape sentinels + outside any measured range: ellipticities and their errors -10, flux + and magnitude errors -1, size errors 1e30. A cut on NGMIX_N_EPOCH > 0 + or on these values removes such a row independently of the flag + columns; the size and flux sentinels (0) lie inside the physical range + and do not. Keep every sentinel out of range when changing one. + """ self._key_ends = ["1M", "1P", "2M", "2P", "NOSHEAR"] diff --git a/src/shapepipe/modules/make_cat_runner.py b/src/shapepipe/modules/make_cat_runner.py index 176341757..405f7def4 100644 --- a/src/shapepipe/modules/make_cat_runner.py +++ b/src/shapepipe/modules/make_cat_runner.py @@ -37,7 +37,15 @@ def make_cat_runner( module_config_sec, w_log, ): - """Define The Make Catalogue Runner.""" + """Define The Make Catalogue Runner. + + @sc [decision:catalogue_assembly.star_galaxy_classification,label:scope] classification-deferred-downstream + The final catalogue carries every detection: no star/galaxy cut is made + here, and separation happens downstream. With SM_DO_CLASSIFICATION on, the + thresholds must come from SM_STAR_THRESH and SM_GAL_THRESH; the function + defaults of :func:`make_cat.save_sm_data` are never used. + + """ # Set input file paths if len(input_file_list) == 3: # No spread model input diff --git a/src/shapepipe/modules/mask_query_runner.py b/src/shapepipe/modules/mask_query_runner.py index a636560b2..9f0c700ea 100644 --- a/src/shapepipe/modules/mask_query_runner.py +++ b/src/shapepipe/modules/mask_query_runner.py @@ -26,7 +26,17 @@ def mask_query_runner( module_config_sec, w_log, ): - """Define The Mask Query Runner.""" + """Define The Mask Query Runner. + + @sc [decision:masking.psf_star_mask_veto,label:scope] mask-query-carries-not-cuts + Without MASK_PATHS the catalogue passes through with no MASK_EXT column; + with it, MASK_EXT is carried for measurement and nothing here removes a + star. The PSF-star veto is a setools edit (``MASK_EXT == 0`` beside + IMAFLAGS_ISO), not a change here. The workflow/rules/exposure.smk docstring + calls the column FLAG_EXT and says setools cuts on it, a [LINT] the + decision record carries. + + """ sexcat_path = input_file_list[0] # Get file prefix (optional) diff --git a/src/shapepipe/modules/psfex_interp_package/psfex_interp.py b/src/shapepipe/modules/psfex_interp_package/psfex_interp.py index b6da6623f..b136e44a2 100644 --- a/src/shapepipe/modules/psfex_interp_package/psfex_interp.py +++ b/src/shapepipe/modules/psfex_interp_package/psfex_interp.py @@ -240,6 +240,12 @@ def interpsfex(self, dotpsfpath, pos): Use PSFEx generated model to perform spatial PSF interpolation. + @sc [decision:star_selection_psf.psf_acceptance_thresholds,label:gate] psf-gate-drops-epoch + A model with ACCEPTED below STAR_THRESH or CHI2 above CHI2_THRESH + yields a failure sentinel, not PSFs, and every caller drops that CCD's + epoch; the gate is the same for the validation and multi-epoch passes. + The thresholds come from config, never from literals here. + Parameters ---------- dotpsfpath : str diff --git a/src/shapepipe/modules/setools_package/setools.py b/src/shapepipe/modules/setools_package/setools.py index 33d8efa8b..7c82c715f 100644 --- a/src/shapepipe/modules/setools_package/setools.py +++ b/src/shapepipe/modules/setools_package/setools.py @@ -618,6 +618,14 @@ def _make_rand_split(self): This function creates mask with random indices corresponding to the specfied ratio. + @sc [decision:star_selection_psf.psf_train_validation_split,label:reproducibility] split-seeded-by-file-number + The permutation is seeded from the digits of the unit's file number, so + a CCD gets the same PSF training and validation stars on every run; + never draw from a global or unseeded generator here. The + ``ratio_`` subset (20 % under RATIO = 20) is the validation + sample and its complement trains PSFEx (``star_split_ratio_80``); + swapping them starves the model of stars below the acceptance gate. + Raises ------ ValueError diff --git a/src/shapepipe/modules/sextractor_package/sextractor_script.py b/src/shapepipe/modules/sextractor_package/sextractor_script.py index ad5a83bcb..0bae59107 100644 --- a/src/shapepipe/modules/sextractor_package/sextractor_script.py +++ b/src/shapepipe/modules/sextractor_package/sextractor_script.py @@ -53,6 +53,12 @@ def ccd_candidate_mask(w, ra, dec, ccd_size, margin_frac=0.5): Flag catalogue positions that lie near a CCD's sky footprint. + @sc [decision:detection.epoch_membership_ccd_bounds,label:invariant] candidate-mask-is-superset + The mask only prefilters the strict CCD_SIZE test in + :func:`make_post_process`; it must keep every position that test could + accept. Tightening it (margin, radius) silently drops epochs and lowers + N_EPOCH instead of just skipping divergent WCS inversions. + The footprint is obtained by forward-projecting (pixel to world) the CCD centre and corners, which is always well defined. The returned mask selects positions within the corner radius plus a fractional @@ -110,6 +116,13 @@ def make_post_process(cat_path, f_wcs_path, pos_params, ccd_size, w_log=None): This function will add one HDU for each epoch to the SExtractor catalogue. Note that this only works for tiles. + @sc [decision:detection.epoch_membership_ccd_bounds,label:convention] epoch-bounds-strict-per-ccd + An object is on a CCD when its inverse-projected pixel lies strictly inside + CCD_SIZE (33, 2080, 1, 4612 committed), and each such CCD adds one to + N_EPOCH. A WCS inversion that fails drops that one CCD's epoch for the + objects near it, never the tile. CCD_N is the 0-based index that split_exp + gives the CCD file and its header entry. + The columns will be: - ``NUMBER``: same as SExtractor NUMBER @@ -363,6 +376,15 @@ def set_input_files( Set up all of the input image files. + @sc [decision:detection.weight_map_usage,label:convention] weight-map-on-both-images + With a weight file, the same map weights detection and measurement (a + separate detection weight only when DETECTION_WEIGHT is set); without + one, the command line forces ``-WEIGHT_TYPE None`` over the config's + MAP_WEIGHT, which changes the detection set. Extra inputs are + positional (image, weight, flag, psf, detection image, detection + weight): a reordering that keeps the count mis-assigns files without + raising. + Parameters ---------- use_weight: bool diff --git a/src/shapepipe/modules/split_exp_package/split_exp.py b/src/shapepipe/modules/split_exp_package/split_exp.py index dfab428f7..5c08b8163 100644 --- a/src/shapepipe/modules/split_exp_package/split_exp.py +++ b/src/shapepipe/modules/split_exp_package/split_exp.py @@ -70,6 +70,19 @@ def create_hdus(self, exp_path, output_suffix, transf_int, save_header): Split a single exposures CCDs into separate files. + @sc [decision:preparation.ccd_split_extent,label:convention] split-all-hdus-or-raise + Every one of the N_HDU (40) CCDs, the ear CCDs 36-39 included, is + written and becomes a candidate epoch; any other HDU count raises + rather than splitting a partial exposure. The file suffix ``-`` + and the header list index are the same 0-based CCD number that becomes + CCD_N downstream. + + @sc [decision:preparation.astrometric_solution_source,label:convention] wcs-from-delivered-header + The stored WCS is ``WCS(header)`` of the delivered CCD header, + unmodified. Every downstream world-to-pixel transform (epoch + membership, stamp placement, position seeding) uses it, so a refit or + header edit here moves every stamp centre and epoch assignment. + Parameters ---------- exp_path : str diff --git a/src/shapepipe/modules/vignetmaker_package/vignetmaker.py b/src/shapepipe/modules/vignetmaker_package/vignetmaker.py index 50f874254..b03712bb1 100644 --- a/src/shapepipe/modules/vignetmaker_package/vignetmaker.py +++ b/src/shapepipe/modules/vignetmaker_package/vignetmaker.py @@ -21,6 +21,13 @@ def get_stamps(image, positions, rad): Extract postage stamps and record their sub-pixel centring. + @sc [decision:preparation.stamp_positioning_and_padding,label:convention] stamps-off-image-raise + Stamps are cut around the rounded pixel with no interpolation, and OFFSET + is the remainder of that same rounding, the Jacobian origin ngmix uses; two + roundings would put the stamp and its centroid prior out of step. Edge + overruns are zero-padded and kept; a centre rounding outside the image + raises, never wraps. + The image is zero-padded by ``rad`` on every side and a ``(2 * rad + 1, 2 * rad + 1)`` stamp is sliced around the integer pixel nearest each position (``numpy.round``). The stamp values are diff --git a/src/shapepipe/pipeline/str_handler.py b/src/shapepipe/pipeline/str_handler.py index 8c9f5dae9..d30573bd2 100644 --- a/src/shapepipe/pipeline/str_handler.py +++ b/src/shapepipe/pipeline/str_handler.py @@ -258,6 +258,13 @@ def _mode(self, input, eps=0.001, iter_max=1000): Compute the mode, the most frequent value of a continuous distribution. + @sc [decision:star_selection_psf.star_selection_box,label:estimator] mode-median-fallback-at-20 + Below 20 objects the result is the median; from 20 up it is the + iterative histogram-zoom mode. The star-selection FWHM box is centred + on this value, so moving the threshold or the binning changes which + stars train the PSF on sparse CCDs. The Returns section below puts the + fallback at 10, a [LINT] the decision record carries. + Parameters ---------- input : numpy.ndarray diff --git a/src/shapepipe/utilities/mask_query.py b/src/shapepipe/utilities/mask_query.py index 982bfd3d3..fe40b3806 100644 --- a/src/shapepipe/utilities/mask_query.py +++ b/src/shapepipe/utilities/mask_query.py @@ -221,6 +221,12 @@ def flag_positions(paths, ra, dec, bits=None, w_log=None): Combine one or more healsparse masks into a single per-object integer flag. + @sc [decision:masking.psf_star_mask_veto,label:convention] off-coverage-is-clean + A position outside a map's coverage contributes 0 for boolean and integer + maps alike, so a map that misses an exposure never flags its stars; the + all-off-coverage case is logged as a warning instead. The flag is the + bitwise OR of the per-map contributions, 0 meaning clean. + Each map contributes at each position: * boolean map: ``1`` where the map is ``True``, ``0`` elsewhere; diff --git a/workflow/scripts/completeness.py b/workflow/scripts/completeness.py index ff776e6e7..0c5253fc0 100644 --- a/workflow/scripts/completeness.py +++ b/workflow/scripts/completeness.py @@ -1,6 +1,14 @@ #!/usr/bin/env python3 """The count-based completeness table — the single failure policy. +@sc [decision:per_unit_completeness,label:policy] exact-counts-fail-the-unit +A mandatory runner below its ``expect`` count fails its whole unit (an +exposure, a tile or an ngmix chunk), so a partial unit never reaches the +catalogue. On the psfex path only exposure-side psfex_interp may fall short (a +CCD rejected by the acceptance gate writes nothing); the never-run MCCD chain +warns throughout. The ``warn`` field note below cites tile psfex_interp as its +example, but that runner is mandatory, a [LINT] the decision record carries. + This is the ported ``complete_check`` count table from the v2.0 bash layer (``run_job_sp_canfar_v2.0.bash`` job dispatch, survey §4). Across smk-g6 (127 exposures, 64 tiles, and 512 ngmix chunks), every non-warning runner From 2b7512148465ab18c2de09678d337aae76a11e44 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 03:31:52 +0200 Subject: [PATCH 10/40] docs(astra): record the header saturation level; note PSF_ACCURACY - detection.saturation_level: SATUR_KEY SATURATE with no SATUR_LEVEL sets the FLAGS saturation bit that star selection rejects on; header presence on exposures and tiles is unverified here - psf_model_complexity: PSF_ACCURACY 0.01 with its anchor Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014bvNTrAmZxcfb1ee83ApPK --- astra.yaml | 35 +++++++++++++++++++++++++++++++---- universes/committed.yaml | 1 + 2 files changed, 32 insertions(+), 4 deletions(-) diff --git a/astra.yaml b/astra.yaml index e40d15b33..c0f078865 100644 --- a/astra.yaml +++ b/astra.yaml @@ -331,7 +331,8 @@ analyses: [detection_threshold_policy, deblending_policy, background_model, weight_map_usage, zero_weight_interpolation, detection_source_mode, epoch_membership_ccd_bounds, photometry_parameters, - spurious_detection_cleaning, blend_photometry_mask_type] + spurious_detection_cleaning, blend_photometry_mask_type, + saturation_level] decisions: detection_threshold_policy: label: Detection significance, minimum area, matched filter @@ -472,6 +473,28 @@ analyses: label: MASK_TYPE CORRECT blank: label: MASK_TYPE BLANK + saturation_level: + label: Saturation level read from each image's header + rationale: >- + SATUR_KEY SATURATE on both passes and no SATUR_LEVEL: SExtractor + takes each image's saturation level from its SATURATE header card and + sets the saturation bit (4) of FLAGS on objects with saturated + pixels. Star selection requires FLAGS == 0 at every step, so the + level decides which bright stars are rejected from the PSF sample; + tile FLAGS reach the catalogue for downstream cuts. Where the card is + absent, SExtractor falls back to SATUR_LEVEL, which neither .sex file + sets, so its built-in default applies (50000 ADU per the SExtractor + documentation). Whether the delivered exposure CCDs and MegaPipe + tiles carry SATURATE is unverified here. + Anchor: workflow/config/cfis/default_exp.sex#SATUR_KEY; + workflow/config/cfis/default_tile.sex#SATUR_KEY; + workflow/config/cfis/star_selection.setools#MASK:star_selection.FLAGS. + default: header_saturate + options: + header_saturate: + label: SATURATE header card; SExtractor default level where absent + fixed_level: + label: Fixed SATUR_LEVEL set in the .sex files photometry_parameters: label: Kron and aperture photometry definitions rationale: >- @@ -857,10 +880,14 @@ analyses: label: PSFEx pixel basis with degree-2 spatial variation per CCD rationale: >- BASIS_TYPE PIXEL, BASIS_NUMBER 20, PSF_SAMPLING 1, PSFVAR_DEGREES 2 - in XWIN/YWIN per CCD. Model flexibility sets the balance between PSF - leakage and overfitting, the dominant additive systematic in cosmic - shear. Stock values; no rationale recorded. + in XWIN/YWIN per CCD. PSF_ACCURACY 0.01 is the fractional accuracy + PSFEx assumes for PSF pixel values, which per the PSFEx documentation + enters the fit weights and so how closely the model follows bright + stars. Model flexibility sets the balance between PSF leakage and + overfitting, the dominant additive systematic in cosmic shear. Stock + values; no rationale recorded. Anchor: workflow/config/cfis/default.psfex#BASIS_TYPE; + workflow/config/cfis/default.psfex#PSF_ACCURACY; workflow/config/cfis/default.psfex#BASIS_NUMBER; workflow/config/cfis/default.psfex#PSFVAR_DEGREES. default: pixel_basis_deg2_per_ccd diff --git a/universes/committed.yaml b/universes/committed.yaml index b250eef95..585730f93 100644 --- a/universes/committed.yaml +++ b/universes/committed.yaml @@ -19,6 +19,7 @@ analyses: zero_weight_interpolation: interp_all spurious_detection_cleaning: clean_1 blend_photometry_mask_type: correct + saturation_level: header_saturate photometry_parameters: kron_25_35 detection_source_mode: sx_nomask_single_image epoch_membership_ccd_bounds: trimmed_bounds_33_2080 From 9dc22f8ca5998ddea8b848075c57e4e1e3ee7fb9 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 03:32:15 +0200 Subject: [PATCH 11/40] test(contracts): utilities import boundary src/shapepipe/utilities/CONTRACTS declares utilities-do-not-import-modules (forbid: shapepipe.utilities.* -> shapepipe.modules.*). test_contracts.py reads the forbid rule from that file and resolves every import under src/shapepipe/utilities with ast, relative imports included. It holds today; loom's check-imports agrees. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014bvNTrAmZxcfb1ee83ApPK --- src/shapepipe/utilities/CONTRACTS | 6 +++ tests/helpers/contracts.py | 85 +++++++++++++++++++++++++++++++ tests/unit/test_contracts.py | 30 +++++++++++ 3 files changed, 121 insertions(+) create mode 100644 src/shapepipe/utilities/CONTRACTS diff --git a/src/shapepipe/utilities/CONTRACTS b/src/shapepipe/utilities/CONTRACTS new file mode 100644 index 000000000..c79c51122 --- /dev/null +++ b/src/shapepipe/utilities/CONTRACTS @@ -0,0 +1,6 @@ +@cc utilities-do-not-import-modules +forbid: shapepipe.utilities.* -> shapepipe.modules.* +Utilities are the primitives module runners share (mask_query is one lookup +with two consumers, the mask_query and make_cat modules). A utility that +imports a module inverts that layering and makes one module's internals a +dependency of every other caller. diff --git a/tests/helpers/contracts.py b/tests/helpers/contracts.py index 2957d1ea4..997bb4d5c 100644 --- a/tests/helpers/contracts.py +++ b/tests/helpers/contracts.py @@ -20,6 +20,7 @@ from dataclasses import dataclass, field import ast +import fnmatch import io from pathlib import Path import re @@ -274,3 +275,87 @@ def coverage_report(contracts, record): continue unanchored.append(contract) return uncovered, unanchored + + +FORBID = re.compile(r"^\s*forbid:\s*(\S+)\s*->\s*(\S+)\s*$") + + +def forbid_rules(contracts_file): + """``(contract_id, source, target)`` for each ``forbid:`` line.""" + + rules, ident = [], None + for line in Path(contracts_file).read_text(encoding="utf-8").splitlines(): + match = VALID.match(line) + if match: + ident = match.group(3) + continue + match = FORBID.match(line) + if match and ident: + rules.append((ident, *match.groups())) + return rules + + +def module_matches(name, pattern): + """Glob match; ``pkg.*`` also matches ``pkg`` itself.""" + + return fnmatch.fnmatchcase(name, pattern) or ( + pattern.endswith(".*") and name == pattern[:-2] + ) + + +def module_name(path, src_root): + """Dotted module name of ``path`` under ``src_root``.""" + + parts = list(Path(path).relative_to(src_root).with_suffix("").parts) + if parts[-1] == "__init__": + parts.pop() + return ".".join(parts) + + +def imported_names(path, module): + """``(line, dotted_name)`` for every import in ``path``. + + Relative imports resolve against ``module``; ``from a import b`` yields + both ``a`` and ``a.b``, since ``b`` may be a submodule. + """ + + with warnings.catch_warnings(): + warnings.simplefilter("ignore", SyntaxWarning) + tree = ast.parse(Path(path).read_text(encoding="utf-8")) + package = module.split(".") + if Path(path).name != "__init__.py": + package = package[:-1] + for node in ast.walk(tree): + if isinstance(node, ast.Import): + for alias in node.names: + yield node.lineno, alias.name + elif isinstance(node, ast.ImportFrom): + base = node.module or "" + if node.level: + parent = package[: len(package) - (node.level - 1)] + base = ".".join([*parent, *([base] if base else [])]) + yield node.lineno, base + for alias in node.names: + if alias.name != "*": + yield node.lineno, f"{base}.{alias.name}" + + +def import_violations(src_root, rules): + """Imports under ``src_root`` that a ``forbid:`` rule rejects.""" + + src_root = Path(src_root) + found = [] + for path in sorted(src_root.rglob("*.py")): + if any(part in SKIP for part in path.parts): + continue + module = module_name(path, src_root) + for ident, source, target in rules: + if not module_matches(module, source): + continue + for line, name in imported_names(path, module): + if module_matches(name, target): + found.append( + f"{path.relative_to(src_root)}:{line}: {module} " + f"imports {name} (contract {ident})" + ) + return found diff --git a/tests/unit/test_contracts.py b/tests/unit/test_contracts.py index 53f108fe2..794423293 100644 --- a/tests/unit/test_contracts.py +++ b/tests/unit/test_contracts.py @@ -10,6 +10,8 @@ coverage_report, decision_errors, decision_ids, + forbid_rules, + import_violations, ) REPO_ROOT = Path(__file__).resolve().parents[2] @@ -175,3 +177,31 @@ def test_contract_coverage_report(): print(f"{len(unanchored)} @sc contracts off the record's anchors:") for contract in unanchored: print(f" {contract.id} at {contract.path}::{contract.scope}") + + +def test_forbidden_import_is_found(tmp_path): + _write(tmp_path, "pkg/utilities/CONTRACTS", """ + @cc no-up-imports + forbid: pkg.utilities.* -> pkg.modules.* + """) + _write(tmp_path, "pkg/utilities/good.py", "import os\n") + _write(tmp_path, "pkg/utilities/bad.py", "from ..modules import runner\n") + _write(tmp_path, "pkg/modules/runner.py", "from pkg.utilities import good\n") + + rules = forbid_rules(tmp_path / "pkg/utilities/CONTRACTS") + violations = import_violations(tmp_path, rules) + + assert rules == [("no-up-imports", "pkg.utilities.*", "pkg.modules.*")] + assert len(violations) == 2 + assert all("bad.py:1" in v and "no-up-imports" in v for v in violations) + + +def test_utilities_do_not_import_modules(): + contracts_file = REPO_ROOT / "src/shapepipe/utilities/CONTRACTS" + rules = forbid_rules(contracts_file) + assert [rule[0] for rule in rules] == ["utilities-do-not-import-modules"] + + violations = import_violations(REPO_ROOT / "src", rules) + + message = "Forbidden imports:\n - " + "\n - ".join(violations) + assert not violations, message From 4100d1a5bb1a33a8db41c106ee447c0d9d2e5d27 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 03:42:16 +0200 Subject: [PATCH 12/40] docs(astra): central_defect_veto; 4-fold symmetrisation for defect_fill - shape_measurement.central_defect_veto: default disabled (committed develop has no veto); radius_10px, implemented on feat/symmetrized-defect-fill, is the smallest radius with |m| < 1% - defect_fill: the recommended option is the 4-fold OR (symmetrized_4fold_noise); a single rot90 leaves coherent c2 of -0.006 to -0.012 for off-centre columns, 4-fold gives |c| < 2e-4 Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014bvNTrAmZxcfb1ee83ApPK --- astra.yaml | 59 +++++++++++++++++++++++++++++----------- universes/committed.yaml | 1 + 2 files changed, 44 insertions(+), 16 deletions(-) diff --git a/astra.yaml b/astra.yaml index c0f078865..176873867 100644 --- a/astra.yaml +++ b/astra.yaml @@ -1045,7 +1045,7 @@ analyses: metacal_scheme, centroid_source, epoch_flux_rescaling, psf_epoch_averaging, galaxy_pixel_weights, psf_likelihood_noise, megacam_ccd_flip, defect_fill, blend_handling, - epoch_masked_fraction_cut] + epoch_masked_fraction_cut, central_defect_veto] decisions: ngmix_seed_mode: label: Per-object RNG seeded from sky position @@ -1287,7 +1287,10 @@ analyses: 90-degree compensating mask. Every DES pipeline symmetrized the defect mask, and ShapePipe does not. The fill should be set independently of blend_handling; the recommended option is - symmetrized_noise. + symmetrized_4fold_noise. A single 90-degree rotation is not enough: + it leaves a coherent c2 of about -0.006 to -0.012 for columns 2-3 px + off-centre, while the 4-fold OR gives |c| < 2e-4 (measured on + feat/symmetrized-defect-fill). Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_ngmix_weights; src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal; src/shapepipe/modules/ngmix_runner.py::ngmix_runner. @@ -1317,17 +1320,20 @@ analyses: reaches this option only as a side effect of BLEND_HANDLING = uberseg. insights: [mask_metacal_acts_on_whole_stamp, mask_des_defect_practice] - symmetrized_noise: - label: 90-degree-symmetrized mask, then noise fill + symmetrized_4fold_noise: + label: 4-fold-symmetrized mask (M | rot90 | rot180 | rot270), then noise fill description: >- - Not implemented. OR the defect mask with its 90-degree rotation - about the stamp centre, zero the weight on the union, and - noise-fill the union as noisefill does. This mirrors the DES - mask symmetrization (Y1/Y3 ngmixer symmetrize_weight; Y6 - symmetrize_masking) and cancels the leading column-aligned - additive term. It roughly doubles the masked area, so the epoch - cut must be applied after symmetrizing. ShapePipe stamps are - square, so rot90 is well defined. The noise image needs no change. + Not implemented on develop; implemented on + feat/symmetrized-defect-fill. OR the defect mask with its 90, + 180 and 270-degree rotations about the stamp centre, zero the + weight on the union, and noise-fill the union as noisefill does. + This extends the DES mask symmetrization (Y1/Y3 ngmixer + symmetrize_weight; Y6 symmetrize_masking, a single rotation) and + cancels the column-aligned additive term that one rotation leaves + for off-centre columns. It can quadruple the masked area, so the + epoch cut must be applied after symmetrizing. ShapePipe stamps + are square, so the rotations are well defined. The noise image + needs no change. insights: [mask_bad_column_symmetrize, mask_des_defect_practice, mask_fixed_orientation] interpolate: label: Symmetrize, then interpolate image and noise image (DES Y6) @@ -1426,6 +1432,28 @@ analyses: simulations, although Sheldon et al. 2020 found similar blend biases with and without MOF. insights: [mask_des_y1_uberseg_only, mask_blend_bias_detection] + central_defect_veto: + label: Per-epoch veto on a defect near the stamp centre + rationale: >- + The committed code has no veto; the option below is implemented on + feat/symmetrized-defect-fill, not on develop. There it drops an epoch + when any defect pixel lies within EPOCH_CENTRAL_DEFECT_RADIUS px of + the stamp centre (0 disables), beside the masked-fraction cut in the + epoch loop. A noise-filled hole near the centre removes the galaxy's + core light, which no fill or symmetrization restores: on a round + galaxy with a 0.7 arcsec PSF, single epoch, it biases m by -6.4% for + a column 8 px from centre and -0.17% at 10 px. + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_postage_stamps. + default: disabled + options: + disabled: + label: No central veto (committed code) + radius_10px: + label: Drop the epoch if a defect lies within 10 px of the centre + description: >- + Implemented on feat/symmetrized-defect-fill, not on develop, and + recommended there: 10 px is the smallest radius with |m| < 1% for + both columns and single pixels. epoch_masked_fraction_cut: label: Per-epoch masked-fraction cut rationale: >- @@ -1445,12 +1473,11 @@ analyses: simulations show avoids calibration bias. Sheldon & Huff 2017 recommend dropping problematic epochs when many are available. The cut interacts with defect_fill: - symmetrizing roughly doubles the masked fraction, so the + symmetrizing multiplies the masked fraction (up to fourfold), so the cut should be applied after symmetrizing. UNIONS has fewer epochs than DES, so the cost in effective number density - has to be measured, not assumed. A central-region veto - (drop the epoch if a defect lies within a few pixels of - the centre) is a cheap refinement. + has to be measured, not assumed. A defect near the centre + is handled separately (central_defect_veto). Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_postage_stamps. default: one_third options: diff --git a/universes/committed.yaml b/universes/committed.yaml index 585730f93..63b6e795b 100644 --- a/universes/committed.yaml +++ b/universes/committed.yaml @@ -54,6 +54,7 @@ analyses: defect_fill: noise blend_handling: none epoch_masked_fraction_cut: one_third + central_defect_veto: disabled catalogue_assembly: decisions: star_galaxy_classification: deferred_downstream From 5e2603d8a5cdac7f6e2e117850badafbb50c57e2 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 03:57:59 +0200 Subject: [PATCH 13/40] test(contracts): contracts on config keys via governs: Resolve semicolon-separated, directory-relative governed refs with the shared ASTRA anchor resolver. Reject empty refs and repeated metadata keys, and include config contracts in the report-only coverage check. Co-Authored-By: GPT-6 Astra --- tests/helpers/contracts.py | 64 +++++++++++++- tests/unit/test_contracts.py | 157 ++++++++++++++++++++++++++++++++++- 2 files changed, 215 insertions(+), 6 deletions(-) diff --git a/tests/helpers/contracts.py b/tests/helpers/contracts.py index 997bb4d5c..f04143c9a 100644 --- a/tests/helpers/contracts.py +++ b/tests/helpers/contracts.py @@ -13,9 +13,17 @@ ``.py`` file is reported as an error rather than silently ignored. * Snakemake (``.smk``, ``Snakefile``): in ``#`` comment blocks. * ``CONTRACTS`` files: anywhere in the file; they govern their directory. + Prose needs no ``only:`` or ``forbid:`` prefix (those are import rules). A ``decision:`` meta names a decision in ``astra.yaml``: a top-level decision by its bare id, a sub-analysis decision as ``.``. +A ``governs:;`` meta names the keys/files a contract constrains. +Refs use the ASTRA anchor grammar, but paths are relative to the contract's +own directory, not the repository root; put cross-directory couplings in +an ancestor's ``CONTRACTS``. Semicolons separate refs without spaces, since +commas separate metadata pairs in ``sc-list``. Repeated metadata keys are +errors, not last-value-wins overrides. Resolution checks existence, not +whether a key is enabled or its value satisfies the contract's prose. """ from dataclasses import dataclass, field @@ -27,7 +35,7 @@ import tokenize import warnings -from tests.helpers.astra_record import extract_anchors +from tests.helpers.astra_record import extract_anchors, resolve_anchor TAG = re.compile(r"^\s*@(sc|cc)\b.*$") VALID = re.compile(r"^\s*@(sc|cc)(?:\s+\[([^\]]*)\])?\s+([\w][\w.-]*)\s*$") @@ -77,6 +85,14 @@ def parse_block(text, path, offset=0, scope=""): errors.append(f"{where}: malformed contract metadata: {meta_text}") continue meta = dict(part.split(":", 1) for part in pairs) + if len(meta) != len(pairs): + errors.append(f"{where}: duplicate metadata key: {meta_text}") + continue + if "governs" in meta and any( + not ref for ref in meta["governs"].split(";") + ): + errors.append(f"{where}: empty governs ref: {meta['governs']}") + continue prose = [] for body in lines[index + 1:]: if not body.strip(): @@ -233,6 +249,38 @@ def decision_errors(contracts, record): ] +def governed_refs(contract): + """Repo-relative anchor refs named by a contract's ``governs:`` meta. + + Paths start at the contract's directory; the shared anchor resolver + rejects absolute paths and parent traversal. The parser has already + rejected empty refs and whitespace in the list. + """ + + if "governs" not in contract.meta: + return () + directory = Path(contract.path).parent + return tuple( + (directory / ref).as_posix() + for ref in contract.meta["governs"].split(";") + ) + + +def governs_errors(contracts, root): + """Diagnostics for every ``governs:`` ref the ASTRA resolver rejects.""" + + errors = [] + for contract in contracts: + for reference in governed_refs(contract): + problem = resolve_anchor(root, reference) + if problem: + errors.append( + f"{contract.path}:{contract.line}: contract {contract.id} " + f"governs {reference!r}: {problem}" + ) + return errors + + def anchored_refs(record): """``(code_symbols, anchored_paths)`` named by the record's anchors. @@ -256,18 +304,26 @@ def coverage_report(contracts, record): """Report-only gaps between the contracts and the record. Returns ``(uncovered, unanchored)``: decision ids no contract cites, and - ``@sc`` contracts whose declaration is not an anchored symbol. A - module-docstring contract counts as anchored when an anchor names its - file or a symbol in it. + ``@sc`` contracts whose declaration or governed refs are not anchored. + A module-docstring contract counts as anchored when an anchor names its + file or a symbol in it. A ``governs:`` contract counts when at least one + ref appears verbatim in the record after rebasing to the repo root; + another key in the same file is not a match. These are report-only + links, not proof that the prose holds or every coupled key is anchored. """ cited = {c.meta["decision"] for c in contracts if "decision" in c.meta} uncovered = sorted(decision_ids(record) - cited) + references = { + ref for anchor in extract_anchors(record) for ref in anchor.references + } symbols, paths = anchored_refs(record) unanchored = [] for contract in contracts: if contract.tag != "sc": continue + if references.intersection(governed_refs(contract)): + continue if contract.scope == "module": if contract.path in paths: continue diff --git a/tests/unit/test_contracts.py b/tests/unit/test_contracts.py index 794423293..16e18828d 100644 --- a/tests/unit/test_contracts.py +++ b/tests/unit/test_contracts.py @@ -4,6 +4,8 @@ from pathlib import Path import textwrap +import pytest + from tests.helpers.astra_record import load_yaml from tests.helpers.contracts import ( collect, @@ -11,6 +13,8 @@ decision_errors, decision_ids, forbid_rules, + governed_refs, + governs_errors, import_violations, ) @@ -156,9 +160,154 @@ def f(): assert decision_ids(RECORD) == {"top_choice", "stage.inner_choice"} +def test_governs_resolves_all_refs_relative_to_the_contract_file(tmp_path): + """A multi-key coupling must not lose refs or resolve them from cwd.""" + + base = "workflow/config/cfis" + _write(tmp_path, f"{base}/default.sex", "DEBLEND_MINCONT 0.002\n") + _write(tmp_path, f"{base}/default.psfex", "PSF_SIZE 51,51\n") + _write(tmp_path, f"{base}/stamps.ini", "[STAMP]\nSIZE = 51\n") + _write(tmp_path, f"{base}/default.param", "VIGNET(51,51)\n") + _write(tmp_path, f"{base}/stars.setools", "[MASK:stars]\nFLAGS == 0\n") + _write(tmp_path, f"{base}/kernel.conv", "CONV NORM\n1 2 1\n") + # Bare-file refs also cover formats the shared resolver cannot select into. + _write(tmp_path, "workflow/config.yaml", "psf_model: psfex\n") + refs = ( + "default.sex#DEBLEND_MINCONT", + "default.psfex#PSF_SIZE", + "stamps.ini#STAMP.SIZE", + "default.param#VIGNET", + "stars.setools#MASK:stars.FLAGS", + "kernel.conv", + ) + _write(tmp_path, f"{base}/CONTRACTS", f""" + @sc [decision:top_choice,governs:{';'.join(refs)}] coupled-config + Keep the apertures coupled. + Ordinary prose is not an import rule. + """) + _write(tmp_path, "workflow/CONTRACTS", """ + @sc [decision:stage.inner_choice,governs:config.yaml] model-choice + Select matching exposure and tile models. + """) + + contracts, errors = collect(tmp_path) + by_id = {c.id: c for c in contracts} + coupled = by_id["coupled-config"] + + assert errors == [] + assert coupled.meta["governs"] == ";".join(refs) + assert coupled.scope == base + assert coupled.line == 2 + assert coupled.prose == ( + "Keep the apertures coupled. Ordinary prose is not an import rule." + ) + assert governed_refs(coupled) == tuple(f"{base}/{ref}" for ref in refs) + assert governs_errors(contracts, tmp_path) == [] + assert decision_errors(contracts, RECORD) == [] + assert forbid_rules(tmp_path / base / "CONTRACTS") == [] + + +@pytest.mark.parametrize("bad_ref", [ + "missing.sex#KEY", + "default.sex#MISSING", + "stamps.ini#STAMP.MISSING", + "stamps.ini#SIZE", # INI selectors need SECTION.KEY. + "stars.setools#MASK:missing.FLAGS", + "../default.sex#KEY", # No escape from the governing directory. + "/absolute/default.sex#KEY", +]) +@pytest.mark.parametrize("bad_first", [True, False]) +def test_every_unresolvable_governs_ref_is_an_error( + tmp_path, bad_ref, bad_first +): + """Checking only the first/last ref silently loses part of a coupling.""" + + base = "workflow/config" + _write(tmp_path, f"{base}/default.sex", "KEY 1\n") + _write(tmp_path, f"{base}/stamps.ini", "[STAMP]\nSIZE = 51\n") + _write(tmp_path, f"{base}/stars.setools", "[MASK:stars]\nFLAGS == 0\n") + refs = [bad_ref, "default.sex#KEY"] + if not bad_first: + refs.reverse() + _write(tmp_path, f"{base}/CONTRACTS", f""" + @sc [governs:{';'.join(refs)}] broken-coupling + Prose. + """) + + contracts, errors = collect(tmp_path) + problems = governs_errors(contracts, tmp_path) + + assert errors == [] + assert len(problems) == 1 + assert f"{base}/CONTRACTS:2" in problems[0] + assert "broken-coupling" in problems[0] + assert bad_ref in problems[0] + + +@pytest.mark.parametrize("metadata, message", [ + ("governs:", "malformed contract metadata"), + ("governs:a.sex#KEY b.sex#KEY", "malformed contract metadata"), + ("governs:a.sex#KEY,b.sex#KEY", "malformed contract metadata"), + ("governs:;a.sex#KEY", "empty governs ref"), + ("governs:a.sex#KEY;", "empty governs ref"), + ("governs:a.sex#KEY;;b.sex#KEY", "empty governs ref"), + ("governs:a.sex#KEY,governs:b.sex#KEY", "duplicate metadata key"), +]) +def test_malformed_governs_does_not_silently_drop_refs( + tmp_path, metadata, message +): + _write(tmp_path, "workflow/config/CONTRACTS", f""" + @sc [{metadata}] malformed-coupling + Prose. + """) + + contracts, errors = collect(tmp_path) + + assert contracts == [] + assert len(errors) == 1 + assert "workflow/config/CONTRACTS:2" in errors[0] + assert message in errors[0] + + +def test_config_coverage_uses_governed_refs_not_the_sidecar_path(tmp_path): + _write(tmp_path, "workflow/config/CONTRACTS", """ + @sc [decision:top_choice,governs:a.sex#KEY;b.ini#S.SIZE] config-coupling + Prose. + + @sc [decision:stage.inner_choice,governs:a.sex#OTHER] off-record-key + A different key in the same file is not the anchored key. + + @sc [governs:config.yaml] whole-file + Prose. + """) + record = { + "decisions": { + "top_choice": { + "rationale": "Anchor: workflow/config/b.ini#S.SIZE." + }, + "uncovered": {}, + }, + "analyses": {"stage": {"decisions": {"inner_choice": { + "rationale": "Anchor: workflow/config/a.sex#KEY." + }}}}, + "description": "Anchor: workflow/config/config.yaml.", + } + + contracts, errors = collect(tmp_path) + uncovered, unanchored = coverage_report(contracts, record) + + assert errors == [] + assert uncovered == ["uncovered"] + assert [c.id for c in unanchored] == ["off-record-key"] + + def test_repository_contracts_are_valid_and_cite_real_decisions(): record, contracts, errors = _repository() - errors = errors + decision_errors(contracts, record) + errors = ( + errors + + decision_errors(contracts, record) + + governs_errors(contracts, REPO_ROOT) + ) message = "Contract problems:\n - " + "\n - ".join(errors) assert not errors, message @@ -176,7 +325,11 @@ def test_contract_coverage_report(): print(f" {decision}") print(f"{len(unanchored)} @sc contracts off the record's anchors:") for contract in unanchored: - print(f" {contract.id} at {contract.path}::{contract.scope}") + targets = governed_refs(contract) + where = "; ".join(targets) if targets else ( + f"{contract.path}::{contract.scope}" + ) + print(f" {contract.id} at {where}") def test_forbidden_import_is_found(tmp_path): From 3d63f0b8d29b6a81a5d1b57d5613dcbde96a6d1f Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 04:01:42 +0200 Subject: [PATCH 14/40] test(astra): assert config values in the anchor grammar Check active config values and static Python literals independently of anchor resolution. Normalize numeric and boolean spellings while retaining list shape and SETools comparison operators. Document the grammar and exercise it on PSF_NOISE. Co-Authored-By: GPT-6 Astra --- astra.yaml | 33 +++- tests/helpers/astra_record.py | 299 +++++++++++++++++++++++++++++-- tests/unit/test_astra_values.py | 301 ++++++++++++++++++++++++++++++++ 3 files changed, 610 insertions(+), 23 deletions(-) create mode 100644 tests/unit/test_astra_values.py diff --git a/astra.yaml b/astra.yaml index 176873867..aee21da7a 100644 --- a/astra.yaml +++ b/astra.yaml @@ -1,16 +1,37 @@ # ASTRA record of ShapePipe's scientific decisions: the choices embedded in # the code and the committed workflow configs (workflow/config/cfis/), why # they stand, and the alternatives. CLAUDE.md says when to amend it; the -# anchor test tests/unit/test_astra_anchors.py keeps it resolvable. +# tests/unit/test_astra_{anchors,values}.py check locations and values separately. # # Conventions: # * Every rationale ends with one sentence "Anchor: ; ." Each ref # is a path relative to the repo root: CODE `path::Symbol` (a def, class # or assignment target, dotted for nesting), CONFIG `path#SECTION.KEY` -# (`path#KEY` for sectionless .sex/.psfex/.param files, where a -# commented-out key still resolves), or FILE `path`. No line numbers. -# Numeric values are stated once, in the rationale, next to the anchor -# that holds them. +# (`path#KEY` for sectionless .sex/.psfex/.param files; .setools uses +# SECTION.KEY), or FILE `path`. No line numbers. Commented-out keys may +# resolve as locations, but only active settings can assert values. +# * Value assertions: `Anchor: path#SECTION.KEY = 1.5; path::NAME = 51.` +# A ref without ` = value` is location-only. Attaching the expectation to +# its locator avoids guessing which file a prose number describes, while +# keeping ordinary, schema-valid prose; no extra ASTRA keys or second +# parameter table. The anchor is the canonical recorded expectation; +# rationale/labels may repeat it for readability. Option ids are stable +# references, never parsed for values. Code/config remains execution truth. +# * Values are numbers, boolean words, strings (quote expressions), or flat +# comma lists, optionally bracketed. Semicolons are reserved for refs. +# Decimal equality is exact (1 = 1.0, 5e-4 = 0.0005), without rounding; +# Y/yes/true/on and N/no/false/off are case-insensitive boolean aliases, +# distinct from 1/0. Trim outer whitespace; other strings are case-sensitive. +# Lists preserve order and length. Only .param VIGNET and .psfex PSF_SIZE +# accept square-size shorthand: 51 = 51,51 (never 51,53). +# * SETools predicates retain their operators as quoted text; repeated cuts +# on one key are an ordered list, e.g. MAG_AUTO = ["> 18.", "< 22."]. +# Expressions compare as text, not algebra. Python selectors may append +# `[key.subkey]` to a named assignment (identifier-like string dict keys); +# only the selected literal is read, including dict(key=value) syntax. +# Ambiguous bindings/settings fail; no imports, calls, arithmetic, argument +# defaults, environment expansion or implicit tool defaults are evaluated. +# These limits keep the check static rather than a second pipeline runtime. # * A decision's `default` is the option the committed code and configs # select; universes/committed.yaml pins it. # * `excluded: true` means considered and rejected. An option the code does @@ -1232,7 +1253,7 @@ analyses: with PSF_NOISE 1e-5; without it the g-prior swamps the PSF likelihood. The recovered PSF shape and size are flat across 1e-4 to 1e-6 on the digital twin. - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::PSF_NOISE; + Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::PSF_NOISE = 1e-5; src/shapepipe/modules/ngmix_package/ngmix.py::make_ngmix_observation. default: psf_noise_1em5 options: diff --git a/tests/helpers/astra_record.py b/tests/helpers/astra_record.py index 1a7123c7d..c95c67caf 100644 --- a/tests/helpers/astra_record.py +++ b/tests/helpers/astra_record.py @@ -4,6 +4,8 @@ import ast import configparser from dataclasses import dataclass +from decimal import Decimal +from functools import lru_cache import json from pathlib import Path import re @@ -83,7 +85,23 @@ def extract_anchors(document): return anchors +def _split_assertion(reference): + """Split the reserved, whitespace-delimited `` = `` (never a cut's ==).""" + + if ";" in reference: + raise ValueError("semicolon is reserved for separating anchor refs") + parts = re.split(r"\s+=\s*", reference.strip(), maxsplit=1) + locator = parts[0] + expected = parts[1].strip() if len(parts) == 2 else None + if not locator or re.search(r"\s|=", locator): + raise ValueError("expected a locator optionally followed by ' = value'") + if expected is not None and (not expected or expected.startswith("=")): + raise ValueError("expected a nonempty value after ' = '") + return locator, expected + + def _parse_reference(reference): + reference, _ = _split_assertion(reference) if "::" in reference: path, symbol = reference.split("::", 1) return "code", path, symbol @@ -116,7 +134,10 @@ def _snakemake_symbol(text, symbol): def resolve_anchor(root, reference): """Return ``None`` if a reference resolves, otherwise a diagnostic.""" - kind, relative, selector = _parse_reference(reference) + try: + kind, relative, selector = _parse_reference(reference) + except ValueError as error: + return str(error) path = Path(relative) if path.is_absolute() or ".." in path.parts: return "path must be relative to the repository root" @@ -139,11 +160,14 @@ def resolve_anchor(root, reference): if target.suffix != ".py": return "code-symbol refs must name a .py file" try: - tree = ast.parse(text, filename=str(target)) - except SyntaxError as error: - return f"cannot parse Python file: {error}" - if not _has_symbol(tree, selector): - return f"no def/class/assignment target named {selector!r}" + tree = _python_tree(text) + symbol, keys = _code_selector(selector) + if not _has_symbol(tree, symbol): + return f"no def/class/assignment target named {symbol!r}" + if keys: + _selected_python_node(tree, selector) + except (SyntaxError, ValueError) as error: + return f"cannot resolve Python selector: {error}" return None suffix = target.suffix.lower() @@ -164,14 +188,11 @@ def _ini_key(text, selector): if "." not in selector: return "INI config ref needs SECTION.KEY" section, key = selector.rsplit(".", 1) - parser = configparser.ConfigParser( - interpolation=None, strict=False, allow_no_value=True - ) try: - parser.read_string(text) + parser = _ini_parser(text, strict=False) except configparser.Error as error: return f"cannot parse INI file: {error}" - if not parser.has_section(section): + if section != parser.default_section and not parser.has_section(section): return f"INI section {section!r} is missing" if not parser.has_option(section, key): return f"INI key {key!r} is missing from section {section!r}" @@ -205,8 +226,9 @@ def _target_names(target): return [] +@lru_cache(maxsize=128) def _bindings(scope): - """Collect declarations and assignment targets in one lexical scope.""" + """Collect all bindings per name; value reads must not pick one silently.""" result = {} @@ -215,7 +237,7 @@ def visit(node): node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef), ): - result[node.name] = node + result.setdefault(node.name, []).append(node) return if isinstance(node, ast.Lambda): return @@ -231,9 +253,10 @@ def visit(node): targets = [] for target in targets: if target is not None: - result.update(dict.fromkeys(_target_names(target), node)) + for name in _target_names(target): + result.setdefault(name, []).append(node) if isinstance(node, ast.ExceptHandler) and node.name: - result[node.name] = node + result.setdefault(node.name, []).append(node) for child in ast.iter_child_nodes(node): visit(child) @@ -246,9 +269,10 @@ def _has_symbol(tree, symbol): scope = tree parts = symbol.split(".") for index, part in enumerate(parts): - declaration = _bindings(scope).get(part) - if declaration is None: + declarations = _bindings(scope).get(part) + if not declarations: return False + declaration = declarations[-1] if index == len(parts) - 1: return True if not isinstance( @@ -260,6 +284,247 @@ def _has_symbol(tree, symbol): return False +def _ini_parser(text, *, strict=True): + parser = configparser.ConfigParser( + interpolation=None, strict=strict, allow_no_value=True + ) + parser.optionxform = str + parser.read_string(text) + return parser + + +@lru_cache(maxsize=16) +def _python_tree(text): + # Cache by source, not path: editing a file must invalidate the read. + return ast.parse(text) + + +def _code_selector(selector): + match = re.fullmatch(r"([\w.]+)(?:\[([\w.]+)\])?", selector) + if not match or any(not p.isidentifier() for p in match[1].split(".")): + raise ValueError(f"invalid Python selector {selector!r}") + keys = tuple(match[2].split(".")) if match[2] else () + if any(not key.isidentifier() for key in keys): + raise ValueError("dict paths need dot-separated identifier keys") + return match[1], keys + + +def _dict_entry(node, key): + """Select syntax, not a runtime value; never execute a dict() call.""" + + if isinstance(node, ast.Dict): + if any( + not isinstance(k, ast.Constant) or not isinstance(k.value, str) + for k in node.keys + ): + raise ValueError("dict selectors need literal string keys, no **") + items = [(k.value, v) for k, v in zip(node.keys, node.values)] + elif ( + isinstance(node, ast.Call) and isinstance(node.func, ast.Name) + and node.func.id == "dict" and not node.args + and all(k.arg is not None for k in node.keywords) + ): + items = [(k.arg, k.value) for k in node.keywords] + else: + raise ValueError("dict selectors need {...} or dict(key=value) syntax") + names = [name for name, _ in items] + if len(names) != len(set(names)): + raise ValueError("ambiguous duplicate dict keys") + if key not in names: + raise ValueError(f"dict key {key!r} is missing") + return dict(items)[key] + + +def _selected_python_node(tree, selector): + symbol, keys = _code_selector(selector) + scope = tree + parts = symbol.split(".") + for index, part in enumerate(parts): + declarations = _bindings(scope).get(part, []) + if len(declarations) != 1: + raise ValueError( + f"{symbol!r} needs one binding; found {len(declarations)} " + f"for {part!r}" + ) + node = declarations[0] + if index < len(parts) - 1: + if not isinstance( + node, (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef) + ): + raise ValueError(f"{part!r} is not a lexical scope") + scope = node + if not isinstance(node, (ast.Assign, ast.AnnAssign)): + raise ValueError(f"{symbol!r} is not a literal assignment") + targets = node.targets if isinstance(node, ast.Assign) else [node.target] + if any(not isinstance(target, ast.Name) for target in targets): + raise ValueError("value assertions need simple named assignment targets") + node = node.value + for key in keys: + node = _dict_entry(node, key) + return node + + +def _line_value(text, selector, suffix): + """Read active lines; SETools repeated predicates form an ordered list.""" + + section = None + if suffix == ".setools": + if "." not in selector: + raise ValueError("SETools config ref needs SECTION.KEY") + section, key = selector.rsplit(".", 1) + else: + key = selector.rsplit(".", 1)[-1] + pattern = re.compile(rf"^{re.escape(key)}(?=$|\s|=|\(|<|>)(.*)$") + active = section is None + values = [] + predicates = [] + for line in text.splitlines(): + line = line.split("#", 1)[0].strip() + if section is not None and line.startswith("[") and line.endswith("]"): + active = line[1:-1].strip() == section + continue + match = pattern.fullmatch(line) if active else None + if not match: + continue + value = match[1].strip() + predicate = suffix == ".setools" and value.startswith( + ("==", "!=", "<", ">") + ) + if suffix == ".param" and value.startswith("(") and value.endswith(")"): + value = value[1:-1] + elif value.startswith("=") and not predicate: + value = value[1:].strip() + values.append(value) + predicates.append(predicate) + if not values or any(not value for value in values): + raise ValueError(f"no active value for {selector!r}") + if len(values) > 1 and not all(predicates): + raise ValueError(f"ambiguous active values for {selector!r}: {values!r}") + return values if len(values) > 1 else values[0] + + +_NUMBER = re.compile(r"[+-]?(?:\d+(?:\.\d*)?|\.\d+)(?:[eE][+-]?\d+)?\Z") +_BOOLEANS = { + "y": True, "yes": True, "true": True, "on": True, + "n": False, "no": False, "false": False, "off": False, +} + + +def _normalise_value(value): + """Use tagged atoms so boolean True cannot compare equal to number 1.""" + + if isinstance(value, bool): + return "bool", value + if isinstance(value, (int, float)): + number = Decimal(str(value)) + if not number.is_finite(): + raise ValueError("numeric values must be finite") + return "number", number + if isinstance(value, (list, tuple)): + elements = tuple(_normalise_value(item) for item in value) + if any(kind == "list" for kind, _ in elements): + raise ValueError("only flat lists are supported") + return "list", elements + if not isinstance(value, str): + raise ValueError("expected a number, boolean, string or flat list") + value = value.strip() + if not value: + raise ValueError("empty values/list elements are not supported") + if value[0] in "[{'\"": + # BaseLoader keeps even 5e-4 and Y as strings, avoiding YAML 1.1's + # inconsistent numeric/boolean coercions. It constructs no objects. + parsed = yaml.load(value, Loader=yaml.BaseLoader) + if isinstance(parsed, list): + return _normalise_value(parsed) + if not isinstance(parsed, str): + raise ValueError("expected a scalar or flat list, not a mapping") + # Quotes protect commas/operators; their contents are a single atom. + value = parsed.strip() + elif "," in value: + return _normalise_value(value.split(",")) + if _NUMBER.fullmatch(value): + return "number", Decimal(value) + if value.lower() in _BOOLEANS: + return "bool", _BOOLEANS[value.lower()] + return "text", value + + +def _square_stamp(value): + if value[0] == "number": + return "list", (value, value) + return value + + +def check_anchor_value(root, reference): + """Return a diagnostic for a mismatched/unreadable assertion, else None. + + A reference without `` = value`` is location-only. Numbers compare + exactly after decimal normalization, not with a tolerance. No imported + code, environment expansion, function calls or expressions are evaluated. + """ + + expected = None + actual = "" + try: + _, expected = _split_assertion(reference) + if expected is None: + return None + kind, relative, selector = _parse_reference(reference) + problem = resolve_anchor(root, reference) + if problem: + raise ValueError(problem) + target = Path(root) / relative + text = target.read_text(encoding="utf-8") + suffix = target.suffix.lower() + if kind == "code" and suffix == ".py": + node = _selected_python_node(_python_tree(text), selector) + try: + actual = ast.literal_eval(node) + except (ValueError, TypeError) as error: + raise ValueError( + "selected Python value is not a literal" + ) from error + elif kind == "config" and suffix == ".ini": + section, key = selector.rsplit(".", 1) + actual = _ini_parser(text).get(section, key) + if actual is None: + raise ValueError("no active value for INI key") + elif kind == "config" and suffix in { + ".sex", ".psfex", ".ww", ".param", ".conf", ".setools" + }: + actual = _line_value(text, selector, suffix) + else: + raise ValueError( + "value assertions need a config key or Python assignment" + ) + want = _normalise_value(expected) + got = _normalise_value(actual) + if (suffix, selector) in {(".param", "VIGNET"), (".psfex", "PSF_SIZE")}: + want, got = _square_stamp(want), _square_stamp(got) + if want == got: + return None + return f"expected {expected!r}, actual {actual!r}" + except ( + ValueError, OSError, SyntaxError, configparser.Error, yaml.YAMLError + ) as error: + return f"expected {expected!r}, actual {actual!r}: {error}" + + +def value_errors(root, record): + """Check assertions, naming decision, ref, expected and actual in errors.""" + + errors = [] + for anchor in extract_anchors(record): + if anchor.error: + errors.append(f"{anchor.location}: {anchor.error}") + continue + for reference in anchor.references: + problem = check_anchor_value(root, reference) + if problem: + errors.append(f"{anchor.location}: {reference}: {problem}") + return errors + + def universe_errors(record, universe): """Check scoped decision IDs and options against the ASTRA record.""" diff --git a/tests/unit/test_astra_values.py b/tests/unit/test_astra_values.py new file mode 100644 index 000000000..d50bca4e1 --- /dev/null +++ b/tests/unit/test_astra_values.py @@ -0,0 +1,301 @@ +"""Check recorded values, not just locations, without importing the pipeline. + +Failure modes: plausible numeric drift, wrong key/section/case, commented or +ambiguous settings, changed cut operators, lost list elements, and executing +Python while trying to inspect it. Fixtures exercise each through the same +reader as the record; location resolution must remain independent of equality. +""" + +from pathlib import Path + +import pytest + +from tests.helpers.astra_record import ( + check_anchor_value, + extract_anchors, + load_yaml, + resolve_anchor, + value_errors, +) + + +REPO_ROOT = Path(__file__).resolve().parents[2] + + +def record_with(reference): + """Put a reference in a scoped decision, as in the real record.""" + + return { + "analyses": { + "detection": { + "decisions": { + "threshold": { + "rationale": f"Affects selection. Anchor: {reference}." + } + } + } + } + } + + +def test_every_astra_value_matches(): + record = load_yaml(REPO_ROOT / "astra.yaml") + anchors = extract_anchors(record) + assert any(" = " in ref for a in anchors for ref in a.references), ( + "astra.yaml contains no value assertions" + ) + errors = value_errors(REPO_ROOT, record) + assert not errors, "ASTRA value mismatches:\n - " + "\n - ".join(errors) + + +def test_anchor_grammar_keeps_values_and_location_only_refs(tmp_path): + (tmp_path / "image.sex").write_text("THRESH 1.25\n", encoding="utf-8") + (tmp_path / "code.py").write_text("def fit():\n pass\n", encoding="utf-8") + record = record_with("image.sex#THRESH = 1.25; code.py::fit; image.sex") + anchor, = extract_anchors(record) + assert anchor.error is None + assert anchor.references == ( + "image.sex#THRESH = 1.25", "code.py::fit", "image.sex" + ) + assert all(resolve_anchor(tmp_path, ref) is None for ref in anchor.references) + assert value_errors(tmp_path, record) == [] + + +@pytest.mark.parametrize( + "actual, expected", + [ + ("1.0", "1"), + ("13.", "13.0"), + ("5e-4", "0.0005"), + ("-1e3", "-1000"), + ("Y", "True"), + ("false", "N"), + ("yes", "on"), + ("off", "No"), + ("2.5, 3.5", "[2.50, 3.500]"), + ("XWIN_IMAGE,YWIN_IMAGE", "XWIN_IMAGE, YWIN_IMAGE"), + ("$ROOT/data{number}.fits", '"$ROOT/data{number}.fits"'), + ], +) +def test_equivalent_spellings(tmp_path, actual, expected): + (tmp_path / "config.ini").write_text(f"[SCIENCE]\nKEY = {actual}\n") + ref = f"config.ini#SCIENCE.KEY = {expected}" + assert check_anchor_value(tmp_path, ref) is None + + +@pytest.mark.parametrize( + "actual, expected", + [ + ("1.000001", "1"), + ("1", "True"), + ("0", "False"), + ("51,51", "51"), # No scalar broadcasting for ordinary keys. + ("51,52", "51,51"), + ("1,2", "2,1"), + ("1,1,1", "1,1"), + ("1", "[1]"), + ("map_weight", "MAP_WEIGHT"), + ], +) +def test_normalization_does_not_hide_drift(tmp_path, actual, expected): + (tmp_path / "config.sex").write_text(f"KEY {actual}\n") + problem = check_anchor_value(tmp_path, f"config.sex#KEY = {expected}") + assert "expected" in problem and "actual" in problem + + +@pytest.mark.parametrize( + "filename, line, selector", + [ + ("default.sex", "THRESH 0.0005 # comment", "THRESH"), + ("default.psfex", "PSF_ACCURACY 0.0005", "PSF_ACCURACY"), + ("default.ww", "WEIGHT_MIN = 0.0005", "WEIGHT_MIN"), + ("default.conf", "THRESH 0.0005", "THRESH"), + ("stars.setools", "[RAND_SPLIT:stars]\nRATIO = 0.0005", + "RAND_SPLIT:stars.RATIO"), + ], +) +def test_line_config_formats(tmp_path, filename, line, selector): + (tmp_path / filename).write_text(line + "\n") + ref = f"{filename}#{selector} = 5e-4" + assert resolve_anchor(tmp_path, ref) is None + assert check_anchor_value(tmp_path, ref) is None + + +@pytest.mark.parametrize( + "filename, line, key", + [ + ("columns.param", "VIGNET(51,51)", "VIGNET"), + ("model.psfex", "PSF_SIZE 51,51", "PSF_SIZE"), + ("model.psfex", "PSF_SIZE 51", "PSF_SIZE"), + ], +) +def test_square_stamp_shorthand(tmp_path, filename, line, key): + target = tmp_path / filename + target.write_text(line + " # stamp\n") + for expected in ("51", "51,51", "[51.0, 51]"): + ref = f"{filename}#{key} = {expected}" + assert check_anchor_value(tmp_path, ref) is None + target.write_text(line.replace("51", "53", 1)) + assert check_anchor_value(tmp_path, f"{filename}#{key} = 51") is not None + + +def test_ini_case_sections_defaults_and_interpolation(tmp_path): + (tmp_path / "config.ini").write_text( + "[DEFAULT]\nENABLED = True\n" + "[SCIENCE]\nKey = 30\nKEY = 31\nPATH = $DATA/%s/file\n" + "[OTHER]\nKEY = 99\n" + ) + for selector, value in ( + ("SCIENCE.Key", "30"), ("SCIENCE.KEY", "31"), + ("OTHER.KEY", "99"), ("SCIENCE.ENABLED", "Y"), + ("DEFAULT.ENABLED", "True"), ("SCIENCE.PATH", "$DATA/%s/file"), + ): + ref = f"config.ini#{selector} = {value}" + assert resolve_anchor(tmp_path, ref) is None + assert check_anchor_value(tmp_path, ref) is None + assert check_anchor_value(tmp_path, "config.ini#SCIENCE.key = 31") is not None + assert check_anchor_value(tmp_path, "config.ini#KEY = 31") is not None + + +@pytest.mark.parametrize( + "filename, contents, selector", + [ + ("config.sex", "# KEY 1\nKEY_EXTRA 1\n", "KEY"), + ("config.param", "# VIGNET(51,51)\n", "VIGNET"), + ("config.setools", "[MASK:stars]\n# FLAGS == 0\n", "MASK:stars.FLAGS"), + ], +) +def test_commented_keys_resolve_but_cannot_assert_active_values( + tmp_path, filename, contents, selector +): + (tmp_path / filename).write_text(contents) + assert resolve_anchor(tmp_path, f"{filename}#{selector}") is None + problem = check_anchor_value(tmp_path, f"{filename}#{selector} = 1") + assert "active" in problem + + +@pytest.mark.parametrize( + "filename, contents, selector", + [ + ("config.sex", "KEY 1\nKEY 2\n", "KEY"), + ("config.ini", "[SCIENCE]\nKEY = 1\nKEY = 2\n", "SCIENCE.KEY"), + ("config.setools", "[RAND_SPLIT:s]\nRATIO = 1\nRATIO = 2\n", + "RAND_SPLIT:s.RATIO"), + ], +) +def test_duplicate_settings_fail_closed(tmp_path, filename, contents, selector): + (tmp_path / filename).write_text(contents) + assert check_anchor_value(tmp_path, f"{filename}#{selector} = 2") is not None + + +def test_setools_cuts_keep_operators_and_all_bounds(tmp_path): + config = tmp_path / "stars.setools" + contents = ( + "[MASK:preselect]\nMAG_AUTO < 21\n" + "[MASK:stars]\nMAG_AUTO > 18.\nMAG_AUTO < 22.\nFLAGS == 0\n" + ) + config.write_text(contents) + ref = 'stars.setools#MASK:stars.MAG_AUTO = ["> 18.", "< 22."]' + assert check_anchor_value(tmp_path, ref) is None + flag_ref = 'stars.setools#MASK:stars.FLAGS = "== 0"' + assert check_anchor_value(tmp_path, flag_ref) is None + for altered in ( + contents.replace("> 18.", ">= 18."), + contents.replace("MAG_AUTO < 22.\n", ""), + ): + config.write_text(altered) + assert check_anchor_value(tmp_path, ref) is not None + + +def test_python_literals_and_dict_paths_without_importing(tmp_path): + (tmp_path / "constants.py").write_text( + "raise RuntimeError('must not execute')\n" + "WIDTH: int = 51\nNOISE = 5e-4\n" + "COMPLETENESS = {'exp_split': {'split_exp_runner': " + "dict(expect=121, warn=True)}}\n" + "class Model:\n WIDTH = 53\n" + "def fit():\n" + " limits = [-1.0, 1.0e3]\n" + " options = {'step': 0.01, 'dynamic': choose_at_runtime()}\n" + ) + for selector, expected in ( + ("WIDTH", "51.0"), ("NOISE", "0.0005"), ("Model.WIDTH", "53"), + ("fit.limits", "-1,1000"), ("fit.options[step]", "1e-2"), + ("COMPLETENESS[exp_split.split_exp_runner.expect]", "121"), + ("COMPLETENESS[exp_split.split_exp_runner.warn]", "Y"), + ): + ref = f"constants.py::{selector} = {expected}" + assert resolve_anchor(tmp_path, ref) is None + assert check_anchor_value(tmp_path, ref) is None + missing = "constants.py::fit.options[missing]" + assert resolve_anchor(tmp_path, missing) is not None + + +@pytest.mark.parametrize( + "contents, selector", + [ + ("X = get_value()", "X"), + ("X = 1 / 3", "X"), + ("X = 1\nX = 2", "X"), + ("X = 1\nX += 1", "X"), + ("X, Y = 1, 2", "X"), + ("def X():\n return 1", "X"), + ("X = {'a': 1, 'a': 2}", "X[a]"), + ("X = dict(**other)", "X[a]"), + ("X = {'a': 1, **other}", "X[a]"), + ("X = {variable: 1}", "X[a]"), + ], +) +def test_python_nonliteral_or_ambiguous_values_fail_closed( + tmp_path, contents, selector +): + (tmp_path / "constants.py").write_text(contents + "\n") + ref = f"constants.py::{selector} = 1" + assert check_anchor_value(tmp_path, ref) is not None + + +@pytest.mark.parametrize( + "reference", + [ + "config.sex#KEY =", "config.sex#KEY == 1", "config.sex = 1", + "config.sex#KEY = [1,", "config.sex#KEY = 1,,2", + "config.sex#KEY = {'value': 1}", "config.sex#KEY = [[1]]", + "../outside.sex#KEY = 1", "/outside.sex#KEY = 1", + "config.sex#KEY = 1; config.sex#KEY = 2", + ], +) +def test_malformed_assertions_are_errors(tmp_path, reference): + (tmp_path / "config.sex").write_text("KEY 1\n") + assert check_anchor_value(tmp_path, reference) is not None + + +def test_injected_drift_reports_decision_ref_expected_actual(tmp_path): + config = tmp_path / "detect.sex" + config.write_text("DETECT_THRESH 1.0\n") + reference = "detect.sex#DETECT_THRESH = 1" + record = record_with(reference) + assert value_errors(tmp_path, record) == [] + + config.write_text("DETECT_THRESH 1.5\n") + assert resolve_anchor(tmp_path, reference) is None + error, = value_errors(tmp_path, record) + for detail in ( + "analyses.detection.decisions.threshold", "detect.sex#DETECT_THRESH", + "expected", "1", "actual", "1.5", + ): + assert detail in error + + +def test_python_drift_is_not_hidden_by_a_cached_ast(tmp_path): + target = tmp_path / "constants.py" + reference = "constants.py::WIDTH = 51" + target.write_text("WIDTH = 51\n") + assert check_anchor_value(tmp_path, reference) is None + target.write_text("WIDTH = 53\n") + assert resolve_anchor(tmp_path, reference) is None + assert check_anchor_value(tmp_path, reference) is not None + + +def test_malformed_anchor_sentence_cannot_silently_skip_values(tmp_path): + record = {"decisions": {"width": {"rationale": "Anchor: config.sex#KEY = 1"}}} + assert value_errors(tmp_path, record) From ecc9169c197f9b3818dbfb4cc761ac5b593c47d0 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 04:06:03 +0200 Subject: [PATCH 15/40] docs(sc): contracts for the config-anchored decisions Keep canonical contracts beside the CFIS configs, with an inherited workflow contract beside the PSF selector. Cover the 17 previously uncovered decisions and the exposure pixel-scale/diagnostic coupling without changing scientific settings; retain the known config inconsistencies explicitly. Match contract coverage against anchor locators through the shared parser, including the value-assertion grammar added concurrently. Keep the tile-overlap config projection visible as a report-only anchor gap. Co-Authored-By: GPT-6 Astra --- tests/helpers/contracts.py | 54 ++++++++++------ tests/unit/test_contracts.py | 9 ++- workflow/CONTRACTS | 13 ++++ workflow/config/cfis/CONTRACTS | 109 +++++++++++++++++++++++++++++++++ 4 files changed, 164 insertions(+), 21 deletions(-) create mode 100644 workflow/CONTRACTS create mode 100644 workflow/config/cfis/CONTRACTS diff --git a/tests/helpers/contracts.py b/tests/helpers/contracts.py index f04143c9a..2f995f474 100644 --- a/tests/helpers/contracts.py +++ b/tests/helpers/contracts.py @@ -18,11 +18,12 @@ A ``decision:`` meta names a decision in ``astra.yaml``: a top-level decision by its bare id, a sub-analysis decision as ``.``. A ``governs:;`` meta names the keys/files a contract constrains. -Refs use the ASTRA anchor grammar, but paths are relative to the contract's +Refs use ASTRA anchor locators, but paths are relative to the contract's own directory, not the repository root; put cross-directory couplings in an ancestor's ``CONTRACTS``. Semicolons separate refs without spaces, since -commas separate metadata pairs in ``sc-list``. Repeated metadata keys are -errors, not last-value-wins overrides. Resolution checks existence, not +commas separate metadata pairs in ``sc-list``. Value assertions stay in +ASTRA, not in whitespace-free ``governs:`` metadata. Repeated metadata keys +are errors, not last-value-wins overrides. Resolution checks existence, not whether a key is enabled or its value satisfies the contract's prose. """ @@ -35,7 +36,12 @@ import tokenize import warnings -from tests.helpers.astra_record import extract_anchors, resolve_anchor +from tests.helpers.astra_record import ( + _parse_reference, + _split_assertion, + extract_anchors, + resolve_anchor, +) TAG = re.compile(r"^\s*@(sc|cc)\b.*$") VALID = re.compile(r"^\s*@(sc|cc)(?:\s+\[([^\]]*)\])?\s+([\w][\w.-]*)\s*$") @@ -281,6 +287,22 @@ def governs_errors(contracts, root): return errors +def _anchor_locators(record): + """Strip optional value assertions using the shared anchor grammar.""" + + locators = set() + for anchor in extract_anchors(record): + for reference in anchor.references: + try: + locator, _ = _split_assertion(reference) + except ValueError: + # The anchor tests diagnose malformed refs; coverage reports + # the remaining links rather than failing to print any gaps. + continue + locators.add(locator) + return locators + + def anchored_refs(record): """``(code_symbols, anchored_paths)`` named by the record's anchors. @@ -289,14 +311,11 @@ def anchored_refs(record): """ symbols, paths = set(), set() - for anchor in extract_anchors(record): - for reference in anchor.references: - if "::" in reference: - path, symbol = reference.split("::", 1) - symbols.add((path, symbol)) - else: - path = reference.split("#", 1)[0] - paths.add(path) + for reference in _anchor_locators(record): + kind, path, selector = _parse_reference(reference) + if kind == "code": + symbols.add((path, selector)) + paths.add(path) return symbols, paths @@ -307,16 +326,15 @@ def coverage_report(contracts, record): ``@sc`` contracts whose declaration or governed refs are not anchored. A module-docstring contract counts as anchored when an anchor names its file or a symbol in it. A ``governs:`` contract counts when at least one - ref appears verbatim in the record after rebasing to the repo root; - another key in the same file is not a match. These are report-only - links, not proof that the prose holds or every coupled key is anchored. + locator appears in the record after rebasing to the repo root and + removing any value assertion from the record's ref; another key in the + same file is not a match. These are report-only links, not proof that + the prose holds or every coupled key is anchored. """ cited = {c.meta["decision"] for c in contracts if "decision" in c.meta} uncovered = sorted(decision_ids(record) - cited) - references = { - ref for anchor in extract_anchors(record) for ref in anchor.references - } + references = _anchor_locators(record) symbols, paths = anchored_refs(record) unanchored = [] for contract in contracts: diff --git a/tests/unit/test_contracts.py b/tests/unit/test_contracts.py index 16e18828d..21b1d72d3 100644 --- a/tests/unit/test_contracts.py +++ b/tests/unit/test_contracts.py @@ -269,9 +269,12 @@ def test_malformed_governs_does_not_silently_drop_refs( assert message in errors[0] -def test_config_coverage_uses_governed_refs_not_the_sidecar_path(tmp_path): +@pytest.mark.parametrize("assertion", ["", " = 51"]) +def test_config_coverage_uses_governed_refs_not_the_sidecar_path( + tmp_path, assertion +): _write(tmp_path, "workflow/config/CONTRACTS", """ - @sc [decision:top_choice,governs:a.sex#KEY;b.ini#S.SIZE] config-coupling + @sc [decision:top_choice,governs:a.sex#UNRECORDED;b.ini#S.SIZE] config-coupling Prose. @sc [decision:stage.inner_choice,governs:a.sex#OTHER] off-record-key @@ -283,7 +286,7 @@ def test_config_coverage_uses_governed_refs_not_the_sidecar_path(tmp_path): record = { "decisions": { "top_choice": { - "rationale": "Anchor: workflow/config/b.ini#S.SIZE." + "rationale": f"Anchor: workflow/config/b.ini#S.SIZE{assertion}." }, "uncovered": {}, }, diff --git a/workflow/CONTRACTS b/workflow/CONTRACTS new file mode 100644 index 000000000..f1bbfb65b --- /dev/null +++ b/workflow/CONTRACTS @@ -0,0 +1,13 @@ +Scientific contracts for workflow-level configuration +===================================================== + +This directory contains config.yaml, so its PSF-selection contract belongs here rather than only beside the CFIS stage configs. +sc-list includes this file when queried for config.yaml or any config below workflow/config/cfis/. +Refs in governs: are relative to this directory and use the shared ASTRA anchor resolver; config.yaml is a whole-file ref because the resolver has no YAML-key selector. +CFIS key-level contracts live in config/cfis/CONTRACTS. + +@sc [decision:star_selection_psf.psf_modelling_software,governs:config.yaml;config/cfis/config_exp_psfex.ini#EXECUTION.MODULE;config/cfis/config_tile_PiViVi_psfex.ini#EXECUTION.MODULE;config/cfis/config_exp_mccd.ini#EXECUTION.MODULE;config/cfis/config_exp_mccd.ini#SEXTRACTOR_RUNNER.INPUT_MODULE;config/cfis/config_tile_PiViVi_mccd.ini#EXECUTION.MODULE;config/cfis/config_MCCD.ini#INSTANCE.FP_GEOMETRY] psf-model-selects-a-matched-chain +The psf_model selector must choose a matching exposure-model producer and tile-interpolation consumer, including the model selected through SP_PSF in downstream inputs. +PSFEx models each CCD independently; MCCD couples the focal plane, so switching software changes the scientific model, not just the executable name. +[LINT] the committed MCCD exposure config still requests the removed mask_runner; repair that input chain and validate the focal-plane model before treating MCCD as a runnable alternative. +Warning-only completeness counts for the untested MCCD chain do not establish scientific equivalence to the PSFEx chain. diff --git a/workflow/config/cfis/CONTRACTS b/workflow/config/cfis/CONTRACTS new file mode 100644 index 000000000..9d06e72c1 --- /dev/null +++ b/workflow/config/cfis/CONTRACTS @@ -0,0 +1,109 @@ +Scientific contracts for the committed CFIS configuration +======================================================= + +Read these before changing a governed key; astra.yaml holds the decision and alternatives. +Query sc-list with a config path (for example workflow/config/cfis/default_tile.sex) to see these contracts and the parent workflow/CONTRACTS. +The directory contracts remain visible to sc-list even though it does not scan config comments. +Sidecars are canonical rather than duplicated in comments, keeping native config inputs unchanged and each constraint in one place. + +A governs: value lists ASTRA-style locators relative to this directory, separated by semicolons without spaces; value assertions stay in astra.yaml. +Use file#KEY for sectionless configs, file#SECTION.KEY for INI/SETools, or a bare file for a whole-file constraint, including absent keys. +Keep each tag on one line and separate contracts with a blank line: sc-list reads the following nonblank lines as prose. +Use plain prose for scientific constraints; only:/forbid: lines are for the separate import-boundary checker. +The contract tests resolve every ref and decision id; they do not enforce the scientific prose or certify that a [LINT] is fixed. +Changing a scientific choice requires updating its ASTRA decision and committed universe, not just editing this file. + +Detection and photometry +------------------------ + +@sc [decision:detection.detection_threshold_policy,governs:default_tile.sex#DETECT_THRESH;default_tile.sex#ANALYSIS_THRESH;default_tile.sex#DETECT_MINAREA;default_tile.sex#THRESH_TYPE;default_tile.sex#FILTER;default_tile.sex#FILTER_NAME;config_tile_Sx.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE;gauss_3.0_7x7.conv;default_exp.sex#DETECT_THRESH;default_exp.sex#ANALYSIS_THRESH;default_exp.sex#DETECT_MINAREA;default_exp.sex#THRESH_TYPE;default_exp.sex#FILTER;default_exp.sex#FILTER_NAME;config_exp_psfex.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE;default.conv] tile-detection-matches-megapipe +Keep tile significance, minimum area and matched filter together as the MegaPipe detection prescription used by Gwyn's UNIONS tile catalogue, in both data and matching image simulations. +Exposure settings serve PSF-star detection and intentionally differ; copying them onto tiles changes the galaxy sample rather than standardising an implementation detail. +The runner's DOT_CONV_FILE overrides FILTER_NAME, so a kernel change must reach the effective command line, not just the .sex file. + +@sc [decision:detection.deblending_policy,governs:default_tile.sex#DEBLEND_MINCONT;default_exp.sex#DEBLEND_MINCONT;default_tile.sex#DEBLEND_NTHRESH;default_exp.sex#DEBLEND_NTHRESH] deblend-contrast-is-stage-specific +Preserve the deliberate tile/exposure difference in DEBLEND_MINCONT: tiles follow MegaPipe, while exposures select PSF-star candidates. +Do not harmonise the contrasts because DEBLEND_NTHRESH is shared; contrast changes object multiplicity, centroids and blend contamination. +A change must propagate to the matching simulations and the recorded detection prescription. + +@sc [decision:detection.background_model,governs:default_tile.sex#BACK_TYPE;default_tile.sex#BACK_SIZE;default_tile.sex#BACK_FILTERSIZE;default_tile.sex#BACKPHOTO_TYPE;default_tile.sex#BACKPHOTO_THICK;default_exp.sex#BACK_TYPE;default_exp.sex#BACK_SIZE;default_exp.sex#BACK_FILTERSIZE;default_exp.sex#BACKPHOTO_TYPE;config_tile_Sx.ini#SEXTRACTOR_RUNNER.BKG_FROM_HEADER;config_exp_psfex.ini#SEXTRACTOR_RUNNER.BKG_FROM_HEADER;config_exp_psfex.ini#SEXTRACTOR_RUNNER.CHECKIMAGE] background-model-follows-image-role +Keep the tile mesh, smoothing and local photometric annulus together as part of MegaPipe detection; the exposure mesh and global photometric background are a different prescription. +Exposure BACKGROUND and BACKGROUND_RMS check images also supply ngmix's epoch subtraction and noise weights, so changing their estimator changes shapes, not just detection photometry. +BKG_FROM_HEADER replaces the AUTO estimate with a manual background; do not enable that override while claiming the same background model. + +@sc [decision:detection.zero_weight_interpolation,governs:default_tile.sex#INTERP_TYPE;default_tile.sex#INTERP_MAXXLAG;default_tile.sex#INTERP_MAXYLAG;default_exp.sex#INTERP_TYPE;default_exp.sex#INTERP_MAXXLAG;default_exp.sex#INTERP_MAXYLAG] interpolated-detections-are-not-repaired-pixels +Treat INTERP_TYPE and both maximum lags as part of the detection and photometry prescription, including in matching simulations. +Flux invented across zero-weight pixels changes objects near defects; it does not make those pixels valid for ngmix or replace its separate defect-fill decision. +No scientific justification for the committed interpolation choice is recorded, so do not present it as a calibrated repair. + +@sc [decision:detection.spurious_detection_cleaning,governs:default_tile.sex#CLEAN;default_tile.sex#CLEAN_PARAM;default_exp.sex#CLEAN;default_exp.sex#CLEAN_PARAM] cleaning-changes-the-detection-sample +Keep CLEAN and CLEAN_PARAM together when reproducing the detection sample in data and simulations. +Cleaning removes detections after deblending, so switching it off or changing its strength changes catalogue membership even with identical detection thresholds and deblending. + +@sc [decision:detection.blend_photometry_mask_type,governs:default_tile.sex#MASK_TYPE;default_exp.sex#MASK_TYPE] blend-photometry-keeps-mirror-correction +The selected CORRECT treatment mirrors neighbour pixels across the target centre for SExtractor photometry; retain that convention when reproducing fluxes and windowed moments. +Replacing it with blanking changes photometry and centroids, and must not be confused with ngmix's separate neighbour-weighting decision. + +@sc [decision:detection.saturation_level,governs:default_tile.sex#SATUR_KEY;default_exp.sex#SATUR_KEY;star_selection.setools#MASK:star_selection.FLAGS] saturation-flags-follow-image-header +SATUR_KEY must refer to the delivered image's saturation card for the intended per-image saturation flags, which FLAGS-based PSF-star rejection consumes. +Do not interpret a missing card as evidence of no saturation: SExtractor then uses its built-in level because no SATUR_LEVEL is pinned here. +Header availability remains unverified; changing the key or adding a fixed level requires checking the bright-star selection, not just whether SExtractor runs. + +@sc [decision:detection.photometry_parameters,governs:default_tile.sex#PHOT_AUTOPARAMS;default_exp.sex#PHOT_AUTOPARAMS;default_tile.sex#PHOT_APERTURES;default_exp.sex#PHOT_APERTURES;default_tile.sex#PHOT_FLUXFRAC;default_exp.sex#PHOT_FLUXFRAC;default.psfex#PHOTFLUX_KEY;default.psfex#PHOTFLUXERR_KEY] kron-definition-links-selection-and-psf-normalisation +Keep the AUTO flux/error definition used for PSFEx normalisation consistent with the MAG_AUTO definition used for star selection and catalogue magnitudes. +A Kron-parameter change requires reconsidering the magnitude window and PSF normalisation together, not silently retaining their old interpretation. +Aperture diameter and flux fraction also define reported measurements; they must match the simulated catalogue rather than be treated as output formatting. + +@sc [decision:detection.detection_source_mode,governs:config_tile_Sx.ini#SEXTRACTOR_RUNNER.DETECTION_IMAGE;config_tile_Sx.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE;config_tile_Sx.ini#SEXTRACTOR_RUNNER.DOT_PARAM_FILE;default_tile.sex#PARAMETERS_NAME;default_noimaflags.param;final_cat.param#IMAFLAGS_ISO] tile-detection-has-no-instrument-flags +Tile detection uses the r-band tile itself, with no separate detection coadd or instrument flag image; keep the requested columns consistent with those inputs. +The runner's DOT_PARAM_FILE overrides PARAMETERS_NAME and deliberately selects the list without IMAFLAGS_ISO. +[LINT] final_cat.param requests IMAFLAGS_ISO although this tile chain never produces it; resolve the export/input mismatch rather than inventing a clean mask column or assuming FLAG_IMAGE alone supplies a mask. + +Shared pixel data and calibration +--------------------------------- + +@sc [decision:masking.pixel_mask_source,governs:config_exp_psfex.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE;config_exp_psfex.ini#SEXTRACTOR_RUNNER.FILE_PATTERN;config_exp_psfex.ini#SEXTRACTOR_RUNNER.INPUT_DIR;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_PATTERN;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_EXP_RUNNERS;star_selection.setools#MASK:star_selection.IMAFLAGS_ISO] instrument-flags-share-pixel-provenance +Exposure IMAFLAGS_ISO and the multi-epoch flag stamps must come from the same delivered instrument flag image split per CCD. +Keep the ME_IMAGE_PATTERN and ME_IMAGE_EXP_RUNNERS lists aligned so a flag stamp cannot silently become an image, weight or sky-mask product. +Sky-fixed healsparse masks remain object-level catalogue information; substituting or rasterising them into this pixel path changes the masking decision. + +@sc [decision:photometric_zeropoint,governs:default_tile.sex#MAG_ZEROPOINT;config_tile_Sx.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER;config_tile_Ng_template.ini#NGMIX_RUNNER.MAG_ZP;config_exp_psfex.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER;config_exp_psfex.ini#SEXTRACTOR_RUNNER.ZP_KEY] tile-and-epoch-photometry-share-flux-scale +Tile MAG_ZEROPOINT, ngmix MAG_ZP and the FSCALE-rescaled epochs must describe the same flux calibration. +Exposure photometry intentionally reads its own header zero-point; do not copy the fixed tile convention to exposures or change a header toggle independently of the calibration. +Equal config values do not verify that delivered MegaPipe stacks have the assumed calibration; that input assumption still needs checking. + +@sc [decision:postage_stamp_size,governs:default_noimaflags.param#VIGNET;default.param#VIGNET;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.STAMP_SIZE;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.STAMP_SIZE;default.psfex#PSF_SIZE] stamp-apertures-move-together +Keep SExtractor VIGNET for galaxies and training stars, both vignetmaker STAMP_SIZEs and the PSFEx PSF_SIZE coupled. +A resize must update the whole set and regenerate cutouts and models; changing only one desynchronises the pixel apertures used by the fit. +Larger stamps also change wing truncation and the measurable galaxy-size range, so this is a scientific change rather than only an allocation change. + +@sc [decision:preparation.object_position_columns,governs:config_tile_Sx.ini#SEXTRACTOR_RUNNER.WORLD_POSITION;config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.POSITION_PARAMS;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.COORD;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.POSITION_PARAMS;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.COORD;config_exp_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS;default.psfex#CENTER_KEYS;default.psfex#PSFVAR_KEYS;final_cat.param#XWIN_WORLD;final_cat.param#YWIN_WORLD] windowed-positions-share-coordinate-frame +Use the same windowed centroid through epoch membership, stamp placement, PSF interpolation and exported positions. +Pair IMAGE columns with pixel coordinates for tile stamps and exposure validation, and WORLD columns with sky coordinates for multi-epoch placement and interpolation. +Changing POSITION_PARAMS without its coordinate frame, or replacing windowed centroids in only one consumer, breaks the shared position underlying stamp offsets, centroid priors and position seeding. + +PSF modelling and diagnostics +---------------------------- + +@sc [decision:star_selection_psf.psf_model_complexity,governs:default.psfex#BASIS_TYPE;default.psfex#BASIS_NUMBER;default.psfex#PSF_SAMPLING;default.psfex#PSF_ACCURACY;default.psfex#PSFVAR_KEYS;default.psfex#PSFVAR_DEGREES] psf-flexibility-requires-star-support +Treat the pixel basis, spatial polynomial, sampling and assumed pixel accuracy as a joint model choice. +Greater flexibility must be supported by the surviving training stars on each CCD and checked on held-out residuals under the acceptance gate; successful optimisation alone does not establish a usable PSF. +PSF_ACCURACY changes bright-star leverage in the fit weights, not merely an optimisation stopping tolerance. + +@sc [decision:star_selection_psf.psfex_candidate_vetting,governs:default.psfex#SAMPLE_AUTOSELECT;default.psfex#BADPIXEL_FILTER;default.psfex#PSF_RECENTER;default.psfex] psfex-vetting-is-not-fully-disabled +Keep explicit stellar-locus selection under SETools control, but do not infer from SAMPLE_AUTOSELECT being off that every other PSFEx candidate cut is inactive. +The omitted SAMPLE_* settings inherit compiled defaults whose effect in this mode is unverified; a PSFEx version change requires inspecting those defaults and surviving candidates before claiming unchanged selection. +BADPIXEL_FILTER and PSF_RECENTER also change the candidate pixels or positions, so enabling them is a scientific change, not a harmless cleanup. + +@sc [decision:star_selection_psf.star_selection_box,governs:config_tile_Ng_template.ini#NGMIX_RUNNER.PIXEL_SCALE;default_exp.sex#PIXEL_SCALE;star_selection.setools#MASK:preselect.FWHM_IMAGE;star_selection.setools#MASK:star_selection.FWHM_IMAGE;star_selection.setools#PLOT:fwhm_field.SCATTER;star_selection.setools] exposure-size-conventions-agree +Conversions of the same exposure pixels must use a consistent exposure scale for stellar-size selection, diagnostics and ngmix's one-pixel centroid prior; tile resampling is not a reason to assign a tile scale to epoch stamps. +Diagnostic labels and statistics must describe the cut actually applied. +[LINT] ngmix and the FWHM map use 0.186 arcsec/px while preselection uses 0.187, and the statistics report a narrower FWHM window than selection applies; reconcile these with the exposure convention rather than propagating the mismatch. + +Catalogue export +---------------- + +@sc [decision:catalogue_assembly.tile_overlap_handling,governs:config_tile_Mc.ini#MAKE_CAT_RUNNER.NUMBER_LIST;final_cat.param#TILE_ID] tile-id-is-provenance-not-deduplication +Preserve TILE_ID when exporting the per-tile catalogues: overlapping tiles can contain separate measurements of the same sky object, and TILE_ID is the supplied provenance, not a uniqueness flag. +NUMBER_LIST selects a tile to assemble, not a disjoint sky region; duplicate removal remains downstream. +The documented TILE_LIST overlap flag is unimplemented, so adding that key cannot make this catalogue unique or flag its overlaps. From d60a5dfab36833cadeebc95e6f9bd0bd33695fa6 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 04:08:10 +0200 Subject: [PATCH 16/40] docs(astra): assert committed values on anchors Assert 130 values across 29 decisions without changing defaults or option ids. Cover coupled stamp sizes, detection, star cuts, PSF settings, and literal ngmix priors/metacal settings. Clarify that the CCD's 2048-index span is inclusive, whereas the committed cut excludes both endpoints. Co-Authored-By: GPT-6 Astra --- astra.yaml | 207 ++++++++++++++++++++++++++++++++++------------------- 1 file changed, 135 insertions(+), 72 deletions(-) diff --git a/astra.yaml b/astra.yaml index aee21da7a..1e32f23f7 100644 --- a/astra.yaml +++ b/astra.yaml @@ -128,10 +128,11 @@ decisions: PSFEx model stamp (PSF_SIZE 51,51). The stamp is the pixel data ngmix fits: it bounds the measurable galaxy size and truncates the wings of large galaxies. No rationale for 51 is recorded. - Anchor: workflow/config/cfis/default_noimaflags.param#VIGNET; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.STAMP_SIZE; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.STAMP_SIZE; - workflow/config/cfis/default.psfex#PSF_SIZE. + Anchor: workflow/config/cfis/default_noimaflags.param#VIGNET = 51; + workflow/config/cfis/default.param#VIGNET = 51; + workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.STAMP_SIZE = 51; + workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.STAMP_SIZE = 51; + workflow/config/cfis/default.psfex#PSF_SIZE = 51. default: px_51 options: px_51: @@ -151,10 +152,11 @@ decisions: calibrated to 30; nothing in the repo checks it. Magnitude cuts (the star-selection window, downstream galaxy cuts) inherit whichever convention their stage uses. - Anchor: workflow/config/cfis/default_tile.sex#MAG_ZEROPOINT; - workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER; - workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.MAG_ZP; - workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.ZP_KEY; + Anchor: workflow/config/cfis/default_tile.sex#MAG_ZEROPOINT = 30.0; + workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER = False; + workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.MAG_ZP = 30.0; + workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER = True; + workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.ZP_KEY = PHOTZP; src/shapepipe/modules/sextractor_package/sextractor_script.py::SExtractorCaller.get_zero_point. default: fixed_30_tiles_header_exposures options: @@ -238,8 +240,8 @@ analyses: (detection.detection_source_mode). Sky-fixed masks never touch pixels: an object inside a star halo is measured from the same pixels as one outside it. - Anchor: workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_PATTERN; + Anchor: workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE = True; + workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_PATTERN = flag, image, weight, background, background_rms; src/shapepipe/modules/ngmix_package/ngmix.py::prepare_ngmix_weights. default: instrument_flags_only options: @@ -272,7 +274,9 @@ analyses: docstring names the column FLAG_EXT and says setools cuts on it; the code writes MASK_EXT and nothing cuts on it. Anchor: workflow/config/cfis/config_exp_psfex.ini; - workflow/config/cfis/star_selection.setools#MASK:star_selection.IMAFLAGS_ISO; + workflow/config/cfis/star_selection.setools#MASK:star_selection.IMAFLAGS_ISO = "== 0"; + workflow/config/cfis/star_selection.setools#MASK:preselect.IMAFLAGS_ISO = "== 0"; + workflow/config/cfis/star_selection.setools#MASK:flag.IMAFLAGS_ISO = "== 0"; src/shapepipe/modules/mask_query_runner.py::mask_query_runner; src/shapepipe/utilities/mask_query.py::flag_positions. default: instrument_flags_only @@ -369,11 +373,20 @@ analyses: only feed star selection. Guinot+22 lists 1.5 sigma, minarea 10 and the default 3x3 kernel, so the tiles differ from the paper on all three. - Anchor: workflow/config/cfis/default_tile.sex#DETECT_THRESH; - workflow/config/cfis/default_tile.sex#DETECT_MINAREA; - workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE; + Anchor: workflow/config/cfis/default_tile.sex#DETECT_THRESH = 1.0; + workflow/config/cfis/default_tile.sex#ANALYSIS_THRESH = 1.0; + workflow/config/cfis/default_tile.sex#DETECT_MINAREA = 3; + workflow/config/cfis/default_tile.sex#FILTER = Y; + workflow/config/cfis/default_tile.sex#SEEING_FWHM = 0.6; + workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE = $SP_CONFIG/gauss_3.0_7x7.conv; workflow/config/cfis/gauss_3.0_7x7.conv; - workflow/config/cfis/default_exp.sex#DETECT_THRESH. + workflow/config/cfis/default_exp.sex#DETECT_THRESH = 1.5; + workflow/config/cfis/default_exp.sex#ANALYSIS_THRESH = 1.5; + workflow/config/cfis/default_exp.sex#DETECT_MINAREA = 5; + workflow/config/cfis/default_exp.sex#FILTER = Y; + workflow/config/cfis/default_exp.sex#SEEING_FWHM = 0.6; + workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE = $SP_CONFIG/default.conv; + workflow/config/cfis/default.conv. default: megapipe_tiles options: megapipe_tiles: @@ -394,8 +407,10 @@ analyses: on tiles (the MegaPipe value) and 0.001 on exposures. Contrast sets object count, centroids, and blend contamination in shapes. Guinot+22 lists 0.001. - Anchor: workflow/config/cfis/default_tile.sex#DEBLEND_MINCONT; - workflow/config/cfis/default_exp.sex#DEBLEND_MINCONT. + Anchor: workflow/config/cfis/default_tile.sex#DEBLEND_MINCONT = 0.002; + workflow/config/cfis/default_exp.sex#DEBLEND_MINCONT = 0.001; + workflow/config/cfis/default_tile.sex#DEBLEND_NTHRESH = 32; + workflow/config/cfis/default_exp.sex#DEBLEND_NTHRESH = 32. default: megapipe_tiles options: megapipe_tiles: @@ -419,10 +434,17 @@ analyses: (shape_measurement.galaxy_pixel_weights). The header background path is off (BKG_FROM_HEADER=False). Residual sky offsets propagate into thresholds, fluxes, completeness and shapes. - Anchor: workflow/config/cfis/default_tile.sex#BACK_SIZE; - workflow/config/cfis/default_tile.sex#BACKPHOTO_TYPE; - workflow/config/cfis/default_exp.sex#BACK_SIZE; - workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.BKG_FROM_HEADER; + Anchor: workflow/config/cfis/default_tile.sex#BACK_TYPE = AUTO; + workflow/config/cfis/default_tile.sex#BACK_SIZE = 512; + workflow/config/cfis/default_tile.sex#BACK_FILTERSIZE = 9; + workflow/config/cfis/default_tile.sex#BACKPHOTO_TYPE = LOCAL; + workflow/config/cfis/default_tile.sex#BACKPHOTO_THICK = 30; + workflow/config/cfis/default_exp.sex#BACK_TYPE = AUTO; + workflow/config/cfis/default_exp.sex#BACK_SIZE = 64; + workflow/config/cfis/default_exp.sex#BACK_FILTERSIZE = 3; + workflow/config/cfis/default_exp.sex#BACKPHOTO_TYPE = GLOBAL; + workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.BKG_FROM_HEADER = False; + workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.BKG_FROM_HEADER = False; src/shapepipe/modules/sextractor_package/sextractor_script.py::SExtractorCaller.get_background. default: auto_megapipe_tiles options: @@ -440,10 +462,14 @@ analyses: WEIGHT_TYPE MAP_WEIGHT on both passes (SExtractor default NONE): the per-pixel variance sets the effective SNR and so the detection set. RESCALE_WEIGHTS and WEIGHT_GAIN are - SExtractor defaults. Guinot+22 keeps every non-tabulated + Y, their SExtractor defaults. Guinot+22 keeps every non-tabulated parameter at its default, which would mean no weight map. - Anchor: workflow/config/cfis/default_tile.sex#WEIGHT_TYPE; - workflow/config/cfis/default_exp.sex#WEIGHT_TYPE; + Anchor: workflow/config/cfis/default_tile.sex#WEIGHT_TYPE = MAP_WEIGHT; + workflow/config/cfis/default_exp.sex#WEIGHT_TYPE = MAP_WEIGHT; + workflow/config/cfis/default_tile.sex#RESCALE_WEIGHTS = Y; + workflow/config/cfis/default_exp.sex#RESCALE_WEIGHTS = Y; + workflow/config/cfis/default_tile.sex#WEIGHT_GAIN = Y; + workflow/config/cfis/default_exp.sex#WEIGHT_GAIN = Y; src/shapepipe/modules/sextractor_package/sextractor_script.py::SExtractorCaller.set_input_files. default: map_weight options: @@ -459,8 +485,12 @@ analyses: INTERP_MAXXLAG/INTERP_MAXYLAG 16: SExtractor invents flux across zero-weight pixels, which changes detections and photometry near masked regions. No rationale is recorded. - Anchor: workflow/config/cfis/default_tile.sex#INTERP_TYPE; - workflow/config/cfis/default_exp.sex#INTERP_TYPE. + Anchor: workflow/config/cfis/default_tile.sex#INTERP_TYPE = ALL; + workflow/config/cfis/default_exp.sex#INTERP_TYPE = ALL; + workflow/config/cfis/default_tile.sex#INTERP_MAXXLAG = 16; + workflow/config/cfis/default_tile.sex#INTERP_MAXYLAG = 16; + workflow/config/cfis/default_exp.sex#INTERP_MAXXLAG = 16; + workflow/config/cfis/default_exp.sex#INTERP_MAXYLAG = 16. default: interp_all options: interp_all: @@ -473,8 +503,10 @@ analyses: CLEAN Y with CLEAN_PARAM 1.0 on both passes deletes detections consistent with being wings of a brighter neighbour, a post-deblend change to the object list. Stock value; no rationale recorded. - Anchor: workflow/config/cfis/default_tile.sex#CLEAN_PARAM; - workflow/config/cfis/default_exp.sex#CLEAN_PARAM. + Anchor: workflow/config/cfis/default_tile.sex#CLEAN_PARAM = 1.0; + workflow/config/cfis/default_exp.sex#CLEAN_PARAM = 1.0; + workflow/config/cfis/default_tile.sex#CLEAN = Y; + workflow/config/cfis/default_exp.sex#CLEAN = Y. default: clean_1 options: clean_1: @@ -486,8 +518,8 @@ analyses: neighbour by their mirror across the object centre during photometry, changing fluxes and windowed moments of blends. Stock value; no rationale recorded. - Anchor: workflow/config/cfis/default_tile.sex#MASK_TYPE; - workflow/config/cfis/default_exp.sex#MASK_TYPE. + Anchor: workflow/config/cfis/default_tile.sex#MASK_TYPE = CORRECT; + workflow/config/cfis/default_exp.sex#MASK_TYPE = CORRECT. default: correct options: correct: @@ -507,9 +539,11 @@ analyses: sets, so its built-in default applies (50000 ADU per the SExtractor documentation). Whether the delivered exposure CCDs and MegaPipe tiles carry SATURATE is unverified here. - Anchor: workflow/config/cfis/default_exp.sex#SATUR_KEY; - workflow/config/cfis/default_tile.sex#SATUR_KEY; - workflow/config/cfis/star_selection.setools#MASK:star_selection.FLAGS. + Anchor: workflow/config/cfis/default_exp.sex#SATUR_KEY = SATURATE; + workflow/config/cfis/default_tile.sex#SATUR_KEY = SATURATE; + workflow/config/cfis/star_selection.setools#MASK:star_selection.FLAGS = "== 0"; + workflow/config/cfis/star_selection.setools#MASK:preselect.FLAGS = "== 0"; + workflow/config/cfis/star_selection.setools#MASK:flag.FLAGS = "== 0". default: header_saturate options: header_saturate: @@ -525,9 +559,13 @@ analyses: FLUX_AUTO is PSFEx's photometric normalisation. A different Kron factor shifts magnitudes and so every magnitude-based cut. Stock values; no rationale recorded. - Anchor: workflow/config/cfis/default_tile.sex#PHOT_AUTOPARAMS; - workflow/config/cfis/default_exp.sex#PHOT_AUTOPARAMS; - workflow/config/cfis/default.psfex#PHOTFLUX_KEY. + Anchor: workflow/config/cfis/default_tile.sex#PHOT_AUTOPARAMS = 2.5,3.5; + workflow/config/cfis/default_exp.sex#PHOT_AUTOPARAMS = 2.5,3.5; + workflow/config/cfis/default_tile.sex#PHOT_APERTURES = 5; + workflow/config/cfis/default_exp.sex#PHOT_APERTURES = 5; + workflow/config/cfis/default_tile.sex#PHOT_FLUXFRAC = 0.5; + workflow/config/cfis/default_exp.sex#PHOT_FLUXFRAC = 0.5; + workflow/config/cfis/default.psfex#PHOTFLUX_KEY = FLUX_AUTO. default: kron_25_35 options: kron_25_35: @@ -541,8 +579,8 @@ analyses: pixel and the tile catalogue carries no IMAFLAGS_ISO. [LINT] final_cat.param, read by the post-processing merge, requests IMAFLAGS_ISO, which the tile chain never produces. - Anchor: workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.DETECTION_IMAGE; - workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE; + Anchor: workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.DETECTION_IMAGE = False; + workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE = False; workflow/config/cfis/default_noimaflags.param; workflow/config/cfis/final_cat.param#IMAFLAGS_ISO. default: sx_nomask_single_image @@ -564,11 +602,12 @@ analyses: CCD_SIZE = 33,2080,1,4612 with strict inequalities bounds each CCD's usable x range, and a WCS inversion failure skips the CCD, lowering N_EPOCH. This sets how many - exposures enter each galaxy's multi-epoch fit. The x range - (33, 2080) spans exactly 2048 px and matches MegaCam's raw - DATASEC, so the excluded strip is likely prescan; that is + exposures enter each galaxy's multi-epoch fit. The x endpoints + 33 and 2080 match MegaCam's raw DATASEC (2048 pixel indices + inclusively), but the strict cut excludes both endpoints. + The excluded strip is likely prescan; that interpretation is unverified here. - Anchor: workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.CCD_SIZE; + Anchor: workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.CCD_SIZE = 33,2080,1,4612; src/shapepipe/modules/sextractor_package/sextractor_script.py::make_post_process; src/shapepipe/modules/sextractor_package/sextractor_script.py::ccd_candidate_mask. default: trimmed_bounds_33_2080 @@ -668,7 +707,7 @@ analyses: N_HDU=40, and any other HDU count raises: every one of the 40 CCDs is a candidate epoch wherever the WCS lands it. Excluding a subset of CCDs is the alternative. - Anchor: workflow/config/cfis/config_exp_Sp.ini#SPLIT_EXP_RUNNER.N_HDU; + Anchor: workflow/config/cfis/config_exp_Sp.ini#SPLIT_EXP_RUNNER.N_HDU = 40; src/shapepipe/modules/split_exp_package/split_exp.py::SplitExposures.create_hdus. default: all_40_hdus options: @@ -690,8 +729,8 @@ analyses: exposures are fit. [LINT] EXP_PREFIX = p is passed to removeprefix, which does nothing to names where p is a suffix, so the key has no effect. - Anchor: workflow/config/cfis/config_tile_Fe.ini#FIND_EXPOSURES_RUNNER.COLNUM; - workflow/config/cfis/config_tile_Fe.ini#FIND_EXPOSURES_RUNNER.EXP_PREFIX; + Anchor: workflow/config/cfis/config_tile_Fe.ini#FIND_EXPOSURES_RUNNER.COLNUM = 3; + workflow/config/cfis/config_tile_Fe.ini#FIND_EXPOSURES_RUNNER.EXP_PREFIX = p; src/shapepipe/modules/find_exposures_package/find_exposures.py::FindExposures.get_exposure_list. default: history_parse options: @@ -710,10 +749,12 @@ analyses: centroids differ systematically for blends and asymmetric galaxies, and the centroid feeds the position seed and the centroid prior. - Anchor: workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.POSITION_PARAMS; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.POSITION_PARAMS; - workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS. + Anchor: workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.POSITION_PARAMS = XWIN_IMAGE,YWIN_IMAGE; + workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.COORD = PIX; + workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS = XWIN_WORLD,YWIN_WORLD; + workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.POSITION_PARAMS = XWIN_WORLD,YWIN_WORLD; + workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.COORD = SPHE; + workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS = XWIN_IMAGE,YWIN_IMAGE. default: xwin_windowed options: xwin_windowed: @@ -800,9 +841,16 @@ analyses: 0.1 px and its plot uses 0.186 arcsec/px, while the applied cut is +- 0.2 px at 0.187; the _mode docstring puts the median fallback at 10 objects, the code at 20. - Anchor: workflow/config/cfis/star_selection.setools#MASK:star_selection.FLAGS; - workflow/config/cfis/star_selection.setools#MASK:preselect.FLAGS; - workflow/config/cfis/default.psfex#SAMPLE_AUTOSELECT; + Anchor: workflow/config/cfis/star_selection.setools#MASK:star_selection.FLAGS = "== 0"; + workflow/config/cfis/star_selection.setools#MASK:preselect.FLAGS = "== 0"; + workflow/config/cfis/star_selection.setools#MASK:star_selection.IMAFLAGS_ISO = "== 0"; + workflow/config/cfis/star_selection.setools#MASK:preselect.IMAFLAGS_ISO = "== 0"; + workflow/config/cfis/star_selection.setools#MASK:star_selection.MAG_AUTO = ["> 18.", "< 22."]; + workflow/config/cfis/star_selection.setools#MASK:star_selection.FWHM_IMAGE = ["<= mode(FWHM_IMAGE{preselect}) + 0.2", ">= mode(FWHM_IMAGE{preselect}) - 0.2"]; + workflow/config/cfis/star_selection.setools#MASK:preselect.MAG_AUTO = ["> 0", "< 21"]; + workflow/config/cfis/star_selection.setools#MASK:preselect.FWHM_IMAGE = ["> 0.3 / 0.187", "< 1.5 / 0.187"]; + workflow/config/cfis/star_selection.setools#PLOT:fwhm_field.SCATTER = "FWHM_IMAGE{star_selection}*0.186"; + workflow/config/cfis/default.psfex#SAMPLE_AUTOSELECT = N; src/shapepipe/pipeline/str_handler.py::StrInterpreter._mode. default: mode_centred_box options: @@ -831,10 +879,11 @@ analyses: It is deterministic: a permutation seeded from the digits of the unit's file number, so a given CCD gets the same split on every run. - Anchor: workflow/config/cfis/star_selection.setools#RAND_SPLIT:star_split.RATIO; + Anchor: workflow/config/cfis/star_selection.setools#RAND_SPLIT:star_split.RATIO = 20; src/shapepipe/modules/setools_package/setools.py::SETools._make_rand_split; - workflow/config/cfis/config_exp_psfex.ini#PSFEX_RUNNER.FILE_PATTERN; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.ME_DOT_PSF_PATTERN. + workflow/config/cfis/config_exp_psfex.ini#PSFEX_RUNNER.FILE_PATTERN = star_split_ratio_80; + workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.FILE_PATTERN = star_split_ratio_80,star_split_ratio_20,psfex_cat; + workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.ME_DOT_PSF_PATTERN = star_split_ratio_80. default: split_80_20_seeded options: split_80_20_seeded: @@ -861,9 +910,9 @@ analyses: unverified. BADPIXEL_FILTER N and PSF_RECENTER N accept flagged star vignets unfiltered and do not recentre candidates. - Anchor: workflow/config/cfis/default.psfex#SAMPLE_AUTOSELECT; - workflow/config/cfis/default.psfex#BADPIXEL_FILTER; - workflow/config/cfis/default.psfex#PSF_RECENTER. + Anchor: workflow/config/cfis/default.psfex#SAMPLE_AUTOSELECT = N; + workflow/config/cfis/default.psfex#BADPIXEL_FILTER = N; + workflow/config/cfis/default.psfex#PSF_RECENTER = N. default: builtin_defaults options: builtin_defaults: @@ -884,10 +933,13 @@ analyses: images from mask_runner, which no longer exists, so the MCCD exposure chain cannot run as committed. Anchor: workflow/config.yaml; - workflow/config/cfis/config_MCCD.ini#INSTANCE.N_COMP_LOC; - workflow/config/cfis/config_MCCD.ini#INSTANCE.FP_GEOMETRY; - workflow/config/cfis/config_MCCD.ini#INPUTS.MIN_N_STARS; - workflow/config/cfis/config_exp_mccd.ini#SEXTRACTOR_RUNNER.INPUT_MODULE; + workflow/config/cfis/config_MCCD.ini#INSTANCE.N_COMP_LOC = 8; + workflow/config/cfis/config_MCCD.ini#INSTANCE.D_COMP_GLOB = 8; + workflow/config/cfis/config_MCCD.ini#INSTANCE.FP_GEOMETRY = CFIS; + workflow/config/cfis/config_MCCD.ini#INSTANCE.RMSE_THRESH = 1.25; + workflow/config/cfis/config_MCCD.ini#INPUTS.MIN_N_STARS = 20; + workflow/config/cfis/config_exp_mccd.ini#SEXTRACTOR_RUNNER.INPUT_MODULE = split_exp_runner,mask_runner; + workflow/config/cfis/config_exp_mccd.ini#SEXTRACTOR_RUNNER.FILE_PATTERN = image,weight,pipeline_flag; src/shapepipe/modules/mccd_package. default: psfex options: @@ -907,10 +959,13 @@ analyses: stars. Model flexibility sets the balance between PSF leakage and overfitting, the dominant additive systematic in cosmic shear. Stock values; no rationale recorded. - Anchor: workflow/config/cfis/default.psfex#BASIS_TYPE; - workflow/config/cfis/default.psfex#PSF_ACCURACY; - workflow/config/cfis/default.psfex#BASIS_NUMBER; - workflow/config/cfis/default.psfex#PSFVAR_DEGREES. + Anchor: workflow/config/cfis/default.psfex#BASIS_TYPE = PIXEL; + workflow/config/cfis/default.psfex#PSF_ACCURACY = 0.01; + workflow/config/cfis/default.psfex#BASIS_NUMBER = 20; + workflow/config/cfis/default.psfex#PSFVAR_DEGREES = 2; + workflow/config/cfis/default.psfex#PSF_SAMPLING = 1; + workflow/config/cfis/default.psfex#PSFVAR_KEYS = XWIN_IMAGE,YWIN_IMAGE; + workflow/config/cfis/default.psfex#MEF_TYPE = INDEPENDENT. default: pixel_basis_deg2_per_ccd options: pixel_basis_deg2_per_ccd: @@ -933,9 +988,10 @@ analyses: Guinot+22 describes discarding the CCD from PSF estimation rather than gating at interpolation. Anchor: src/shapepipe/modules/psfex_interp_package/psfex_interp.py::PSFExInterpolator.interpsfex; - workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.CHI2_THRESH. + workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH = 22; + workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.CHI2_THRESH = 2; + workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH = 22; + workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.CHI2_THRESH = 2. default: stars22_chi2_2 options: stars22_chi2_2: @@ -1146,8 +1202,10 @@ analyses: ngmix_runner comment says pixel scale also sets a noise window, but get_noise is never called. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::get_prior; + src/shapepipe/modules/ngmix_package/ngmix.py::get_prior.T_range = -1,1e3; + src/shapepipe/modules/ngmix_package/ngmix.py::get_prior.F_range = -100,1e9; src/shapepipe/modules/ngmix_package/ngmix.py::make_runners; - workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.PIXEL_SCALE. + workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.PIXEL_SCALE = 0.186. default: gpriorba04_flat options: gpriorba04_flat: @@ -1169,6 +1227,11 @@ analyses: directly. No sheared-PSF types run, so the catalogue has no PSF response term. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal; + src/shapepipe/modules/ngmix_package/ngmix.py::METACAL_TYPES = noshear,1p,1m,2p,2m; + src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal.metacal_pars[types] = noshear,1p,1m,2p,2m; + src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal.metacal_pars[step] = 0.01; + src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal.metacal_pars[fixnoise] = True; + src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal.metacal_pars[use_noise_image] = True; src/shapepipe/modules/ngmix_runner.py::ngmix_runner. default: five_types_step001_fitgauss options: @@ -1237,7 +1300,7 @@ analyses: False). Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_ngmix_weights; src/shapepipe/modules/ngmix_package/ngmix.py::background_subtract; - workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.BKG_RMS_VIGNET_PATH. + workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.BKG_RMS_VIGNET_PATH = $NGMIX_VIGNET_DIR/vignetmaker_runner_run_2/output/background_rms_vignet{file_number_string}.sqlite. default: rms_vignet_weights options: rms_vignet_weights: @@ -1917,7 +1980,7 @@ analyses: sigma_s > 0.0003 together with s > 0 and 20 < MAG_AUTO < 26; the dormant code implements only the spread-model test. - Anchor: workflow/config/cfis/config_tile_Mc.ini#MAKE_CAT_RUNNER.SM_DO_CLASSIFICATION; + Anchor: workflow/config/cfis/config_tile_Mc.ini#MAKE_CAT_RUNNER.SM_DO_CLASSIFICATION = False; src/shapepipe/modules/make_cat_runner.py::make_cat_runner; src/shapepipe/modules/make_cat_package/make_cat.py::save_sm_data; workflow/config/cfis/final_cat.param#SPREAD_CLASS. From 586a86986b5e8e5a69324bed8c2bfe6b0bd9bbab Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 10:50:48 +0200 Subject: [PATCH 17/40] docs(astra): defect fill, central veto and masked-fraction cut follow the measured design - defect_fill: noise on the unsymmetrized defect set stays default; interpolate (feat/defect-interpolation) describes the bounded-run fill with quarter-turn weight orbit; four-fold symmetrization is excluded on its measured m and c1; the model option is dropped - central_defect_veto: fixed radii (10 px noise, 7 px interpolated) on the defect mask only; size-scaled radius excluded; calibration and known limits stated; default stays disabled - epoch_masked_fraction_cut: the branch counts the raw defect set, with EPOCH_MASKED_FRACTION_CUT configurable; default stays 1/3 Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014bvNTrAmZxcfb1ee83ApPK --- astra.yaml | 244 ++++++++++++++++++++++++++--------------------------- 1 file changed, 118 insertions(+), 126 deletions(-) diff --git a/astra.yaml b/astra.yaml index 1e32f23f7..e6a9d98e6 100644 --- a/astra.yaml +++ b/astra.yaml @@ -1336,110 +1336,88 @@ analyses: megapipe_flip: label: Flip CCDs < 18 and 36/37 (MegaPipe orientation) defect_fill: - label: Image content of flagged / zero-weight pixels before metacal + label: Image content of defect pixels before metacal rationale: >- - prepare_ngmix_weights gives weight 0 to every pixel with a nonzero - instrument flag, zero exposure weight or invalid background RMS. What - the IMAGE holds in those pixels still matters, because metacal never - looks at weights: ngmix builds a galsim InterpolatedImage from the - whole observation image, deconvolves, shears and reconvolves it, and - copies the weight map through unchanged. Whatever sits in a - zero-weight pixel is therefore spread into the weighted pixels within - about a PSF width, with ringing at sharp features. DES's own - corrector (ngmixer) says why it fills: "it may be important for codes - that take moments or use FFTs". The fill is coupled to the neighbour - treatment through the single BLEND_HANDLING key, which the committed - config leaves at its default. Under noisefill, masked pixels get an - independent noise realisation at the per-pixel RMS. Under uberseg, - the fill is skipped, so raw bad columns, bleeds, cosmic rays and - bright-star light enter metacal. [LINT] the prepare_ngmix_weights - docstring says noisefill keeps the weight of filled pixels (the code - zeroes it), and the ngmix_runner comment says noisefill fills - neighbour pixels (it fills flagged pixels and leaves neighbours - untouched). No DES metacal pipeline passed raw defects through - metacal: Y1 dropped every epoch with a masked pixel - (max_zero_weight_frac 0.0), Y3 filled symmetrized defects with the - best-fit central model (both candidate Y3 configs do this; which one - was production is not recorded), and Y6 interpolated symmetrized - defects in the image and in every noise image. The residual cost of - any fill is anisotropy. A filled bad column that crosses the galaxy - removes or misplaces light along one detector axis. The - reconvolution spreads that into an additive e1-type term, coherent on - the sky because CFHT/MegaCam, like DECam, has a fixed sky orientation - (inferred from CFHT's equatorial mount, no derotator). Sheldon & Huff - 2017 saw a large additive e1 even with model fill, removed by a - 90-degree compensating mask. Every DES pipeline symmetrized the - defect mask, and ShapePipe does not. The fill should be set - independently of blend_handling; the recommended option is - symmetrized_4fold_noise. A single 90-degree rotation is not enough: - it leaves a coherent c2 of about -0.006 to -0.012 for columns 2-3 px - off-centre, while the 4-fold OR gives |c| < 2e-4 (measured on - feat/symmetrized-defect-fill). + A defect pixel is one with a nonzero instrument flag, zero exposure + weight or invalid background RMS; prepare_ngmix_weights gives it + weight 0. What the IMAGE holds there still matters, because metacal + never looks at weights: ngmix builds a galsim InterpolatedImage from + the whole observation image, deconvolves, shears and reconvolves it, + and copies the weight map through unchanged, so whatever sits in a + zero-weight pixel spreads into the weighted pixels within about a PSF + width. DES's own corrector (ngmixer) fills for that reason: "it may + be important for codes that take moments or use FFTs". On develop + the fill rides on BLEND_HANDLING, which the committed config leaves + at noisefill: defect pixels get independent noise at the per-pixel + RMS; under uberseg the fill is skipped and raw defects enter + metacal. [LINT] the prepare_ngmix_weights docstring says noisefill + keeps the weight of filled pixels (the code zeroes it), and the + ngmix_runner comment says noisefill fills neighbour pixels (it fills + flagged pixels and leaves neighbours untouched). On + feat/defect-fill-veto every defect pixel is zero-weighted and filled + before metacal under every BLEND_HANDLING; under uberseg, + neighbour-side pixels only lose weight. The fill uses the + unsymmetrized defect set. DES symmetrized its defect masks, but + measured on feat/defect-fill-veto, four-fold symmetrization + quadruples the multiplicative bias (m = -2.7% against -0.64% for a + 3-px bleed 10 px from a galaxy of half-light radius 0.5 arcsec + through a 0.7 arcsec PSF) and still leaves c1 = 7.3e-4 through an + elliptical PSF. The residual cost of an unsymmetrized fill is a hole + in the galaxy light; the central veto bounds it + (central_defect_veto). Noise stays the default until a survey A/B + against interpolation. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_ngmix_weights; src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal; src/shapepipe/modules/ngmix_runner.py::ngmix_runner. default: noise options: noise: - label: Noise fill (BLEND_HANDLING = noisefill) + label: Independent noise on the unsymmetrized defect set description: >- - Masked pixels are replaced by an independent noise realisation at - the per-pixel background RMS and keep weight 0. Consistent with - metacal's fixnoise noise image, which covers every pixel. Removes - defects, but leaves an unsymmetrized hole in the galaxy light - wherever a defect crosses the object, which can give an e1-type - additive term. + Defect pixels are replaced by an independent noise realisation at + the per-pixel background RMS and keep weight 0, consistent with + metacal's fixnoise noise image, which covers every pixel. On + develop this happens only under BLEND_HANDLING = noisefill; on + the fill branches it is DEFECT_FILL = noise under every + BLEND_HANDLING. insights: [mask_metacal_acts_on_whole_stamp, mask_bad_column_symmetrize] + interpolate: + label: Interpolate short bounded runs; noise-fill the rest + description: >- + Implemented on feat/defect-interpolation (a stacked draft), not on + develop; DEFECT_FILL = interpolate. Only short bounded runs (at + most 3 px across, clean on both sides) are interpolated; + everything else is noise-filled. The interpolation is + Clough-Tocher from clean pixels within 4 px, averaged over the + four quarter turns, and shared with the fixnoise image. Weights + are also zeroed on the quarter-turn orbit of each interpolated + pixel while those pixels keep their light, which cancels the + weight term: c1 goes from -1.3e-3 to 2e-6 for a column 8 px from + a 0.5 arcsec galaxy. The fill itself is never symmetrized: + symmetrizing it gives m = +0.89% against +0.19% for a 3-px bleed + at 6 px. Measured on feat/defect-interpolation. + insights: [mask_interpolate_with_noise, mask_sharp_edges_ring] + symmetrized_4fold_noise: + label: Four-fold-symmetrized defect set (M | rot90 | rot180 | rot270), then noise fill + excluded: true + excluded_reason: >- + Measured on feat/defect-fill-veto, it quadruples m (-2.7% against + -0.64% for a 3-px bleed at 10 px, half-light radius 0.5 arcsec, + PSF 0.7 arcsec) and still leaves c1 = 7.3e-4 through an + elliptical PSF. + insights: [mask_bad_column_symmetrize, mask_des_defect_practice, mask_fixed_orientation] raw: - label: "No fill: raw defect values (BLEND_HANDLING = uberseg)" + label: "No fill: raw defect values (develop under BLEND_HANDLING = uberseg)" description: >- - Masked pixels keep weight 0, but their raw values (bad columns, + Defect pixels keep weight 0, but their raw values (bad columns, saturation, bleeds, cosmic rays, bright-star light) stay in the image that metacal deconvolves, shears and reconvolves. excluded: true excluded_reason: >- Metacal acts on every pixel regardless of weight, so raw defects leak into the weighted pixels. No published metacal pipeline does - this: DES dropped, model-filled or interpolated defects. The code - reaches this option only as a side effect of BLEND_HANDLING = - uberseg. + this: DES dropped, model-filled or interpolated defects. insights: [mask_metacal_acts_on_whole_stamp, mask_des_defect_practice] - symmetrized_4fold_noise: - label: 4-fold-symmetrized mask (M | rot90 | rot180 | rot270), then noise fill - description: >- - Not implemented on develop; implemented on - feat/symmetrized-defect-fill. OR the defect mask with its 90, - 180 and 270-degree rotations about the stamp centre, zero the - weight on the union, and noise-fill the union as noisefill does. - This extends the DES mask symmetrization (Y1/Y3 ngmixer - symmetrize_weight; Y6 symmetrize_masking, a single rotation) and - cancels the column-aligned additive term that one rotation leaves - for off-centre columns. It can quadruple the masked area, so the - epoch cut must be applied after symmetrizing. ShapePipe stamps - are square, so the rotations are well defined. The noise image - needs no change. - insights: [mask_bad_column_symmetrize, mask_des_defect_practice, mask_fixed_orientation] - interpolate: - label: Symmetrize, then interpolate image and noise image (DES Y6) - description: >- - Not implemented. Symmetrize as above, then fill the union by 2D - Clough-Tocher interpolation (scipy) of the image and, identically, - of the fixnoise noise image. This restores galaxy light across - narrow defects instead of leaving a hole. It is the DES Y6 and - Rubin metadetect practice. Poor for large holes (star masks), - which Y6 zeroes with apodized edges. - insights: [mask_interpolate_with_noise, mask_bad_column_symmetrize, mask_sharp_edges_ring] - model: - label: Symmetrize, then fill with the best-fit central model (DES Y3) - description: >- - Not implemented. Fill symmetrized defects with the PSF-convolved - best-fit model of the central object from a pre-metacal fit - (ngmix v1.3.9 replace_masked_pixels). The uberseg-only Y3 config - adds no noise (add_noise=False); the MOF-corrector config adds - it. This restores galaxy light; without noise it leaves - noise-free patches that the full-stamp fixnoise noise image does - not mirror. - insights: [mask_bad_column_symmetrize, mask_des_defect_practice] blend_handling: label: Neighbour treatment before metacal rationale: >- @@ -1519,63 +1497,77 @@ analyses: central_defect_veto: label: Per-epoch veto on a defect near the stamp centre rationale: >- - The committed code has no veto; the option below is implemented on - feat/symmetrized-defect-fill, not on develop. There it drops an epoch - when any defect pixel lies within EPOCH_CENTRAL_DEFECT_RADIUS px of - the stamp centre (0 disables), beside the masked-fraction cut in the - epoch loop. A noise-filled hole near the centre removes the galaxy's - core light, which no fill or symmetrization restores: on a round - galaxy with a 0.7 arcsec PSF, single epoch, it biases m by -6.4% for - a column 8 px from centre and -0.17% at 10 px. + The committed code has no veto; the veto is implemented on + feat/defect-fill-veto (7777181b), not on develop. There an epoch is + dropped when a defect pixel lies strictly closer to the stamp centre + than its fill's radius, beside the masked-fraction cut in the epoch + loop. It reads only the defect mask, so it selects on nothing + shear-responsive. The radii are fixed, not scaled by galaxy size, + because a size-scaled veto would select on a shear-responsive + quantity: 10 px for noise-filled pixels + (EPOCH_CENTRAL_DEFECT_RADIUS) and 7 px for interpolated pixels + (EPOCH_INTERPOLATED_DEFECT_RADIUS, feat/defect-interpolation). They + are calibrated on 51-px stamps at known RMS and high S/N, through a + round PSF and a (0.05, 0.02) elliptical PSF: for noise fill on + galaxies of half-light radius 0.3 and 0.5 arcsec through a 0.7 arcsec + PSF, and for interpolation also on 0.7 and 0.9 arcsec galaxies. + Known limits: at 10 px, wide defects through the elliptical PSF sit + at the 1% bound (m11 = -0.98%, and -0.24% at 11 px), and noise fill + needs 14 px for 0.7 and 0.9 arcsec galaxies. Measured on + feat/defect-fill-veto and feat/defect-interpolation. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_postage_stamps. default: disabled options: disabled: label: No central veto (committed code) - radius_10px: - label: Drop the epoch if a defect lies within 10 px of the centre + fixed_radii: + label: Fixed radii, 10 px for noise-filled and 7 px for interpolated defects description: >- - Implemented on feat/symmetrized-defect-fill, not on develop, and - recommended there: 10 px is the smallest radius with |m| < 1% for - both columns and single pixels. + Implemented on feat/defect-fill-veto and feat/defect-interpolation, + not on develop. + size_scaled_radius: + label: Veto radius scaled by galaxy size + excluded: true + excluded_reason: >- + Galaxy size responds to shear, so a size-scaled veto selects on + a shear-responsive quantity. epoch_masked_fraction_cut: label: Per-epoch masked-fraction cut rationale: >- - [HARDCODED] an epoch whose stamp has more than 1/3 of its - pixels flagged (any nonzero flag bit, including the - tile-coverage bit 2**10 set where the tile vignet is - off-image) is dropped from the multi-epoch fit; an object - with no surviving epoch has no shape. The cut counts flag - pixels only, not zero-weight or invalid-RMS pixels. Before - it, an epoch is dropped silently if its galaxy stamp is - all zeros or its background-subtracted noise estimate - (sigma_mad) is not positive. DES was stricter. Y1 rejected - any epoch with a masked or zero-weight pixel, and any - whose central 4-pixel region was masked. Y3 cut at 10% of - raw zero-weight pixels. Y6 dropped images more than 10% - missing and cut objects at mfrac < 0.1, which its - simulations show avoids calibration bias. Sheldon & Huff - 2017 recommend dropping problematic epochs when many are - available. The cut interacts with defect_fill: - symmetrizing multiplies the masked fraction (up to fourfold), so the - cut should be applied after symmetrizing. UNIONS has fewer - epochs than DES, so the cost in effective number density - has to be measured, not assumed. A defect near the centre - is handled separately (central_defect_veto). + [HARDCODED] on develop, an epoch whose stamp has more than 1/3 of its + pixels flagged (any nonzero flag bit, including the tile-coverage bit + 2**10 set where the tile vignet is off-image) is dropped from the + multi-epoch fit; an object with no surviving epoch has no shape. The + develop cut counts flag pixels only, not zero-weight or invalid-RMS + pixels. On feat/defect-fill-veto it counts the raw defect set + (flagged, zero-weight and invalid-RMS pixels, not symmetrized), with + the threshold configurable as EPOCH_MASKED_FRACTION_CUT, default + 1/3. Before the cut, an epoch is dropped silently if its galaxy stamp + is all zeros or its background-subtracted noise estimate (sigma_mad) + is not positive. DES was stricter. Y1 rejected any epoch with a + masked or zero-weight pixel, and any whose central 4-pixel region was + masked. Y3 cut at 10% of raw zero-weight pixels. Y6 dropped images + more than 10% missing and cut objects at mfrac < 0.1, which its + simulations show avoids calibration bias. Sheldon & Huff 2017 + recommend dropping problematic epochs when many are available. + UNIONS has fewer epochs than DES, so the cost in effective number + density has to be measured, not assumed. A defect near the centre is + handled separately (central_defect_veto). Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_postage_stamps. default: one_third options: one_third: - label: 1/3 of the stamp flagged + label: 1/3 of the stamp in the defect set description: >- - Drop an epoch only if more than 1/3 of the stamp pixels carry a - nonzero flag. + Drop an epoch only if more than 1/3 of the stamp pixels are + defects (on develop, flagged pixels). ten_percent: label: 10% (DES Y3 / Y6) description: >- - Not implemented. Drop an epoch if more than 10% of the - (symmetrized) stamp is masked, matching DES Y3 - max_zero_weight_frac and Y6 max_masked_fraction. + Not implemented as a default; EPOCH_MASKED_FRACTION_CUT = 0.1 on + feat/defect-fill-veto. Drop an epoch if more than 10% of the stamp + is in the defect set, matching DES Y3 max_zero_weight_frac and Y6 + max_masked_fraction. insights: [mask_multi_epoch_drop] any_masked: label: Any masked pixel (DES Y1) From 7ada933fc59fa0823309c49896409a93c4a8182d Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 10:59:33 +0200 Subject: [PATCH 18/40] test(astra): assert gate keys and absent keys; ambiguous subscript bindings; INI booleans per getboolean - Assert the booleans that make asserted values live: WEIGHT_IMAGE (weight map), MAKE_POST_PROCESS (CCD_SIZE), vignetmaker MASKING (STAMP_SIZE). - `= absent` asserts a config key has no active line; the record uses it for MASK_EXT in every star-selection mask block and SATUR_LEVEL in both .sex files. New contract psf-stars-vetoed-on-instrument-flags-only. - NAME[...] = / NAME.attr = in the binding's scope makes a value read of NAME ambiguous. - INI booleans follow ConfigParser.getboolean; Y/N only for .sex/.psfex. - A real-record mutation test covers each drift that previously passed. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014bvNTrAmZxcfb1ee83ApPK --- astra.yaml | 28 ++++-- tests/helpers/astra_record.py | 172 +++++++++++++++++++++++++++++--- tests/unit/test_astra_values.py | 140 +++++++++++++++++++++++++- workflow/config/cfis/CONTRACTS | 6 +- 4 files changed, 322 insertions(+), 24 deletions(-) diff --git a/astra.yaml b/astra.yaml index e6a9d98e6..f43222744 100644 --- a/astra.yaml +++ b/astra.yaml @@ -10,6 +10,8 @@ # (`path#KEY` for sectionless .sex/.psfex/.param files; .setools uses # SECTION.KEY), or FILE `path`. No line numbers. Commented-out keys may # resolve as locations, but only active settings can assert values. +# `= absent` asserts a config key has no active line (its file and any +# section must exist); quote it ("absent") to mean the text. # * Value assertions: `Anchor: path#SECTION.KEY = 1.5; path::NAME = 51.` # A ref without ` = value` is location-only. Attaching the expectation to # its locator avoids guessing which file a prose number describes, while @@ -19,9 +21,11 @@ # references, never parsed for values. Code/config remains execution truth. # * Values are numbers, boolean words, strings (quote expressions), or flat # comma lists, optionally bracketed. Semicolons are reserved for refs. -# Decimal equality is exact (1 = 1.0, 5e-4 = 0.0005), without rounding; -# Y/yes/true/on and N/no/false/off are case-insensitive boolean aliases, -# distinct from 1/0. Trim outer whitespace; other strings are case-sensitive. +# Decimal equality is exact (1 = 1.0, 5e-4 = 0.0005), without rounding. +# Boolean words follow the file's reader, case-insensitively: INI as +# ConfigParser.getboolean (yes/true/on/1, no/false/off/0; Y/N are text), +# .sex/.psfex also Y/N, Python True/False; elsewhere words are text and +# 1/0 are numbers. Trim outer whitespace; other strings are case-sensitive. # Lists preserve order and length. Only .param VIGNET and .psfex PSF_SIZE # accept square-size shorthand: 51 = 51,51 (never 51,53). # * SETools predicates retain their operators as quoted text; repeated cuts @@ -29,8 +33,10 @@ # Expressions compare as text, not algebra. Python selectors may append # `[key.subkey]` to a named assignment (identifier-like string dict keys); # only the selected literal is read, including dict(key=value) syntax. -# Ambiguous bindings/settings fail; no imports, calls, arithmetic, argument -# defaults, environment expansion or implicit tool defaults are evaluated. +# Ambiguous bindings/settings fail, including NAME[...] = or NAME.attr = +# in the same scope; .update()-style calls and mutation from other scopes +# are not seen. No imports, calls, arithmetic, argument defaults, +# environment expansion or implicit tool defaults are evaluated. # These limits keep the check static rather than a second pipeline runtime. # * A decision's `default` is the option the committed code and configs # select; universes/committed.yaml pins it. @@ -132,6 +138,8 @@ decisions: workflow/config/cfis/default.param#VIGNET = 51; workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.STAMP_SIZE = 51; workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.STAMP_SIZE = 51; + workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.MASKING = False; + workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.MASKING = False; workflow/config/cfis/default.psfex#PSF_SIZE = 51. default: px_51 options: @@ -277,6 +285,9 @@ analyses: workflow/config/cfis/star_selection.setools#MASK:star_selection.IMAFLAGS_ISO = "== 0"; workflow/config/cfis/star_selection.setools#MASK:preselect.IMAFLAGS_ISO = "== 0"; workflow/config/cfis/star_selection.setools#MASK:flag.IMAFLAGS_ISO = "== 0"; + workflow/config/cfis/star_selection.setools#MASK:star_selection.MASK_EXT = absent; + workflow/config/cfis/star_selection.setools#MASK:preselect.MASK_EXT = absent; + workflow/config/cfis/star_selection.setools#MASK:flag.MASK_EXT = absent; src/shapepipe/modules/mask_query_runner.py::mask_query_runner; src/shapepipe/utilities/mask_query.py::flag_positions. default: instrument_flags_only @@ -470,6 +481,8 @@ analyses: workflow/config/cfis/default_exp.sex#RESCALE_WEIGHTS = Y; workflow/config/cfis/default_tile.sex#WEIGHT_GAIN = Y; workflow/config/cfis/default_exp.sex#WEIGHT_GAIN = Y; + workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.WEIGHT_IMAGE = True; + workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.WEIGHT_IMAGE = True; src/shapepipe/modules/sextractor_package/sextractor_script.py::SExtractorCaller.set_input_files. default: map_weight options: @@ -543,7 +556,9 @@ analyses: workflow/config/cfis/default_tile.sex#SATUR_KEY = SATURATE; workflow/config/cfis/star_selection.setools#MASK:star_selection.FLAGS = "== 0"; workflow/config/cfis/star_selection.setools#MASK:preselect.FLAGS = "== 0"; - workflow/config/cfis/star_selection.setools#MASK:flag.FLAGS = "== 0". + workflow/config/cfis/star_selection.setools#MASK:flag.FLAGS = "== 0"; + workflow/config/cfis/default_tile.sex#SATUR_LEVEL = absent; + workflow/config/cfis/default_exp.sex#SATUR_LEVEL = absent. default: header_saturate options: header_saturate: @@ -608,6 +623,7 @@ analyses: The excluded strip is likely prescan; that interpretation is unverified here. Anchor: workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.CCD_SIZE = 33,2080,1,4612; + workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.MAKE_POST_PROCESS = True; src/shapepipe/modules/sextractor_package/sextractor_script.py::make_post_process; src/shapepipe/modules/sextractor_package/sextractor_script.py::ccd_candidate_mask. default: trimmed_bounds_33_2080 diff --git a/tests/helpers/astra_record.py b/tests/helpers/astra_record.py index c95c67caf..e864c7b02 100644 --- a/tests/helpers/astra_record.py +++ b/tests/helpers/astra_record.py @@ -136,6 +136,7 @@ def resolve_anchor(root, reference): try: kind, relative, selector = _parse_reference(reference) + _, expected = _split_assertion(reference) except ValueError as error: return str(error) path = Path(relative) @@ -154,6 +155,8 @@ def resolve_anchor(root, reference): except (OSError, UnicodeError) as error: return f"cannot read file: {error}" + if kind == "code" and expected == ABSENT: + return "absent assertions need a config key, not a code symbol" if kind == "code": if _is_snakemake_file(target): return _snakemake_symbol(text, selector) @@ -171,6 +174,8 @@ def resolve_anchor(root, reference): return None suffix = target.suffix.lower() + if expected == ABSENT: + return _absent_scope(text, selector, suffix) if suffix == ".ini": return _ini_key(text, selector) if suffix == ".setools": @@ -184,6 +189,39 @@ def resolve_anchor(root, reference): return f"unsupported config-key file type {suffix or '(no extension)'}" +ABSENT = "absent" +_LINE_SUFFIXES = {".sex", ".psfex", ".ww", ".param", ".conf", ".setools"} +_SECTIONED = {".ini", ".setools"} + + +def _absent_scope(text, selector, suffix): + """An absent key still needs a real file type and, if sectioned, section. + + Without the section check a renamed section would make every absence + trivially true. + """ + + if suffix not in _LINE_SUFFIXES | {".ini"}: + return f"unsupported config-key file type {suffix or '(no extension)'}" + if suffix not in _SECTIONED: + return None + if "." not in selector: + return "sectioned config ref needs SECTION.KEY" + section = selector.rsplit(".", 1)[0] + if suffix == ".ini": + try: + parser = _ini_parser(text, strict=False) + except configparser.Error as error: + return f"cannot parse INI file: {error}" + if section == parser.default_section or parser.has_section(section): + return None + elif any( + line.strip() == f"[{section}]" for line in text.splitlines() + ): + return None + return f"section {section!r} is missing" + + def _ini_key(text, selector): if "." not in selector: return "INI config ref needs SECTION.KEY" @@ -265,6 +303,62 @@ def visit(node): return result +def _mutated_name(target): + """Base name of ``NAME[...] =`` / ``NAME.attr =`` (nested included).""" + + while isinstance(target, (ast.Subscript, ast.Attribute)): + target = target.value + if isinstance(target, ast.Name): + return target.id + return None + + +@lru_cache(maxsize=128) +def _mutations(scope): + """Names whose bound object is item- or attribute-assigned in ``scope``. + + Same lexical scope only, like ``_bindings``; method calls such as + ``.update()`` and mutation from other scopes are not seen. + """ + + names = set() + + def visit(node): + if isinstance( + node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef, ast.Lambda) + ): + return + if isinstance(node, ast.Assign): + targets = node.targets + elif isinstance(node, (ast.AnnAssign, ast.AugAssign)): + targets = [node.target] + elif isinstance(node, ast.Delete): + targets = node.targets + elif isinstance(node, (ast.For, ast.AsyncFor)): + targets = [node.target] + elif isinstance(node, (ast.With, ast.AsyncWith)): + targets = [item.optional_vars for item in node.items] + else: + targets = [] + stack = [target for target in targets if target is not None] + while stack: + target = stack.pop() + if isinstance(target, (ast.Tuple, ast.List)): + stack.extend(target.elts) + elif isinstance(target, ast.Starred): + stack.append(target.value) + else: + name = _mutated_name(target) + if name: + names.add(name) + for child in ast.iter_child_nodes(node): + visit(child) + + for statement in scope.body: + visit(statement) + return frozenset(names) + + def _has_symbol(tree, symbol): scope = tree parts = symbol.split(".") @@ -347,6 +441,10 @@ def _selected_python_node(tree, selector): f"for {part!r}" ) node = declarations[0] + if index == len(parts) - 1 and part in _mutations(scope): + raise ValueError( + f"{symbol!r} is item- or attribute-assigned after binding" + ) if index < len(parts) - 1: if not isinstance( node, (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef) @@ -364,8 +462,8 @@ def _selected_python_node(tree, selector): return node -def _line_value(text, selector, suffix): - """Read active lines; SETools repeated predicates form an ordered list.""" +def _active_lines(text, selector, suffix): + """Every active (uncommented) setting of the key, as (value, predicate).""" section = None if suffix == ".setools": @@ -396,6 +494,13 @@ def _line_value(text, selector, suffix): value = value[1:].strip() values.append(value) predicates.append(predicate) + return values, predicates + + +def _line_value(text, selector, suffix): + """Read active lines; SETools repeated predicates form an ordered list.""" + + values, predicates = _active_lines(text, selector, suffix) if not values or any(not value for value in values): raise ValueError(f"no active value for {selector!r}") if len(values) > 1 and not all(predicates): @@ -404,13 +509,38 @@ def _line_value(text, selector, suffix): _NUMBER = re.compile(r"[+-]?(?:\d+(?:\.\d*)?|\.\d+)(?:[eE][+-]?\d+)?\Z") -_BOOLEANS = { - "y": True, "yes": True, "true": True, "on": True, - "n": False, "no": False, "false": False, "off": False, +# Boolean words each reader accepts. INI follows ConfigParser.getboolean, +# whose 1/0 spellings are handled at comparison (see _ini_bool); SExtractor +# and PSFEx add Y/N; Python literals are already bool, and the record spells +# them True/False. Elsewhere words stay text. +_INI_BOOLEANS = { + "yes": True, "true": True, "on": True, + "no": False, "false": False, "off": False, } +_ASTROMATIC_BOOLEANS = {**_INI_BOOLEANS, "y": True, "n": False} +_PYTHON_BOOLEANS = {"true": True, "false": False} +_NO_BOOLEANS = {} + + +def _booleans_for(suffix): + if suffix == ".ini": + return _INI_BOOLEANS + if suffix in {".sex", ".psfex"}: + return _ASTROMATIC_BOOLEANS + if suffix == ".py": + return _PYTHON_BOOLEANS + return _NO_BOOLEANS + + +def _ini_bool(want, got): + """getboolean also reads 1/0; accept them only against a bool expectation.""" + + if want[0] == "bool" and got[0] == "number" and got[1] in (0, 1): + return "bool", got[1] == 1 + return got -def _normalise_value(value): +def _normalise_value(value, booleans=_ASTROMATIC_BOOLEANS): """Use tagged atoms so boolean True cannot compare equal to number 1.""" if isinstance(value, bool): @@ -421,7 +551,7 @@ def _normalise_value(value): raise ValueError("numeric values must be finite") return "number", number if isinstance(value, (list, tuple)): - elements = tuple(_normalise_value(item) for item in value) + elements = tuple(_normalise_value(item, booleans) for item in value) if any(kind == "list" for kind, _ in elements): raise ValueError("only flat lists are supported") return "list", elements @@ -435,17 +565,17 @@ def _normalise_value(value): # inconsistent numeric/boolean coercions. It constructs no objects. parsed = yaml.load(value, Loader=yaml.BaseLoader) if isinstance(parsed, list): - return _normalise_value(parsed) + return _normalise_value(parsed, booleans) if not isinstance(parsed, str): raise ValueError("expected a scalar or flat list, not a mapping") # Quotes protect commas/operators; their contents are a single atom. value = parsed.strip() elif "," in value: - return _normalise_value(value.split(",")) + return _normalise_value(value.split(","), booleans) if _NUMBER.fullmatch(value): return "number", Decimal(value) - if value.lower() in _BOOLEANS: - return "bool", _BOOLEANS[value.lower()] + if value.lower() in booleans: + return "bool", booleans[value.lower()] return "text", value @@ -476,6 +606,19 @@ def check_anchor_value(root, reference): target = Path(root) / relative text = target.read_text(encoding="utf-8") suffix = target.suffix.lower() + if expected == ABSENT: + if suffix == ".ini": + section, key = selector.rsplit(".", 1) + parser = _ini_parser(text) + if not parser.has_option(section, key): + return None + actual = parser.get(section, key) + else: + active, _ = _active_lines(text, selector, suffix) + if not active: + return None + actual = active if len(active) > 1 else active[0] + return f"expected no active setting, actual {actual!r}" if kind == "code" and suffix == ".py": node = _selected_python_node(_python_tree(text), selector) try: @@ -497,8 +640,11 @@ def check_anchor_value(root, reference): raise ValueError( "value assertions need a config key or Python assignment" ) - want = _normalise_value(expected) - got = _normalise_value(actual) + booleans = _booleans_for(suffix) + want = _normalise_value(expected, booleans) + got = _normalise_value(actual, booleans) + if suffix == ".ini": + got = _ini_bool(want, got) if (suffix, selector) in {(".param", "VIGNET"), (".psfex", "PSF_SIZE")}: want, got = _square_stamp(want), _square_stamp(got) if want == got: diff --git a/tests/unit/test_astra_values.py b/tests/unit/test_astra_values.py index d50bca4e1..17160d355 100644 --- a/tests/unit/test_astra_values.py +++ b/tests/unit/test_astra_values.py @@ -68,10 +68,10 @@ def test_anchor_grammar_keeps_values_and_location_only_refs(tmp_path): ("13.", "13.0"), ("5e-4", "0.0005"), ("-1e3", "-1000"), - ("Y", "True"), - ("false", "N"), ("yes", "on"), ("off", "No"), + ("1", "True"), # ConfigParser.getboolean reads 1/0 as booleans. + ("0", "false"), ("2.5, 3.5", "[2.50, 3.500]"), ("XWIN_IMAGE,YWIN_IMAGE", "XWIN_IMAGE, YWIN_IMAGE"), ("$ROOT/data{number}.fits", '"$ROOT/data{number}.fits"'), @@ -83,6 +83,35 @@ def test_equivalent_spellings(tmp_path, actual, expected): assert check_anchor_value(tmp_path, ref) is None +@pytest.mark.parametrize("filename", ["config.sex", "model.psfex"]) +@pytest.mark.parametrize( + "actual, expected", [("Y", "True"), ("false", "N"), ("n", "off")] +) +def test_astromatic_formats_accept_y_n(tmp_path, filename, actual, expected): + (tmp_path / filename).write_text(f"KEY {actual}\n") + assert check_anchor_value(tmp_path, f"{filename}#KEY = {expected}") is None + + +@pytest.mark.parametrize( + "filename, contents, selector, expected", + [ + # getboolean raises on Y/N, so the record must not call them True/False. + ("config.ini", "[S]\nKEY = Y\n", "S.KEY", "True"), + ("config.ini", "[S]\nKEY = True\n", "S.KEY", "Y"), + ("config.ini", "[S]\nKEY = 2\n", "S.KEY", "True"), + ("stars.setools", "[RAND_SPLIT:s]\nKEY = Y\n", "RAND_SPLIT:s.KEY", "True"), + ("constants.py", "KEY = True\n", "KEY", "Y"), + ], +) +def test_boolean_words_follow_each_reader( + tmp_path, filename, contents, selector, expected +): + (tmp_path / filename).write_text(contents) + separator = "::" if filename.endswith(".py") else "#" + ref = f"{filename}{separator}{selector} = {expected}" + assert check_anchor_value(tmp_path, ref) is not None + + @pytest.mark.parametrize( "actual, expected", [ @@ -147,7 +176,7 @@ def test_ini_case_sections_defaults_and_interpolation(tmp_path): ) for selector, value in ( ("SCIENCE.Key", "30"), ("SCIENCE.KEY", "31"), - ("OTHER.KEY", "99"), ("SCIENCE.ENABLED", "Y"), + ("OTHER.KEY", "99"), ("SCIENCE.ENABLED", "yes"), ("DEFAULT.ENABLED", "True"), ("SCIENCE.PATH", "$DATA/%s/file"), ): ref = f"config.ini#{selector} = {value}" @@ -222,7 +251,7 @@ def test_python_literals_and_dict_paths_without_importing(tmp_path): ("WIDTH", "51.0"), ("NOISE", "0.0005"), ("Model.WIDTH", "53"), ("fit.limits", "-1,1000"), ("fit.options[step]", "1e-2"), ("COMPLETENESS[exp_split.split_exp_runner.expect]", "121"), - ("COMPLETENESS[exp_split.split_exp_runner.warn]", "Y"), + ("COMPLETENESS[exp_split.split_exp_runner.warn]", "True"), ): ref = f"constants.py::{selector} = {expected}" assert resolve_anchor(tmp_path, ref) is None @@ -244,6 +273,12 @@ def test_python_literals_and_dict_paths_without_importing(tmp_path): ("X = dict(**other)", "X[a]"), ("X = {'a': 1, **other}", "X[a]"), ("X = {variable: 1}", "X[a]"), + ("X = {'a': 1}\nX['a'] = 2", "X[a]"), + ("X = {'a': 1}\nX['b']['c'] = 2", "X[a]"), + ("X = {'a': 1}\nX['a'] += 1", "X[a]"), + ("X = {'a': 1}\ndel X['a']", "X[a]"), + ("X = 1\nX.attr = 2", "X"), + ("def f():\n X = {'a': 1}\n X['a'] = 2", "f.X[a]"), ], ) def test_python_nonliteral_or_ambiguous_values_fail_closed( @@ -299,3 +334,100 @@ def test_python_drift_is_not_hidden_by_a_cached_ast(tmp_path): def test_malformed_anchor_sentence_cannot_silently_skip_values(tmp_path): record = {"decisions": {"width": {"rationale": "Anchor: config.sex#KEY = 1"}}} assert value_errors(tmp_path, record) + + +@pytest.mark.parametrize( + "filename, contents, selector", + [ + ("config.sex", "SATUR_KEY SATURATE\n# SATUR_LEVEL 50000\n", "SATUR_LEVEL"), + ("config.ini", "[DEFAULT]\nA = 1\n[S]\nKEY = 1\n", "S.OTHER"), + ("stars.setools", "[MASK:other]\nMASK_EXT == 0\n[MASK:stars]\n" + "IMAFLAGS_ISO == 0\n# MASK_EXT == 0\n", "MASK:stars.MASK_EXT"), + ], +) +def test_absent_passes_when_no_active_setting(tmp_path, filename, contents, selector): + (tmp_path / filename).write_text(contents) + ref = f"{filename}#{selector} = absent" + assert resolve_anchor(tmp_path, ref) is None + assert check_anchor_value(tmp_path, ref) is None + + +@pytest.mark.parametrize( + "filename, contents, selector", + [ + ("config.sex", "SATUR_LEVEL 50000\n", "SATUR_LEVEL"), + ("config.sex", "SATUR_LEVEL\n", "SATUR_LEVEL"), + ("config.ini", "[S]\nKEY = 1\n", "S.KEY"), + ("config.ini", "[DEFAULT]\nKEY = 1\n[S]\n", "S.KEY"), + ("stars.setools", "[MASK:stars]\nIMAFLAGS_ISO == 0\nMASK_EXT == 0\n", + "MASK:stars.MASK_EXT"), + ], +) +def test_absent_fails_on_an_active_setting(tmp_path, filename, contents, selector): + (tmp_path / filename).write_text(contents) + problem = check_anchor_value(tmp_path, f"{filename}#{selector} = absent") + assert "expected no active setting" in problem + + +@pytest.mark.parametrize( + "filename, contents, reference", + [ + # A renamed section must not make every absence trivially true. + ("config.ini", "[S]\nKEY = 1\n", "config.ini#RENAMED.KEY = absent"), + ("stars.setools", "[MASK:stars]\nFLAGS == 0\n", + "stars.setools#MASK:renamed.MASK_EXT = absent"), + ("constants.py", "X = 1\n", "constants.py::Y = absent"), + ("missing.sex", None, "missing.sex#KEY = absent"), + ], +) +def test_absent_needs_a_real_file_and_section(tmp_path, filename, contents, reference): + if contents is not None: + (tmp_path / filename).write_text(contents) + assert resolve_anchor(tmp_path, reference) is not None + assert check_anchor_value(tmp_path, reference) is not None + + +def test_quoted_absent_is_ordinary_text(tmp_path): + (tmp_path / "config.sex").write_text("KEY absent\n") + assert check_anchor_value(tmp_path, 'config.sex#KEY = "absent"') is None + assert check_anchor_value(tmp_path, "config.sex#KEY = absent") is not None + + +@pytest.mark.parametrize( + "path, pattern, replacement", + [ + ("workflow/config/cfis/config_tile_Sx.ini", + "WEIGHT_IMAGE = True", "WEIGHT_IMAGE = False"), + ("workflow/config/cfis/config_tile_Sx.ini", + "MAKE_POST_PROCESS = True", "MAKE_POST_PROCESS = False"), + ("workflow/config/cfis/star_selection.setools", + "[MASK:star_selection]\n", "[MASK:star_selection]\nMASK_EXT == 0\n"), + ("workflow/config/cfis/default_tile.sex", + "SATUR_KEY", "SATUR_LEVEL 50000\nSATUR_KEY"), + ("src/shapepipe/modules/ngmix_package/ngmix.py", + "\n boot = ngmix.metacal.", + "\n metacal_pars['step'] = 0.02\n boot = ngmix.metacal."), + ], +) +def test_real_record_catches_gate_and_absence_drift( + tmp_path, path, pattern, replacement +): + """Each mutation once passed the record; each must now be reported.""" + + record = load_yaml(REPO_ROOT / "astra.yaml") + for ref in { + ref.split(" = ", 1)[0].split("#", 1)[0].split("::", 1)[0] + for anchor in extract_anchors(record) for ref in anchor.references + }: + source = REPO_ROOT / ref + if source.is_file(): + target = tmp_path / ref + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(source.read_text(encoding="utf-8")) + assert value_errors(tmp_path, record) == [] + + target = tmp_path / path + text = target.read_text(encoding="utf-8") + assert text.count(pattern) == 1 + target.write_text(text.replace(pattern, replacement)) + assert value_errors(tmp_path, record) diff --git a/workflow/config/cfis/CONTRACTS b/workflow/config/cfis/CONTRACTS index 9d06e72c1..52af90545 100644 --- a/workflow/config/cfis/CONTRACTS +++ b/workflow/config/cfis/CONTRACTS @@ -44,7 +44,7 @@ Cleaning removes detections after deblending, so switching it off or changing it The selected CORRECT treatment mirrors neighbour pixels across the target centre for SExtractor photometry; retain that convention when reproducing fluxes and windowed moments. Replacing it with blanking changes photometry and centroids, and must not be confused with ngmix's separate neighbour-weighting decision. -@sc [decision:detection.saturation_level,governs:default_tile.sex#SATUR_KEY;default_exp.sex#SATUR_KEY;star_selection.setools#MASK:star_selection.FLAGS] saturation-flags-follow-image-header +@sc [decision:detection.saturation_level,governs:default_tile.sex#SATUR_KEY;default_exp.sex#SATUR_KEY;star_selection.setools#MASK:star_selection.FLAGS;default_tile.sex;default_exp.sex] saturation-flags-follow-image-header SATUR_KEY must refer to the delivered image's saturation card for the intended per-image saturation flags, which FLAGS-based PSF-star rejection consumes. Do not interpret a missing card as evidence of no saturation: SExtractor then uses its built-in level because no SATUR_LEVEL is pinned here. Header availability remains unverified; changing the key or adding a fixed level requires checking the bright-star selection, not just whether SExtractor runs. @@ -67,6 +67,10 @@ Exposure IMAFLAGS_ISO and the multi-epoch flag stamps must come from the same de Keep the ME_IMAGE_PATTERN and ME_IMAGE_EXP_RUNNERS lists aligned so a flag stamp cannot silently become an image, weight or sky-mask product. Sky-fixed healsparse masks remain object-level catalogue information; substituting or rasterising them into this pixel path changes the masking decision. +@sc [decision:masking.psf_star_mask_veto,governs:star_selection.setools#MASK:star_selection.IMAFLAGS_ISO;star_selection.setools#MASK:preselect.IMAFLAGS_ISO;star_selection.setools#MASK:flag.IMAFLAGS_ISO;star_selection.setools] psf-stars-vetoed-on-instrument-flags-only +PSF-star candidates are rejected on the instrument flags (IMAFLAGS_ISO == 0) and on no sky-fixed map: no mask block cuts on MASK_EXT. +A MASK_EXT cut changes the PSF training sample and is a change to this decision, not a cleanup; querying maps through MASK_PATHS without the cut records MASK_EXT and changes no star. + @sc [decision:photometric_zeropoint,governs:default_tile.sex#MAG_ZEROPOINT;config_tile_Sx.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER;config_tile_Ng_template.ini#NGMIX_RUNNER.MAG_ZP;config_exp_psfex.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER;config_exp_psfex.ini#SEXTRACTOR_RUNNER.ZP_KEY] tile-and-epoch-photometry-share-flux-scale Tile MAG_ZEROPOINT, ngmix MAG_ZP and the FSCALE-rescaled epochs must describe the same flux calibration. Exposure photometry intentionally reads its own header zero-point; do not copy the fixed tile convention to exposures or change a header toggle independently of the calibration. From b01d233e94b36e715225e198e4ce92bd715508bf Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 11:00:38 +0200 Subject: [PATCH 19/40] docs(astra): PSFEx compiled SAMPLE_* defaults, verified by psfex -dd psfex_candidate_vetting and psfex-vetting-is-not-fully-disabled state the PSFEx 3.21.1 compiled defaults (psfex -dd in the develop-runtime image) and assert each omitted SAMPLE_* key absent from default.psfex, so pinning one is visible to the record. Which cuts act with SAMPLE_AUTOSELECT N is marked as from the PSFEx source, not re-read here. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014bvNTrAmZxcfb1ee83ApPK --- astra.yaml | 33 +++++++++++++++++++++++---------- workflow/config/cfis/CONTRACTS | 4 +++- 2 files changed, 26 insertions(+), 11 deletions(-) diff --git a/astra.yaml b/astra.yaml index f43222744..5f5cf5e83 100644 --- a/astra.yaml +++ b/astra.yaml @@ -918,21 +918,34 @@ analyses: psfex_candidate_vetting: label: PSFEx built-in candidate cuts, unpinned rationale: >- - default.psfex sets SAMPLE_AUTOSELECT N but omits - SAMPLE_MINSN, SAMPLE_MAXELLIP, SAMPLE_FWHMRANGE and - SAMPLE_VARIABILITY, so [HARDCODED] whatever PSFEx compiles - in for them governs, changing with the PSFEx version; - which of these cuts still act with SAMPLE_AUTOSELECT N is - unverified. BADPIXEL_FILTER N and PSF_RECENTER N accept - flagged star vignets unfiltered and do not recentre - candidates. + default.psfex sets SAMPLE_AUTOSELECT N (compiled default Y) + and omits every other SAMPLE_* key, so [HARDCODED] PSFEx's + compiled defaults govern them and can change with its + version. psfex -dd (PSFEx 3.21.1 in the develop-runtime + image) gives SAMPLE_FWHMRANGE 2.0,10.0, SAMPLE_VARIABILITY + 0.2, SAMPLE_MINSN 20, SAMPLE_MAXELLIP 0.3, SAMPLE_FLAGMASK + 0x00fe, SAMPLE_WFLAGMASK 0x0000 and SAMPLE_IMAFLAGMASK 0x0. + From the PSFEx source, not re-read here, MINSN, MAXELLIP, + FLAGMASK, FWHMRANGE and VARIABILITY still cut candidates + with SAMPLE_AUTOSELECT N; SETools cuts neither signal-to-noise + nor ellipticity, so MINSN and MAXELLIP can reject stars it + kept. BADPIXEL_FILTER N and PSF_RECENTER N, also the compiled + defaults, accept flagged star vignets unfiltered and do not + recentre candidates. Anchor: workflow/config/cfis/default.psfex#SAMPLE_AUTOSELECT = N; workflow/config/cfis/default.psfex#BADPIXEL_FILTER = N; - workflow/config/cfis/default.psfex#PSF_RECENTER = N. + workflow/config/cfis/default.psfex#PSF_RECENTER = N; + workflow/config/cfis/default.psfex#SAMPLE_FWHMRANGE = absent; + workflow/config/cfis/default.psfex#SAMPLE_VARIABILITY = absent; + workflow/config/cfis/default.psfex#SAMPLE_MINSN = absent; + workflow/config/cfis/default.psfex#SAMPLE_MAXELLIP = absent; + workflow/config/cfis/default.psfex#SAMPLE_FLAGMASK = absent; + workflow/config/cfis/default.psfex#SAMPLE_WFLAGMASK = absent; + workflow/config/cfis/default.psfex#SAMPLE_IMAFLAGMASK = absent. default: builtin_defaults options: builtin_defaults: - label: PSFEx built-in SAMPLE_* values, no bad-pixel filter + label: PSFEx 3.21.1 built-in SAMPLE_* values, no bad-pixel filter pinned_explicit: label: Write the SAMPLE_* values explicitly into default.psfex psf_modelling_software: diff --git a/workflow/config/cfis/CONTRACTS b/workflow/config/cfis/CONTRACTS index 52af90545..5cb2fc458 100644 --- a/workflow/config/cfis/CONTRACTS +++ b/workflow/config/cfis/CONTRACTS @@ -96,7 +96,9 @@ PSF_ACCURACY changes bright-star leverage in the fit weights, not merely an opti @sc [decision:star_selection_psf.psfex_candidate_vetting,governs:default.psfex#SAMPLE_AUTOSELECT;default.psfex#BADPIXEL_FILTER;default.psfex#PSF_RECENTER;default.psfex] psfex-vetting-is-not-fully-disabled Keep explicit stellar-locus selection under SETools control, but do not infer from SAMPLE_AUTOSELECT being off that every other PSFEx candidate cut is inactive. -The omitted SAMPLE_* settings inherit compiled defaults whose effect in this mode is unverified; a PSFEx version change requires inspecting those defaults and surviving candidates before claiming unchanged selection. +The omitted SAMPLE_* keys take PSFEx's compiled defaults (psfex -dd, PSFEx 3.21.1 in the develop-runtime image): FWHMRANGE 2.0,10.0, VARIABILITY 0.2, MINSN 20, MAXELLIP 0.3, FLAGMASK 0x00fe, WFLAGMASK 0x0000, IMAFLAGMASK 0x0. +From the PSFEx source, not re-read here, MINSN, MAXELLIP, FLAGMASK, FWHMRANGE and VARIABILITY still cut candidates with SAMPLE_AUTOSELECT N; SETools applies no signal-to-noise or ellipticity cut, so these can reject stars it kept. +A PSFEx version change can move these defaults: re-run psfex -dd and compare surviving candidates before claiming unchanged selection. BADPIXEL_FILTER and PSF_RECENTER also change the candidate pixels or positions, so enabling them is a scientific change, not a harmless cleanup. @sc [decision:star_selection_psf.star_selection_box,governs:config_tile_Ng_template.ini#NGMIX_RUNNER.PIXEL_SCALE;default_exp.sex#PIXEL_SCALE;star_selection.setools#MASK:preselect.FWHM_IMAGE;star_selection.setools#MASK:star_selection.FWHM_IMAGE;star_selection.setools#PLOT:fwhm_field.SCATTER;star_selection.setools] exposure-size-conventions-agree From 9eef18094e85146844fecc3fbd78ca948b73a8bb Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Sat, 26 Sep 2026 11:01:13 +0200 Subject: [PATCH 20/40] docs(astra): PSFEx vetting points at #919; compiled values are not assertable Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014bvNTrAmZxcfb1ee83ApPK --- astra.yaml | 4 +++- workflow/config/cfis/CONTRACTS | 1 + 2 files changed, 4 insertions(+), 1 deletion(-) diff --git a/astra.yaml b/astra.yaml index 5f5cf5e83..aa99432ce 100644 --- a/astra.yaml +++ b/astra.yaml @@ -931,7 +931,9 @@ analyses: nor ellipticity, so MINSN and MAXELLIP can reject stars it kept. BADPIXEL_FILTER N and PSF_RECENTER N, also the compiled defaults, accept flagged star vignets unfiltered and do not - recentre candidates. + recentre candidates. Compiled-in values admit no `= value` + assertion, so the anchors assert only that default.psfex omits + each SAMPLE_* key; issue #919 pins them in default.psfex. Anchor: workflow/config/cfis/default.psfex#SAMPLE_AUTOSELECT = N; workflow/config/cfis/default.psfex#BADPIXEL_FILTER = N; workflow/config/cfis/default.psfex#PSF_RECENTER = N; diff --git a/workflow/config/cfis/CONTRACTS b/workflow/config/cfis/CONTRACTS index 5cb2fc458..b3502589b 100644 --- a/workflow/config/cfis/CONTRACTS +++ b/workflow/config/cfis/CONTRACTS @@ -99,6 +99,7 @@ Keep explicit stellar-locus selection under SETools control, but do not infer fr The omitted SAMPLE_* keys take PSFEx's compiled defaults (psfex -dd, PSFEx 3.21.1 in the develop-runtime image): FWHMRANGE 2.0,10.0, VARIABILITY 0.2, MINSN 20, MAXELLIP 0.3, FLAGMASK 0x00fe, WFLAGMASK 0x0000, IMAFLAGMASK 0x0. From the PSFEx source, not re-read here, MINSN, MAXELLIP, FLAGMASK, FWHMRANGE and VARIABILITY still cut candidates with SAMPLE_AUTOSELECT N; SETools applies no signal-to-noise or ellipticity cut, so these can reject stars it kept. A PSFEx version change can move these defaults: re-run psfex -dd and compare surviving candidates before claiming unchanged selection. +Compiled-in values cannot be asserted from the config, only their omission from default.psfex; issue #919 pins them there. BADPIXEL_FILTER and PSF_RECENTER also change the candidate pixels or positions, so enabling them is a scientific change, not a harmless cleanup. @sc [decision:star_selection_psf.star_selection_box,governs:config_tile_Ng_template.ini#NGMIX_RUNNER.PIXEL_SCALE;default_exp.sex#PIXEL_SCALE;star_selection.setools#MASK:preselect.FWHM_IMAGE;star_selection.setools#MASK:star_selection.FWHM_IMAGE;star_selection.setools#PLOT:fwhm_field.SCATTER;star_selection.setools] exposure-size-conventions-agree From 39ae11eceedf73f08d9896ac4e6a33bac7dd067d Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 02:56:25 +0200 Subject: [PATCH 21/40] docs(astra): retire lints resolved by #907, #909, #918 The merged fixes made six record lints false and broke four value assertions and one contract ref. Re-anchor the MCCD exposure chain to what it now reads (split image/weight/flag, mask_query before setools), the setools FWHM plot to 0.187, and CFIS EXP_PREFIX to a location-only ref (blank). Pin the MCCD completeness counts the rationale now names. Drop the resolved lints from the record, the @sc blocks and CONTRACTS; the IMAFLAGS_ISO export (#912), ngmix's 0.186 pixel scale against star selection's 0.187, and the ngmix noisefill/noise-window doc lints remain. Co-Authored-By: Claude Opus 5.5 --- astra.yaml | 66 +++++++++---------- .../find_exposures_package/find_exposures.py | 5 +- src/shapepipe/modules/mask_query_runner.py | 4 +- src/shapepipe/pipeline/str_handler.py | 3 +- workflow/CONTRACTS | 5 +- workflow/config/cfis/CONTRACTS | 10 +-- workflow/scripts/completeness.py | 3 +- 7 files changed, 43 insertions(+), 53 deletions(-) diff --git a/astra.yaml b/astra.yaml index aa99432ce..e6a783063 100644 --- a/astra.yaml +++ b/astra.yaml @@ -97,15 +97,17 @@ decisions: appear at all. The one tolerated shortfall is the exposure-side psfex_interp VALIDATION output, where a CCD whose model fails the acceptance gate (star_selection_psf.psf_acceptance_thresholds) - produces nothing and the unit only warns; the MCCD chain, never - run in a campaign, warns on every runner. Science-path PSF - rejection does not go through this table: psfex_interp drops the - epoch per object inside the tile run. [LINT] the completeness.py - docstring gives tile psfex_interp as its warn example, but that - runner is mandatory, and a workflow/rules/exposure.smk comment - says a floor's :warn tolerates setools rejecting a sparse CCD, but - setools has a mandatory count and no floor exists. - Anchor: workflow/scripts/completeness.py::COMPLETENESS. + produces nothing and the unit only warns. The MCCD chain, never + run in a campaign, warns on every runner; its counts follow + config_exp_mccd.ini, with one mask_query catalogue per CCD and + preprocessing merging an exposure's stars into one train and one + test catalogue. Science-path PSF rejection does not go through + this table: psfex_interp drops the epoch per object inside the + tile run. + Anchor: workflow/scripts/completeness.py::COMPLETENESS; + workflow/scripts/completeness.py::COMPLETENESS[exp_psf.psfex.psfex_interp_runner.warn] = True; + workflow/scripts/completeness.py::COMPLETENESS[exp_psf.mccd.mask_query_runner.expect] = 40; + workflow/scripts/completeness.py::COMPLETENESS[exp_psf.mccd.mccd_preprocessing_runner.expect] = 2. default: exact_counts options: exact_counts: @@ -278,9 +280,7 @@ analyses: bits 0 and 1 are excluded because halos say nothing about whether a star is a good PSF sample. Imposing the veto is one line per mask block in star_selection.setools - (MASK_EXT == 0). [LINT] the workflow/rules/exposure.smk - docstring names the column FLAG_EXT and says setools cuts - on it; the code writes MASK_EXT and nothing cuts on it. + (MASK_EXT == 0). Anchor: workflow/config/cfis/config_exp_psfex.ini; workflow/config/cfis/star_selection.setools#MASK:star_selection.IMAFLAGS_ISO = "== 0"; workflow/config/cfis/star_selection.setools#MASK:preselect.IMAFLAGS_ISO = "== 0"; @@ -593,7 +593,7 @@ analyses: image and no detection coadd exists, so detection sees every tile pixel and the tile catalogue carries no IMAFLAGS_ISO. [LINT] final_cat.param, read by the post-processing merge, requests - IMAFLAGS_ISO, which the tile chain never produces. + IMAFLAGS_ISO, which the tile chain never produces (issue #912). Anchor: workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.DETECTION_IMAGE = False; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE = False; workflow/config/cfis/default_noimaflags.param; @@ -738,20 +738,20 @@ analyses: rationale: >- A tile's contributing exposures are the file names in column 3 (COLNUM) of each HISTORY line, stripped of their - extension and deduplicated: the coadd's own provenance is - trusted as the epoch list. Names keep their trailing p - (2243881p); downstream code drops it when it needs the - bare exposure ID. A mis-parse changes N_EPOCH and which - exposures are fit. [LINT] EXP_PREFIX = p is passed to - removeprefix, which does nothing to names where p is a - suffix, so the key has no effect. + full extension and deduplicated: the coadd's own provenance + is trusted as the epoch list, and a tile whose header cannot + be read fails rather than yielding an empty list. Names keep + their trailing p (2243881p), an epoch letter rather than a + prefix, so EXP_PREFIX is blank in the CFIS config; + downstream code drops the p when it needs the bare exposure + ID. A mis-parse changes N_EPOCH and which exposures are fit. Anchor: workflow/config/cfis/config_tile_Fe.ini#FIND_EXPOSURES_RUNNER.COLNUM = 3; - workflow/config/cfis/config_tile_Fe.ini#FIND_EXPOSURES_RUNNER.EXP_PREFIX = p; + workflow/config/cfis/config_tile_Fe.ini#FIND_EXPOSURES_RUNNER.EXP_PREFIX; src/shapepipe/modules/find_exposures_package/find_exposures.py::FindExposures.get_exposure_list. default: history_parse options: history_parse: - label: HISTORY column 3, prefix p, deduplicated + label: HISTORY column 3, no prefix stripped, deduplicated object_position_columns: label: Windowed centroids (XWIN/YWIN) define every position rationale: >- @@ -853,10 +853,6 @@ analyses: behaviour changes selection on sparse CCDs. PSFEx's own selection is off (SAMPLE_AUTOSELECT N); see psfex_candidate_vetting for what PSFEx may still apply. - [LINT] the file's statistics log the FWHM cut as mode +- - 0.1 px and its plot uses 0.186 arcsec/px, while the - applied cut is +- 0.2 px at 0.187; the _mode docstring - puts the median fallback at 10 objects, the code at 20. Anchor: workflow/config/cfis/star_selection.setools#MASK:star_selection.FLAGS = "== 0"; workflow/config/cfis/star_selection.setools#MASK:preselect.FLAGS = "== 0"; workflow/config/cfis/star_selection.setools#MASK:star_selection.IMAFLAGS_ISO = "== 0"; @@ -865,7 +861,7 @@ analyses: workflow/config/cfis/star_selection.setools#MASK:star_selection.FWHM_IMAGE = ["<= mode(FWHM_IMAGE{preselect}) + 0.2", ">= mode(FWHM_IMAGE{preselect}) - 0.2"]; workflow/config/cfis/star_selection.setools#MASK:preselect.MAG_AUTO = ["> 0", "< 21"]; workflow/config/cfis/star_selection.setools#MASK:preselect.FWHM_IMAGE = ["> 0.3 / 0.187", "< 1.5 / 0.187"]; - workflow/config/cfis/star_selection.setools#PLOT:fwhm_field.SCATTER = "FWHM_IMAGE{star_selection}*0.186"; + workflow/config/cfis/star_selection.setools#PLOT:fwhm_field.SCATTER = "FWHM_IMAGE{star_selection}*0.187"; workflow/config/cfis/default.psfex#SAMPLE_AUTOSELECT = N; src/shapepipe/pipeline/str_handler.py::StrInterpreter._mode. default: mode_centred_box @@ -960,17 +956,20 @@ analyses: (FP_GEOMETRY CFIS; N_COMP_LOC 8, D_COMP_GLOB 8, MIN_N_STARS 20, RMSE_THRESH 1.25); the completeness table treats its counts as warnings because no campaign has run - it. [LINT] config_exp_mccd.ini still reads pipeline_flag - images from mask_runner, which no longer exists, so the - MCCD exposure chain cannot run as committed. + it. Up to the model, its exposure chain matches PSFEx's: + SExtractor reads the split image, weight and instrument + flag directly, and mask_query sits between SExtractor and + setools. Anchor: workflow/config.yaml; workflow/config/cfis/config_MCCD.ini#INSTANCE.N_COMP_LOC = 8; workflow/config/cfis/config_MCCD.ini#INSTANCE.D_COMP_GLOB = 8; workflow/config/cfis/config_MCCD.ini#INSTANCE.FP_GEOMETRY = CFIS; workflow/config/cfis/config_MCCD.ini#INSTANCE.RMSE_THRESH = 1.25; workflow/config/cfis/config_MCCD.ini#INPUTS.MIN_N_STARS = 20; - workflow/config/cfis/config_exp_mccd.ini#SEXTRACTOR_RUNNER.INPUT_MODULE = split_exp_runner,mask_runner; - workflow/config/cfis/config_exp_mccd.ini#SEXTRACTOR_RUNNER.FILE_PATTERN = image,weight,pipeline_flag; + workflow/config/cfis/config_exp_mccd.ini#EXECUTION.MODULE = sextractor_runner,mask_query_runner,setools_runner,mccd_preprocessing_runner,mccd_fit_val_runner,merge_starcat_runner,mccd_plots_runner; + workflow/config/cfis/config_exp_mccd.ini#SEXTRACTOR_RUNNER.INPUT_DIR = $SP_RUN/output/run_sp_exp_Sp/split_exp_runner/output; + workflow/config/cfis/config_exp_mccd.ini#SEXTRACTOR_RUNNER.FILE_PATTERN = image,weight,flag; + workflow/config/cfis/config_exp_mccd.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE = True; src/shapepipe/modules/mccd_package. default: psfex options: @@ -2025,9 +2024,6 @@ analyses: measured in both. [HARDCODED] make_cat attaches only TILE_ID; there is no unique-object rule and no overlap flag, so duplicates are left to downstream selection. - [LINT] the make_cat package docstring documents a - TILE_LIST key that flags overlap objects; nothing - implements it. Anchor: src/shapepipe/modules/make_cat_package/make_cat.py::save_sextractor_data; src/shapepipe/modules/make_cat_package/__init__.py. default: no_dedup_in_pipeline diff --git a/src/shapepipe/modules/find_exposures_package/find_exposures.py b/src/shapepipe/modules/find_exposures_package/find_exposures.py index a41dde98e..0f3815a93 100644 --- a/src/shapepipe/modules/find_exposures_package/find_exposures.py +++ b/src/shapepipe/modules/find_exposures_package/find_exposures.py @@ -68,9 +68,8 @@ def get_exposure_list(self): The epoch list is the deduplicated file names in HISTORY column COLNUM with the extension stripped and the trailing ``p`` kept; a tile whose header cannot be read must fail, never yield an empty or partial list. - EXP_PREFIX is meant to strip a name prefix; removeprefix does nothing - to CFIS names, where ``p`` is a suffix, a [LINT] the decision record - carries. + EXP_PREFIX strips only a leading prefix; CFIS names carry none, so the + CFIS config leaves it blank and must not set it to the ``p`` suffix. Returns ------- diff --git a/src/shapepipe/modules/mask_query_runner.py b/src/shapepipe/modules/mask_query_runner.py index 9f0c700ea..f4bc97941 100644 --- a/src/shapepipe/modules/mask_query_runner.py +++ b/src/shapepipe/modules/mask_query_runner.py @@ -32,9 +32,7 @@ def mask_query_runner( Without MASK_PATHS the catalogue passes through with no MASK_EXT column; with it, MASK_EXT is carried for measurement and nothing here removes a star. The PSF-star veto is a setools edit (``MASK_EXT == 0`` beside - IMAFLAGS_ISO), not a change here. The workflow/rules/exposure.smk docstring - calls the column FLAG_EXT and says setools cuts on it, a [LINT] the - decision record carries. + IMAFLAGS_ISO), not a change here. """ sexcat_path = input_file_list[0] diff --git a/src/shapepipe/pipeline/str_handler.py b/src/shapepipe/pipeline/str_handler.py index 701026f06..7e6f5d5f3 100644 --- a/src/shapepipe/pipeline/str_handler.py +++ b/src/shapepipe/pipeline/str_handler.py @@ -262,8 +262,7 @@ def _mode(self, input, eps=0.001, iter_max=1000): Below 20 objects the result is the median; from 20 up it is the iterative histogram-zoom mode. The star-selection FWHM box is centred on this value, so moving the threshold or the binning changes which - stars train the PSF on sparse CCDs. The Returns section below puts the - fallback at 10, a [LINT] the decision record carries. + stars train the PSF on sparse CCDs. Parameters ---------- diff --git a/workflow/CONTRACTS b/workflow/CONTRACTS index f1bbfb65b..0f6488cbd 100644 --- a/workflow/CONTRACTS +++ b/workflow/CONTRACTS @@ -6,8 +6,7 @@ sc-list includes this file when queried for config.yaml or any config below work Refs in governs: are relative to this directory and use the shared ASTRA anchor resolver; config.yaml is a whole-file ref because the resolver has no YAML-key selector. CFIS key-level contracts live in config/cfis/CONTRACTS. -@sc [decision:star_selection_psf.psf_modelling_software,governs:config.yaml;config/cfis/config_exp_psfex.ini#EXECUTION.MODULE;config/cfis/config_tile_PiViVi_psfex.ini#EXECUTION.MODULE;config/cfis/config_exp_mccd.ini#EXECUTION.MODULE;config/cfis/config_exp_mccd.ini#SEXTRACTOR_RUNNER.INPUT_MODULE;config/cfis/config_tile_PiViVi_mccd.ini#EXECUTION.MODULE;config/cfis/config_MCCD.ini#INSTANCE.FP_GEOMETRY] psf-model-selects-a-matched-chain +@sc [decision:star_selection_psf.psf_modelling_software,governs:config.yaml;config/cfis/config_exp_psfex.ini#EXECUTION.MODULE;config/cfis/config_tile_PiViVi_psfex.ini#EXECUTION.MODULE;config/cfis/config_exp_mccd.ini#EXECUTION.MODULE;config/cfis/config_tile_PiViVi_mccd.ini#EXECUTION.MODULE;config/cfis/config_MCCD.ini#INSTANCE.FP_GEOMETRY] psf-model-selects-a-matched-chain The psf_model selector must choose a matching exposure-model producer and tile-interpolation consumer, including the model selected through SP_PSF in downstream inputs. PSFEx models each CCD independently; MCCD couples the focal plane, so switching software changes the scientific model, not just the executable name. -[LINT] the committed MCCD exposure config still requests the removed mask_runner; repair that input chain and validate the focal-plane model before treating MCCD as a runnable alternative. -Warning-only completeness counts for the untested MCCD chain do not establish scientific equivalence to the PSFEx chain. +The MCCD exposure chain shares PSFEx's detection inputs, mask_query and star selection, but no campaign has validated its focal-plane model; its warning-only completeness counts do not establish scientific equivalence to the PSFEx chain. diff --git a/workflow/config/cfis/CONTRACTS b/workflow/config/cfis/CONTRACTS index b3502589b..af1f1bd44 100644 --- a/workflow/config/cfis/CONTRACTS +++ b/workflow/config/cfis/CONTRACTS @@ -57,13 +57,13 @@ Aperture diameter and flux fraction also define reported measurements; they must @sc [decision:detection.detection_source_mode,governs:config_tile_Sx.ini#SEXTRACTOR_RUNNER.DETECTION_IMAGE;config_tile_Sx.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE;config_tile_Sx.ini#SEXTRACTOR_RUNNER.DOT_PARAM_FILE;default_tile.sex#PARAMETERS_NAME;default_noimaflags.param;final_cat.param#IMAFLAGS_ISO] tile-detection-has-no-instrument-flags Tile detection uses the r-band tile itself, with no separate detection coadd or instrument flag image; keep the requested columns consistent with those inputs. The runner's DOT_PARAM_FILE overrides PARAMETERS_NAME and deliberately selects the list without IMAFLAGS_ISO. -[LINT] final_cat.param requests IMAFLAGS_ISO although this tile chain never produces it; resolve the export/input mismatch rather than inventing a clean mask column or assuming FLAG_IMAGE alone supplies a mask. +[LINT] final_cat.param requests IMAFLAGS_ISO although this tile chain never produces it (issue #912); resolve the export/input mismatch rather than inventing a clean mask column or assuming FLAG_IMAGE alone supplies a mask. Shared pixel data and calibration --------------------------------- -@sc [decision:masking.pixel_mask_source,governs:config_exp_psfex.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE;config_exp_psfex.ini#SEXTRACTOR_RUNNER.FILE_PATTERN;config_exp_psfex.ini#SEXTRACTOR_RUNNER.INPUT_DIR;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_PATTERN;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_EXP_RUNNERS;star_selection.setools#MASK:star_selection.IMAFLAGS_ISO] instrument-flags-share-pixel-provenance -Exposure IMAFLAGS_ISO and the multi-epoch flag stamps must come from the same delivered instrument flag image split per CCD. +@sc [decision:masking.pixel_mask_source,governs:config_exp_psfex.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE;config_exp_psfex.ini#SEXTRACTOR_RUNNER.FILE_PATTERN;config_exp_psfex.ini#SEXTRACTOR_RUNNER.INPUT_DIR;config_exp_mccd.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE;config_exp_mccd.ini#SEXTRACTOR_RUNNER.FILE_PATTERN;config_exp_mccd.ini#SEXTRACTOR_RUNNER.INPUT_DIR;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_PATTERN;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_EXP_RUNNERS;star_selection.setools#MASK:star_selection.IMAFLAGS_ISO] instrument-flags-share-pixel-provenance +Exposure IMAFLAGS_ISO, under either PSF chain, and the multi-epoch flag stamps must come from the same delivered instrument flag image split per CCD. Keep the ME_IMAGE_PATTERN and ME_IMAGE_EXP_RUNNERS lists aligned so a flag stamp cannot silently become an image, weight or sky-mask product. Sky-fixed healsparse masks remain object-level catalogue information; substituting or rasterising them into this pixel path changes the masking decision. @@ -105,7 +105,7 @@ BADPIXEL_FILTER and PSF_RECENTER also change the candidate pixels or positions, @sc [decision:star_selection_psf.star_selection_box,governs:config_tile_Ng_template.ini#NGMIX_RUNNER.PIXEL_SCALE;default_exp.sex#PIXEL_SCALE;star_selection.setools#MASK:preselect.FWHM_IMAGE;star_selection.setools#MASK:star_selection.FWHM_IMAGE;star_selection.setools#PLOT:fwhm_field.SCATTER;star_selection.setools] exposure-size-conventions-agree Conversions of the same exposure pixels must use a consistent exposure scale for stellar-size selection, diagnostics and ngmix's one-pixel centroid prior; tile resampling is not a reason to assign a tile scale to epoch stamps. Diagnostic labels and statistics must describe the cut actually applied. -[LINT] ngmix and the FWHM map use 0.186 arcsec/px while preselection uses 0.187, and the statistics report a narrower FWHM window than selection applies; reconcile these with the exposure convention rather than propagating the mismatch. +[LINT] ngmix's PIXEL_SCALE is 0.186 arcsec/px while star selection's cuts, plot and statistics use 0.187; reconcile ngmix with the exposure convention rather than propagating the mismatch. Catalogue export ---------------- @@ -113,4 +113,4 @@ Catalogue export @sc [decision:catalogue_assembly.tile_overlap_handling,governs:config_tile_Mc.ini#MAKE_CAT_RUNNER.NUMBER_LIST;final_cat.param#TILE_ID] tile-id-is-provenance-not-deduplication Preserve TILE_ID when exporting the per-tile catalogues: overlapping tiles can contain separate measurements of the same sky object, and TILE_ID is the supplied provenance, not a uniqueness flag. NUMBER_LIST selects a tile to assemble, not a disjoint sky region; duplicate removal remains downstream. -The documented TILE_LIST overlap flag is unimplemented, so adding that key cannot make this catalogue unique or flag its overlaps. +No key flags or removes overlap objects, so making this catalogue unique or flagging its overlaps needs new code, not a config change. diff --git a/workflow/scripts/completeness.py b/workflow/scripts/completeness.py index a0a420aab..41fbc87d2 100644 --- a/workflow/scripts/completeness.py +++ b/workflow/scripts/completeness.py @@ -6,8 +6,7 @@ exposure, a tile or an ngmix chunk), so a partial unit never reaches the catalogue. On the psfex path only exposure-side psfex_interp may fall short (a CCD rejected by the acceptance gate writes nothing); the never-run MCCD chain -warns throughout. The ``warn`` field note below cites tile psfex_interp as its -example, but that runner is mandatory, a [LINT] the decision record carries. +warns throughout. This is the ported ``complete_check`` count table from the v2.0 bash layer (``run_job_sp_canfar_v2.0.bash`` job dispatch, survey §4): every non-warning From 950544b106fed7bf9d5925c588004172375c2146 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 03:31:25 +0200 Subject: [PATCH 22/40] docs(astra): tighten the record's prose Rationales no longer restate values their Anchor sentence asserts, and literature comparisons that a cited insight already carries become pointers. The header keeps the anchor grammar and markers; the value grammar's fine print moves to tests/helpers/astra_record.py, beside the parser that enforces it. Anchors, ids, options and evidence are unchanged (206 tests, astra validate). Corrected while tightening: the fit_initialisation default label (the galaxy guess takes its flux from a PSF-flux fit), and the blend_handling `none` description, which now claims only what Jarvis et al. 2016 support. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01AkrwnKzU1wfUaHUPt3yYk3 --- astra.yaml | 974 +++++++++++++++------------------- tests/helpers/astra_record.py | 31 +- 2 files changed, 460 insertions(+), 545 deletions(-) diff --git a/astra.yaml b/astra.yaml index e6a783063..ad56721c2 100644 --- a/astra.yaml +++ b/astra.yaml @@ -1,51 +1,24 @@ # ASTRA record of ShapePipe's scientific decisions: the choices embedded in # the code and the committed workflow configs (workflow/config/cfis/), why -# they stand, and the alternatives. CLAUDE.md says when to amend it; the -# tests/unit/test_astra_{anchors,values}.py check locations and values separately. +# they stand, and the alternatives. CLAUDE.md says when to amend it. # -# Conventions: -# * Every rationale ends with one sentence "Anchor: ; ." Each ref -# is a path relative to the repo root: CODE `path::Symbol` (a def, class -# or assignment target, dotted for nesting), CONFIG `path#SECTION.KEY` -# (`path#KEY` for sectionless .sex/.psfex/.param files; .setools uses -# SECTION.KEY), or FILE `path`. No line numbers. Commented-out keys may -# resolve as locations, but only active settings can assert values. -# `= absent` asserts a config key has no active line (its file and any -# section must exist); quote it ("absent") to mean the text. -# * Value assertions: `Anchor: path#SECTION.KEY = 1.5; path::NAME = 51.` -# A ref without ` = value` is location-only. Attaching the expectation to -# its locator avoids guessing which file a prose number describes, while -# keeping ordinary, schema-valid prose; no extra ASTRA keys or second -# parameter table. The anchor is the canonical recorded expectation; -# rationale/labels may repeat it for readability. Option ids are stable -# references, never parsed for values. Code/config remains execution truth. -# * Values are numbers, boolean words, strings (quote expressions), or flat -# comma lists, optionally bracketed. Semicolons are reserved for refs. -# Decimal equality is exact (1 = 1.0, 5e-4 = 0.0005), without rounding. -# Boolean words follow the file's reader, case-insensitively: INI as -# ConfigParser.getboolean (yes/true/on/1, no/false/off/0; Y/N are text), -# .sex/.psfex also Y/N, Python True/False; elsewhere words are text and -# 1/0 are numbers. Trim outer whitespace; other strings are case-sensitive. -# Lists preserve order and length. Only .param VIGNET and .psfex PSF_SIZE -# accept square-size shorthand: 51 = 51,51 (never 51,53). -# * SETools predicates retain their operators as quoted text; repeated cuts -# on one key are an ordered list, e.g. MAG_AUTO = ["> 18.", "< 22."]. -# Expressions compare as text, not algebra. Python selectors may append -# `[key.subkey]` to a named assignment (identifier-like string dict keys); -# only the selected literal is read, including dict(key=value) syntax. -# Ambiguous bindings/settings fail, including NAME[...] = or NAME.attr = -# in the same scope; .update()-style calls and mutation from other scopes -# are not seen. No imports, calls, arithmetic, argument defaults, -# environment expansion or implicit tool defaults are evaluated. -# These limits keep the check static rather than a second pipeline runtime. +# Conventions (tests/helpers/astra_record.py parses them and documents the +# full value grammar; tests/unit/test_astra_{anchors,values}.py enforce it): +# * Every rationale ends with one sentence "Anchor: ; ." Refs are +# repo-relative: CODE `path::Symbol` (a def, class or assignment target, +# dotted for nesting, `[key]` into a dict literal), CONFIG +# `path#SECTION.KEY` (`path#KEY` for sectionless .sex/.psfex/.param +# files), or FILE `path`. No line numbers. +# * `ref = value` asserts the value at that location; a bare ref asserts +# only that it exists. `= absent` asserts a config key has no active +# line. The anchor is the canonical expectation; prose may repeat a value +# for readability, and the code and configs remain execution truth. # * A decision's `default` is the option the committed code and configs # select; universes/committed.yaml pins it. -# * `excluded: true` means considered and rejected. An option the code does -# not implement says so in its description and is not excluded. -# * [HARDCODED] in a rationale marks a committed choice fixed in code with -# no config key to change it. -# * [LINT] in a rationale marks a place where the code disagrees with itself -# or with its own documentation. +# * `excluded: true` means considered and rejected. An option the code +# does not implement says so in its description and is not excluded. +# * [HARDCODED]: the committed choice is fixed in code, with no config key. +# [LINT]: the code disagrees with itself or its own documentation. # * A prior insight repeated inside a sub-analysis carries a `_local` # suffix, because insight ids are scoped. @@ -90,20 +63,18 @@ decisions: per_unit_completeness: label: Per-unit completeness gate rationale: >- - Every rule that runs shapepipe_run checks its products against a - nominal per-runner count (tile_ngmix per chunk): a runner below - its count fails the unit (an exposure or a tile), so a partial - unit never enters the catalogue; the missing unit's objects do not - appear at all. The one tolerated shortfall is the exposure-side - psfex_interp VALIDATION output, where a CCD whose model fails the - acceptance gate (star_selection_psf.psf_acceptance_thresholds) - produces nothing and the unit only warns. The MCCD chain, never - run in a campaign, warns on every runner; its counts follow - config_exp_mccd.ini, with one mask_query catalogue per CCD and - preprocessing merging an exposure's stars into one train and one - test catalogue. Science-path PSF rejection does not go through - this table: psfex_interp drops the epoch per object inside the - tile run. + Every shapepipe_run rule checks its products against a nominal + per-runner count (per chunk for tile_ngmix). A runner short of its count + fails the unit (exposure or tile), so a partial unit never enters the + catalogue and a failed unit's objects are absent. The one tolerated + shortfall is the exposure-side psfex_interp VALIDATION output: a CCD + whose model fails the acceptance gate + (star_selection_psf.psf_acceptance_thresholds) produces nothing and the + unit only warns. The MCCD chain, never run in a campaign, warns on every + runner; its counts follow config_exp_mccd.ini (one mask_query catalogue + per CCD; preprocessing merges an exposure's stars into one train and one + test catalogue). Science-path PSF rejection bypasses this table: + psfex_interp drops the epoch per object inside the tile run. Anchor: workflow/scripts/completeness.py::COMPLETENESS; workflow/scripts/completeness.py::COMPLETENESS[exp_psf.psfex.psfex_interp_runner.warn] = True; workflow/scripts/completeness.py::COMPLETENESS[exp_psf.mccd.mask_query_runner.expect] = 40; @@ -130,12 +101,11 @@ decisions: postage_stamp_size: label: Postage-stamp size shared by vignets, ngmix stamps and PSF models rationale: >- - One number, 51 px (about 9.5 arcsec), pins three coupled apertures: - the SExtractor VIGNET(51,51) around each detection, the vignetmaker - stamps that feed ngmix (STAMP_SIZE in both vignetmaker runs), and the - PSFEx model stamp (PSF_SIZE 51,51). The stamp is the pixel data ngmix - fits: it bounds the measurable galaxy size and truncates the wings of - large galaxies. No rationale for 51 is recorded. + One size pins three coupled apertures: the SExtractor VIGNET around each + detection, the vignetmaker stamps ngmix fits, and the PSFEx model + stamp. The stamp bounds the measurable galaxy size and truncates the + wings of large galaxies. No rationale for 51 px (about 9.5 arcsec) is + recorded. Anchor: workflow/config/cfis/default_noimaflags.param#VIGNET = 51; workflow/config/cfis/default.param#VIGNET = 51; workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.STAMP_SIZE = 51; @@ -150,18 +120,17 @@ decisions: larger_adaptive: label: Larger or size-adaptive stamps description: >- - Not implemented; needs the vignet, stamp and PSF sizes changed - together, since changing one alone desynchronises them. + Not implemented; the vignet, stamp and PSF sizes must change + together. photometric_zeropoint: label: Magnitude zero-point convention rationale: >- - Tiles use a fixed MAG_ZEROPOINT 30.0 (ZP_FROM_HEADER=False), and - ngmix repeats it in MAG_ZP; exposures read the per-image header - PHOTZP. The fixed tile value assumes the MegaPipe stacks are - calibrated to 30; nothing in the repo checks it. Magnitude cuts - (the star-selection window, downstream galaxy cuts) inherit - whichever convention their stage uses. + Tiles use a fixed zero-point of 30 (ZP_FROM_HEADER=False), repeated in + ngmix's MAG_ZP; exposures read the per-image header PHOTZP. The fixed + value assumes the MegaPipe stacks are calibrated to 30; nothing in the + repo checks it. Magnitude cuts (the star-selection window, downstream + galaxy cuts) inherit their stage's convention. Anchor: workflow/config/cfis/default_tile.sex#MAG_ZEROPOINT = 30.0; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER = False; workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.MAG_ZP = 30.0; @@ -181,10 +150,9 @@ decisions: prior_insights: des_psf_blacklist: claim: >- - DES enters a CCD's PSF model into a blacklist rather than failing the - exposure - in Y3, any CCD with fewer than 25 stars surviving outlier - rejection is blacklisted and excluded downstream (~2% of data removed), - and processing proceeds. + DES blacklists a CCD's PSF model rather than failing the exposure: in + Y3, a CCD with fewer than 25 stars surviving outlier rejection is + excluded downstream (~2% of data removed) and processing proceeds. created_at: "2026-07-16T00:00:00Z" evidence: - id: ev_jarvis_y3 @@ -213,12 +181,11 @@ analyses: masking: description: >- Which masks reach the measurement, and where. ShapePipe generates no - masks. The instrument flag image delivered with each exposure is the - only mask that reaches pixels. Sky-fixed masks (star halos and bodies, - manual regions, missing bands) are healsparse maps built outside - ShapePipe; their geometry is decided there. Inside ShapePipe they are - only queried at object positions into catalogue columns, and no stage - cuts on those columns. + masks; the instrument flag image delivered with each exposure is the + only one that reaches pixels. Sky-fixed masks (star halos and bodies, + manual regions, missing bands) are healsparse maps built and designed + outside ShapePipe, which only queries them at object positions into + catalogue columns that no stage cuts on. inputs: - id: exposure_flags type: data @@ -240,16 +207,16 @@ analyses: pixel_mask_source: label: Only the instrument flag image reaches pixels rationale: >- - On exposures SExtractor reads the split flag image (FLAG_IMAGE=True), - producing IMAFLAGS_ISO, which the PSF star selection requires to be - zero. The multi-epoch vignet run cuts flag stamps from the same split - flag image, and ngmix gives weight 0 to every flagged pixel and drops - epochs that are mostly flagged (shape_measurement.defect_fill, + On exposures SExtractor reads the split flag image, producing + IMAFLAGS_ISO, which the PSF star selection requires to be zero. The + multi-epoch vignet run cuts flag stamps from the same image; ngmix + gives flagged pixels weight 0 and drops mostly-flagged epochs + (shape_measurement.defect_fill, shape_measurement.epoch_masked_fraction_cut). Tiles have no flag image, so tile detection runs unflagged (detection.detection_source_mode). Sky-fixed masks never touch pixels: an object inside a star halo is measured from the same - pixels as one outside it. + unmodified pixels as one outside it. Anchor: workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE = True; workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_PATTERN = flag, image, weight, background, background_rms; src/shapepipe/modules/ngmix_package/ngmix.py::prepare_ngmix_weights. @@ -263,24 +230,19 @@ analyses: excluded: true excluded_reason: >- Sky-fixed masks say where an object sits, not that its pixels are - corrupted, so what to do about them is an analysis decision. - Rasterising them would bake one mask version into every shape; - catalogue columns leave the choice downstream. + corrupted, so acting on them is an analysis decision. + Rasterising would bake one mask version into every shape. psf_star_mask_veto: label: PSF-star candidates rejected on instrument flags only rationale: >- - The star selection cuts IMAFLAGS_ISO == 0 and nothing else - from the masks. mask_query sits in the exposure module - chain between SExtractor and setools: when MASK_PATHS - names maps it writes MASK_EXT (0 clean, nonzero flagged; - off-coverage counts as clean) onto each CCD's catalogue. - MASK_PATHS ships commented out, so the committed module - passes the catalogue through with no MASK_EXT column. The - intended map is the UNIONS star-body product (bit 2); halo - bits 0 and 1 are excluded because halos say nothing about - whether a star is a good PSF sample. Imposing the veto is - one line per mask block in star_selection.setools - (MASK_EXT == 0). + Of the masks, star selection cuts only on IMAFLAGS_ISO. mask_query sits + between SExtractor and setools in the exposure chain; when + MASK_PATHS names maps it writes MASK_EXT (0 clean, nonzero flagged, + off-coverage clean) onto each CCD's catalogue. MASK_PATHS ships + commented out, so the module passes catalogues through without + MASK_EXT. The intended map is the UNIONS star-body product (bit 2); + halo bits 0 and 1 are left out because halos say nothing about + whether a star is a good PSF sample. Anchor: workflow/config/cfis/config_exp_psfex.ini; workflow/config/cfis/star_selection.setools#MASK:star_selection.IMAFLAGS_ISO = "== 0"; workflow/config/cfis/star_selection.setools#MASK:preselect.IMAFLAGS_ISO = "== 0"; @@ -298,8 +260,9 @@ analyses: label: Also reject candidates on the star-body map (MASK_EXT == 0) description: >- Set MASK_PATHS to the star-body map and add MASK_EXT == 0 beside - each IMAFLAGS_ISO cut. Querying without the cut records MASK_EXT - and changes no star. + each IMAFLAGS_ISO cut in star_selection.setools (one line per + mask block). Querying without the cut records MASK_EXT and + changes no star. star_body_and_halo_veto: label: Also reject candidates inside star halos excluded: true @@ -311,12 +274,11 @@ analyses: rationale: >- The final catalogue ships every detected object. When MASK_EXT_PATHS lists band:path pairs, make_cat queries each healsparse map at the - object's windowed position and writes one MASK_ column holding - the map value verbatim; an object off a map's coverage gets that - map's sentinel (False for boolean maps, which reads as unmasked; - typically -1 for integer maps). The committed make_cat config sets no - MASK_EXT_PATHS, so no mask column is written and every mask cut - happens downstream against the maps themselves. + object's windowed position and writes the map value verbatim into a + MASK_ column; an object off coverage gets the map's sentinel + (False for boolean maps, reading as unmasked; typically -1 for + integer maps). The committed config sets no MASK_EXT_PATHS, so no + mask column is written and all mask cuts happen downstream. Anchor: src/shapepipe/modules/make_cat_runner.py::make_cat_runner; src/shapepipe/modules/make_cat_package/make_cat.py::save_mask_ext_data; src/shapepipe/utilities/mask_query.py::query_map. @@ -373,17 +335,12 @@ analyses: detection_threshold_policy: label: Detection significance, minimum area, matched filter rationale: >- - Tiles: DETECT_THRESH and ANALYSIS_THRESH 1.0 sigma, - DETECT_MINAREA 3, filtered with a 7x7 Gaussian of FWHM 3 - px (gauss_3.0_7x7.conv), near the CFIS average seeing of - 0.65 arcsec (about 3.5 px at 0.187 arcsec/px; the .sex - files set SEEING_FWHM 0.6). These are the parameters of - the MegaPipe tile catalogue, so ShapePipe's galaxy sample - matches the catalogue UNIONS adopts. Exposures: 1.5 sigma, - minarea 5, the 3x3 FWHM 2 px kernel (default.conv); they - only feed star selection. Guinot+22 lists 1.5 sigma, - minarea 10 and the default 3x3 kernel, so the tiles differ - from the paper on all three. + Tiles use the MegaPipe tile-catalogue threshold, minimum area and + filter, so ShapePipe's galaxy sample matches the catalogue UNIONS + adopts; the 7x7 Gaussian of FWHM 3 px is near the CFIS average seeing + of 0.65 arcsec (about 3.5 px at 0.187 arcsec/px). Exposures only feed + star selection and keep stock values with the 3x3 FWHM 2 px kernel. + The tiles differ from Guinot+22 in all three. Anchor: workflow/config/cfis/default_tile.sex#DETECT_THRESH = 1.0; workflow/config/cfis/default_tile.sex#ANALYSIS_THRESH = 1.0; workflow/config/cfis/default_tile.sex#DETECT_MINAREA = 3; @@ -414,10 +371,9 @@ analyses: deblending_policy: label: Deblending contrast rationale: >- - DEBLEND_NTHRESH 32 on both passes; DEBLEND_MINCONT 0.002 - on tiles (the MegaPipe value) and 0.001 on exposures. - Contrast sets object count, centroids, and blend - contamination in shapes. Guinot+22 lists 0.001. + Tiles use the MegaPipe DEBLEND_MINCONT; exposures use a lower one, + the value Guinot+22 lists. Both share DEBLEND_NTHRESH. Contrast sets + object count, centroids, and blend contamination in shapes. Anchor: workflow/config/cfis/default_tile.sex#DEBLEND_MINCONT = 0.002; workflow/config/cfis/default_exp.sex#DEBLEND_MINCONT = 0.001; workflow/config/cfis/default_tile.sex#DEBLEND_NTHRESH = 32; @@ -436,14 +392,12 @@ analyses: background_model: label: Background estimation and photometric background rationale: >- - Both passes estimate the background with SExtractor AUTO. Tiles use - the MegaPipe mesh BACK_SIZE 512 with BACK_FILTERSIZE 9 and a LOCAL - photometric background (annulus BACKPHOTO_THICK 30); exposures use - mesh 64, filter 3 and a GLOBAL photometric background. The exposure - BACKGROUND and BACKGROUND_RMS maps are also what ngmix subtracts from - each epoch and weights its pixels by - (shape_measurement.galaxy_pixel_weights). The header background path - is off (BKG_FROM_HEADER=False). Residual sky offsets propagate into + Both passes use SExtractor AUTO backgrounds: tiles with the MegaPipe + mesh and filter sizes and a LOCAL photometric background, exposures + with a finer mesh and a GLOBAL one. ngmix subtracts the exposure + BACKGROUND map from each epoch and weights its pixels by + BACKGROUND_RMS (shape_measurement.galaxy_pixel_weights). The header + background path is off. Residual sky offsets propagate into thresholds, fluxes, completeness and shapes. Anchor: workflow/config/cfis/default_tile.sex#BACK_TYPE = AUTO; workflow/config/cfis/default_tile.sex#BACK_SIZE = 512; @@ -470,11 +424,11 @@ analyses: weight_map_usage: label: Weight map as inverse variance for detection rationale: >- - WEIGHT_TYPE MAP_WEIGHT on both passes (SExtractor default - NONE): the per-pixel variance sets the effective SNR and - so the detection set. RESCALE_WEIGHTS and WEIGHT_GAIN are - Y, their SExtractor defaults. Guinot+22 keeps every non-tabulated - parameter at its default, which would mean no weight map. + MAP_WEIGHT on both passes (SExtractor default NONE): the per-pixel + variance sets the effective SNR and so the detection set. + RESCALE_WEIGHTS and WEIGHT_GAIN keep their SExtractor defaults. + Guinot+22 keeps every non-tabulated parameter at its default, which + would mean no weight map. Anchor: workflow/config/cfis/default_tile.sex#WEIGHT_TYPE = MAP_WEIGHT; workflow/config/cfis/default_exp.sex#WEIGHT_TYPE = MAP_WEIGHT; workflow/config/cfis/default_tile.sex#RESCALE_WEIGHTS = Y; @@ -494,10 +448,9 @@ analyses: zero_weight_interpolation: label: Interpolation across zero-weight pixels rationale: >- - INTERP_TYPE ALL on both passes (SExtractor default NONE), with - INTERP_MAXXLAG/INTERP_MAXYLAG 16: SExtractor invents flux across - zero-weight pixels, which changes detections and photometry near - masked regions. No rationale is recorded. + INTERP_TYPE ALL on both passes (SExtractor default NONE): SExtractor + invents flux across zero-weight pixels, changing detections and + photometry near masked regions. No rationale recorded. Anchor: workflow/config/cfis/default_tile.sex#INTERP_TYPE = ALL; workflow/config/cfis/default_exp.sex#INTERP_TYPE = ALL; workflow/config/cfis/default_tile.sex#INTERP_MAXXLAG = 16; @@ -513,7 +466,7 @@ analyses: spurious_detection_cleaning: label: Cleaning of spurious detections rationale: >- - CLEAN Y with CLEAN_PARAM 1.0 on both passes deletes detections + CLEAN on both passes deletes detections consistent with being wings of a brighter neighbour, a post-deblend change to the object list. Stock value; no rationale recorded. Anchor: workflow/config/cfis/default_tile.sex#CLEAN_PARAM = 1.0; @@ -527,10 +480,9 @@ analyses: blend_photometry_mask_type: label: Neighbour pixels in blend photometry rationale: >- - MASK_TYPE CORRECT on both passes replaces pixels belonging to a - neighbour by their mirror across the object centre during - photometry, changing fluxes and windowed moments of blends. Stock - value; no rationale recorded. + MASK_TYPE CORRECT on both passes replaces neighbour pixels by their + mirror across the object centre during photometry, changing blend + fluxes and windowed moments. Stock value; no rationale recorded. Anchor: workflow/config/cfis/default_tile.sex#MASK_TYPE = CORRECT; workflow/config/cfis/default_exp.sex#MASK_TYPE = CORRECT. default: correct @@ -542,16 +494,15 @@ analyses: saturation_level: label: Saturation level read from each image's header rationale: >- - SATUR_KEY SATURATE on both passes and no SATUR_LEVEL: SExtractor - takes each image's saturation level from its SATURATE header card and - sets the saturation bit (4) of FLAGS on objects with saturated - pixels. Star selection requires FLAGS == 0 at every step, so the - level decides which bright stars are rejected from the PSF sample; - tile FLAGS reach the catalogue for downstream cuts. Where the card is - absent, SExtractor falls back to SATUR_LEVEL, which neither .sex file - sets, so its built-in default applies (50000 ADU per the SExtractor - documentation). Whether the delivered exposure CCDs and MegaPipe - tiles carry SATURATE is unverified here. + Both passes take each image's saturation level from its SATURATE + header card and set FLAGS bit 4 on objects with saturated pixels. + Star selection requires FLAGS == 0 at every step, so the level decides + which bright stars leave the PSF sample; tile FLAGS reach the + catalogue for downstream cuts. Where the card is absent, SExtractor + falls back to its built-in SATUR_LEVEL (50000 ADU per its + documentation), since neither .sex file sets one. Whether the + delivered exposure CCDs and MegaPipe tiles carry SATURATE is + unverified. Anchor: workflow/config/cfis/default_exp.sex#SATUR_KEY = SATURATE; workflow/config/cfis/default_tile.sex#SATUR_KEY = SATURATE; workflow/config/cfis/star_selection.setools#MASK:star_selection.FLAGS = "== 0"; @@ -568,12 +519,11 @@ analyses: photometry_parameters: label: Kron and aperture photometry definitions rationale: >- - PHOT_AUTOPARAMS 2.5,3.5 (Kron factor, minimum radius), PHOT_APERTURES - 5 px and PHOT_FLUXFRAC 0.5 on both passes. MAG_AUTO is the axis of - the star-selection magnitude window and the catalogue magnitude; - FLUX_AUTO is PSFEx's photometric normalisation. A different Kron - factor shifts magnitudes and so every magnitude-based cut. Stock - values; no rationale recorded. + Stock Kron, aperture and flux-fraction settings on both passes; no + rationale recorded. MAG_AUTO is the axis of the star-selection + magnitude window and the catalogue magnitude; FLUX_AUTO is PSFEx's + photometric normalisation. A different Kron factor shifts magnitudes + and so every magnitude-based cut. Anchor: workflow/config/cfis/default_tile.sex#PHOT_AUTOPARAMS = 2.5,3.5; workflow/config/cfis/default_exp.sex#PHOT_AUTOPARAMS = 2.5,3.5; workflow/config/cfis/default_tile.sex#PHOT_APERTURES = 5; @@ -588,10 +538,10 @@ analyses: detection_source_mode: label: Single-image, unflagged detection on the r-band tile rationale: >- - DETECTION_IMAGE=False and FLAG_IMAGE=False with the - default_noimaflags.param column list: tiles have no instrument flag - image and no detection coadd exists, so detection sees every tile - pixel and the tile catalogue carries no IMAFLAGS_ISO. [LINT] + No detection image, no flag image, and the default_noimaflags.param + column list: tiles have no instrument flag image and no detection + coadd exists, so detection sees every tile pixel and the tile + catalogue carries no IMAFLAGS_ISO. [LINT] final_cat.param, read by the post-processing merge, requests IMAFLAGS_ISO, which the tile chain never produces (issue #912). Anchor: workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.DETECTION_IMAGE = False; @@ -614,14 +564,12 @@ analyses: epoch_membership_ccd_bounds: label: Which exposure CCDs an object belongs to (N_EPOCH) rationale: >- - CCD_SIZE = 33,2080,1,4612 with strict inequalities bounds - each CCD's usable x range, and a WCS inversion failure - skips the CCD, lowering N_EPOCH. This sets how many - exposures enter each galaxy's multi-epoch fit. The x endpoints - 33 and 2080 match MegaCam's raw DATASEC (2048 pixel indices - inclusively), but the strict cut excludes both endpoints. - The excluded strip is likely prescan; that interpretation is - unverified here. + CCD_SIZE bounds each CCD's usable x range with strict inequalities, + and a WCS inversion failure skips the CCD, lowering N_EPOCH; this + sets how many exposures enter each galaxy's multi-epoch fit. The x + endpoints match MegaCam's raw DATASEC (2048 pixel indices + inclusively), but the strict cut excludes both. The excluded strip is + likely prescan (unverified). Anchor: workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.CCD_SIZE = 33,2080,1,4612; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.MAKE_POST_PROCESS = True; src/shapepipe/modules/sextractor_package/sextractor_script.py::make_post_process; @@ -701,12 +649,11 @@ analyses: astrometric_solution_source: label: Astrometry taken verbatim from delivered per-CCD headers rationale: >- - split_exp builds WCS(header) from each raw CCD header, and - merge_headers stores the lot; every downstream world-to-pixel - transform (stamp placement, epoch membership, position seeding) uses - that solution. [HARDCODED] no re-derivation or astrometric refinement - exists. A joint re-fit would move every stamp centre and position - seed. + split_exp builds WCS(header) from each raw CCD header and + merge_headers stores them; every downstream world-to-pixel transform + (stamp placement, epoch membership, position seeding) uses that + solution. [HARDCODED] no astrometric re-derivation or refinement + exists; a joint re-fit would move every stamp centre and position seed. Anchor: src/shapepipe/modules/split_exp_package/split_exp.py::SplitExposures.create_hdus; src/shapepipe/modules/merge_headers_package/merge_headers.py::merge_headers. default: delivered_headers @@ -720,9 +667,8 @@ analyses: ccd_split_extent: label: All 40 MegaCam HDUs split and carried as candidate epochs rationale: >- - N_HDU=40, and any other HDU count raises: every one of the - 40 CCDs is a candidate epoch wherever the WCS lands it. - Excluding a subset of CCDs is the alternative. + Any HDU count other than N_HDU raises; every CCD is a candidate + epoch wherever the WCS lands it. Anchor: workflow/config/cfis/config_exp_Sp.ini#SPLIT_EXP_RUNNER.N_HDU = 40; src/shapepipe/modules/split_exp_package/split_exp.py::SplitExposures.create_hdus. default: all_40_hdus @@ -736,15 +682,13 @@ analyses: epoch_provenance_from_tile_history: label: Epoch sets parsed from tile FITS HISTORY cards rationale: >- - A tile's contributing exposures are the file names in - column 3 (COLNUM) of each HISTORY line, stripped of their - full extension and deduplicated: the coadd's own provenance - is trusted as the epoch list, and a tile whose header cannot - be read fails rather than yielding an empty list. Names keep - their trailing p (2243881p), an epoch letter rather than a - prefix, so EXP_PREFIX is blank in the CFIS config; - downstream code drops the p when it needs the bare exposure - ID. A mis-parse changes N_EPOCH and which exposures are fit. + The coadd's own provenance is the epoch list: the file names in + column COLNUM of each HISTORY line, stripped of their full extension + and deduplicated. A tile whose header cannot be read fails rather + than yielding an empty list. Names keep their trailing p (2243881p), + an epoch letter rather than a prefix, so EXP_PREFIX is blank; + downstream code drops the p when it needs the bare exposure ID. A + mis-parse changes N_EPOCH and which exposures are fit. Anchor: workflow/config/cfis/config_tile_Fe.ini#FIND_EXPOSURES_RUNNER.COLNUM = 3; workflow/config/cfis/config_tile_Fe.ini#FIND_EXPOSURES_RUNNER.EXP_PREFIX; src/shapepipe/modules/find_exposures_package/find_exposures.py::FindExposures.get_exposure_list. @@ -755,16 +699,13 @@ analyses: object_position_columns: label: Windowed centroids (XWIN/YWIN) define every position rationale: >- - PSF interpolation sites, tile and multi-epoch stamp - centres, and the catalogue position all use SExtractor's - windowed centroid. Tile stamps are cut at - XWIN_IMAGE/YWIN_IMAGE in tile pixels (COORD = PIX); PSF - interpolation and multi-epoch stamps use - XWIN_WORLD/YWIN_WORLD; exposure-side PSF validation uses - XWIN_IMAGE/YWIN_IMAGE. Windowed, isophotal and model - centroids differ systematically for blends and asymmetric - galaxies, and the centroid feeds the position seed and the - centroid prior. + PSF interpolation sites, tile and multi-epoch stamp centres, and the + catalogue position all use SExtractor's windowed centroid: in pixels + for tile stamps and exposure-side PSF validation, in world + coordinates for tile-side PSF interpolation and multi-epoch stamps. + Windowed, isophotal and model centroids differ systematically for + blends and asymmetric galaxies, and the centroid feeds the position + seed and the centroid prior. Anchor: workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.POSITION_PARAMS = XWIN_IMAGE,YWIN_IMAGE; workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.COORD = PIX; workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS = XWIN_WORLD,YWIN_WORLD; @@ -778,17 +719,15 @@ analyses: stamp_positioning_and_padding: label: Nearest-pixel stamp extraction with zero padding rationale: >- - [HARDCODED] stamps are cut around the pixel nearest the - object's position, with no sub-pixel interpolation; the - sub-pixel remainder is stored as the stamp's OFFSET, which - ngmix uses as the Jacobian origin - (shape_measurement.centroid_source), so extraction and - centroid prior share one rounding. Multi-epoch stamps take - the position from the tile world coordinate through the - stored per-CCD WCS. Objects whose stamp overruns an image - edge are kept, with out-of-image pixels zero-filled. A - stamp centre that rounds outside the image raises, which - fails the vignet run for the whole tile. + [HARDCODED] stamps are cut around the pixel nearest the object's + position, with no sub-pixel interpolation; the sub-pixel remainder + is stored as the stamp's OFFSET, which ngmix uses as the Jacobian + origin (shape_measurement.centroid_source), so extraction and + centroid prior share one rounding. Multi-epoch stamps map the tile + world coordinate through the stored per-CCD WCS. Stamps overrunning + an image edge are kept, zero-filled outside the image; a stamp + centre that rounds outside the image raises and fails the vignet + run for the whole tile. Anchor: src/shapepipe/modules/vignetmaker_package/vignetmaker.py::get_stamps; src/shapepipe/modules/vignetmaker_package/vignetmaker.py::VignetMaker._get_stamp_me. default: round_and_zero_pad @@ -845,14 +784,13 @@ analyses: star_selection_box: label: Stellar-locus selection, magnitude window and FWHM window around the mode rationale: >- - 18 < MAG_AUTO < 22, |FWHM - mode| <= 0.2 px, FLAGS == 0 - and IMAFLAGS_ISO == 0. The mode is computed on a - preselection (MAG_AUTO < 21, FWHM 0.3-1.5 arcsec at 0.187 - arcsec/px) by an iterative histogram-zoom estimator that - falls back to the median below 20 objects, so small-N - behaviour changes selection on sparse CCDs. PSFEx's own - selection is off (SAMPLE_AUTOSELECT N); see - psfex_candidate_vetting for what PSFEx may still apply. + Stars are flag-free objects in a MAG_AUTO window whose FWHM lies + within 0.2 px of the mode of a looser, size-limited preselection + (FWHM 0.3-1.5 arcsec at 0.187 arcsec/px). The mode estimator is an + iterative histogram zoom that falls back to the median below 20 + objects, so small-N behaviour changes selection on sparse CCDs. + PSFEx's own selection is off; see psfex_candidate_vetting for what + PSFEx may still apply. Anchor: workflow/config/cfis/star_selection.setools#MASK:star_selection.FLAGS = "== 0"; workflow/config/cfis/star_selection.setools#MASK:preselect.FLAGS = "== 0"; workflow/config/cfis/star_selection.setools#MASK:star_selection.IMAFLAGS_ISO = "== 0"; @@ -882,14 +820,12 @@ analyses: psf_train_validation_split: label: Seeded 80/20 star split, model fit vs held-out validation rationale: >- - RAND_SPLIT RATIO 20: the 80% sample fits the PSFEx model - and feeds the tile multi-epoch interpolation - (ME_DOT_PSF_PATTERN); the 20% sample is the independent - residual diagnostic (psfex_interp VALIDATION mode). The - split trades training stars per CCD, which interacts with - the acceptance gate, against an independent residual test. - It is deterministic: a permutation seeded from the digits - of the unit's file number, so a given CCD gets the same + The 80% sample fits the PSFEx model and feeds the tile multi-epoch + interpolation; the 20% sample is the independent residual + diagnostic (psfex_interp VALIDATION mode). The split trades + training stars per CCD, which interacts with the acceptance gate, + against an independent residual test. The permutation is seeded + from the digits of the unit's file number, so a CCD gets the same split on every run. Anchor: workflow/config/cfis/star_selection.setools#RAND_SPLIT:star_split.RATIO = 20; src/shapepipe/modules/setools_package/setools.py::SETools._make_rand_split; @@ -914,22 +850,21 @@ analyses: psfex_candidate_vetting: label: PSFEx built-in candidate cuts, unpinned rationale: >- - default.psfex sets SAMPLE_AUTOSELECT N (compiled default Y) - and omits every other SAMPLE_* key, so [HARDCODED] PSFEx's - compiled defaults govern them and can change with its - version. psfex -dd (PSFEx 3.21.1 in the develop-runtime - image) gives SAMPLE_FWHMRANGE 2.0,10.0, SAMPLE_VARIABILITY - 0.2, SAMPLE_MINSN 20, SAMPLE_MAXELLIP 0.3, SAMPLE_FLAGMASK - 0x00fe, SAMPLE_WFLAGMASK 0x0000 and SAMPLE_IMAFLAGMASK 0x0. - From the PSFEx source, not re-read here, MINSN, MAXELLIP, - FLAGMASK, FWHMRANGE and VARIABILITY still cut candidates - with SAMPLE_AUTOSELECT N; SETools cuts neither signal-to-noise - nor ellipticity, so MINSN and MAXELLIP can reject stars it - kept. BADPIXEL_FILTER N and PSF_RECENTER N, also the compiled - defaults, accept flagged star vignets unfiltered and do not - recentre candidates. Compiled-in values admit no `= value` - assertion, so the anchors assert only that default.psfex omits - each SAMPLE_* key; issue #919 pins them in default.psfex. + default.psfex sets SAMPLE_AUTOSELECT N (compiled default Y) and + omits every other SAMPLE_* key, so [HARDCODED] PSFEx's compiled + defaults govern them and can change with its version. psfex -dd + (PSFEx 3.21.1, develop-runtime image) gives SAMPLE_FWHMRANGE + 2.0,10.0, SAMPLE_VARIABILITY 0.2, SAMPLE_MINSN 20, SAMPLE_MAXELLIP + 0.3, SAMPLE_FLAGMASK 0x00fe, SAMPLE_WFLAGMASK 0x0000 and + SAMPLE_IMAFLAGMASK 0x0. Per the PSFEx source (not re-read here), + MINSN, MAXELLIP, FLAGMASK, FWHMRANGE and VARIABILITY still cut + candidates with autoselect off; SETools cuts neither S/N nor + ellipticity, so MINSN and MAXELLIP can reject stars it kept. + BADPIXEL_FILTER N and PSF_RECENTER N, also compiled defaults, leave + flagged star vignets unfiltered and candidates unrecentred. + Compiled-in values admit no `= value` assertion, so the anchors + assert only that each SAMPLE_* key is absent; open issue #919 would + pin them in default.psfex. Anchor: workflow/config/cfis/default.psfex#SAMPLE_AUTOSELECT = N; workflow/config/cfis/default.psfex#BADPIXEL_FILTER = N; workflow/config/cfis/default.psfex#PSF_RECENTER = N; @@ -949,17 +884,14 @@ analyses: psf_modelling_software: label: PSF model, PSFEx per CCD or MCCD over the focal plane rationale: >- - workflow/config.yaml psf_model selects the exposure and - tile config pair; the committed value is psfex, which fits - each CCD independently. MCCD (Liaudat+2021) fits one - hybrid local+global model over the focal plane - (FP_GEOMETRY CFIS; N_COMP_LOC 8, D_COMP_GLOB 8, - MIN_N_STARS 20, RMSE_THRESH 1.25); the completeness table - treats its counts as warnings because no campaign has run - it. Up to the model, its exposure chain matches PSFEx's: - SExtractor reads the split image, weight and instrument - flag directly, and mask_query sits between SExtractor and - setools. + psf_model in workflow/config.yaml selects the exposure and tile + config pair; the committed value is psfex, which fits each CCD + independently. MCCD (Liaudat+2021) fits one hybrid local+global + model over the focal plane; the completeness table treats its + counts as warnings because no campaign has run it. Up to the model, + its exposure chain matches PSFEx's: SExtractor reads the split + image, weight and instrument flag directly, and mask_query sits + between SExtractor and setools. Anchor: workflow/config.yaml; workflow/config/cfis/config_MCCD.ini#INSTANCE.N_COMP_LOC = 8; workflow/config/cfis/config_MCCD.ini#INSTANCE.D_COMP_GLOB = 8; @@ -982,13 +914,13 @@ analyses: psf_model_complexity: label: PSFEx pixel basis with degree-2 spatial variation per CCD rationale: >- - BASIS_TYPE PIXEL, BASIS_NUMBER 20, PSF_SAMPLING 1, PSFVAR_DEGREES 2 - in XWIN/YWIN per CCD. PSF_ACCURACY 0.01 is the fractional accuracy - PSFEx assumes for PSF pixel values, which per the PSFEx documentation - enters the fit weights and so how closely the model follows bright - stars. Model flexibility sets the balance between PSF leakage and - overfitting, the dominant additive systematic in cosmic shear. Stock - values; no rationale recorded. + A pixel basis with polynomial variation in XWIN/YWIN, fit per CCD + at native sampling. PSF_ACCURACY is the fractional accuracy PSFEx + assumes for PSF pixel values; per the PSFEx documentation it enters + the fit weights, and so how closely the model follows bright stars. + Model flexibility trades overfitting against PSF leakage, the + dominant additive systematic in cosmic shear. + Stock values; no rationale recorded. Anchor: workflow/config/cfis/default.psfex#BASIS_TYPE = PIXEL; workflow/config/cfis/default.psfex#PSF_ACCURACY = 0.01; workflow/config/cfis/default.psfex#BASIS_NUMBER = 20; @@ -1007,16 +939,14 @@ analyses: psf_acceptance_thresholds: label: Per-CCD PSF-model quality gate rationale: >- - A CCD whose model has ACCEPTED < STAR_THRESH = 22 or CHI2 - > CHI2_THRESH = 2 is not interpolated, on both the - validation and the multi-epoch pass; 22 applies the - published floor to the 80% training sample. In the science - path the CCD's epoch is dropped for every object on it; an - object left with no epoch has no shape. There is no - minimum-epoch floor in the pipeline: NGMIX_N_EPOCH records - what survived and epoch-count cuts happen downstream. - Guinot+22 describes discarding the CCD from PSF estimation - rather than gating at interpolation. + A CCD whose model has ACCEPTED < STAR_THRESH or CHI2 > CHI2_THRESH + is not interpolated, on both the validation and the multi-epoch + pass; STAR_THRESH applies the published 22-star floor to the 80% + training sample. In the science path the CCD's epoch is dropped for + every object on it, and an object left with no epoch has no shape. + The pipeline has no minimum-epoch floor: NGMIX_N_EPOCH records what + survived and epoch-count cuts happen downstream. Guinot+22 discards + the CCD from PSF estimation rather than gating at interpolation. Anchor: src/shapepipe/modules/psfex_interp_package/psfex_interp.py::PSFExInterpolator.interpsfex; workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH = 22; workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.CHI2_THRESH = 2; @@ -1086,9 +1016,9 @@ analyses: location: { page: 7 } guinot22_psfex_preselection_off: claim: >- - PSFEx's internal pre-selection is deliberately disabled so that the - pipeline's own star selection is the only one, the paper describing - the model as fit on the entire star sample. + PSFEx's internal pre-selection is disabled because the pipeline + does its own star selection; the PSF is fit on the entire star + sample. created_at: "2022-04-01T00:00:00Z" evidence: - id: ev_guinot22_preselection_off @@ -1134,9 +1064,8 @@ analyses: # ═════════════════════════════════════════════════════════════════════════ shape_measurement: description: >- - Galaxy shape estimation: joint multi-epoch ngmix Gaussian fits with - metacalibration. Most choices here are fixed in code; this is where the - silent-default risk concentrates. + Joint multi-epoch ngmix Gaussian fits with metacalibration. Most + choices are fixed in code, so silent defaults concentrate here. inputs: - id: vignets type: data @@ -1159,10 +1088,10 @@ analyses: rationale: >- Each object's RNG (noise realisations, guesses, priors) is seeded from its position: [HARDCODED] 3-arcsec sky boxes offset by the - first epoch's CCD number, folded and Cantor-paired mod 2^32. Every - stream is a function of sky position, so results do not depend on - how a tile is chunked, and metacal's fixnoise counter-noise cancels - across image-simulation branches that share an object's box. + first epoch's CCD number, folded and Cantor-paired mod 2^32. Results + therefore do not depend on how a tile is chunked, and metacal's + fixnoise counter-noise cancels across image-simulation branches that + share an object's box. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::position_seed; src/shapepipe/modules/ngmix_package/ngmix.py::Ngmix.process. default: position_seed @@ -1178,10 +1107,9 @@ analyses: galaxy_model: label: Single-Gaussian galaxy and PSF models rationale: >- - [HARDCODED] ngmix Fitter(model='gauss') for both the galaxy and the - PSF. Under metacalibration, model bias largely cancels in the - response, which is the standard defence of the Gaussian; the code - does not state it. + [HARDCODED] ngmix Fitter(model='gauss') for galaxy and PSF. The + standard defence, that model bias largely cancels in the metacal + response, is not stated in the code. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::make_runners. default: gauss options: @@ -1194,23 +1122,19 @@ analyses: fit_initialisation: label: Fit guesses and retries rationale: >- - [HARDCODED] the galaxy guesser (TPSFFluxAndPriorGuesser) - starts from T = 0.25 with a flux taken from a PSF-flux - fit; the PSF guesser (TFluxGuesser) starts from T = 0.25 - and the catalogue flux. The galaxy runner retries 5 times, - the PSF runner twice. With a non-convex likelihood the - guess and retries decide which objects converge. A failed - ngmix fit is flagged; any exception during an object's fit - drops it with no ngmix row, leaving it to the catalogue - sentinels (catalogue_assembly.failure_sentinels). - Guinot+22 initialised the whole guess vector from HSM - adaptive moments on each sheared image; the code does not. + [HARDCODED] the galaxy guesser (TPSFFluxAndPriorGuesser) starts + from T = 0.25 and a PSF-flux fit; the PSF guesser (TFluxGuesser) + from T = 0.25 and the catalogue flux. The galaxy runner tries 5 + times, the PSF runner twice. With a non-convex likelihood, guess and + retries decide which objects converge. A failed fit is flagged; an + exception during an object's fit drops it with no ngmix row, leaving + it to the catalogue sentinels (catalogue_assembly.failure_sentinels). Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::make_runners; src/shapepipe/modules/ngmix_package/ngmix.py::Ngmix.process. default: prior_guess_t025_ntry5_2 options: prior_guess_t025_ntry5_2: - label: T guess 0.25 + catalogue flux, ntry 5 / 2 + label: T guess 0.25; PSF-flux (galaxy) / catalogue flux (PSF); ntry 5 / 2 hsm_initialisation: label: Guesses from HSM adaptive moments (Guinot+22) insights: [guinot22_hsm_initialisation] @@ -1218,19 +1142,16 @@ analyses: fit_priors: label: ngmix joint prior rationale: >- - [HARDCODED] ellipticity GPriorBA with sigma 0.4; flat T in - [-1, 1e3] and flat F in [-100, 1e9], with negative support - (the bounds decide which noisy fits survive and which - rail); a centroid prior of width one pixel scale, - PIXEL_SCALE 0.186 arcsec (derived from the WCS when the - key is absent). The same joint prior also constrains the - PSF fits: the PSF fitter is built with the galaxy prior. - Prior width drives noise bias; no rationale recorded. - Guinot+22 states a flat F in [-1e4, 1e9] and a flat prior - on the half-light radius r50 rather than on T. [LINT] the epoch stamps are exposure - pixels, and star selection uses 0.187 arcsec/px; the - ngmix_runner comment says pixel scale also sets a noise - window, but get_noise is never called. + [HARDCODED] ellipticity GPriorBA with sigma 0.4; flat T and F + priors with negative support (the bounds decide which noisy fits + survive and which rail); a centroid prior one pixel scale wide + (PIXEL_SCALE, derived from the WCS when the key is absent). The PSF + fitter is built with the same joint prior. Prior width drives noise + bias; no rationale recorded. Guinot+22 used a wider flux prior and a + flat prior on r50 rather than T. [LINT] the epoch stamps are + exposure pixels, and star selection uses 0.187 arcsec/px; the + ngmix_runner comment says pixel scale also sets a noise window, but + get_noise is never called. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::get_prior; src/shapepipe/modules/ngmix_package/ngmix.py::get_prior.T_range = -1,1e3; src/shapepipe/modules/ngmix_package/ngmix.py::get_prior.F_range = -100,1e9; @@ -1250,10 +1171,10 @@ analyses: metacal_scheme: label: Metacalibration protocol rationale: >- - [HARDCODED] types noshear, 1p, 1m, 2p, 2m with step 0.01, fixnoise - with the noise image, and ignore_failed_psf (an epoch whose PSF fit - fails is dropped). The reconvolution kernel is METACAL_PSF, default - fitgauss, not set in the committed config; it moves the response + [HARDCODED] the five metacal types and step, fixnoise with the noise + image, and ignore_failed_psf (an epoch whose PSF fit fails is + dropped). The reconvolution kernel, METACAL_PSF, defaults to + fitgauss and is unset in the committed config; it moves the response directly. No sheared-PSF types run, so the catalogue has no PSF response term. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal; @@ -1275,10 +1196,10 @@ analyses: label: Jacobian origin at the coadd centroid rationale: >- The Jacobian origin, where the centroid prior centres, is the - sub-pixel offset the stamp extractor stored when it cut the stamp + sub-pixel offset the stamp extractor stored when cutting the stamp ("wcs", the default at every level; the committed config sets no - CENTROID_SOURCE). One projection and one rounding serve both - extraction and prior, so they cannot disagree near a rounding tie. + CENTROID_SOURCE). Extraction and prior share one projection and one + rounding, so they cannot disagree near a rounding tie. "hsm" re-centres on adaptive moments measured from the stamp. Anchor: src/shapepipe/modules/ngmix_runner.py::ngmix_runner; src/shapepipe/modules/ngmix_package/ngmix.py::make_ngmix_observation. @@ -1307,11 +1228,10 @@ analyses: psf_epoch_averaging: label: Catalogue PSF quantities averaged over epochs by galaxy weight rationale: >- - [HARDCODED] the PSF shape and size written to the catalogue (the - original image PSF and the metacal reconvolution kernel) are averages - over epochs weighted by the summed galaxy inverse variance of each - epoch; epochs whose PSF fit failed are left out. These columns feed - PSF-leakage estimates downstream. + [HARDCODED] the catalogue PSF shape and size (original image PSF + and metacal reconvolution kernel) are epoch averages weighted by each + epoch's summed galaxy inverse variance; epochs whose PSF fit failed + are left out. These columns feed downstream PSF-leakage estimates. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::_average_psf_fits; src/shapepipe/modules/ngmix_package/ngmix.py::average_original_psf. default: galaxy_weight_sum @@ -1324,10 +1244,9 @@ analyses: With BKG_RMS_VIGNET_PATH set, each pixel's weight is 1/rms^2 from the SExtractor background-RMS map (all or nothing; a missing file raises), and the noise realisations use the same per-pixel RMS; the - fallback is a scalar 1/sigma_mad^2. A scalar sigma mis-reports errors - wherever the RMS varies. Each epoch is background-subtracted with the - SExtractor background vignet (BKG_SUB, on unless the key is set - False). + fallback is a scalar 1/sigma_mad^2. Each epoch is + background-subtracted with the SExtractor background vignet + (BKG_SUB, on unless set False). Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_ngmix_weights; src/shapepipe/modules/ngmix_package/ngmix.py::background_subtract; workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.BKG_RMS_VIGNET_PATH = $NGMIX_VIGNET_DIR/vignetmaker_runner_run_2/output/background_rms_vignet{file_number_string}.sqlite. @@ -1342,10 +1261,10 @@ analyses: psf_likelihood_noise: label: Flat PSF-observation weight rationale: >- - [HARDCODED] the PSF observation carries a flat weight 1/PSF_NOISE^2 - with PSF_NOISE 1e-5; without it the g-prior swamps the PSF - likelihood. The recovered PSF shape and size are flat across 1e-4 to - 1e-6 on the digital twin. + [HARDCODED] the PSF observation carries a flat weight 1/PSF_NOISE^2; + without it the g-prior swamps the PSF likelihood. The recovered PSF + shape and size are flat for PSF_NOISE from 1e-4 to 1e-6 on the + digital twin. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::PSF_NOISE = 1e-5; src/shapepipe/modules/ngmix_package/ngmix.py::make_ngmix_observation. default: psf_noise_1em5 @@ -1359,7 +1278,7 @@ analyses: and segmentation stamp are rotated to register with the epoch stamp. A wrong flip mis-registers the tile coverage flag against the epoch, changing flagged pixels and the masked-fraction cut. The docstring - warns it gives incorrect results for THELI CCDs. + warns it is wrong for THELI CCDs. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::Ngmix.MegaCamFlip. default: megapipe_flip options: @@ -1368,34 +1287,28 @@ analyses: defect_fill: label: Image content of defect pixels before metacal rationale: >- - A defect pixel is one with a nonzero instrument flag, zero exposure - weight or invalid background RMS; prepare_ngmix_weights gives it - weight 0. What the IMAGE holds there still matters, because metacal - never looks at weights: ngmix builds a galsim InterpolatedImage from - the whole observation image, deconvolves, shears and reconvolves it, - and copies the weight map through unchanged, so whatever sits in a - zero-weight pixel spreads into the weighted pixels within about a PSF - width. DES's own corrector (ngmixer) fills for that reason: "it may - be important for codes that take moments or use FFTs". On develop - the fill rides on BLEND_HANDLING, which the committed config leaves - at noisefill: defect pixels get independent noise at the per-pixel - RMS; under uberseg the fill is skipped and raw defects enter - metacal. [LINT] the prepare_ngmix_weights docstring says noisefill - keeps the weight of filled pixels (the code zeroes it), and the - ngmix_runner comment says noisefill fills neighbour pixels (it fills - flagged pixels and leaves neighbours untouched). On + A defect pixel (nonzero instrument flag, zero exposure weight or + invalid background RMS) gets weight 0 in prepare_ngmix_weights. Its + image value still matters: ngmix's metacal deconvolves, shears and + reconvolves an InterpolatedImage of the whole image and copies the + weights through, so a zero-weight pixel's content spreads into the + weighted pixels within about a PSF width. DES's ngmixer fills for + that reason: "it may be important for codes that take moments or use + FFTs". On develop the fill rides on BLEND_HANDLING, which the + committed config leaves at noisefill: defect pixels get independent + noise at the per-pixel RMS; under uberseg the fill is skipped and raw + defects enter metacal. [LINT] the prepare_ngmix_weights docstring + says noisefill keeps the weight of filled pixels (the code zeroes + it), and the ngmix_runner comment says noisefill fills neighbour + pixels (it fills flagged pixels and leaves neighbours untouched). On feat/defect-fill-veto every defect pixel is zero-weighted and filled - before metacal under every BLEND_HANDLING; under uberseg, - neighbour-side pixels only lose weight. The fill uses the - unsymmetrized defect set. DES symmetrized its defect masks, but - measured on feat/defect-fill-veto, four-fold symmetrization - quadruples the multiplicative bias (m = -2.7% against -0.64% for a - 3-px bleed 10 px from a galaxy of half-light radius 0.5 arcsec - through a 0.7 arcsec PSF) and still leaves c1 = 7.3e-4 through an - elliptical PSF. The residual cost of an unsymmetrized fill is a hole - in the galaxy light; the central veto bounds it - (central_defect_veto). Noise stays the default until a survey A/B - against interpolation. + before metacal under every BLEND_HANDLING; uberseg only removes + weight from neighbour-side pixels. The fill uses the unsymmetrized + defect set: DES symmetrized its masks, but four-fold symmetrization + quadruples m and still leaves an additive c1 + (symmetrized_4fold_noise). The cost of not symmetrizing, a hole in + the galaxy light, is bounded by central_defect_veto. Noise stays the + default until a survey A/B against interpolation. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_ngmix_weights; src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal; src/shapepipe/modules/ngmix_runner.py::ngmix_runner. @@ -1404,28 +1317,25 @@ analyses: noise: label: Independent noise on the unsymmetrized defect set description: >- - Defect pixels are replaced by an independent noise realisation at - the per-pixel background RMS and keep weight 0, consistent with - metacal's fixnoise noise image, which covers every pixel. On - develop this happens only under BLEND_HANDLING = noisefill; on - the fill branches it is DEFECT_FILL = noise under every - BLEND_HANDLING. + Defect pixels get independent noise at the per-pixel background + RMS and keep weight 0, consistent with metacal's fixnoise noise + image, which covers every pixel. On the fill branches this is + DEFECT_FILL = noise. insights: [mask_metacal_acts_on_whole_stamp, mask_bad_column_symmetrize] interpolate: label: Interpolate short bounded runs; noise-fill the rest description: >- - Implemented on feat/defect-interpolation (a stacked draft), not on - develop; DEFECT_FILL = interpolate. Only short bounded runs (at - most 3 px across, clean on both sides) are interpolated; - everything else is noise-filled. The interpolation is - Clough-Tocher from clean pixels within 4 px, averaged over the - four quarter turns, and shared with the fixnoise image. Weights - are also zeroed on the quarter-turn orbit of each interpolated - pixel while those pixels keep their light, which cancels the - weight term: c1 goes from -1.3e-3 to 2e-6 for a column 8 px from - a 0.5 arcsec galaxy. The fill itself is never symmetrized: - symmetrizing it gives m = +0.89% against +0.19% for a 3-px bleed - at 6 px. Measured on feat/defect-interpolation. + On feat/defect-interpolation (a stacked draft), not develop; + DEFECT_FILL = interpolate. Only short bounded runs (at most 3 px + across, clean on both sides) are interpolated, Clough-Tocher from + clean pixels within 4 px, averaged over the four quarter turns + and shared with the fixnoise image; everything else is + noise-filled. Weights are also zeroed on the quarter-turn orbit + of each interpolated pixel while those pixels keep their light, + which cancels the weight term: c1 goes from -1.3e-3 to 2e-6 for a + column 8 px from a 0.5 arcsec galaxy. The fill itself is never + symmetrized: that gives m = +0.89% against +0.19% for a 3-px + bleed at 6 px. Measured on feat/defect-interpolation. insights: [mask_interpolate_with_noise, mask_sharp_edges_ring] symmetrized_4fold_noise: label: Four-fold-symmetrized defect set (M | rot90 | rot180 | rot270), then noise fill @@ -1451,34 +1361,30 @@ analyses: blend_handling: label: Neighbour treatment before metacal rationale: >- - This decision covers how pixels shared with a neighbour are treated, - and only that; defect fill is the separate decision above. The two - are coupled through BLEND_HANDLING: noisefill (the default; the - committed config sets no key) leaves neighbours fully weighted and - untouched, while uberseg zeroes the weight of pixels nearer a - neighbour's coadd segmentation footprint than the target's - (DILATE_NEIGHBOUR, default 1) and leaves the image untouched. - Official uberseg is weight-only: esheldon/meds get_uberseg returns a - weight map and never modifies the image (a nearest-segment-pixel - Voronoi split). DES Y1's fiducial metacal ran on uberseg-weighted - stamps with the raw neighbour light still in the image. The last Y3 - config does the same, though an earlier Y3 config subtracted MOF - neighbours first. DES Y6 does not mask neighbours before the shear - step and uses uberseg only as the weight of the fit after metacal. - None of the DES or Rubin metacal/metadetect pipelines noise-fills the - neighbour side. Leaving neighbour light raw is consistent with - metacal: it is real sky, and the artificial shear shears it along - with the target, as the real shear does. The reconvolution spreads it - slightly further across the Voronoi boundary than the PSF already - had. The known residual is about +2% m for uberseg-only against MOF - subtraction in DES Y1 simulations. Separately, Sheldon et al. 2020 - find that the blending bias of per-stamp metacal is dominated by - shear-dependent detection, which no pixel treatment fixes and which - DES Y3 calibrated with simulations. Noise-filling the neighbour side - would instead cut the target's own light along an unsheared - boundary, a sharp edge that rings in the FFTs. The recommended - comparison arm is uberseg (weight-only), with defect_fill held equal - across arms. + Covers only pixels shared with a neighbour; defect_fill is coupled to + it through BLEND_HANDLING. noisefill (the default; the committed + config sets no key) leaves neighbours fully weighted and untouched. + uberseg zeroes the weight of pixels nearer a neighbour's coadd + segmentation footprint than the target's (DILATE_NEIGHBOUR, default 1) + and leaves the image untouched, as official uberseg does: + esheldon/meds get_uberseg returns a weight map (a + nearest-segment-pixel Voronoi split). DES Y1's fiducial metacal and + the last Y3 config ran on uberseg-weighted stamps with raw neighbour + light (an earlier Y3 config subtracted MOF neighbours first); Y6 masks + no neighbours before the shear step, using uberseg only as the fit + weight after metacal. No DES or Rubin metacal/metadetect pipeline + noise-fills the neighbour side. Raw neighbour light is consistent with + metacal: it is real sky, and the artificial shear shears it with the + target, as the real shear does; the reconvolution spreads it slightly + further across the Voronoi boundary than the PSF already had. Known + residuals: about +2% m for uberseg-only relative to MOF subtraction in DES Y1 + simulations, and a blending bias dominated by shear-dependent + detection, which no pixel treatment fixes and which DES Y3 calibrated + with simulations (mask_des_y1_uberseg_only, + mask_blend_bias_detection). Noise-filling the neighbour side would + instead cut the target's own light along an unsheared boundary, a + sharp edge that rings in the FFTs. The recommended comparison arm is + uberseg (weight-only), with defect_fill held equal across arms. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::uberseg_weight; src/shapepipe/modules/ngmix_package/ngmix.py::prepare_ngmix_weights; src/shapepipe/modules/ngmix_runner.py::ngmix_runner. @@ -1487,64 +1393,56 @@ analyses: none: label: No neighbour treatment (BLEND_HANDLING = noisefill) description: >- - Neighbour pixels keep their full weight and image values. The fit - sees all neighbour light in the stamp, which is the configuration - Jarvis et al. 2016 found biased toward neighbours (worse than - their segmentation-only mask). It is a candidate cause of the - FLAGS=2 B-modes investigated in #814. + Neighbour pixels keep their full weight and image values, so the + fit sees all neighbour light, which biases shapes toward + neighbours (Jarvis et al. 2016); this masks less than even their + plain segmentation map. A candidate cause of the FLAGS=2 B-modes + investigated in #814. insights: [mask_uberseg_neighbour_bias] uberseg: label: UberSeg, weight-only description: >- - Zero the weight of pixels nearer a neighbour footprint than the - target, as DES Y1/Y3 did. The image is untouched there, so the - neighbour's light is sheared coherently with the target. Needs - the coadd segmentation stamp (SEG_VIGNET_PATH). DILATE_NEIGHBOUR=1 - absorbs the coadd-vs-epoch overlay offset, because ShapePipe - reuses one coadd seg stamp for every epoch where MEDS reprojects - it. + As in DES Y1/Y3. Needs the coadd segmentation stamp + (SEG_VIGNET_PATH). DILATE_NEIGHBOUR absorbs the coadd-vs-epoch + overlay offset: ShapePipe reuses one coadd seg stamp for every + epoch where MEDS reprojects it. insights: [mask_uberseg_weight_only, mask_uberseg_neighbour_bias, mask_des_y1_uberseg_only, mask_blend_bias_detection] uberseg_fill: label: UberSeg plus noise fill of neighbour-side pixels description: >- - Not implemented. Additionally replace the neighbour-side pixels - with noise. This removes neighbour light from metacal, but cuts - the target's own wings along an unsheared Voronoi boundary: a - sharp edge (FFT ringing) that does not respond to the artificial - shear as sky does. No published pipeline does this. Useful only as - a diagnostic arm; a smooth (apodized) taper would be the less - damaging variant. + Not implemented. Also replace neighbour-side pixels with noise, + removing neighbour light from metacal at the cost of a sharp, + unsheared edge through the target's wings (FFT ringing). No + published pipeline does this; useful only as a diagnostic arm, + where a smooth (apodized) taper would do less damage. insights: [mask_sharp_edges_ring, mask_blend_bias_detection] mof_subtract: label: Subtract neighbour models, then UberSeg (DES Y1 alternative) description: >- Not implemented. Subtract multi-object-fit models of the - neighbours from the stamp before metacal and keep uberseg - weights. This removed the ~2% uberseg-only bias in DES Y1 - simulations, although Sheldon et al. 2020 found similar blend - biases with and without MOF. + neighbours before metacal and keep uberseg weights. This removed + the ~2% uberseg-only bias in DES Y1 simulations, although Sheldon + et al. 2020 found similar blend biases with and without MOF. insights: [mask_des_y1_uberseg_only, mask_blend_bias_detection] central_defect_veto: label: Per-epoch veto on a defect near the stamp centre rationale: >- - The committed code has no veto; the veto is implemented on - feat/defect-fill-veto (7777181b), not on develop. There an epoch is - dropped when a defect pixel lies strictly closer to the stamp centre - than its fill's radius, beside the masked-fraction cut in the epoch - loop. It reads only the defect mask, so it selects on nothing - shear-responsive. The radii are fixed, not scaled by galaxy size, - because a size-scaled veto would select on a shear-responsive - quantity: 10 px for noise-filled pixels - (EPOCH_CENTRAL_DEFECT_RADIUS) and 7 px for interpolated pixels - (EPOCH_INTERPOLATED_DEFECT_RADIUS, feat/defect-interpolation). They - are calibrated on 51-px stamps at known RMS and high S/N, through a - round PSF and a (0.05, 0.02) elliptical PSF: for noise fill on - galaxies of half-light radius 0.3 and 0.5 arcsec through a 0.7 arcsec - PSF, and for interpolation also on 0.7 and 0.9 arcsec galaxies. - Known limits: at 10 px, wide defects through the elliptical PSF sit - at the 1% bound (m11 = -0.98%, and -0.24% at 11 px), and noise fill - needs 14 px for 0.7 and 0.9 arcsec galaxies. Measured on - feat/defect-fill-veto and feat/defect-interpolation. + Not on develop; implemented on feat/defect-fill-veto (7777181b). + There an epoch is dropped when a defect pixel lies strictly closer + to the stamp centre than its fill's radius, beside the + masked-fraction cut in the epoch loop. The veto reads only the + defect mask, so it selects on nothing shear-responsive; for the same + reason the radii are fixed rather than scaled by galaxy size: 10 px + for noise-filled pixels (EPOCH_CENTRAL_DEFECT_RADIUS) and 7 px for + interpolated ones (EPOCH_INTERPOLATED_DEFECT_RADIUS, + feat/defect-interpolation). Calibrated on 51-px stamps at known RMS + and high S/N, through a round and a (0.05, 0.02) elliptical 0.7 + arcsec PSF, on galaxies of half-light radius 0.3 and 0.5 arcsec, and + for interpolation also 0.7 and 0.9 arcsec. Known limits: at 10 px, + wide defects through the elliptical PSF sit at the 1% bound (m11 = + -0.98%; -0.24% at 11 px), and noise fill needs 14 px for 0.7 and 0.9 + arcsec galaxies. Measured on feat/defect-fill-veto and + feat/defect-interpolation. Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_postage_stamps. default: disabled options: @@ -1567,22 +1465,20 @@ analyses: [HARDCODED] on develop, an epoch whose stamp has more than 1/3 of its pixels flagged (any nonzero flag bit, including the tile-coverage bit 2**10 set where the tile vignet is off-image) is dropped from the - multi-epoch fit; an object with no surviving epoch has no shape. The - develop cut counts flag pixels only, not zero-weight or invalid-RMS - pixels. On feat/defect-fill-veto it counts the raw defect set - (flagged, zero-weight and invalid-RMS pixels, not symmetrized), with - the threshold configurable as EPOCH_MASKED_FRACTION_CUT, default - 1/3. Before the cut, an epoch is dropped silently if its galaxy stamp - is all zeros or its background-subtracted noise estimate (sigma_mad) - is not positive. DES was stricter. Y1 rejected any epoch with a - masked or zero-weight pixel, and any whose central 4-pixel region was - masked. Y3 cut at 10% of raw zero-weight pixels. Y6 dropped images - more than 10% missing and cut objects at mfrac < 0.1, which its - simulations show avoids calibration bias. Sheldon & Huff 2017 - recommend dropping problematic epochs when many are available. - UNIONS has fewer epochs than DES, so the cost in effective number - density has to be measured, not assumed. A defect near the centre is - handled separately (central_defect_veto). + multi-epoch fit; an object with no surviving epoch has no shape. + Zero-weight and invalid-RMS pixels are not counted. On + feat/defect-fill-veto the cut counts the raw, unsymmetrized defect + set (flagged, zero-weight and invalid-RMS pixels) against + EPOCH_MASKED_FRACTION_CUT, default 1/3. Before the cut, an epoch is + dropped silently if its galaxy stamp is all zeros or its + background-subtracted sigma_mad is not positive. DES was stricter: + Y1 rejected any epoch with a masked or zero-weight pixel, or with + its central 4-pixel region masked; Y3 cut at 10% raw zero-weight + pixels; Y6 drops images more than 10% missing and keeps objects at + mfrac < 0.1 (mask_multi_epoch_drop). UNIONS has fewer epochs than + DES, so the cost in effective number density has to be measured, + not assumed. A defect near the centre is handled separately + (central_defect_veto). Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_postage_stamps. default: one_third options: @@ -1594,17 +1490,16 @@ analyses: ten_percent: label: 10% (DES Y3 / Y6) description: >- - Not implemented as a default; EPOCH_MASKED_FRACTION_CUT = 0.1 on - feat/defect-fill-veto. Drop an epoch if more than 10% of the stamp - is in the defect set, matching DES Y3 max_zero_weight_frac and Y6 - max_masked_fraction. + Not a default; EPOCH_MASKED_FRACTION_CUT = 0.1 on + feat/defect-fill-veto. Matches DES Y3 max_zero_weight_frac and + Y6 max_masked_fraction. insights: [mask_multi_epoch_drop] any_masked: label: Any masked pixel (DES Y1) description: >- - Not implemented. Drop an epoch if any stamp pixel is masked, so - that no fill is ever needed. DES could afford this with about 10 - epochs per band; UNIONS likely cannot. + Not implemented. Drop an epoch with any masked stamp pixel, so no + fill is ever needed. DES could afford this with about 10 epochs + per band; UNIONS likely cannot. insights: [mask_multi_epoch_drop, mask_des_defect_practice] prior_insights: guinot22_ngmix_priors: @@ -1669,13 +1564,12 @@ analyses: mask_uberseg_weight_only: label: UberSeg is a weight-map operation (Jarvis 2016) claim: >- - UberSeg, introduced for DES SV, zeroes the WEIGHT of pixels that - belong to another object's coadd segmentation footprint or lie - closer to another object than to the target; MEDS weights are also - zeroed wherever a mask flag is set. The image is not modified: in - the forward-model fits it was designed for, a zero-weight pixel - drops out of the likelihood exactly. esheldon/meds get_uberseg - matches, which returns a weight map. + UberSeg (DES SV) zeroes the WEIGHT of pixels in another object's + coadd segmentation footprint or closer to another object than to the + target; MEDS weights are also zeroed wherever a mask flag is set. + The image is not modified: in the forward-model fits it was designed + for, a zero-weight pixel drops out of the likelihood exactly. + esheldon/meds get_uberseg, which returns a weight map, matches. created_at: "2026-09-26T00:00:00Z" derived: true evidence: @@ -1693,10 +1587,10 @@ analyses: label: Neighbour light biases shapes toward neighbours (Jarvis 2016) claim: >- With a plain segmentation-map mask, light from a bright neighbour - just outside its footprint entered the fit and biased the shape - toward the neighbour; UberSeg made that bias undetectable in - end-to-end simulations. ShapePipe's default (no neighbour treatment) - masks less than even the plain segmentation map. + just outside its footprint biased the shape toward the neighbour; + UberSeg made that bias undetectable in end-to-end simulations. + ShapePipe's default (no neighbour treatment) masks less than even + the plain segmentation map. created_at: "2026-09-26T00:00:00Z" derived: false evidence: @@ -1714,11 +1608,11 @@ analyses: label: Metacal transforms every stamp pixel, weights unseen claim: >- Metacal builds an interpolated image of the whole postage stamp, - deconvolves, shears and reconvolves it; its Fourier transforms - cannot accommodate missing data. So the image values of zero-weight - pixels feed the sheared images. The method papers leave masking - unaddressed. ngmix implements exactly this: an InterpolatedImage of - obs.image, with the weights copied through. + then deconvolves, shears and reconvolves it; its Fourier transforms + cannot accommodate missing data, so zero-weight pixels' image values + feed the sheared images. The method papers leave masking + unaddressed. ngmix does exactly this: an InterpolatedImage of + obs.image, weights copied through. created_at: "2026-09-26T00:00:00Z" derived: true evidence: @@ -1742,11 +1636,11 @@ analyses: claim: >- Sheldon & Huff 2017 found that filling bad columns (with the best-fit model, not even noise) gave a large additive e1 bias and a - few-per-mille multiplicative bias. Both vanished when a compensating - column rotated by 90 degrees about the stamp centre was added. - Sheldon et al. 2020 recommend the same compensating mask, and DES Y6 - OR-s each bad-pixel mask with its 90-degree rotation, as previous - DES pipelines did, to cancel additive biases. + few-per-mille multiplicative bias, both removed by adding a + compensating column rotated 90 degrees about the stamp centre. + Sheldon et al. 2020 recommend the same compensating mask; DES Y6, + like previous DES pipelines, OR-s each bad-pixel mask with its + 90-degree rotation to cancel additive biases. created_at: "2026-09-26T00:00:00Z" derived: true evidence: @@ -1781,10 +1675,10 @@ analyses: Every DES metacal generation removed defects from the image before metacal: Y1 dropped any epoch with a masked pixel, Y3 filled symmetrized defects with the best-fit central model (ngmix-y1-config - / ngmix-y3-config, see the defect_fill rationale), and Y6 + / ngmix-y3-config), and Y6 interpolated symmetrized defects following "previous DES shear measurement pipelines". None left raw defect values in a zero-weight - pixel, which is what ShapePipe's uberseg path does today. + pixel, as ShapePipe's uberseg path does. created_at: "2026-09-26T00:00:00Z" derived: true evidence: @@ -1831,11 +1725,10 @@ analyses: mask_sharp_edges_ring: label: Sharp mask edges ring through metacal; apodize large masks claim: >- - Masks with sharp edges cause ringing in the metacal FFTs, so - bright-star regions are set to zero with an apodized (smoothly - tapered) edge, and the noise image gets the same masking. The same - reasoning argues against a hard noise fill cut along a neighbour's - Voronoi boundary. + Sharp mask edges ring in the metacal FFTs, so bright-star regions + are zeroed with an apodized (smoothly tapered) edge, and the noise + image gets the same masking. The same reasoning argues against a + hard noise fill cut along a neighbour's Voronoi boundary. created_at: "2026-09-26T00:00:00Z" derived: true evidence: @@ -1860,8 +1753,8 @@ analyses: With many dithered epochs, problematic data can simply be dropped (Sheldon & Huff 2017). DES Y6 drops input images more than 10% missing and cuts objects at masked fraction mfrac < 0.1, which its - image simulations show is enough to avoid shear calibration bias. - ShapePipe's cut is 1/3. + image simulations show avoids shear calibration bias. ShapePipe's + cut is 1/3. created_at: "2026-09-26T00:00:00Z" derived: true evidence: @@ -1885,9 +1778,9 @@ analyses: claim: >- DES Y1 metacal's fiducial catalogue handled neighbours with uberseg only (raw neighbour light in the image). In dense deblending - simulations it carried m of about +2% (2.18 +/- 0.16 per cent at S/N - > 10), removed by subtracting MOF models of the neighbours. In data - the relative uberseg-vs-MOF m was 0.023 +/- 0.009. + simulations it carried m = 2.18 +/- 0.16 per cent at S/N > 10, + removed by subtracting MOF models of the neighbours; in data the + relative uberseg-vs-MOF m was 0.023 +/- 0.009. created_at: "2026-09-26T00:00:00Z" derived: false evidence: @@ -1904,13 +1797,12 @@ analyses: mask_blend_bias_detection: label: Stamp-metacal blend bias is mostly shear-dependent detection claim: >- - Sheldon et al. 2020 find that the few-percent blending bias of - per-stamp metacal comes from shear-dependent detection, not from - blended light itself: it is similar with and without neighbour - subtraction, and in metacal the space between objects is sheared - coherently. DES Y3 kept Y1's approach (Gatti et al. list no change - to neighbour or mask handling) and calibrated the resulting 2-3% - with image simulations. + Sheldon et al. 2020 find that per-stamp metacal's few-percent + blending bias comes from shear-dependent detection, not blended + light: it is similar with and without neighbour subtraction, and + metacal shears the space between objects coherently. DES Y3 kept Y1's + approach (Gatti et al. list no change to neighbour or mask handling) + and calibrated the resulting 2-3% with image simulations. created_at: "2026-09-26T00:00:00Z" derived: true evidence: @@ -1944,11 +1836,11 @@ analyses: claim: >- Masks have a preferred direction (columns, bleeds, spikes). On a camera with a fixed sky orientation (DECam, and CFHT/MegaCam on its - equatorial mount), mask-induced shape errors therefore add up - coherently; with Rubin's camera rotation, unmasked trails averaged - away. DES Y3 has an unexplained mean e1 of 3.5e-4 that its - simulations, which include the real bad-pixel masks, do not - reproduce; whether masks cause it is not established. + equatorial mount), mask-induced shape errors add up coherently; + with Rubin's camera rotation, unmasked trails averaged away. DES Y3 has + an unexplained mean e1 of 3.5e-4 that its simulations, which include + the real bad-pixel masks, do not reproduce; whether masks cause it is + not established. created_at: "2026-09-26T00:00:00Z" derived: true evidence: @@ -1990,18 +1882,14 @@ analyses: star_galaxy_classification: label: Star/galaxy separation deferred out of the pipeline rationale: >- - SM_DO_CLASSIFICATION=False and no spread-model input is - wired, so the catalogue ships every object and separation - happens downstream. The dormant make_cat classifier uses - class = sm + 2 sm_err with stars at |class| < - SM_STAR_THRESH and galaxies at class > SM_GAL_THRESH, read - from config when classification is on; the function - defaults are 0.003 and 0.01, and the committed config sets - neither key, so enabling classification alone raises. - Guinot+22 selected galaxies in the pipeline at s + 2 - sigma_s > 0.0003 together with s > 0 and 20 < MAG_AUTO < - 26; the dormant code implements only the spread-model - test. + Classification is off and no spread-model input is wired, so the + catalogue ships every object and separation happens downstream. The + dormant make_cat classifier computes class = sm + 2 sm_err, with + stars at |class| < SM_STAR_THRESH and galaxies at class > + SM_GAL_THRESH, both read from config (the function defaults, 0.003 + and 0.01, are bypassed); the committed config sets neither key, so + enabling classification alone raises. Unlike Guinot+22 + (guinot22_spread_model_cut), it applies no s > 0 or magnitude cut. Anchor: workflow/config/cfis/config_tile_Mc.ini#MAKE_CAT_RUNNER.SM_DO_CLASSIFICATION = False; src/shapepipe/modules/make_cat_runner.py::make_cat_runner; src/shapepipe/modules/make_cat_package/make_cat.py::save_sm_data; @@ -2014,16 +1902,16 @@ analyses: label: spread_model classification in make_cat insights: [guinot22_spread_model_cut] description: >- - Not wired in the workflow; needs spread_model_runner after the - PSF interpolation, its output added to make_cat's inputs, and - SM_STAR_THRESH / SM_GAL_THRESH set. + Not wired; needs spread_model_runner after the PSF + interpolation, its output added to make_cat's inputs, and both + thresholds set. tile_overlap_handling: label: Tile-overlap duplicates neither removed nor flagged rationale: >- - Adjacent tiles overlap, and objects in the overlap are - measured in both. [HARDCODED] make_cat attaches only - TILE_ID; there is no unique-object rule and no overlap - flag, so duplicates are left to downstream selection. + Adjacent tiles overlap, and objects in the overlap are measured in + both. [HARDCODED] make_cat attaches only TILE_ID; with no + unique-object rule or overlap flag, duplicates are left to + downstream selection. Anchor: src/shapepipe/modules/make_cat_package/make_cat.py::save_sextractor_data; src/shapepipe/modules/make_cat_package/__init__.py. default: no_dedup_in_pipeline @@ -2042,15 +1930,13 @@ analyses: failure_sentinels: label: Objects without shape measurements kept, with sentinel values rationale: >- - [HARDCODED] detections with no ngmix row (no surviving - epoch, or an exception during the fit) stay in the - catalogue with sentinels: sizes, fluxes, magnitudes and - flags 0, flux and magnitude errors -1, ellipticities and - their errors -10, size errors 1e30, and NGMIX_N_EPOCH 0. - The sentinels define what a downstream cut must exclude: a - failed object's NGMIX_MCAL_FLAGS reads 0, the success - value, so a cut on flags alone keeps it; a cut on - NGMIX_N_EPOCH > 0 removes it. + [HARDCODED] detections with no ngmix row (no surviving epoch, or an + exception during the fit) stay in the catalogue with sentinels: + sizes, fluxes, magnitudes and flags 0, flux and magnitude errors -1, + ellipticities and their errors -10, size errors 1e30, + NGMIX_N_EPOCH 0. A failed object's NGMIX_MCAL_FLAGS reads 0, the + success value, so a cut on flags alone keeps it; NGMIX_N_EPOCH > 0 + removes it. Anchor: src/shapepipe/modules/make_cat_package/make_cat.py::SaveCatalogue._save_ngmix_data. default: sentinel_values options: diff --git a/tests/helpers/astra_record.py b/tests/helpers/astra_record.py index e864c7b02..deb67d3b9 100644 --- a/tests/helpers/astra_record.py +++ b/tests/helpers/astra_record.py @@ -1,4 +1,33 @@ -"""Reusable parsing and resolution helpers for ShapePipe's ASTRA record.""" +"""Parse and resolve the anchors in ShapePipe's ASTRA record (astra.yaml). + +The record's header states the anchor grammar. This module is the reference +for the value grammar behind ``ref = value``: + +* Values are numbers, boolean words, strings (quote expressions), or flat + comma lists, optionally bracketed. Semicolons separate refs. Decimal + equality is exact (1 = 1.0, 5e-4 = 0.0005), without rounding. +* Boolean words follow the file's reader, case-insensitively: INI as + ConfigParser.getboolean (yes/true/on/1, no/false/off/0; Y/N are text), + .sex/.psfex also Y/N, Python True/False; elsewhere words are text and + 1/0 are numbers. Outer whitespace is trimmed; other strings are + case-sensitive. Lists preserve order and length. Only .param VIGNET and + .psfex PSF_SIZE accept square-size shorthand: 51 = 51,51. +* ``= absent`` asserts a config key has no active line (its file and any + section must exist); quote it ("absent") to mean the text. Commented-out + keys resolve as locations but never assert values. +* .setools refs use ``SECTION.KEY``. Predicates keep their operators as + quoted text; repeated cuts on one key are an ordered list, e.g. + ``MAG_AUTO = ["> 18.", "< 22."]``. Expressions compare as text. +* Python selectors may append ``[key.subkey]`` to a named assignment + (identifier-like string dict keys); only the selected literal is read, + including ``dict(key=value)`` syntax. Ambiguous bindings fail, including + ``NAME[...] =`` or ``NAME.attr =`` in the same scope; ``.update()`` calls + and mutation from other scopes are not seen. No imports, calls, + arithmetic, argument defaults, environment expansion or implicit tool + defaults are evaluated: the check stays static rather than becoming a + second pipeline runtime. +* Option ids are stable references and are never parsed for values. +""" import argparse import ast From 843fdb483b606f4aceed2fa0f0a12d65e717399a Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 03:51:12 +0200 Subject: [PATCH 23/40] test: link science guardrails to ASTRA decisions --- CLAUDE.md | 5 +- pyproject.toml | 1 + tests/README.md | 3 + tests/helpers/contracts.py | 108 ++++++++++++++++++++++-- tests/science/test_additive_null.py | 3 + tests/science/test_mbias.py | 2 + tests/science/test_resolution_ladder.py | 3 + tests/science/test_star_response.py | 2 + tests/science/test_symmetry.py | 3 + tests/unit/test_contracts.py | 71 ++++++++++++++-- 10 files changed, 184 insertions(+), 17 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index b5140cac6..2ed249e6f 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -126,7 +126,10 @@ where the change lives — and *scientific* decisions in `astra.yaml`, below. consequential scientific choice embedded in the code and the committed configs, with its rationale, its alternatives, and an anchor to the code or config that implements it. `universes/committed.yaml` pins the option the committed -configuration selects for every decision. The format is ASTRA; +configuration selects for every decision. The format is ASTRA. +The `decision` pytest marker links each `tests/science/` guardrail to the +ASTRA decisions it protects, and `tests/unit/test_contracts.py` validates +those IDs alongside `@sc` references. `uvx astra-tools@0.2.17 guide` is the briefing and `uvx astra-tools@0.2.17 spec` the field reference. The file's header states its conventions (anchor grammar, `[HARDCODED]`, `[LINT]`). diff --git a/pyproject.toml b/pyproject.toml index a87d17b9d..50ae8dac8 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -148,4 +148,5 @@ markers = [ "slow: heavy compute (minutes); excluded from the fast inner loop.", "candide: needs the candide cluster and/or its real data; auto-skipped elsewhere.", "unions: uses UNIONS-specific survey layout or data.", + "decision(*ids): the astra.yaml decisions this test protects.", ] diff --git a/tests/README.md b/tests/README.md index 0b0df1524..50439f247 100644 --- a/tests/README.md +++ b/tests/README.md @@ -27,9 +27,12 @@ policy — it applies everywhere. ``` slow heavy compute (minutes); excluded from the fast inner loop candide needs the candide cluster and/or its real data; auto-skipped elsewhere +decision(*ids) ASTRA decisions protected by a test; ids checked against astra.yaml ``` `--strict-markers` is on, so a typo'd marker is an error, not a silent no-op. +Science guardrails use `decision(*ids)` at module or test level; AST parsing +checks every ID against `astra.yaml` without importing the tests. Use `pytest -m "not unions"` to run the survey-generic tests. A `candide`-marked test is **collected everywhere** (so `--collect-only` shows diff --git a/tests/helpers/contracts.py b/tests/helpers/contracts.py index 2f995f474..9efb80b4a 100644 --- a/tests/helpers/contracts.py +++ b/tests/helpers/contracts.py @@ -27,10 +27,10 @@ whether a key is enabled or its value satisfies the contract's prose. """ -from dataclasses import dataclass, field import ast import fnmatch import io +from dataclasses import dataclass, field from pathlib import Path import re import tokenize @@ -65,6 +65,15 @@ class Contract: scope: str = "" +@dataclass(frozen=True) +class DecisionMarker: + """One pytest marker linking a test to an ASTRA decision.""" + + decision: str + path: str + line: int + + def parse_block(text, path, offset=0, scope=""): """Parse every contract in ``text``; return ``(contracts, errors)``.""" @@ -255,6 +264,89 @@ def decision_errors(contracts, record): ] +def _is_decision_marker_call(node): + """Whether ``node`` calls ``pytest.mark.decision(...)``.""" + + function = node.func + return ( + isinstance(function, ast.Attribute) + and function.attr == "decision" + and isinstance(function.value, ast.Attribute) + and function.value.attr == "mark" + and isinstance(function.value.value, ast.Name) + and function.value.value.id == "pytest" + ) + + +def decision_markers(root): + """Parse literal ``pytest.mark.decision`` ids from Python files in tests/. + + The AST scan includes decorators and module-level ``pytestmark`` values + (including lists) without importing test modules. Non-literal ids are + errors so dynamic expressions cannot evade record validation. + """ + + root = Path(root) + tests_root = root / "tests" + markers, errors = [], [] + if not tests_root.is_dir(): + return markers, errors + for path in sorted(tests_root.rglob("*.py")): + if any(part in SKIP for part in path.parts): + continue + relative = path.relative_to(root) + try: + source = path.read_text(encoding="utf-8") + tree = ast.parse(source, filename=str(relative)) + except (OSError, SyntaxError) as error: + errors.append( + f"{relative}: cannot parse decision markers: {error}" + ) + continue + calls = sorted( + (node for node in ast.walk(tree) + if isinstance(node, ast.Call) and _is_decision_marker_call(node)), + key=lambda node: (node.lineno, node.col_offset), + ) + for call in calls: + where = f"{relative}:{call.lineno}" + if not call.args: + errors.append( + f"{where}: decision marker needs literal string ids" + ) + if call.keywords: + errors.append( + f"{where}: decision markers accept positional ids only" + ) + for argument in call.args: + if ( + isinstance(argument, ast.Constant) + and isinstance(argument.value, str) + ): + markers.append( + DecisionMarker( + argument.value, relative.as_posix(), call.lineno + ) + ) + else: + errors.append( + f"{where}: decision marker ids must be literal strings" + ) + return markers, errors + + +def decision_marker_errors(markers, record): + """Markers whose decision id is absent from the ASTRA record.""" + + known = decision_ids(record) + return [ + f"{marker.path}:{marker.line}: decision marker cites unknown decision " + f"{marker.decision!r}" + for marker in markers + if marker.decision not in known + ] + + def governed_refs(contract): """Repo-relative anchor refs named by a contract's ``governs:`` meta. @@ -319,20 +411,22 @@ def anchored_refs(record): return symbols, paths -def coverage_report(contracts, record): +def coverage_report(contracts, record, decision_markers=()): """Report-only gaps between the contracts and the record. - Returns ``(uncovered, unanchored)``: decision ids no contract cites, and - ``@sc`` contracts whose declaration or governed refs are not anchored. - A module-docstring contract counts as anchored when an anchor names its - file or a symbol in it. A ``governs:`` contract counts when at least one - locator appears in the record after rebasing to the repo root and + Returns ``(uncovered, unanchored)``: decision ids cited by neither a + contract nor a test marker, and ``@sc`` contracts whose declaration or + governed refs are not anchored. A module-docstring contract counts as + anchored when an anchor names its file or a symbol in it. A ``governs:`` + contract counts when at least one locator appears in the record after + rebasing to the repo root and removing any value assertion from the record's ref; another key in the same file is not a match. These are report-only links, not proof that the prose holds or every coupled key is anchored. """ cited = {c.meta["decision"] for c in contracts if "decision" in c.meta} + cited.update(marker.decision for marker in decision_markers) uncovered = sorted(decision_ids(record) - cited) references = _anchor_locators(record) symbols, paths = anchored_refs(record) diff --git a/tests/science/test_additive_null.py b/tests/science/test_additive_null.py index c8440071b..001a24982 100644 --- a/tests/science/test_additive_null.py +++ b/tests/science/test_additive_null.py @@ -30,9 +30,12 @@ """ import numpy as np +import pytest from tests.helpers.metacal_sim import recover +pytestmark = pytest.mark.decision("shape_measurement.metacal_scheme") + SEEDS = list(range(8)) # deterministic ensemble; additive bias is statistical PSF_E1 = 0.05 # true PSF ellipticity the deconvolution must remove C_TOL = 1e-3 # |c| null (twin's published bound); mean ~5e-6, worst seed ~5e-5 diff --git a/tests/science/test_mbias.py b/tests/science/test_mbias.py index 4efb7c3dd..b9a7981fb 100644 --- a/tests/science/test_mbias.py +++ b/tests/science/test_mbias.py @@ -26,6 +26,8 @@ from tests.helpers.artifacts import emit_mbias_artifacts +pytestmark = pytest.mark.decision("shape_measurement.metacal_scheme") + # The GitHub Pages publish seam (see tests/_artifacts/README.md). _ARTIFACTS_DIR = Path(__file__).resolve().parents[1] / "_artifacts" diff --git a/tests/science/test_resolution_ladder.py b/tests/science/test_resolution_ladder.py index c1e0a6c2b..eb1e68e87 100644 --- a/tests/science/test_resolution_ladder.py +++ b/tests/science/test_resolution_ladder.py @@ -147,6 +147,7 @@ def rungs(): return ladder +@pytest.mark.decision("shape_measurement.metacal_scheme") def test_estimator_has_power(rungs): """Every resolved rung's σ_m is small enough that its assert has teeth. @@ -168,6 +169,7 @@ def test_estimator_has_power(rungs): ) +@pytest.mark.decision("shape_measurement.metacal_scheme") def test_resolved_rungs_unbiased(rungs): """On resolved rungs (ratio >= 0.5), ``|m|`` stays below a few x 1e-3. @@ -189,6 +191,7 @@ def test_resolved_rungs_unbiased(rungs): ) +@pytest.mark.decision("shape_measurement.metacal_scheme") def test_response_positive_all_rungs(rungs): """Every rung has a non-degenerate positive response ``R11 > 0.1``. diff --git a/tests/science/test_star_response.py b/tests/science/test_star_response.py index 23a984b6f..d52d3c251 100644 --- a/tests/science/test_star_response.py +++ b/tests/science/test_star_response.py @@ -33,7 +33,9 @@ """ import numpy as np +import pytest +pytestmark = pytest.mark.decision("shape_measurement.metacal_scheme") PSF_E1 = 0.05 # true PSF ellipticity the deconvolution must remove METACAL_STEP = 0.01 # ngmix MetacalBootstrapper default shear step diff --git a/tests/science/test_symmetry.py b/tests/science/test_symmetry.py index ca019543d..e44e769e3 100644 --- a/tests/science/test_symmetry.py +++ b/tests/science/test_symmetry.py @@ -22,9 +22,12 @@ class of bug a single-component m-bias test cannot see: a g1<->g2 swap, a status; this is a pure tripwire. """ import numpy as np +import pytest from tests.helpers.metacal_sim import recover +pytestmark = pytest.mark.decision("shape_measurement.metacal_scheme") + SEED = 42 INJECTED = 0.02 # per-component injected shear magnitude for the arms diff --git a/tests/unit/test_contracts.py b/tests/unit/test_contracts.py index 21b1d72d3..cb5f3f12c 100644 --- a/tests/unit/test_contracts.py +++ b/tests/unit/test_contracts.py @@ -1,17 +1,20 @@ """Keep the @sc contracts well-formed and tied to the ASTRA decision record.""" +import textwrap from functools import cache from pathlib import Path -import textwrap import pytest from tests.helpers.astra_record import load_yaml from tests.helpers.contracts import ( + DecisionMarker, collect, coverage_report, decision_errors, decision_ids, + decision_marker_errors, + decision_markers, forbid_rules, governed_refs, governs_errors, @@ -30,7 +33,8 @@ def _repository(): record = load_yaml(REPO_ROOT / "astra.yaml") contracts, errors = collect(REPO_ROOT) - return record, contracts, errors + markers, marker_errors = decision_markers(REPO_ROOT) + return record, contracts, markers, errors + marker_errors def _write(root, relative, text): @@ -160,6 +164,47 @@ def f(): assert decision_ids(RECORD) == {"top_choice", "stage.inner_choice"} +def test_parser_reads_decision_markers_from_decorators_and_pytestmark( + tmp_path, +): + """Find IDs in decorators and module-level pytestmark lists.""" + _write(tmp_path, "tests/test_markers.py", ''' + import pytest + + pytestmark = [pytest.mark.decision("top_choice")] + + @pytest.mark.decision("stage.inner_choice", "top_choice") + def test_it(): + pass + ''') + + markers, errors = decision_markers(tmp_path) + + assert errors == [] + assert [(m.decision, m.path, m.line) for m in markers] == [ + ("top_choice", "tests/test_markers.py", 4), + ("stage.inner_choice", "tests/test_markers.py", 6), + ("top_choice", "tests/test_markers.py", 6), + ] + assert decision_marker_errors(markers, RECORD) == [] + + +def test_unknown_decision_marker_is_an_error(tmp_path): + """Reject decision IDs missing from the ASTRA record.""" + _write(tmp_path, "tests/test_bad_marker.py", ''' + import pytest + pytestmark = [pytest.mark.decision("missing_choice")] + ''') + + markers, errors = decision_markers(tmp_path) + + assert errors == [] + assert decision_marker_errors(markers, RECORD) == [ + "tests/test_bad_marker.py:3: decision marker cites unknown decision " + "'missing_choice'" + ] + + def test_governs_resolves_all_refs_relative_to_the_contract_file(tmp_path): """A multi-key coupling must not lose refs or resolve them from cwd.""" @@ -298,17 +343,24 @@ def test_config_coverage_uses_governed_refs_not_the_sidecar_path( contracts, errors = collect(tmp_path) uncovered, unanchored = coverage_report(contracts, record) + covered, _ = coverage_report( + contracts, + record, + [DecisionMarker("uncovered", "tests/science/test_x.py", 1)], + ) assert errors == [] assert uncovered == ["uncovered"] + assert covered == [] assert [c.id for c in unanchored] == ["off-record-key"] def test_repository_contracts_are_valid_and_cite_real_decisions(): - record, contracts, errors = _repository() + record, contracts, markers, errors = _repository() errors = ( errors + decision_errors(contracts, record) + + decision_marker_errors(markers, record) + governs_errors(contracts, REPO_ROOT) ) @@ -317,13 +369,14 @@ def test_repository_contracts_are_valid_and_cite_real_decisions(): def test_contract_coverage_report(): - """Report-only: print record decisions and contracts that lack a partner.""" - - record, contracts, _ = _repository() - uncovered, unanchored = coverage_report(contracts, record) + """Report ASTRA decision gaps and unanchored contracts.""" + record, contracts, markers, _ = _repository() + uncovered, unanchored = coverage_report(contracts, record, markers) - print(f"\n{len(contracts)} contracts; " - f"{len(uncovered)} decisions cited by no contract:") + print( + f"\n{len(contracts)} contracts; {len(markers)} test decision markers; " + f"{len(uncovered)} decisions cited by neither:" + ) for decision in uncovered: print(f" {decision}") print(f"{len(unanchored)} @sc contracts off the record's anchors:") From ba687dcb44c514e90b11e20b7fa177e0904d0ac3 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 04:28:49 +0200 Subject: [PATCH 24/40] feat(decisions): add site-tag checker and Values resolver --- tests/helpers/astra_record.py | 826 ------------------- tests/helpers/contracts.py | 529 ------------ tests/helpers/decisions.py | 1285 ++++++++++++++++++++++++++++++ tests/unit/test_astra_anchors.py | 75 -- tests/unit/test_astra_values.py | 433 ---------- tests/unit/test_contracts.py | 381 +-------- tests/unit/test_decisions.py | 366 +++++++++ 7 files changed, 1655 insertions(+), 2240 deletions(-) delete mode 100644 tests/helpers/astra_record.py delete mode 100644 tests/helpers/contracts.py create mode 100644 tests/helpers/decisions.py delete mode 100644 tests/unit/test_astra_anchors.py delete mode 100644 tests/unit/test_astra_values.py create mode 100644 tests/unit/test_decisions.py diff --git a/tests/helpers/astra_record.py b/tests/helpers/astra_record.py deleted file mode 100644 index deb67d3b9..000000000 --- a/tests/helpers/astra_record.py +++ /dev/null @@ -1,826 +0,0 @@ -"""Parse and resolve the anchors in ShapePipe's ASTRA record (astra.yaml). - -The record's header states the anchor grammar. This module is the reference -for the value grammar behind ``ref = value``: - -* Values are numbers, boolean words, strings (quote expressions), or flat - comma lists, optionally bracketed. Semicolons separate refs. Decimal - equality is exact (1 = 1.0, 5e-4 = 0.0005), without rounding. -* Boolean words follow the file's reader, case-insensitively: INI as - ConfigParser.getboolean (yes/true/on/1, no/false/off/0; Y/N are text), - .sex/.psfex also Y/N, Python True/False; elsewhere words are text and - 1/0 are numbers. Outer whitespace is trimmed; other strings are - case-sensitive. Lists preserve order and length. Only .param VIGNET and - .psfex PSF_SIZE accept square-size shorthand: 51 = 51,51. -* ``= absent`` asserts a config key has no active line (its file and any - section must exist); quote it ("absent") to mean the text. Commented-out - keys resolve as locations but never assert values. -* .setools refs use ``SECTION.KEY``. Predicates keep their operators as - quoted text; repeated cuts on one key are an ordered list, e.g. - ``MAG_AUTO = ["> 18.", "< 22."]``. Expressions compare as text. -* Python selectors may append ``[key.subkey]`` to a named assignment - (identifier-like string dict keys); only the selected literal is read, - including ``dict(key=value)`` syntax. Ambiguous bindings fail, including - ``NAME[...] =`` or ``NAME.attr =`` in the same scope; ``.update()`` calls - and mutation from other scopes are not seen. No imports, calls, - arithmetic, argument defaults, environment expansion or implicit tool - defaults are evaluated: the check stays static rather than becoming a - second pipeline runtime. -* Option ids are stable references and are never parsed for values. -""" - -import argparse -import ast -import configparser -from dataclasses import dataclass -from decimal import Decimal -from functools import lru_cache -import json -from pathlib import Path -import re -import subprocess -import sys - -import yaml - - -@dataclass(frozen=True) -class Anchor: - """An anchor sentence found in a YAML value.""" - - location: str - references: tuple[str, ...] - error: str | None = None - - -def load_yaml(path): - """Load YAML with PyYAML's safe loader.""" - - return yaml.safe_load(Path(path).read_text(encoding="utf-8")) - - -def _walk(value, location=""): - if isinstance(value, dict): - for key, child in value.items(): - path = f"{location}.{key}" if location else str(key) - yield from _walk(child, path) - elif isinstance(value, list): - for index, child in enumerate(value): - yield from _walk(child, f"{location}[{index}]") - else: - yield location, value - - -def _rationales(document): - for location, value in _walk(document): - if location.endswith(".rationale"): - yield location, value - - -def extract_anchors(document): - """Parse all ``Anchor:`` sentences and check every rationale has one.""" - - anchors = [] - for location, value in _walk(document): - if not isinstance(value, str) or "Anchor:" not in value: - continue - tail = value.split("Anchor:", 1)[1].strip() - error = None - refs = () - if value.count("Anchor:") != 1: - count = value.count("Anchor:") - error = f"expected one Anchor: marker, found {count}" - elif not tail.endswith("."): - error = "anchor sentence must end with a period" - else: - refs = tuple(part.strip() for part in tail[:-1].split(";")) - if not refs or any(not ref for ref in refs): - error = "anchor sentence contains an empty ref" - anchors.append(Anchor(location, refs, error)) - - for location, value in _rationales(document): - if ( - not isinstance(value, str) - or value.count("Anchor:") != 1 - or not value.rstrip().endswith(".") - ): - anchors.append( - Anchor( - location, - (), - "rationale must end with exactly one Anchor: sentence", - ) - ) - return anchors - - -def _split_assertion(reference): - """Split the reserved, whitespace-delimited `` = `` (never a cut's ==).""" - - if ";" in reference: - raise ValueError("semicolon is reserved for separating anchor refs") - parts = re.split(r"\s+=\s*", reference.strip(), maxsplit=1) - locator = parts[0] - expected = parts[1].strip() if len(parts) == 2 else None - if not locator or re.search(r"\s|=", locator): - raise ValueError("expected a locator optionally followed by ' = value'") - if expected is not None and (not expected or expected.startswith("=")): - raise ValueError("expected a nonempty value after ' = '") - return locator, expected - - -def _parse_reference(reference): - reference, _ = _split_assertion(reference) - if "::" in reference: - path, symbol = reference.split("::", 1) - return "code", path, symbol - if "#" in reference: - path, key = reference.split("#", 1) - return "config", path, key - return "path", reference, "" - - -_SNAKEMAKE_SUFFIXES = {".smk"} -_SNAKEFILE_NAMES = {"Snakefile"} - - -def _is_snakemake_file(target): - return target.suffix in _SNAKEMAKE_SUFFIXES or target.name in _SNAKEFILE_NAMES - - -def _snakemake_symbol(text, symbol): - rule_pattern = re.compile( - rf"^\s*(?:rule|checkpoint)\s+{re.escape(symbol)}\s*:", re.MULTILINE - ) - if rule_pattern.search(text): - return None - def_pattern = re.compile(rf"^\s*def\s+{re.escape(symbol)}\(", re.MULTILINE) - if def_pattern.search(text): - return None - return f"no rule/checkpoint/def named {symbol!r}" - - -def resolve_anchor(root, reference): - """Return ``None`` if a reference resolves, otherwise a diagnostic.""" - - try: - kind, relative, selector = _parse_reference(reference) - _, expected = _split_assertion(reference) - except ValueError as error: - return str(error) - path = Path(relative) - if path.is_absolute() or ".." in path.parts: - return "path must be relative to the repository root" - target = Path(root) / path - if not target.exists(): - return "path does not exist" - if kind == "path": - return None - if not target.is_file(): - return "code/config refs must name a file" - - try: - text = target.read_text(encoding="utf-8") - except (OSError, UnicodeError) as error: - return f"cannot read file: {error}" - - if kind == "code" and expected == ABSENT: - return "absent assertions need a config key, not a code symbol" - if kind == "code": - if _is_snakemake_file(target): - return _snakemake_symbol(text, selector) - if target.suffix != ".py": - return "code-symbol refs must name a .py file" - try: - tree = _python_tree(text) - symbol, keys = _code_selector(selector) - if not _has_symbol(tree, symbol): - return f"no def/class/assignment target named {symbol!r}" - if keys: - _selected_python_node(tree, selector) - except (SyntaxError, ValueError) as error: - return f"cannot resolve Python selector: {error}" - return None - - suffix = target.suffix.lower() - if expected == ABSENT: - return _absent_scope(text, selector, suffix) - if suffix == ".ini": - return _ini_key(text, selector) - if suffix == ".setools": - return _setools_key(text, selector) - if suffix in {".sex", ".psfex", ".ww", ".param", ".conf"}: - key = selector.rsplit(".", 1)[-1] - pattern = re.compile(rf"^\s*(?:#\s*)?{re.escape(key)}(?=$|\s|=|\()") - if any(pattern.search(line) for line in text.splitlines()): - return None - return f"no line starts with key {key!r} (commented keys are allowed)" - return f"unsupported config-key file type {suffix or '(no extension)'}" - - -ABSENT = "absent" -_LINE_SUFFIXES = {".sex", ".psfex", ".ww", ".param", ".conf", ".setools"} -_SECTIONED = {".ini", ".setools"} - - -def _absent_scope(text, selector, suffix): - """An absent key still needs a real file type and, if sectioned, section. - - Without the section check a renamed section would make every absence - trivially true. - """ - - if suffix not in _LINE_SUFFIXES | {".ini"}: - return f"unsupported config-key file type {suffix or '(no extension)'}" - if suffix not in _SECTIONED: - return None - if "." not in selector: - return "sectioned config ref needs SECTION.KEY" - section = selector.rsplit(".", 1)[0] - if suffix == ".ini": - try: - parser = _ini_parser(text, strict=False) - except configparser.Error as error: - return f"cannot parse INI file: {error}" - if section == parser.default_section or parser.has_section(section): - return None - elif any( - line.strip() == f"[{section}]" for line in text.splitlines() - ): - return None - return f"section {section!r} is missing" - - -def _ini_key(text, selector): - if "." not in selector: - return "INI config ref needs SECTION.KEY" - section, key = selector.rsplit(".", 1) - try: - parser = _ini_parser(text, strict=False) - except configparser.Error as error: - return f"cannot parse INI file: {error}" - if section != parser.default_section and not parser.has_section(section): - return f"INI section {section!r} is missing" - if not parser.has_option(section, key): - return f"INI key {key!r} is missing from section {section!r}" - return None - - -def _setools_key(text, selector): - if "." not in selector: - return "SETools config ref needs SECTION.KEY" - section, key = selector.rsplit(".", 1) - pattern = re.compile(rf"^\s*(?:#\s*)?{re.escape(key)}(?=$|\s|=|<|>)") - active = False - for line in text.splitlines(): - stripped = line.strip() - if stripped.startswith("[") and stripped.endswith("]"): - active = stripped[1:-1].strip() == section - elif active and pattern.search(line): - return None - return f"SETools key {key!r} is missing from section {section!r}" - - -def _target_names(target): - if isinstance(target, ast.Name): - return [target.id] - if isinstance(target, ast.Attribute): - return [target.attr] - if isinstance(target, (ast.Tuple, ast.List)): - return [name for item in target.elts for name in _target_names(item)] - if isinstance(target, ast.Starred): - return _target_names(target.value) - return [] - - -@lru_cache(maxsize=128) -def _bindings(scope): - """Collect all bindings per name; value reads must not pick one silently.""" - - result = {} - - def visit(node): - if isinstance( - node, - (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef), - ): - result.setdefault(node.name, []).append(node) - return - if isinstance(node, ast.Lambda): - return - if isinstance(node, ast.Assign): - targets = node.targets - elif isinstance(node, (ast.AnnAssign, ast.AugAssign, ast.NamedExpr)): - targets = [node.target] - elif isinstance(node, (ast.For, ast.AsyncFor)): - targets = [node.target] - elif isinstance(node, (ast.With, ast.AsyncWith)): - targets = [item.optional_vars for item in node.items] - else: - targets = [] - for target in targets: - if target is not None: - for name in _target_names(target): - result.setdefault(name, []).append(node) - if isinstance(node, ast.ExceptHandler) and node.name: - result.setdefault(node.name, []).append(node) - for child in ast.iter_child_nodes(node): - visit(child) - - for statement in scope.body: - visit(statement) - return result - - -def _mutated_name(target): - """Base name of ``NAME[...] =`` / ``NAME.attr =`` (nested included).""" - - while isinstance(target, (ast.Subscript, ast.Attribute)): - target = target.value - if isinstance(target, ast.Name): - return target.id - return None - - -@lru_cache(maxsize=128) -def _mutations(scope): - """Names whose bound object is item- or attribute-assigned in ``scope``. - - Same lexical scope only, like ``_bindings``; method calls such as - ``.update()`` and mutation from other scopes are not seen. - """ - - names = set() - - def visit(node): - if isinstance( - node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef, ast.Lambda) - ): - return - if isinstance(node, ast.Assign): - targets = node.targets - elif isinstance(node, (ast.AnnAssign, ast.AugAssign)): - targets = [node.target] - elif isinstance(node, ast.Delete): - targets = node.targets - elif isinstance(node, (ast.For, ast.AsyncFor)): - targets = [node.target] - elif isinstance(node, (ast.With, ast.AsyncWith)): - targets = [item.optional_vars for item in node.items] - else: - targets = [] - stack = [target for target in targets if target is not None] - while stack: - target = stack.pop() - if isinstance(target, (ast.Tuple, ast.List)): - stack.extend(target.elts) - elif isinstance(target, ast.Starred): - stack.append(target.value) - else: - name = _mutated_name(target) - if name: - names.add(name) - for child in ast.iter_child_nodes(node): - visit(child) - - for statement in scope.body: - visit(statement) - return frozenset(names) - - -def _has_symbol(tree, symbol): - scope = tree - parts = symbol.split(".") - for index, part in enumerate(parts): - declarations = _bindings(scope).get(part) - if not declarations: - return False - declaration = declarations[-1] - if index == len(parts) - 1: - return True - if not isinstance( - declaration, - (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef), - ): - return False - scope = declaration - return False - - -def _ini_parser(text, *, strict=True): - parser = configparser.ConfigParser( - interpolation=None, strict=strict, allow_no_value=True - ) - parser.optionxform = str - parser.read_string(text) - return parser - - -@lru_cache(maxsize=16) -def _python_tree(text): - # Cache by source, not path: editing a file must invalidate the read. - return ast.parse(text) - - -def _code_selector(selector): - match = re.fullmatch(r"([\w.]+)(?:\[([\w.]+)\])?", selector) - if not match or any(not p.isidentifier() for p in match[1].split(".")): - raise ValueError(f"invalid Python selector {selector!r}") - keys = tuple(match[2].split(".")) if match[2] else () - if any(not key.isidentifier() for key in keys): - raise ValueError("dict paths need dot-separated identifier keys") - return match[1], keys - - -def _dict_entry(node, key): - """Select syntax, not a runtime value; never execute a dict() call.""" - - if isinstance(node, ast.Dict): - if any( - not isinstance(k, ast.Constant) or not isinstance(k.value, str) - for k in node.keys - ): - raise ValueError("dict selectors need literal string keys, no **") - items = [(k.value, v) for k, v in zip(node.keys, node.values)] - elif ( - isinstance(node, ast.Call) and isinstance(node.func, ast.Name) - and node.func.id == "dict" and not node.args - and all(k.arg is not None for k in node.keywords) - ): - items = [(k.arg, k.value) for k in node.keywords] - else: - raise ValueError("dict selectors need {...} or dict(key=value) syntax") - names = [name for name, _ in items] - if len(names) != len(set(names)): - raise ValueError("ambiguous duplicate dict keys") - if key not in names: - raise ValueError(f"dict key {key!r} is missing") - return dict(items)[key] - - -def _selected_python_node(tree, selector): - symbol, keys = _code_selector(selector) - scope = tree - parts = symbol.split(".") - for index, part in enumerate(parts): - declarations = _bindings(scope).get(part, []) - if len(declarations) != 1: - raise ValueError( - f"{symbol!r} needs one binding; found {len(declarations)} " - f"for {part!r}" - ) - node = declarations[0] - if index == len(parts) - 1 and part in _mutations(scope): - raise ValueError( - f"{symbol!r} is item- or attribute-assigned after binding" - ) - if index < len(parts) - 1: - if not isinstance( - node, (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef) - ): - raise ValueError(f"{part!r} is not a lexical scope") - scope = node - if not isinstance(node, (ast.Assign, ast.AnnAssign)): - raise ValueError(f"{symbol!r} is not a literal assignment") - targets = node.targets if isinstance(node, ast.Assign) else [node.target] - if any(not isinstance(target, ast.Name) for target in targets): - raise ValueError("value assertions need simple named assignment targets") - node = node.value - for key in keys: - node = _dict_entry(node, key) - return node - - -def _active_lines(text, selector, suffix): - """Every active (uncommented) setting of the key, as (value, predicate).""" - - section = None - if suffix == ".setools": - if "." not in selector: - raise ValueError("SETools config ref needs SECTION.KEY") - section, key = selector.rsplit(".", 1) - else: - key = selector.rsplit(".", 1)[-1] - pattern = re.compile(rf"^{re.escape(key)}(?=$|\s|=|\(|<|>)(.*)$") - active = section is None - values = [] - predicates = [] - for line in text.splitlines(): - line = line.split("#", 1)[0].strip() - if section is not None and line.startswith("[") and line.endswith("]"): - active = line[1:-1].strip() == section - continue - match = pattern.fullmatch(line) if active else None - if not match: - continue - value = match[1].strip() - predicate = suffix == ".setools" and value.startswith( - ("==", "!=", "<", ">") - ) - if suffix == ".param" and value.startswith("(") and value.endswith(")"): - value = value[1:-1] - elif value.startswith("=") and not predicate: - value = value[1:].strip() - values.append(value) - predicates.append(predicate) - return values, predicates - - -def _line_value(text, selector, suffix): - """Read active lines; SETools repeated predicates form an ordered list.""" - - values, predicates = _active_lines(text, selector, suffix) - if not values or any(not value for value in values): - raise ValueError(f"no active value for {selector!r}") - if len(values) > 1 and not all(predicates): - raise ValueError(f"ambiguous active values for {selector!r}: {values!r}") - return values if len(values) > 1 else values[0] - - -_NUMBER = re.compile(r"[+-]?(?:\d+(?:\.\d*)?|\.\d+)(?:[eE][+-]?\d+)?\Z") -# Boolean words each reader accepts. INI follows ConfigParser.getboolean, -# whose 1/0 spellings are handled at comparison (see _ini_bool); SExtractor -# and PSFEx add Y/N; Python literals are already bool, and the record spells -# them True/False. Elsewhere words stay text. -_INI_BOOLEANS = { - "yes": True, "true": True, "on": True, - "no": False, "false": False, "off": False, -} -_ASTROMATIC_BOOLEANS = {**_INI_BOOLEANS, "y": True, "n": False} -_PYTHON_BOOLEANS = {"true": True, "false": False} -_NO_BOOLEANS = {} - - -def _booleans_for(suffix): - if suffix == ".ini": - return _INI_BOOLEANS - if suffix in {".sex", ".psfex"}: - return _ASTROMATIC_BOOLEANS - if suffix == ".py": - return _PYTHON_BOOLEANS - return _NO_BOOLEANS - - -def _ini_bool(want, got): - """getboolean also reads 1/0; accept them only against a bool expectation.""" - - if want[0] == "bool" and got[0] == "number" and got[1] in (0, 1): - return "bool", got[1] == 1 - return got - - -def _normalise_value(value, booleans=_ASTROMATIC_BOOLEANS): - """Use tagged atoms so boolean True cannot compare equal to number 1.""" - - if isinstance(value, bool): - return "bool", value - if isinstance(value, (int, float)): - number = Decimal(str(value)) - if not number.is_finite(): - raise ValueError("numeric values must be finite") - return "number", number - if isinstance(value, (list, tuple)): - elements = tuple(_normalise_value(item, booleans) for item in value) - if any(kind == "list" for kind, _ in elements): - raise ValueError("only flat lists are supported") - return "list", elements - if not isinstance(value, str): - raise ValueError("expected a number, boolean, string or flat list") - value = value.strip() - if not value: - raise ValueError("empty values/list elements are not supported") - if value[0] in "[{'\"": - # BaseLoader keeps even 5e-4 and Y as strings, avoiding YAML 1.1's - # inconsistent numeric/boolean coercions. It constructs no objects. - parsed = yaml.load(value, Loader=yaml.BaseLoader) - if isinstance(parsed, list): - return _normalise_value(parsed, booleans) - if not isinstance(parsed, str): - raise ValueError("expected a scalar or flat list, not a mapping") - # Quotes protect commas/operators; their contents are a single atom. - value = parsed.strip() - elif "," in value: - return _normalise_value(value.split(","), booleans) - if _NUMBER.fullmatch(value): - return "number", Decimal(value) - if value.lower() in booleans: - return "bool", booleans[value.lower()] - return "text", value - - -def _square_stamp(value): - if value[0] == "number": - return "list", (value, value) - return value - - -def check_anchor_value(root, reference): - """Return a diagnostic for a mismatched/unreadable assertion, else None. - - A reference without `` = value`` is location-only. Numbers compare - exactly after decimal normalization, not with a tolerance. No imported - code, environment expansion, function calls or expressions are evaluated. - """ - - expected = None - actual = "" - try: - _, expected = _split_assertion(reference) - if expected is None: - return None - kind, relative, selector = _parse_reference(reference) - problem = resolve_anchor(root, reference) - if problem: - raise ValueError(problem) - target = Path(root) / relative - text = target.read_text(encoding="utf-8") - suffix = target.suffix.lower() - if expected == ABSENT: - if suffix == ".ini": - section, key = selector.rsplit(".", 1) - parser = _ini_parser(text) - if not parser.has_option(section, key): - return None - actual = parser.get(section, key) - else: - active, _ = _active_lines(text, selector, suffix) - if not active: - return None - actual = active if len(active) > 1 else active[0] - return f"expected no active setting, actual {actual!r}" - if kind == "code" and suffix == ".py": - node = _selected_python_node(_python_tree(text), selector) - try: - actual = ast.literal_eval(node) - except (ValueError, TypeError) as error: - raise ValueError( - "selected Python value is not a literal" - ) from error - elif kind == "config" and suffix == ".ini": - section, key = selector.rsplit(".", 1) - actual = _ini_parser(text).get(section, key) - if actual is None: - raise ValueError("no active value for INI key") - elif kind == "config" and suffix in { - ".sex", ".psfex", ".ww", ".param", ".conf", ".setools" - }: - actual = _line_value(text, selector, suffix) - else: - raise ValueError( - "value assertions need a config key or Python assignment" - ) - booleans = _booleans_for(suffix) - want = _normalise_value(expected, booleans) - got = _normalise_value(actual, booleans) - if suffix == ".ini": - got = _ini_bool(want, got) - if (suffix, selector) in {(".param", "VIGNET"), (".psfex", "PSF_SIZE")}: - want, got = _square_stamp(want), _square_stamp(got) - if want == got: - return None - return f"expected {expected!r}, actual {actual!r}" - except ( - ValueError, OSError, SyntaxError, configparser.Error, yaml.YAMLError - ) as error: - return f"expected {expected!r}, actual {actual!r}: {error}" - - -def value_errors(root, record): - """Check assertions, naming decision, ref, expected and actual in errors.""" - - errors = [] - for anchor in extract_anchors(record): - if anchor.error: - errors.append(f"{anchor.location}: {anchor.error}") - continue - for reference in anchor.references: - problem = check_anchor_value(root, reference) - if problem: - errors.append(f"{anchor.location}: {reference}: {problem}") - return errors - - -def universe_errors(record, universe): - """Check scoped decision IDs and options against the ASTRA record.""" - - record_decisions = _decisions(record) - pinned = _decisions(universe) - errors = [] - for location in sorted(pinned.keys() - record_decisions.keys()): - errors.append( - f"{location}: universe decision is absent from astra.yaml" - ) - for location in sorted(record_decisions.keys() - pinned.keys()): - errors.append( - f"{location}: astra.yaml decision is not pinned in the universe" - ) - for location in sorted(record_decisions.keys() & pinned.keys()): - definition = record_decisions[location] - options = ( - definition.get("options", {}) - if isinstance(definition, dict) - else {} - ) - if not isinstance(options, dict) or pinned[location] not in options: - errors.append( - f"{location}: pinned option {pinned[location]!r} is not in " - "ASTRA options" - ) - return errors - - -def _decisions(document, location=""): - scope = document if isinstance(document, dict) else {} - result = {} - for decision_id, definition in (scope.get("decisions") or {}).items(): - key = ( - f"{location}.decisions.{decision_id}" - if location - else f"decisions.{decision_id}" - ) - result[key] = definition - for analysis_id, analysis in (scope.get("analyses") or {}).items(): - child = ( - f"{location}.analyses.{analysis_id}" - if location - else f"analyses.{analysis_id}" - ) - if isinstance(analysis, dict): - result.update(_decisions(analysis, child)) - return result - - -def _git_sha(root): - try: - return subprocess.run( - ["git", "rev-parse", "HEAD"], - cwd=root, - capture_output=True, - check=True, - text=True, - ).stdout.strip() - except (OSError, subprocess.CalledProcessError): - return None - - -def build_report(root): - """Resolve every anchor and universe pin under ``root`` into a report dict.""" - - root = Path(root) - astra_yaml = root / "astra.yaml" - record = load_yaml(astra_yaml) - anchors = extract_anchors(record) - - unresolved = [] - for anchor in anchors: - if anchor.error: - unresolved.append( - {"location": anchor.location, "ref": None, "problem": anchor.error} - ) - continue - for reference in anchor.references: - problem = resolve_anchor(root, reference) - if problem: - unresolved.append( - { - "location": anchor.location, - "ref": reference, - "problem": problem, - } - ) - - universe = load_yaml(root / "universes" / "committed.yaml") - errors = universe_errors(record, universe) - - return { - "astra_yaml": str(astra_yaml), - "git_sha": _git_sha(root), - "anchors_total": len(anchors), - "unresolved": unresolved, - "universe_errors": errors, - "ok": not unresolved and not errors, - } - - -def main(argv=None): - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument( - "--report", required=True, help="path to write the JSON report to" - ) - parser.add_argument( - "--root", - default=Path(__file__).resolve().parents[2], - help="repository root (default: repo root inferred from this file)", - ) - args = parser.parse_args(argv) - - report = build_report(args.root) - report_path = Path(args.report) - report_path.parent.mkdir(parents=True, exist_ok=True) - report_path.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") - - return 0 if report["ok"] else 1 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/tests/helpers/contracts.py b/tests/helpers/contracts.py deleted file mode 100644 index 9efb80b4a..000000000 --- a/tests/helpers/contracts.py +++ /dev/null @@ -1,529 +0,0 @@ -"""Parse and validate the repository's scientific contracts. - -A contract is a tagged block colocated with the code it governs: a tag line -(``@sc`` or ``@cc``, an optional bracketed ``key:value`` meta list, then a -stable id) followed by prose up to the next blank line. The line grammar is -the loom ``sc-list`` reference parser's, regex for regex, so a contract that -parses here parses there. - -Where contracts live: - -* Python: in the docstring of the module, class or function they govern. - ``sc-list`` reads docstrings only, so a tag line in a ``#`` comment of a - ``.py`` file is reported as an error rather than silently ignored. -* Snakemake (``.smk``, ``Snakefile``): in ``#`` comment blocks. -* ``CONTRACTS`` files: anywhere in the file; they govern their directory. - Prose needs no ``only:`` or ``forbid:`` prefix (those are import rules). - -A ``decision:`` meta names a decision in ``astra.yaml``: a top-level -decision by its bare id, a sub-analysis decision as ``.``. -A ``governs:;`` meta names the keys/files a contract constrains. -Refs use ASTRA anchor locators, but paths are relative to the contract's -own directory, not the repository root; put cross-directory couplings in -an ancestor's ``CONTRACTS``. Semicolons separate refs without spaces, since -commas separate metadata pairs in ``sc-list``. Value assertions stay in -ASTRA, not in whitespace-free ``governs:`` metadata. Repeated metadata keys -are errors, not last-value-wins overrides. Resolution checks existence, not -whether a key is enabled or its value satisfies the contract's prose. -""" - -import ast -import fnmatch -import io -from dataclasses import dataclass, field -from pathlib import Path -import re -import tokenize -import warnings - -from tests.helpers.astra_record import ( - _parse_reference, - _split_assertion, - extract_anchors, - resolve_anchor, -) - -TAG = re.compile(r"^\s*@(sc|cc)\b.*$") -VALID = re.compile(r"^\s*@(sc|cc)(?:\s+\[([^\]]*)\])?\s+([\w][\w.-]*)\s*$") -META_PAIR = re.compile(r"[\w.-]+:[^,\s]+") -MISSING_ID = re.compile(r"\s*@(sc|cc)(?:\s+\[[^\]]*\])?\s*") - -SCAN_ROOTS = ("src", "workflow", "scripts") -SKIP = {".git", ".venv", "venv", "__pycache__", "node_modules", ".felt"} - - -@dataclass -class Contract: - """One parsed contract.""" - - tag: str - id: str - meta: dict = field(default_factory=dict) - prose: str = "" - path: str = "" - line: int = 0 - scope: str = "" - - -@dataclass(frozen=True) -class DecisionMarker: - """One pytest marker linking a test to an ASTRA decision.""" - - decision: str - path: str - line: int - - -def parse_block(text, path, offset=0, scope=""): - """Parse every contract in ``text``; return ``(contracts, errors)``.""" - - lines = text.splitlines() - found, errors = [], [] - for index, line in enumerate(lines): - if not TAG.match(line): - continue - where = f"{path}:{index + 1 + offset}" - match = VALID.match(line) - if not match: - detail = ( - "missing contract id" - if MISSING_ID.fullmatch(line) - else "malformed contract line" - ) - errors.append(f"{where}: {detail}: {line.strip()}") - continue - tag, meta_text, ident = match.groups() - meta = {} - if meta_text is not None: - pairs = [part.strip() for part in meta_text.split(",")] - if any(not META_PAIR.fullmatch(part) for part in pairs): - errors.append(f"{where}: malformed contract metadata: {meta_text}") - continue - meta = dict(part.split(":", 1) for part in pairs) - if len(meta) != len(pairs): - errors.append(f"{where}: duplicate metadata key: {meta_text}") - continue - if "governs" in meta and any( - not ref for ref in meta["governs"].split(";") - ): - errors.append(f"{where}: empty governs ref: {meta['governs']}") - continue - prose = [] - for body in lines[index + 1:]: - if not body.strip(): - break - prose.append(body.strip()) - found.append( - Contract(tag, ident, meta, " ".join(prose), str(path), - index + 1 + offset, scope) - ) - return found, errors - - -def _declarations(tree): - parents = { - child: parent - for parent in ast.walk(tree) - for child in ast.iter_child_nodes(parent) - } - yield tree, "module" - for node in ast.walk(tree): - if not isinstance( - node, (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef) - ): - continue - names, parent = [node.name], parents.get(node) - while parent is not None and not isinstance(parent, ast.Module): - if isinstance( - parent, (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef) - ): - names.append(parent.name) - parent = parents.get(parent) - yield node, ".".join(reversed(names)) - - -def python_contracts(path, source): - """Contracts in a Python file's docstrings; tags in comments are errors.""" - - try: - with warnings.catch_warnings(): - warnings.simplefilter("ignore", SyntaxWarning) - tree = ast.parse(source, filename=str(path)) - except SyntaxError as error: - return [], [f"{path}: cannot parse: {error}"] - found, errors = [], [] - for node, scope in _declarations(tree): - doc = ast.get_docstring(node, clean=False) - if not doc: - continue - first = 1 if node is tree else node.body[0].lineno - records, issues = parse_block(doc, path, first - 1, scope) - found.extend(records) - errors.extend(issues) - try: - tokens = tokenize.generate_tokens(io.StringIO(source).readline) - for token in tokens: - if token.type == tokenize.COMMENT and TAG.match( - token.string.lstrip("#") - ): - errors.append( - f"{path}:{token.start[0]}: contract in a comment; " - "move it into the governing docstring" - ) - except (tokenize.TokenError, SyntaxError): - pass - return found, errors - - -def snakemake_contracts(path, source): - """Contracts in a Snakemake file's ``#`` comment blocks.""" - - stripped = [] - for line in source.splitlines(): - text = line.strip() - stripped.append(text[1:] if text.startswith("#") else "") - return parse_block("\n".join(stripped), path, 0, "file") - - -def _is_snakemake(path): - return path.suffix == ".smk" or path.name == "Snakefile" - - -def contract_files(root, scan_roots=SCAN_ROOTS): - """Yield every file under ``scan_roots`` that can carry contracts.""" - - root = Path(root) - for top in scan_roots: - base = root / top - if not base.is_dir(): - continue - for path in sorted(base.rglob("*")): - if any(part in SKIP for part in path.parts) or not path.is_file(): - continue - if ( - path.name == "CONTRACTS" - or path.suffix == ".py" - or _is_snakemake(path) - ): - yield path - - -def collect(root, scan_roots=SCAN_ROOTS): - """Parse every contract under ``root``; return ``(contracts, errors)``. - - Errors cover malformed tag lines, malformed meta, missing ids and ids - used more than once. Paths are relative to ``root``. - """ - - root = Path(root) - contracts, errors = [], [] - for path in contract_files(root, scan_roots): - relative = path.relative_to(root) - text = path.read_text(encoding="utf-8") - if path.name == "CONTRACTS": - found, issues = parse_block( - text, relative, 0, str(relative.parent) - ) - elif _is_snakemake(path): - found, issues = snakemake_contracts(relative, text) - else: - found, issues = python_contracts(relative, text) - contracts.extend(found) - errors.extend(issues) - seen = {} - for contract in contracts: - seen.setdefault(contract.id, []).append(contract) - for ident, items in seen.items(): - if len(items) > 1: - where = ", ".join(f"{c.path}:{c.line}" for c in items) - errors.append(f"duplicate contract id {ident}: {where}") - return contracts, errors - - -def decision_ids(record, prefix=""): - """Scoped decision ids: bare at top level, dotted inside sub-analyses.""" - - ids = set() - for decision in (record.get("decisions") or {}): - ids.add(f"{prefix}{decision}") - for name, analysis in (record.get("analyses") or {}).items(): - if isinstance(analysis, dict): - ids |= decision_ids(analysis, f"{prefix}{name}.") - return ids - - -def decision_errors(contracts, record): - """Contracts whose ``decision:`` meta names no decision in the record.""" - - known = decision_ids(record) - return [ - f"{c.path}:{c.line}: contract {c.id} cites unknown decision " - f"{c.meta['decision']!r}" - for c in contracts - if "decision" in c.meta and c.meta["decision"] not in known - ] - - -def _is_decision_marker_call(node): - """Whether ``node`` calls ``pytest.mark.decision(...)``.""" - - function = node.func - return ( - isinstance(function, ast.Attribute) - and function.attr == "decision" - and isinstance(function.value, ast.Attribute) - and function.value.attr == "mark" - and isinstance(function.value.value, ast.Name) - and function.value.value.id == "pytest" - ) - - -def decision_markers(root): - """Parse literal ``pytest.mark.decision`` ids from Python files in tests/. - - The AST scan includes decorators and module-level ``pytestmark`` values - (including lists) without importing test modules. Non-literal ids are - errors so dynamic expressions cannot evade record validation. - """ - - root = Path(root) - tests_root = root / "tests" - markers, errors = [], [] - if not tests_root.is_dir(): - return markers, errors - for path in sorted(tests_root.rglob("*.py")): - if any(part in SKIP for part in path.parts): - continue - relative = path.relative_to(root) - try: - source = path.read_text(encoding="utf-8") - tree = ast.parse(source, filename=str(relative)) - except (OSError, SyntaxError) as error: - errors.append( - f"{relative}: cannot parse decision markers: {error}" - ) - continue - calls = sorted( - (node for node in ast.walk(tree) - if isinstance(node, ast.Call) and _is_decision_marker_call(node)), - key=lambda node: (node.lineno, node.col_offset), - ) - for call in calls: - where = f"{relative}:{call.lineno}" - if not call.args: - errors.append( - f"{where}: decision marker needs literal string ids" - ) - if call.keywords: - errors.append( - f"{where}: decision markers accept positional ids only" - ) - for argument in call.args: - if ( - isinstance(argument, ast.Constant) - and isinstance(argument.value, str) - ): - markers.append( - DecisionMarker( - argument.value, relative.as_posix(), call.lineno - ) - ) - else: - errors.append( - f"{where}: decision marker ids must be literal strings" - ) - return markers, errors - - -def decision_marker_errors(markers, record): - """Markers whose decision id is absent from the ASTRA record.""" - - known = decision_ids(record) - return [ - f"{marker.path}:{marker.line}: decision marker cites unknown decision " - f"{marker.decision!r}" - for marker in markers - if marker.decision not in known - ] - - -def governed_refs(contract): - """Repo-relative anchor refs named by a contract's ``governs:`` meta. - - Paths start at the contract's directory; the shared anchor resolver - rejects absolute paths and parent traversal. The parser has already - rejected empty refs and whitespace in the list. - """ - - if "governs" not in contract.meta: - return () - directory = Path(contract.path).parent - return tuple( - (directory / ref).as_posix() - for ref in contract.meta["governs"].split(";") - ) - - -def governs_errors(contracts, root): - """Diagnostics for every ``governs:`` ref the ASTRA resolver rejects.""" - - errors = [] - for contract in contracts: - for reference in governed_refs(contract): - problem = resolve_anchor(root, reference) - if problem: - errors.append( - f"{contract.path}:{contract.line}: contract {contract.id} " - f"governs {reference!r}: {problem}" - ) - return errors - - -def _anchor_locators(record): - """Strip optional value assertions using the shared anchor grammar.""" - - locators = set() - for anchor in extract_anchors(record): - for reference in anchor.references: - try: - locator, _ = _split_assertion(reference) - except ValueError: - # The anchor tests diagnose malformed refs; coverage reports - # the remaining links rather than failing to print any gaps. - continue - locators.add(locator) - return locators - - -def anchored_refs(record): - """``(code_symbols, anchored_paths)`` named by the record's anchors. - - ``code_symbols`` holds ``(path, Symbol)`` pairs from ``path::Symbol`` - refs; ``anchored_paths`` holds every path any ref names. - """ - - symbols, paths = set(), set() - for reference in _anchor_locators(record): - kind, path, selector = _parse_reference(reference) - if kind == "code": - symbols.add((path, selector)) - paths.add(path) - return symbols, paths - - -def coverage_report(contracts, record, decision_markers=()): - """Report-only gaps between the contracts and the record. - - Returns ``(uncovered, unanchored)``: decision ids cited by neither a - contract nor a test marker, and ``@sc`` contracts whose declaration or - governed refs are not anchored. A module-docstring contract counts as - anchored when an anchor names its file or a symbol in it. A ``governs:`` - contract counts when at least one locator appears in the record after - rebasing to the repo root and - removing any value assertion from the record's ref; another key in the - same file is not a match. These are report-only links, not proof that - the prose holds or every coupled key is anchored. - """ - - cited = {c.meta["decision"] for c in contracts if "decision" in c.meta} - cited.update(marker.decision for marker in decision_markers) - uncovered = sorted(decision_ids(record) - cited) - references = _anchor_locators(record) - symbols, paths = anchored_refs(record) - unanchored = [] - for contract in contracts: - if contract.tag != "sc": - continue - if references.intersection(governed_refs(contract)): - continue - if contract.scope == "module": - if contract.path in paths: - continue - elif (contract.path, contract.scope) in symbols: - continue - unanchored.append(contract) - return uncovered, unanchored - - -FORBID = re.compile(r"^\s*forbid:\s*(\S+)\s*->\s*(\S+)\s*$") - - -def forbid_rules(contracts_file): - """``(contract_id, source, target)`` for each ``forbid:`` line.""" - - rules, ident = [], None - for line in Path(contracts_file).read_text(encoding="utf-8").splitlines(): - match = VALID.match(line) - if match: - ident = match.group(3) - continue - match = FORBID.match(line) - if match and ident: - rules.append((ident, *match.groups())) - return rules - - -def module_matches(name, pattern): - """Glob match; ``pkg.*`` also matches ``pkg`` itself.""" - - return fnmatch.fnmatchcase(name, pattern) or ( - pattern.endswith(".*") and name == pattern[:-2] - ) - - -def module_name(path, src_root): - """Dotted module name of ``path`` under ``src_root``.""" - - parts = list(Path(path).relative_to(src_root).with_suffix("").parts) - if parts[-1] == "__init__": - parts.pop() - return ".".join(parts) - - -def imported_names(path, module): - """``(line, dotted_name)`` for every import in ``path``. - - Relative imports resolve against ``module``; ``from a import b`` yields - both ``a`` and ``a.b``, since ``b`` may be a submodule. - """ - - with warnings.catch_warnings(): - warnings.simplefilter("ignore", SyntaxWarning) - tree = ast.parse(Path(path).read_text(encoding="utf-8")) - package = module.split(".") - if Path(path).name != "__init__.py": - package = package[:-1] - for node in ast.walk(tree): - if isinstance(node, ast.Import): - for alias in node.names: - yield node.lineno, alias.name - elif isinstance(node, ast.ImportFrom): - base = node.module or "" - if node.level: - parent = package[: len(package) - (node.level - 1)] - base = ".".join([*parent, *([base] if base else [])]) - yield node.lineno, base - for alias in node.names: - if alias.name != "*": - yield node.lineno, f"{base}.{alias.name}" - - -def import_violations(src_root, rules): - """Imports under ``src_root`` that a ``forbid:`` rule rejects.""" - - src_root = Path(src_root) - found = [] - for path in sorted(src_root.rglob("*.py")): - if any(part in SKIP for part in path.parts): - continue - module = module_name(path, src_root) - for ident, source, target in rules: - if not module_matches(module, source): - continue - for line, name in imported_names(path, module): - if module_matches(name, target): - found.append( - f"{path.relative_to(src_root)}:{line}: {module} " - f"imports {name} (contract {ident})" - ) - return found diff --git a/tests/helpers/decisions.py b/tests/helpers/decisions.py new file mode 100644 index 000000000..f55fcf2f8 --- /dev/null +++ b/tests/helpers/decisions.py @@ -0,0 +1,1285 @@ +"""Parse ShapePipe's decision tags, ASTRA Values and local contracts. + +The value grammar behind ``Values: ref = value`` is: + +* Values are numbers, boolean words, strings (quote expressions), or flat + comma lists, optionally bracketed. Semicolons separate refs. Decimal + equality is exact (1 = 1.0, 5e-4 = 0.0005), without rounding. +* Boolean words follow the file's reader, case-insensitively: INI as + ConfigParser.getboolean (yes/true/on/1, no/false/off/0; Y/N are text), + .sex/.psfex also Y/N, Python True/False; elsewhere words are text and + 1/0 are numbers. Outer whitespace is trimmed; other strings are + case-sensitive. Lists preserve order and length. Only .param VIGNET and + .psfex PSF_SIZE accept square-size shorthand: 51 = 51,51. +* ``= absent`` asserts a config key has no active line; its tagged file or + section scope must still exist. Quote it ("absent") to mean the text. +* .setools refs use ``SECTION.KEY``. Predicates keep their operators as + quoted text; repeated cuts on one key are an ordered list, e.g. + ``MAG_AUTO = ["> 18.", "< 22."]``. Expressions compare as text. +* Python selectors may append ``[key.subkey]`` to a named assignment + (identifier-like string dict keys); only the selected literal is read, + including ``dict(key=value)`` syntax. Ambiguous bindings fail, including + ``NAME[...] =`` or ``NAME.attr =`` in the same scope; ``.update()`` calls + and mutation from other scopes are not seen. No imports, calls, + arithmetic, argument defaults, environment expansion or implicit tool + defaults are evaluated: the check stays static rather than becoming a + second pipeline runtime. +* Option ids are stable references and are never parsed for values. +""" + +import argparse +import ast +import configparser +import fnmatch +import re +import sys +import tokenize +import warnings +from dataclasses import dataclass +from decimal import Decimal +from functools import lru_cache +from io import StringIO +from pathlib import Path + +import yaml + + +@dataclass(frozen=True) +class Site: + """One code declaration, statement, config paragraph, or config scope.""" + + path: str + start: int + end: int + kind: str + symbol: str = "" + section: str = "" + scope: str = "" + + @property + def identity(self): + return (self.path, self.start, self.end, self.kind, self.symbol, self.section) + + +@dataclass(frozen=True) +class Tag: + """One parsed ``@sc`` tag and the site it governs.""" + + path: str + line: int + decisions: tuple[str, ...] + ident: str | None + meta: dict + prose: str + site: Site | None + + +@dataclass(frozen=True) +class DecisionMarker: + """One pytest marker linking a test to an ASTRA decision.""" + + decision: str + path: str + line: int + + +def load_yaml(path): + """Load YAML with PyYAML's safe loader.""" + + return yaml.safe_load(Path(path).read_text(encoding="utf-8")) + + +ABSENT = "absent" +_NO_SETTING = object() +_TAG_LINE = re.compile(r"^\s*@sc(?:\s+\[([^\]]*)\])?(?:\s+(\S+))?\s*$") +_META_KEY = re.compile(r"[A-Za-z_][\w.-]*:[^,\s]+\Z") +_ID = re.compile(r"[A-Za-z_][\w.-]*\Z") +_CONFIG_SUFFIXES = { + ".ini", ".sex", ".param", ".conv", ".psfex", ".setools", + ".ww", ".conf", ".yaml", ".yml", +} +_SKIP = {".git", ".venv", "venv", "__pycache__", "node_modules", ".felt"} + + +def _tag_line(text): + """Parse one ``@sc`` line into metadata and an optional local id.""" + + match = _TAG_LINE.fullmatch(text.strip()) + if not match: + return None, None, f"malformed @sc tag: {text.strip()}" + meta_text, ident = match.groups() + meta = {} + if meta_text is not None: + pairs = [part.strip() for part in meta_text.split(",")] + if not pairs or any(not _META_KEY.fullmatch(part) for part in pairs): + return None, None, f"malformed @sc metadata: {meta_text}" + for pair in pairs: + key, value = pair.split(":", 1) + if key != "decision" and key in meta: + return None, None, f"duplicate @sc metadata key: {key}" + meta.setdefault(key, []).append(value) + if ident is not None and not _ID.fullmatch(ident): + return None, None, f"malformed local-contract id {ident!r}" + if "scope" in meta and meta["scope"] != ["file"]: + return None, None, "scope metadata must be scope:file" + if "decision" not in meta and ident is None: + return None, None, "@sc needs a decision citation or local-contract id" + return meta, ident, None + + +def _comment_body(line): + stripped = line.lstrip() + if not stripped.startswith("#"): + return None + return stripped[1:].lstrip() + + +def _comment_prose(lines, index, end=None): + stop = len(lines) if end is None else end + prose = [] + for line in lines[index + 1:stop]: + if not line.strip(): + break + body = _comment_body(line) + if body is None or body.startswith("@sc"): + break + if body: + prose.append(body.strip()) + return " ".join(prose) + + +def _ast_symbol(node, parents): + parts = [node.name] + parent = parents.get(node) + while parent is not None and not isinstance(parent, ast.Module): + if isinstance(parent, (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef)): + parts.append(parent.name) + parent = parents.get(parent) + return ".".join(reversed(parts)) + + +def _python_docstring_tags(path, source, tree): + lines = source.splitlines() + parents = {child: parent for parent in ast.walk(tree) + for child in ast.iter_child_nodes(parent)} + owners = [(tree, "")] + owners.extend((node, _ast_symbol(node, parents)) for node in ast.walk(tree) + if isinstance(node, (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef))) + tags, errors, consumed = [], [], set() + for owner, symbol in owners: + body = getattr(owner, "body", []) + if not body or not isinstance(body[0], ast.Expr) or not isinstance( + body[0].value, ast.Constant + ) or not isinstance(body[0].value.value, str): + continue + doc_node = body[0].value + for offset, doc_line in enumerate(doc_node.value.splitlines()): + if "@sc" not in doc_line: + continue + line_no = doc_node.lineno + offset + consumed.add(line_no) + meta, ident, error = _tag_line(doc_line) + if error: + errors.append(f"{path}:{line_no}: {error}") + continue + prose = [] + for part in doc_node.value.splitlines()[offset + 1:]: + if not part.strip() or part.strip().startswith("@sc"): + break + prose.append(part.strip()) + prose_text = " ".join(prose) + if ident is not None and not prose_text: + errors.append(f"{path}:{line_no}: local contract {ident} has no prose") + site = Site(path, getattr(owner, "lineno", 1), + getattr(owner, "end_lineno", len(lines)), + "python_declaration", symbol=symbol) + tags.append(Tag(path, line_no, tuple(meta.get("decision", ())), ident, + {k: v for k, v in meta.items() if k != "decision"}, + prose_text, site)) + return tags, errors, consumed + + +def _comment_site(path, lines, index, meta, tree=None, *, snakemake=False): + line_no = index + 1 + if meta.get("scope") == ["file"]: + if tree is not None or snakemake: + return None + has_content = any( + line.strip() and _comment_body(line) is None for line in lines + ) + if not has_content: + return None + return Site(path, 1, len(lines), "config", scope="file") + if snakemake: + for n in range(index + 1, len(lines)): + line = lines[n] + if not line.strip(): + return None + if line.lstrip().startswith("#"): + continue + start = n + 1 + if re.match(r"^\s*(?:rule|checkpoint)\s+[\w.-]+\s*:", line): + end = len(lines) + for j in range(n + 1, len(lines)): + if lines[j].strip() and not lines[j].startswith((" ", "\t", "#")): + end = j + break + return Site(path, start, end, "snakemake") + return Site(path, start, start, "snakemake") + return None + if tree is not None: + node = next((n for n in tree.body if getattr(n, "lineno", 0) > line_no), None) + if node is None: + return None + for between in lines[index + 1:node.lineno - 1]: + if not between.strip() or not between.lstrip().startswith("#"): + return None + names = [] + for target in getattr(node, "targets", []): + names.extend(_target_names(target)) + if hasattr(node, "target"): + names.extend(_target_names(node.target)) + return Site(path, node.lineno, getattr(node, "end_lineno", node.lineno), + "python_statement", symbol=names[0] if names else "") + + # Config paragraphs end at blank lines; comments before/inside a run are + # skipped, but a blank before the first setting leaves an empty site. + for n in range(index + 1, len(lines)): + line = lines[n] + if not line.strip(): + return None + comment = _comment_body(line) + if comment is not None: + if comment.startswith("@sc"): + return None + continue + start = n + 1 + header = re.match(r"^\s*\[([^]]+)\]\s*(?:[#;].*)?$", line) + if header: + end = len(lines) + for j in range(n + 1, len(lines)): + comment = _comment_body(lines[j]) + if comment is not None and comment.startswith("@sc"): + end = j + break + if re.match(r"^\s*\[[^]]+\]\s*(?:[#;].*)?$", lines[j]): + end = j + break + return Site(path, start, end, "config", section=header.group(1).strip(), scope="section") + end = len(lines) + for j in range(n + 1, len(lines)): + if not lines[j].strip(): + end = j + break + comment = _comment_body(lines[j]) + if comment is not None and comment.startswith("@sc"): + end = j + break + return Site(path, start, end, "config", scope="paragraph") + return None + + +def _parse_file_tags(path, root): + relative = Path(path).relative_to(root).as_posix() + source = Path(path).read_text(encoding="utf-8") + lines = source.splitlines() + suffix = Path(path).suffix.lower() + tags, errors = [], [] + is_snakemake = suffix == ".smk" or Path(path).name == "Snakefile" + tree = None + consumed = set() + comment_tokens = {} + if suffix == ".py": + try: + tree = ast.parse(source, filename=relative) + except SyntaxError as error: + return [], [f"{relative}: cannot parse tagged Python: {error}"] + found, issues, consumed = _python_docstring_tags(relative, source, tree) + tags.extend(found) + errors.extend(issues) + try: + comment_tokens = { + token.start[0]: (token.start[1], token.string) + for token in tokenize.generate_tokens(StringIO(source).readline) + if token.type == tokenize.COMMENT + } + except tokenize.TokenError as error: + return tags, errors + [f"{relative}: cannot tokenize Python comments: {error}"] + if suffix == ".py" or is_snakemake or suffix in _CONFIG_SUFFIXES: + if suffix == ".py": + tag_comments = [ + (line_no, column, comment[1:].lstrip()) + for line_no, (column, comment) in comment_tokens.items() + if "@sc" in comment + ] + else: + tag_comments = [ + (index + 1, len(line) - len(line.lstrip()), body) + for index, line in enumerate(lines) + if (body := _comment_body(line)) is not None and "@sc" in body + ] + for line_no, column, body in tag_comments: + index = line_no - 1 + if suffix == ".py" and line_no in consumed: + continue + if suffix == ".py" and column != 0: + errors.append(f"{relative}:{line_no}: Python statement tags must be module-level comments") + continue + meta, ident, error = _tag_line(body) + if error: + errors.append(f"{relative}:{line_no}: {error}") + continue + site = _comment_site(relative, lines, index, meta, tree, + snakemake=is_snakemake) + if site is None: + errors.append(f"{relative}:{line_no}: @sc tag governs no site") + continue + prose = _comment_prose(lines, index) + if ident is not None and not prose: + errors.append(f"{relative}:{line_no}: local contract {ident} has no prose") + tags.append(Tag(relative, line_no, tuple(meta.get("decision", ())), ident, + {k: v for k, v in meta.items() if k != "decision"}, + prose, site)) + return tags, errors + + +def scan_tags(root): + """Collect all supported-site tags and report malformed/duplicate tags.""" + + root = Path(root) + tags, errors = [], [] + for path in sorted(root.rglob("*")): + if not path.is_file() or any(part in _SKIP for part in path.parts): + continue + if path.name == "CONTRACTS": + continue + if path.suffix.lower() not in _CONFIG_SUFFIXES | {".py", ".smk"} and path.name != "Snakefile": + continue + found, issues = _parse_file_tags(path, root) + tags.extend(found) + errors.extend(issues) + by_id = {} + for tag in tags: + if tag.ident: + by_id.setdefault(tag.ident, []).append(tag) + for ident, found in sorted(by_id.items()): + if len(found) > 1: + where = ", ".join(f"{tag.path}:{tag.line}" for tag in found) + errors.append(f"duplicate local-contract id {ident}: {where}") + return tags, errors + + +def decision_ids(record, prefix=""): + """Scoped decision IDs: bare at top level, dotted in sub-analyses.""" + + if not isinstance(record, dict): + return set() + result = {f"{prefix}{key}" for key in (record.get("decisions") or {})} + for name, analysis in (record.get("analyses") or {}).items(): + if isinstance(analysis, dict): + result.update(decision_ids(analysis, f"{prefix}{name}.")) + return result + + +def tag_errors(tags, record): + """Check citations and require every ASTRA decision to have a site.""" + + known = decision_ids(record) + cited = set() + errors = [] + for tag in tags: + for decision in tag.decisions: + cited.add(decision) + if decision not in known: + errors.append(f"{tag.path}:{tag.line}: @sc cites unknown decision {decision!r}") + errors.extend(f"{decision}: decision has no tagged site" for decision in sorted(known - cited)) + return errors + + +def _parse_values(rationale): + if not isinstance(rationale, str) or "Values:" not in rationale: + return (), None + if rationale.count("Values:") != 1: + return (), "rationale must contain at most one Values: sentence" + tail = rationale.rsplit("Values:", 1)[1].strip() + if not tail.endswith("."): + return (), "Values: sentence must end with a period" + entries = tuple(part.strip() for part in tail[:-1].split(";")) + if not entries or any(not entry for entry in entries): + return (), "Values: sentence contains an empty ref" + return entries, None + + +def _split_assertion(reference): + """Split one ``ref = value`` entry (not a quoted cut's ``==``).""" + + parts = re.split(r"\s+=\s*", reference.strip(), maxsplit=1) + ref = parts[0] + expected = parts[1].strip() if len(parts) == 2 else None + if not ref or re.search(r"\s|=", ref): + raise ValueError("expected a ref optionally followed by ' = value'") + if expected is None or not expected or expected.startswith("="): + raise ValueError("Values entries require a nonempty ' = value'") + return ref, expected + + +def _parse_reference(reference): + if "::" in reference: + path, selector = reference.split("::", 1) + return "python", path, selector + if "#" in reference: + path, selector = reference.split("#", 1) + return "config", path, selector + return "bare", "", reference + + +def _target_names(target): + if isinstance(target, ast.Name): + return [target.id] + if isinstance(target, ast.Attribute): + return [target.attr] + if isinstance(target, (ast.Tuple, ast.List)): + return [name for item in target.elts for name in _target_names(item)] + if isinstance(target, ast.Starred): + return _target_names(target.value) + return [] + + +@lru_cache(maxsize=128) +def _bindings(scope): + """Collect all bindings per name; value reads must not pick one silently.""" + + result = {} + + def visit(node): + if isinstance( + node, + (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef), + ): + result.setdefault(node.name, []).append(node) + return + if isinstance(node, ast.Lambda): + return + if isinstance(node, ast.Assign): + targets = node.targets + elif isinstance(node, (ast.AnnAssign, ast.AugAssign, ast.NamedExpr)): + targets = [node.target] + elif isinstance(node, (ast.For, ast.AsyncFor)): + targets = [node.target] + elif isinstance(node, (ast.With, ast.AsyncWith)): + targets = [item.optional_vars for item in node.items] + else: + targets = [] + for target in targets: + if target is not None: + for name in _target_names(target): + result.setdefault(name, []).append(node) + if isinstance(node, ast.ExceptHandler) and node.name: + result.setdefault(node.name, []).append(node) + for child in ast.iter_child_nodes(node): + visit(child) + + for statement in scope.body: + visit(statement) + return result + + +def _mutated_name(target): + """Base name of ``NAME[...] =`` / ``NAME.attr =`` (nested included).""" + + while isinstance(target, (ast.Subscript, ast.Attribute)): + target = target.value + if isinstance(target, ast.Name): + return target.id + return None + + +@lru_cache(maxsize=128) +def _mutations(scope): + """Names whose bound object is item- or attribute-assigned in ``scope``. + + Same lexical scope only, like ``_bindings``; method calls such as + ``.update()`` and mutation from other scopes are not seen. + """ + + names = set() + + def visit(node): + if isinstance( + node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef, ast.Lambda) + ): + return + if isinstance(node, ast.Assign): + targets = node.targets + elif isinstance(node, (ast.AnnAssign, ast.AugAssign)): + targets = [node.target] + elif isinstance(node, ast.Delete): + targets = node.targets + elif isinstance(node, (ast.For, ast.AsyncFor)): + targets = [node.target] + elif isinstance(node, (ast.With, ast.AsyncWith)): + targets = [item.optional_vars for item in node.items] + else: + targets = [] + stack = [target for target in targets if target is not None] + while stack: + target = stack.pop() + if isinstance(target, (ast.Tuple, ast.List)): + stack.extend(target.elts) + elif isinstance(target, ast.Starred): + stack.append(target.value) + else: + name = _mutated_name(target) + if name: + names.add(name) + for child in ast.iter_child_nodes(node): + visit(child) + + for statement in scope.body: + visit(statement) + return frozenset(names) + + +def _ini_parser(text, *, strict=True): + parser = configparser.ConfigParser( + interpolation=None, strict=strict, allow_no_value=True + ) + parser.optionxform = str + parser.read_string(text) + return parser + + +@lru_cache(maxsize=16) +def _python_tree(text): + # Cache by source, not path: editing a file must invalidate the read. + return ast.parse(text) + + +def _code_selector(selector): + match = re.fullmatch(r"([\w.]+)(?:\[([\w.]+)\])?", selector) + if not match or any(not p.isidentifier() for p in match[1].split(".")): + raise ValueError(f"invalid Python selector {selector!r}") + keys = tuple(match[2].split(".")) if match[2] else () + if any(not key.isidentifier() for key in keys): + raise ValueError("dict paths need dot-separated identifier keys") + return match[1], keys + + +def _dict_entry(node, key): + """Select syntax, not a runtime value; never execute a dict() call.""" + + if isinstance(node, ast.Dict): + if any( + not isinstance(k, ast.Constant) or not isinstance(k.value, str) + for k in node.keys + ): + raise ValueError("dict selectors need literal string keys, no **") + items = [(k.value, v) for k, v in zip(node.keys, node.values)] + elif ( + isinstance(node, ast.Call) and isinstance(node.func, ast.Name) + and node.func.id == "dict" and not node.args + and all(k.arg is not None for k in node.keywords) + ): + items = [(k.arg, k.value) for k in node.keywords] + else: + raise ValueError("dict selectors need {...} or dict(key=value) syntax") + names = [name for name, _ in items] + if len(names) != len(set(names)): + raise ValueError("ambiguous duplicate dict keys") + if key not in names: + raise ValueError(f"dict key {key!r} is missing") + return dict(items)[key] + + +def _selected_python_node(tree, selector): + symbol, keys = _code_selector(selector) + scope = tree + parts = symbol.split(".") + for index, part in enumerate(parts): + declarations = _bindings(scope).get(part, []) + if len(declarations) != 1: + raise ValueError( + f"{symbol!r} needs one binding; found {len(declarations)} " + f"for {part!r}" + ) + node = declarations[0] + if index == len(parts) - 1 and part in _mutations(scope): + raise ValueError( + f"{symbol!r} is item- or attribute-assigned after binding" + ) + if index < len(parts) - 1: + if not isinstance( + node, (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef) + ): + raise ValueError(f"{part!r} is not a lexical scope") + scope = node + if not isinstance(node, (ast.Assign, ast.AnnAssign)): + raise ValueError(f"{symbol!r} is not a literal assignment") + targets = node.targets if isinstance(node, ast.Assign) else [node.target] + if any(not isinstance(target, ast.Name) for target in targets): + raise ValueError("value assertions need simple named assignment targets") + node = node.value + for key in keys: + node = _dict_entry(node, key) + return node + + +_NUMBER = re.compile(r"[+-]?(?:\d+(?:\.\d*)?|\.\d+)(?:[eE][+-]?\d+)?\Z") +# Boolean words each reader accepts. INI follows ConfigParser.getboolean, +# whose 1/0 spellings are handled at comparison (see _ini_bool); SExtractor +# and PSFEx add Y/N; Python literals are already bool, and the record spells +# them True/False. Elsewhere words stay text. +_INI_BOOLEANS = { + "yes": True, "true": True, "on": True, + "no": False, "false": False, "off": False, +} +_ASTROMATIC_BOOLEANS = {**_INI_BOOLEANS, "y": True, "n": False} +_PYTHON_BOOLEANS = {"true": True, "false": False} +_YAML_BOOLEANS = {"true": True, "false": False, "yes": True, "no": False, "on": True, "off": False} +_NO_BOOLEANS = {} + + +def _booleans_for(suffix): + if suffix == ".ini": + return _INI_BOOLEANS + if suffix in {".sex", ".psfex"}: + return _ASTROMATIC_BOOLEANS + if suffix == ".py": + return _PYTHON_BOOLEANS + if suffix in {".yaml", ".yml"}: + return _YAML_BOOLEANS + return _NO_BOOLEANS + + +def _ini_bool(want, got): + """getboolean also reads 1/0; accept them only against a bool expectation.""" + + if want[0] == "bool" and got[0] == "number" and got[1] in (0, 1): + return "bool", got[1] == 1 + return got + + +def _normalise_value(value, booleans=_ASTROMATIC_BOOLEANS): + """Use tagged atoms so boolean True cannot compare equal to number 1.""" + + if isinstance(value, bool): + return "bool", value + if isinstance(value, (int, float)): + number = Decimal(str(value)) + if not number.is_finite(): + raise ValueError("numeric values must be finite") + return "number", number + if isinstance(value, (list, tuple)): + elements = tuple(_normalise_value(item, booleans) for item in value) + if any(kind == "list" for kind, _ in elements): + raise ValueError("only flat lists are supported") + return "list", elements + if not isinstance(value, str): + raise ValueError("expected a number, boolean, string or flat list") + value = value.strip() + if not value: + raise ValueError("empty values/list elements are not supported") + if value[0] in "[{'\"": + # BaseLoader keeps even 5e-4 and Y as strings, avoiding YAML 1.1's + # inconsistent numeric/boolean coercions. It constructs no objects. + parsed = yaml.load(value, Loader=yaml.BaseLoader) + if isinstance(parsed, list): + return _normalise_value(parsed, booleans) + if not isinstance(parsed, str): + raise ValueError("expected a scalar or flat list, not a mapping") + # Quotes protect commas/operators; their contents are a single atom. + value = parsed.strip() + elif "," in value: + return _normalise_value(value.split(","), booleans) + if _NUMBER.fullmatch(value): + return "number", Decimal(value) + if value.lower() in booleans: + return "bool", booleans[value.lower()] + return "text", value + + +def _square_stamp(value): + if value[0] == "number": + return "list", (value, value) + return value + + +def universe_errors(record, universe): + """Check scoped decision IDs and options against the ASTRA record.""" + + record_decisions = _decisions(record) + pinned = _decisions(universe) + errors = [] + for location in sorted(pinned.keys() - record_decisions.keys()): + errors.append( + f"{location}: universe decision is absent from astra.yaml" + ) + for location in sorted(record_decisions.keys() - pinned.keys()): + errors.append( + f"{location}: astra.yaml decision is not pinned in the universe" + ) + for location in sorted(record_decisions.keys() & pinned.keys()): + definition = record_decisions[location] + options = ( + definition.get("options", {}) + if isinstance(definition, dict) + else {} + ) + if not isinstance(options, dict) or pinned[location] not in options: + errors.append( + f"{location}: pinned option {pinned[location]!r} is not in " + "ASTRA options" + ) + return errors + + +def _decisions(document, location=""): + scope = document if isinstance(document, dict) else {} + result = {} + for decision_id, definition in (scope.get("decisions") or {}).items(): + key = ( + f"{location}.decisions.{decision_id}" + if location + else f"decisions.{decision_id}" + ) + result[key] = definition + for analysis_id, analysis in (scope.get("analyses") or {}).items(): + child = ( + f"{location}.analyses.{analysis_id}" + if location + else f"analyses.{analysis_id}" + ) + if isinstance(analysis, dict): + result.update(_decisions(analysis, child)) + return result + + +def _iter_decision_defs(document, prefix=""): + if not isinstance(document, dict): + return + for ident, definition in (document.get("decisions") or {}).items(): + yield f"{prefix}{ident}", definition + for name, analysis in (document.get("analyses") or {}).items(): + if isinstance(analysis, dict): + yield from _iter_decision_defs(analysis, f"{prefix}{name}.") + + +def _sites_for(tags, decision): + unique = {} + for tag in tags: + if decision in tag.decisions and tag.site is not None: + unique[tag.site.identity] = tag.site + return list(unique.values()) + + +def _path_matches(site_path, qualifier): + path = Path(qualifier) + if path.is_absolute() or ".." in path.parts or not qualifier: + return False + normalized = path.as_posix().lstrip("./") + return site_path == normalized or site_path.endswith("/" + normalized) + + +def _split_config_key(selector, suffix, site): + if suffix in {".ini", ".setools"} and "." in selector: + return selector.rsplit(".", 1) + section = site.section if site.scope == "section" else "" + return section, selector.rsplit(".", 1)[-1] + + +def _strip_config_comment(line, suffix): + stripped = line.strip() + if not stripped or stripped.startswith(("#", ";")): + return "" + if suffix != ".ini" and "#" in stripped: + stripped = stripped.split("#", 1)[0].strip() + return stripped + + +def _config_values_in_site(root, site, selector): + """Return this config site's value for selector, or None if it doesn't.""" + + target = Path(root) / site.path + text = target.read_text(encoding="utf-8") + suffix = target.suffix.lower() + if suffix in {".yaml", ".yml"}: + data = yaml.safe_load(text) + parts = selector.split(".") + node = data + for part in parts: + if not isinstance(node, dict) or part not in node: + return None + node = node[part] + key = parts[-1] + line_matches = any( + re.match(rf"^\s*{re.escape(key)}\s*:", line) + for line in text.splitlines()[site.start - 1:site.end] + ) + return node if line_matches else None + + lines = text.splitlines() + section, key = _split_config_key(selector, suffix, site) + current_section = "DEFAULT" if suffix == ".ini" else "" + values, predicates = [], [] + for number, raw in enumerate(lines, 1): + stripped = _strip_config_comment(raw, suffix) + header = re.match(r"^\[([^]]+)\]$", stripped) + if header: + current_section = header.group(1).strip() + continue + if number < site.start or number > site.end or not stripped: + continue + if suffix in {".ini", ".setools"}: + wanted_section = section or (site.section if site.scope == "section" else "") + if wanted_section and current_section != wanted_section: + continue + if suffix == ".ini": + match = re.match(r"^([^:=\s][^:=]*?)\s*[:=]\s*(.*)$", stripped) + if not match or match.group(1).strip() != key: + continue + value = match.group(2).strip() + predicate = False + elif suffix == ".setools": + match = re.match(rf"^{re.escape(key)}(?=$|\s|=|<|>)(.*)$", stripped) + if not match: + continue + value = match.group(1).strip() + predicate = value.startswith(("==", "!=", "<", ">")) + if value.startswith("=") and not predicate: + value = value[1:].strip() + elif suffix == ".param": + match = re.match(rf"^{re.escape(key)}(?:\s*\(([^)]*)\)|\s+(.*))?$", stripped) + if not match: + continue + value = match.group(1) if match.group(1) is not None else (match.group(2) or "") + predicate = False + else: + match = re.match(rf"^{re.escape(key)}(?=$|\s|=|\()(.*)$", stripped) + if not match: + continue + value = match.group(1).strip() + predicate = False + if value.startswith("="): + value = value[1:].strip() + if not value: + raise ValueError(f"no active value for {selector!r}") + values.append(value) + predicates.append(predicate) + if not values: + return None + if len(values) > 1 and not all(predicates): + raise ValueError(f"ambiguous active values for {selector!r}: {values!r}") + return values if len(values) > 1 else values[0] + + +def _section_exists(text, suffix, section): + if suffix == ".ini": + try: + parser = _ini_parser(text, strict=False) + except configparser.Error as error: + raise ValueError(f"cannot parse INI file: {error}") from error + return section == parser.default_section or parser.has_section(section) + return any( + line.strip() == f"[{section}]" for line in text.splitlines() + ) + + +def _absent_scope_matches(root, site, selector): + target = Path(root) / site.path + suffix = target.suffix.lower() + if site.scope not in {"file", "section"}: + return None + text = target.read_text(encoding="utf-8") + section = selector.rsplit(".", 1)[0] if suffix in {".ini", ".setools"} and "." in selector else "" + if site.scope == "section": + if section and section != site.section: + return None + section = site.section + if section and not _section_exists(text, suffix, section): + return None + try: + actual = _config_values_in_site(root, site, selector) + except ValueError as error: + # A no-value option is still an active setting for an absence check. + return str(error) + return _NO_SETTING if actual is None else actual + + +def _python_value_in_site(root, site, selector): + source = (Path(root) / site.path).read_text(encoding="utf-8") + tree = _python_tree(source) + symbol, _ = _code_selector(selector) + if site.symbol and (symbol == site.symbol or symbol.startswith(site.symbol + ".")): + full_selector = selector + elif site.symbol: + full_selector = f"{site.symbol}.{selector}" + else: + full_selector = selector + full_symbol, _ = _code_selector(full_selector) + if site.symbol and full_symbol != site.symbol and not full_symbol.startswith(site.symbol + "."): + return None + try: + node = _selected_python_node(tree, full_selector) + except ValueError as error: + if "needs one binding" in str(error) or "missing" in str(error): + return None + raise + try: + return ast.literal_eval(node) + except (ValueError, TypeError) as error: + raise ValueError("selected Python value is not a literal") from error + + +def _site_actual(root, site, kind, selector): + suffix = Path(site.path).suffix.lower() + if kind == "python": + if suffix != ".py": + return None + return _python_value_in_site(root, site, selector) + if suffix not in _CONFIG_SUFFIXES: + return None + return _config_values_in_site(root, site, selector) + + +def _value_comparison(expected, actual, suffix, selector): + booleans = _booleans_for(suffix) + want = _normalise_value(expected, booleans) + got = _normalise_value(actual, booleans) + if suffix == ".ini": + got = _ini_bool(want, got) + if (suffix, selector.rsplit(".", 1)[-1]) in { + (".param", "VIGNET"), (".psfex", "PSF_SIZE") + }: + want, got = _square_stamp(want), _square_stamp(got) + return want == got + + +def _check_value_entry(root, decision, reference, tags): + try: + ref, expected = _split_assertion(reference) + kind, qualifier, selector = _parse_reference(ref) + if kind == "bare": + if not selector or re.search(r"\s", selector): + raise ValueError("invalid unqualified ref") + elif not qualifier or Path(qualifier).is_absolute() or ".." in Path(qualifier).parts: + raise ValueError("path qualifier must be a relative path suffix") + except ValueError as error: + return f"{decision}: ref {reference!r}: expected , actual : {error}" + + candidates = [] + for site in _sites_for(tags, decision): + suffix = Path(site.path).suffix.lower() + expected_kind = "python" if suffix == ".py" else "config" + if kind != "bare" and kind != expected_kind: + continue + if qualifier and not _path_matches(site.path, qualifier): + continue + try: + if expected == ABSENT: + actual = _absent_scope_matches(root, site, selector) + else: + actual = _site_actual(root, site, expected_kind, selector) + except (ValueError, OSError, SyntaxError, configparser.Error, yaml.YAMLError) as error: + return f"{decision}: ref {ref!r}: expected {expected!r}, actual : {error}" + if actual is not None: + candidates.append((site, actual)) + + if len(candidates) != 1: + actual = "" if not candidates else f"" + return f"{decision}: ref {ref!r}: expected {expected!r}, actual {actual} (ref must resolve to exactly one tagged site; found {len(candidates)})" + + site, actual = candidates[0] + if expected == ABSENT: + if actual is _NO_SETTING: + return None + return f"{decision}: ref {ref!r}: expected no active setting, actual {actual!r}" + try: + suffix = Path(site.path).suffix.lower() + if _value_comparison(expected, actual, suffix, selector): + return None + return f"{decision}: ref {ref!r}: expected {expected!r}, actual {actual!r}" + except (ValueError, OSError, SyntaxError, configparser.Error, yaml.YAMLError) as error: + return f"{decision}: ref {ref!r}: expected {expected!r}, actual {actual!r}: {error}" + + +def value_errors(root, record, tags=None): + """Check every Values entry against exactly one site tagged for its decision.""" + + root = Path(root) + if tags is None: + tags, _ = scan_tags(root) + errors = [] + for decision, definition in _iter_decision_defs(record): + rationale = definition.get("rationale") if isinstance(definition, dict) else None + if not isinstance(rationale, str): + continue + if "Anchor:" in rationale: + errors.append(f"{decision}: legacy Anchor: sentence remains in rationale") + continue + entries, problem = _parse_values(rationale) + if problem: + errors.append(f"{decision}: {problem}") + continue + for entry in entries: + try: + _split_assertion(entry) + except ValueError as error: + errors.append(f"{decision}: Values ref {entry!r}: {error}") + continue + result = _check_value_entry(root, decision, entry, tags) + if result: + errors.append(result) + return errors + + +def _is_decision_marker_call(node): + function = node.func + return ( + isinstance(function, ast.Attribute) and function.attr == "decision" + and isinstance(function.value, ast.Attribute) and function.value.attr == "mark" + and isinstance(function.value.value, ast.Name) + and function.value.value.id == "pytest" + ) + + +def decision_markers(root): + root = Path(root) + markers, errors = [], [] + test_root = root / "tests" + if not test_root.is_dir(): + return markers, errors + for path in sorted(test_root.rglob("*.py")): + if any(part in _SKIP for part in path.parts): + continue + relative = path.relative_to(root).as_posix() + try: + tree = ast.parse(path.read_text(encoding="utf-8"), filename=relative) + except (OSError, SyntaxError) as error: + errors.append(f"{relative}: cannot parse decision markers: {error}") + continue + calls = sorted( + (node for node in ast.walk(tree) if isinstance(node, ast.Call) + and _is_decision_marker_call(node)), + key=lambda node: (node.lineno, node.col_offset), + ) + for call in calls: + where = f"{relative}:{call.lineno}" + if not call.args: + errors.append(f"{where}: decision marker needs literal string ids") + if call.keywords: + errors.append(f"{where}: decision markers accept positional ids only") + for argument in call.args: + if isinstance(argument, ast.Constant) and isinstance(argument.value, str): + markers.append(DecisionMarker(argument.value, relative, call.lineno)) + else: + errors.append(f"{where}: decision marker ids must be literal strings") + return markers, errors + + +def decision_marker_errors(markers, record): + known = decision_ids(record) + return [f"{marker.path}:{marker.line}: decision marker cites unknown decision {marker.decision!r}" + for marker in markers if marker.decision not in known] + + +def _rationale_sentence(definition): + text = str(definition.get("rationale", "")).strip() + text = re.sub(r"\s+", " ", text) + parts = re.split(r"(?<=[.!?])\s+", text, maxsplit=1) + return parts[0] if parts else "" + + +def _value_sentence(definition): + entries, problem = _parse_values(definition.get("rationale", "")) + return "Values: " + "; ".join(entries) + "." if entries and not problem else "" + + +def repository_errors(root): + """Run the bidirectional tag/value and marker checks for a repository.""" + + root = Path(root) + record = load_yaml(root / "astra.yaml") + tags, errors = scan_tags(root) + markers, marker_parse_errors = decision_markers(root) + errors = list(errors) + errors.extend(tag_errors(tags, record)) + errors.extend(value_errors(root, record, tags)) + errors.extend(marker_parse_errors) + errors.extend(decision_marker_errors(markers, record)) + universe_path = root / "universes" / "committed.yaml" + if universe_path.is_file(): + errors.extend(universe_errors(record, load_yaml(universe_path))) + return errors + + +_FORBID = re.compile(r"^\s*forbid:\s*(\S+)\s*->\s*(\S+)\s*$") +_CONTRACT_TAG = re.compile(r"^\s*@(sc|cc)(?:\s+\[[^]]*\])?\s+([\w][\w.-]*)\s*$") + + +def forbid_rules(contracts_file): + """Return ``(contract_id, source, target)`` import-boundary rules.""" + + rules, ident = [], None + for line in Path(contracts_file).read_text(encoding="utf-8").splitlines(): + match = _CONTRACT_TAG.match(line) + if match: + ident = match.group(2) + continue + match = _FORBID.match(line) + if match and ident: + rules.append((ident, *match.groups())) + return rules + + +def module_matches(name, pattern): + return fnmatch.fnmatchcase(name, pattern) or ( + pattern.endswith(".*") and name == pattern[:-2] + ) + + +def module_name(path, src_root): + parts = list(Path(path).relative_to(src_root).with_suffix("").parts) + if parts[-1] == "__init__": + parts.pop() + return ".".join(parts) + + +def imported_names(path, module): + """Yield ``(line, dotted_name)`` for imports in a Python module.""" + + with warnings.catch_warnings(): + warnings.simplefilter("ignore", SyntaxWarning) + tree = ast.parse(Path(path).read_text(encoding="utf-8")) + package = module.split(".") + if Path(path).name != "__init__.py": + package = package[:-1] + for node in ast.walk(tree): + if isinstance(node, ast.Import): + for alias in node.names: + yield node.lineno, alias.name + elif isinstance(node, ast.ImportFrom): + base = node.module or "" + if node.level: + parent = package[:len(package) - (node.level - 1)] + base = ".".join([*parent, *([base] if base else [])]) + yield node.lineno, base + for alias in node.names: + if alias.name != "*": + yield node.lineno, f"{base}.{alias.name}" + + +def import_violations(src_root, rules): + """Find imports forbidden by ``CONTRACTS`` rules.""" + + src_root = Path(src_root) + found = [] + for path in sorted(src_root.rglob("*.py")): + if any(part in _SKIP for part in path.parts): + continue + module = module_name(path, src_root) + for ident, source, target in rules: + if not module_matches(module, source): + continue + for line, name in imported_names(path, module): + if module_matches(name, target): + found.append( + f"{path.relative_to(src_root)}:{line}: {module} imports " + f"{name} (contract {ident})" + ) + return found + + +def _decision_description(record, decision): + for ident, definition in _iter_decision_defs(record): + if ident == decision: + label = definition.get("label", decision) if isinstance(definition, dict) else decision + return label, _rationale_sentence(definition), _value_sentence(definition) + return decision, "", "" + + +def _location_path(root, value): + raw = Path(value) + if raw.is_absolute(): + try: + return raw.resolve().relative_to(Path(root).resolve()).as_posix() + except ValueError: + return None + return raw.as_posix().lstrip("./") + + +def _print_location(root, record, tags, location): + line = None + match = re.match(r"^(.*):(\d+)$", location) + if match: + location, line = match.group(1), int(match.group(2)) + relative = _location_path(root, location) + if relative is None: + print(f"{location}: outside repository") + return + matches = [] + for tag in tags: + site = tag.site + if site is None or site.path != relative: + continue + if line is not None and not (site.start <= line <= site.end or tag.line == line): + continue + matches.append(tag) + by_decision = sorted({decision for tag in matches for decision in tag.decisions}) + print(f"{relative}" + (f":{line}" if line is not None else "")) + for decision in by_decision: + label, rationale, values = _decision_description(record, decision) + print(f" {decision} — {label}") + if rationale: + print(f" {rationale}") + if values: + print(f" {values}") + for tag in matches: + if tag.ident: + label = tag.meta.get("label", [""])[0] + suffix = f" [{label}]" if label else "" + print(f" Local contract {tag.ident}{suffix}: {tag.prose}") + if not matches: + print(" No tagged site governs this location.") + + +def _print_decision_sites(record, tags, decision): + label, rationale, values = _decision_description(record, decision) + print(f"{decision} — {label}") + if rationale: + print(f" {rationale}") + if values: + print(f" {values}") + sites = _sites_for(tags, decision) + if not sites: + print(" No tagged sites.") + for site in sorted(sites, key=lambda item: (item.path, item.start)): + print(f" {site.path}:{site.start}-{site.end} ({site.kind})") + for tag in tags: + if decision in tag.decisions and tag.site and tag.site.identity == site.identity and tag.ident: + print(f" Local contract {tag.ident}: {tag.prose}") + + +def main(argv=None): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("location", nargs="?", help="path[:line] to inspect") + parser.add_argument("--decision", help="list all sites for one decision") + parser.add_argument("--root", default=Path(__file__).resolve().parents[2]) + args = parser.parse_args(argv) + root = Path(args.root) + record = load_yaml(root / "astra.yaml") + tags, errors = scan_tags(root) + if args.decision: + _print_decision_sites(record, tags, args.decision) + elif args.location: + _print_location(root, record, tags, args.location) + else: + parser.error("supply a path[:line] or --decision ") + if errors: + print(f"\n{len(errors)} tag parse error(s):") + for error in errors: + print(f" {error}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/unit/test_astra_anchors.py b/tests/unit/test_astra_anchors.py deleted file mode 100644 index 38b334f12..000000000 --- a/tests/unit/test_astra_anchors.py +++ /dev/null @@ -1,75 +0,0 @@ -"""Keep ASTRA decision anchors and the committed universe resolvable.""" - -from pathlib import Path - -from tests.helpers.astra_record import ( - extract_anchors, - load_yaml, - resolve_anchor, - universe_errors, -) - - -REPO_ROOT = Path(__file__).resolve().parents[2] - - -def test_snakemake_rule_and_function_anchors_resolve(tmp_path): - rules_dir = tmp_path / "workflow" / "rules" - rules_dir.mkdir(parents=True) - (rules_dir / "example.smk").write_text( - "def tile_local(tile):\n return tile\n\n\n" - "rule tile_detect:\n input: 'a'\n output: 'b'\n", - encoding="utf-8", - ) - - assert resolve_anchor(tmp_path, "workflow/rules/example.smk::tile_detect") is None - assert resolve_anchor(tmp_path, "workflow/rules/example.smk::tile_local") is None - - problem = resolve_anchor(tmp_path, "workflow/rules/example.smk::no_such_rule") - assert problem == "no rule/checkpoint/def named 'no_such_rule'" - - -def test_snakefile_rule_anchor_resolves(tmp_path): - (tmp_path / "workflow").mkdir() - (tmp_path / "workflow" / "Snakefile").write_text( - "checkpoint plan:\n input: 'a'\n", encoding="utf-8" - ) - - assert resolve_anchor(tmp_path, "workflow/Snakefile::plan") is None - - -def test_every_astra_anchor_resolves(): - record = load_yaml(REPO_ROOT / "astra.yaml") - anchors = extract_anchors(record) - errors = [] - - assert anchors, "astra.yaml contains no Anchor: sentences" - for anchor in anchors: - if anchor.error: - errors.append(f"{anchor.location}: {anchor.error}") - continue - for reference in anchor.references: - problem = resolve_anchor(REPO_ROOT, reference) - if problem: - errors.append( - f"{anchor.location}: {reference}: {problem}" - ) - - message = ( - "Unresolved ASTRA anchors or rationales:\n - " - + "\n - ".join(errors) - ) - assert not errors, message - - -def test_committed_universe_matches_astra_decisions(): - record = load_yaml(REPO_ROOT / "astra.yaml") - universe = load_yaml(REPO_ROOT / "universes" / "committed.yaml") - - errors = universe_errors(record, universe) - - message = ( - "ASTRA / committed universe mismatch:\n - " - + "\n - ".join(errors) - ) - assert not errors, message diff --git a/tests/unit/test_astra_values.py b/tests/unit/test_astra_values.py deleted file mode 100644 index 17160d355..000000000 --- a/tests/unit/test_astra_values.py +++ /dev/null @@ -1,433 +0,0 @@ -"""Check recorded values, not just locations, without importing the pipeline. - -Failure modes: plausible numeric drift, wrong key/section/case, commented or -ambiguous settings, changed cut operators, lost list elements, and executing -Python while trying to inspect it. Fixtures exercise each through the same -reader as the record; location resolution must remain independent of equality. -""" - -from pathlib import Path - -import pytest - -from tests.helpers.astra_record import ( - check_anchor_value, - extract_anchors, - load_yaml, - resolve_anchor, - value_errors, -) - - -REPO_ROOT = Path(__file__).resolve().parents[2] - - -def record_with(reference): - """Put a reference in a scoped decision, as in the real record.""" - - return { - "analyses": { - "detection": { - "decisions": { - "threshold": { - "rationale": f"Affects selection. Anchor: {reference}." - } - } - } - } - } - - -def test_every_astra_value_matches(): - record = load_yaml(REPO_ROOT / "astra.yaml") - anchors = extract_anchors(record) - assert any(" = " in ref for a in anchors for ref in a.references), ( - "astra.yaml contains no value assertions" - ) - errors = value_errors(REPO_ROOT, record) - assert not errors, "ASTRA value mismatches:\n - " + "\n - ".join(errors) - - -def test_anchor_grammar_keeps_values_and_location_only_refs(tmp_path): - (tmp_path / "image.sex").write_text("THRESH 1.25\n", encoding="utf-8") - (tmp_path / "code.py").write_text("def fit():\n pass\n", encoding="utf-8") - record = record_with("image.sex#THRESH = 1.25; code.py::fit; image.sex") - anchor, = extract_anchors(record) - assert anchor.error is None - assert anchor.references == ( - "image.sex#THRESH = 1.25", "code.py::fit", "image.sex" - ) - assert all(resolve_anchor(tmp_path, ref) is None for ref in anchor.references) - assert value_errors(tmp_path, record) == [] - - -@pytest.mark.parametrize( - "actual, expected", - [ - ("1.0", "1"), - ("13.", "13.0"), - ("5e-4", "0.0005"), - ("-1e3", "-1000"), - ("yes", "on"), - ("off", "No"), - ("1", "True"), # ConfigParser.getboolean reads 1/0 as booleans. - ("0", "false"), - ("2.5, 3.5", "[2.50, 3.500]"), - ("XWIN_IMAGE,YWIN_IMAGE", "XWIN_IMAGE, YWIN_IMAGE"), - ("$ROOT/data{number}.fits", '"$ROOT/data{number}.fits"'), - ], -) -def test_equivalent_spellings(tmp_path, actual, expected): - (tmp_path / "config.ini").write_text(f"[SCIENCE]\nKEY = {actual}\n") - ref = f"config.ini#SCIENCE.KEY = {expected}" - assert check_anchor_value(tmp_path, ref) is None - - -@pytest.mark.parametrize("filename", ["config.sex", "model.psfex"]) -@pytest.mark.parametrize( - "actual, expected", [("Y", "True"), ("false", "N"), ("n", "off")] -) -def test_astromatic_formats_accept_y_n(tmp_path, filename, actual, expected): - (tmp_path / filename).write_text(f"KEY {actual}\n") - assert check_anchor_value(tmp_path, f"{filename}#KEY = {expected}") is None - - -@pytest.mark.parametrize( - "filename, contents, selector, expected", - [ - # getboolean raises on Y/N, so the record must not call them True/False. - ("config.ini", "[S]\nKEY = Y\n", "S.KEY", "True"), - ("config.ini", "[S]\nKEY = True\n", "S.KEY", "Y"), - ("config.ini", "[S]\nKEY = 2\n", "S.KEY", "True"), - ("stars.setools", "[RAND_SPLIT:s]\nKEY = Y\n", "RAND_SPLIT:s.KEY", "True"), - ("constants.py", "KEY = True\n", "KEY", "Y"), - ], -) -def test_boolean_words_follow_each_reader( - tmp_path, filename, contents, selector, expected -): - (tmp_path / filename).write_text(contents) - separator = "::" if filename.endswith(".py") else "#" - ref = f"{filename}{separator}{selector} = {expected}" - assert check_anchor_value(tmp_path, ref) is not None - - -@pytest.mark.parametrize( - "actual, expected", - [ - ("1.000001", "1"), - ("1", "True"), - ("0", "False"), - ("51,51", "51"), # No scalar broadcasting for ordinary keys. - ("51,52", "51,51"), - ("1,2", "2,1"), - ("1,1,1", "1,1"), - ("1", "[1]"), - ("map_weight", "MAP_WEIGHT"), - ], -) -def test_normalization_does_not_hide_drift(tmp_path, actual, expected): - (tmp_path / "config.sex").write_text(f"KEY {actual}\n") - problem = check_anchor_value(tmp_path, f"config.sex#KEY = {expected}") - assert "expected" in problem and "actual" in problem - - -@pytest.mark.parametrize( - "filename, line, selector", - [ - ("default.sex", "THRESH 0.0005 # comment", "THRESH"), - ("default.psfex", "PSF_ACCURACY 0.0005", "PSF_ACCURACY"), - ("default.ww", "WEIGHT_MIN = 0.0005", "WEIGHT_MIN"), - ("default.conf", "THRESH 0.0005", "THRESH"), - ("stars.setools", "[RAND_SPLIT:stars]\nRATIO = 0.0005", - "RAND_SPLIT:stars.RATIO"), - ], -) -def test_line_config_formats(tmp_path, filename, line, selector): - (tmp_path / filename).write_text(line + "\n") - ref = f"{filename}#{selector} = 5e-4" - assert resolve_anchor(tmp_path, ref) is None - assert check_anchor_value(tmp_path, ref) is None - - -@pytest.mark.parametrize( - "filename, line, key", - [ - ("columns.param", "VIGNET(51,51)", "VIGNET"), - ("model.psfex", "PSF_SIZE 51,51", "PSF_SIZE"), - ("model.psfex", "PSF_SIZE 51", "PSF_SIZE"), - ], -) -def test_square_stamp_shorthand(tmp_path, filename, line, key): - target = tmp_path / filename - target.write_text(line + " # stamp\n") - for expected in ("51", "51,51", "[51.0, 51]"): - ref = f"{filename}#{key} = {expected}" - assert check_anchor_value(tmp_path, ref) is None - target.write_text(line.replace("51", "53", 1)) - assert check_anchor_value(tmp_path, f"{filename}#{key} = 51") is not None - - -def test_ini_case_sections_defaults_and_interpolation(tmp_path): - (tmp_path / "config.ini").write_text( - "[DEFAULT]\nENABLED = True\n" - "[SCIENCE]\nKey = 30\nKEY = 31\nPATH = $DATA/%s/file\n" - "[OTHER]\nKEY = 99\n" - ) - for selector, value in ( - ("SCIENCE.Key", "30"), ("SCIENCE.KEY", "31"), - ("OTHER.KEY", "99"), ("SCIENCE.ENABLED", "yes"), - ("DEFAULT.ENABLED", "True"), ("SCIENCE.PATH", "$DATA/%s/file"), - ): - ref = f"config.ini#{selector} = {value}" - assert resolve_anchor(tmp_path, ref) is None - assert check_anchor_value(tmp_path, ref) is None - assert check_anchor_value(tmp_path, "config.ini#SCIENCE.key = 31") is not None - assert check_anchor_value(tmp_path, "config.ini#KEY = 31") is not None - - -@pytest.mark.parametrize( - "filename, contents, selector", - [ - ("config.sex", "# KEY 1\nKEY_EXTRA 1\n", "KEY"), - ("config.param", "# VIGNET(51,51)\n", "VIGNET"), - ("config.setools", "[MASK:stars]\n# FLAGS == 0\n", "MASK:stars.FLAGS"), - ], -) -def test_commented_keys_resolve_but_cannot_assert_active_values( - tmp_path, filename, contents, selector -): - (tmp_path / filename).write_text(contents) - assert resolve_anchor(tmp_path, f"{filename}#{selector}") is None - problem = check_anchor_value(tmp_path, f"{filename}#{selector} = 1") - assert "active" in problem - - -@pytest.mark.parametrize( - "filename, contents, selector", - [ - ("config.sex", "KEY 1\nKEY 2\n", "KEY"), - ("config.ini", "[SCIENCE]\nKEY = 1\nKEY = 2\n", "SCIENCE.KEY"), - ("config.setools", "[RAND_SPLIT:s]\nRATIO = 1\nRATIO = 2\n", - "RAND_SPLIT:s.RATIO"), - ], -) -def test_duplicate_settings_fail_closed(tmp_path, filename, contents, selector): - (tmp_path / filename).write_text(contents) - assert check_anchor_value(tmp_path, f"{filename}#{selector} = 2") is not None - - -def test_setools_cuts_keep_operators_and_all_bounds(tmp_path): - config = tmp_path / "stars.setools" - contents = ( - "[MASK:preselect]\nMAG_AUTO < 21\n" - "[MASK:stars]\nMAG_AUTO > 18.\nMAG_AUTO < 22.\nFLAGS == 0\n" - ) - config.write_text(contents) - ref = 'stars.setools#MASK:stars.MAG_AUTO = ["> 18.", "< 22."]' - assert check_anchor_value(tmp_path, ref) is None - flag_ref = 'stars.setools#MASK:stars.FLAGS = "== 0"' - assert check_anchor_value(tmp_path, flag_ref) is None - for altered in ( - contents.replace("> 18.", ">= 18."), - contents.replace("MAG_AUTO < 22.\n", ""), - ): - config.write_text(altered) - assert check_anchor_value(tmp_path, ref) is not None - - -def test_python_literals_and_dict_paths_without_importing(tmp_path): - (tmp_path / "constants.py").write_text( - "raise RuntimeError('must not execute')\n" - "WIDTH: int = 51\nNOISE = 5e-4\n" - "COMPLETENESS = {'exp_split': {'split_exp_runner': " - "dict(expect=121, warn=True)}}\n" - "class Model:\n WIDTH = 53\n" - "def fit():\n" - " limits = [-1.0, 1.0e3]\n" - " options = {'step': 0.01, 'dynamic': choose_at_runtime()}\n" - ) - for selector, expected in ( - ("WIDTH", "51.0"), ("NOISE", "0.0005"), ("Model.WIDTH", "53"), - ("fit.limits", "-1,1000"), ("fit.options[step]", "1e-2"), - ("COMPLETENESS[exp_split.split_exp_runner.expect]", "121"), - ("COMPLETENESS[exp_split.split_exp_runner.warn]", "True"), - ): - ref = f"constants.py::{selector} = {expected}" - assert resolve_anchor(tmp_path, ref) is None - assert check_anchor_value(tmp_path, ref) is None - missing = "constants.py::fit.options[missing]" - assert resolve_anchor(tmp_path, missing) is not None - - -@pytest.mark.parametrize( - "contents, selector", - [ - ("X = get_value()", "X"), - ("X = 1 / 3", "X"), - ("X = 1\nX = 2", "X"), - ("X = 1\nX += 1", "X"), - ("X, Y = 1, 2", "X"), - ("def X():\n return 1", "X"), - ("X = {'a': 1, 'a': 2}", "X[a]"), - ("X = dict(**other)", "X[a]"), - ("X = {'a': 1, **other}", "X[a]"), - ("X = {variable: 1}", "X[a]"), - ("X = {'a': 1}\nX['a'] = 2", "X[a]"), - ("X = {'a': 1}\nX['b']['c'] = 2", "X[a]"), - ("X = {'a': 1}\nX['a'] += 1", "X[a]"), - ("X = {'a': 1}\ndel X['a']", "X[a]"), - ("X = 1\nX.attr = 2", "X"), - ("def f():\n X = {'a': 1}\n X['a'] = 2", "f.X[a]"), - ], -) -def test_python_nonliteral_or_ambiguous_values_fail_closed( - tmp_path, contents, selector -): - (tmp_path / "constants.py").write_text(contents + "\n") - ref = f"constants.py::{selector} = 1" - assert check_anchor_value(tmp_path, ref) is not None - - -@pytest.mark.parametrize( - "reference", - [ - "config.sex#KEY =", "config.sex#KEY == 1", "config.sex = 1", - "config.sex#KEY = [1,", "config.sex#KEY = 1,,2", - "config.sex#KEY = {'value': 1}", "config.sex#KEY = [[1]]", - "../outside.sex#KEY = 1", "/outside.sex#KEY = 1", - "config.sex#KEY = 1; config.sex#KEY = 2", - ], -) -def test_malformed_assertions_are_errors(tmp_path, reference): - (tmp_path / "config.sex").write_text("KEY 1\n") - assert check_anchor_value(tmp_path, reference) is not None - - -def test_injected_drift_reports_decision_ref_expected_actual(tmp_path): - config = tmp_path / "detect.sex" - config.write_text("DETECT_THRESH 1.0\n") - reference = "detect.sex#DETECT_THRESH = 1" - record = record_with(reference) - assert value_errors(tmp_path, record) == [] - - config.write_text("DETECT_THRESH 1.5\n") - assert resolve_anchor(tmp_path, reference) is None - error, = value_errors(tmp_path, record) - for detail in ( - "analyses.detection.decisions.threshold", "detect.sex#DETECT_THRESH", - "expected", "1", "actual", "1.5", - ): - assert detail in error - - -def test_python_drift_is_not_hidden_by_a_cached_ast(tmp_path): - target = tmp_path / "constants.py" - reference = "constants.py::WIDTH = 51" - target.write_text("WIDTH = 51\n") - assert check_anchor_value(tmp_path, reference) is None - target.write_text("WIDTH = 53\n") - assert resolve_anchor(tmp_path, reference) is None - assert check_anchor_value(tmp_path, reference) is not None - - -def test_malformed_anchor_sentence_cannot_silently_skip_values(tmp_path): - record = {"decisions": {"width": {"rationale": "Anchor: config.sex#KEY = 1"}}} - assert value_errors(tmp_path, record) - - -@pytest.mark.parametrize( - "filename, contents, selector", - [ - ("config.sex", "SATUR_KEY SATURATE\n# SATUR_LEVEL 50000\n", "SATUR_LEVEL"), - ("config.ini", "[DEFAULT]\nA = 1\n[S]\nKEY = 1\n", "S.OTHER"), - ("stars.setools", "[MASK:other]\nMASK_EXT == 0\n[MASK:stars]\n" - "IMAFLAGS_ISO == 0\n# MASK_EXT == 0\n", "MASK:stars.MASK_EXT"), - ], -) -def test_absent_passes_when_no_active_setting(tmp_path, filename, contents, selector): - (tmp_path / filename).write_text(contents) - ref = f"{filename}#{selector} = absent" - assert resolve_anchor(tmp_path, ref) is None - assert check_anchor_value(tmp_path, ref) is None - - -@pytest.mark.parametrize( - "filename, contents, selector", - [ - ("config.sex", "SATUR_LEVEL 50000\n", "SATUR_LEVEL"), - ("config.sex", "SATUR_LEVEL\n", "SATUR_LEVEL"), - ("config.ini", "[S]\nKEY = 1\n", "S.KEY"), - ("config.ini", "[DEFAULT]\nKEY = 1\n[S]\n", "S.KEY"), - ("stars.setools", "[MASK:stars]\nIMAFLAGS_ISO == 0\nMASK_EXT == 0\n", - "MASK:stars.MASK_EXT"), - ], -) -def test_absent_fails_on_an_active_setting(tmp_path, filename, contents, selector): - (tmp_path / filename).write_text(contents) - problem = check_anchor_value(tmp_path, f"{filename}#{selector} = absent") - assert "expected no active setting" in problem - - -@pytest.mark.parametrize( - "filename, contents, reference", - [ - # A renamed section must not make every absence trivially true. - ("config.ini", "[S]\nKEY = 1\n", "config.ini#RENAMED.KEY = absent"), - ("stars.setools", "[MASK:stars]\nFLAGS == 0\n", - "stars.setools#MASK:renamed.MASK_EXT = absent"), - ("constants.py", "X = 1\n", "constants.py::Y = absent"), - ("missing.sex", None, "missing.sex#KEY = absent"), - ], -) -def test_absent_needs_a_real_file_and_section(tmp_path, filename, contents, reference): - if contents is not None: - (tmp_path / filename).write_text(contents) - assert resolve_anchor(tmp_path, reference) is not None - assert check_anchor_value(tmp_path, reference) is not None - - -def test_quoted_absent_is_ordinary_text(tmp_path): - (tmp_path / "config.sex").write_text("KEY absent\n") - assert check_anchor_value(tmp_path, 'config.sex#KEY = "absent"') is None - assert check_anchor_value(tmp_path, "config.sex#KEY = absent") is not None - - -@pytest.mark.parametrize( - "path, pattern, replacement", - [ - ("workflow/config/cfis/config_tile_Sx.ini", - "WEIGHT_IMAGE = True", "WEIGHT_IMAGE = False"), - ("workflow/config/cfis/config_tile_Sx.ini", - "MAKE_POST_PROCESS = True", "MAKE_POST_PROCESS = False"), - ("workflow/config/cfis/star_selection.setools", - "[MASK:star_selection]\n", "[MASK:star_selection]\nMASK_EXT == 0\n"), - ("workflow/config/cfis/default_tile.sex", - "SATUR_KEY", "SATUR_LEVEL 50000\nSATUR_KEY"), - ("src/shapepipe/modules/ngmix_package/ngmix.py", - "\n boot = ngmix.metacal.", - "\n metacal_pars['step'] = 0.02\n boot = ngmix.metacal."), - ], -) -def test_real_record_catches_gate_and_absence_drift( - tmp_path, path, pattern, replacement -): - """Each mutation once passed the record; each must now be reported.""" - - record = load_yaml(REPO_ROOT / "astra.yaml") - for ref in { - ref.split(" = ", 1)[0].split("#", 1)[0].split("::", 1)[0] - for anchor in extract_anchors(record) for ref in anchor.references - }: - source = REPO_ROOT / ref - if source.is_file(): - target = tmp_path / ref - target.parent.mkdir(parents=True, exist_ok=True) - target.write_text(source.read_text(encoding="utf-8")) - assert value_errors(tmp_path, record) == [] - - target = tmp_path / path - text = target.read_text(encoding="utf-8") - assert text.count(pattern) == 1 - target.write_text(text.replace(pattern, replacement)) - assert value_errors(tmp_path, record) diff --git a/tests/unit/test_contracts.py b/tests/unit/test_contracts.py index cb5f3f12c..4091adb5e 100644 --- a/tests/unit/test_contracts.py +++ b/tests/unit/test_contracts.py @@ -1,41 +1,13 @@ -"""Keep the @sc contracts well-formed and tied to the ASTRA decision record.""" +"""Preserve the utilities import-boundary contract.""" import textwrap -from functools import cache from pathlib import Path -import pytest +from tests.helpers.decisions import forbid_rules, import_violations -from tests.helpers.astra_record import load_yaml -from tests.helpers.contracts import ( - DecisionMarker, - collect, - coverage_report, - decision_errors, - decision_ids, - decision_marker_errors, - decision_markers, - forbid_rules, - governed_refs, - governs_errors, - import_violations, -) REPO_ROOT = Path(__file__).resolve().parents[2] -RECORD = { - "decisions": {"top_choice": {}}, - "analyses": {"stage": {"decisions": {"inner_choice": {}}}}, -} - - -@cache -def _repository(): - record = load_yaml(REPO_ROOT / "astra.yaml") - contracts, errors = collect(REPO_ROOT) - markers, marker_errors = decision_markers(REPO_ROOT) - return record, contracts, markers, errors + marker_errors - def _write(root, relative, text): path = root / relative @@ -43,351 +15,6 @@ def _write(root, relative, text): path.write_text(textwrap.dedent(text), encoding="utf-8") -def test_parser_reads_docstrings_contracts_files_and_snakemake(tmp_path): - _write(tmp_path, "src/pkg/mod.py", ''' - """Module.""" - - - class Thing: - def method(self): - """Do it. - - @sc [decision:stage.inner_choice,label:convention] method-rule - The method keeps its promise. - Second prose line. - - Returns - ------- - None - """ - ''') - _write(tmp_path, "src/pkg/CONTRACTS", """ - @cc boundary-rule - forbid: pkg.a.* -> pkg.b.* - """) - _write(tmp_path, "workflow/rules/x.smk", """ - # @sc [decision:top_choice] rule-contract - # The rule keeps its promise. - - rule x: - output: "a" - """) - - contracts, errors = collect(tmp_path) - - assert errors == [] - by_id = {c.id: c for c in contracts} - assert set(by_id) == {"method-rule", "boundary-rule", "rule-contract"} - method = by_id["method-rule"] - assert method.scope == "Thing.method" - assert method.meta == { - "decision": "stage.inner_choice", - "label": "convention", - } - assert method.prose == "The method keeps its promise. Second prose line." - assert method.line == 9 - assert by_id["boundary-rule"].scope == "src/pkg" - assert decision_errors(contracts, RECORD) == [] - - -def test_malformed_meta_and_missing_id_are_errors(tmp_path): - _write(tmp_path, "src/mod.py", ''' - def f(): - """F. - - @sc [decision stage.inner_choice] bad-meta - Prose. - - @sc [label:x] - Prose. - - @sc two words - Prose. - """ - ''') - - contracts, errors = collect(tmp_path) - - assert contracts == [] - assert len(errors) == 3 - assert "malformed contract metadata" in errors[0] - assert "missing contract id" in errors[1] - assert "malformed contract line" in errors[2] - - -def test_duplicate_ids_and_comment_contracts_are_errors(tmp_path): - _write(tmp_path, "src/a.py", ''' - def f(): - """F. - - @sc same-id - Prose. - """ - ''') - _write(tmp_path, "scripts/b.py", ''' - def g(): - """G. - - @sc same-id - Prose. - """ - # @sc hidden-id - return None - ''') - - _, errors = collect(tmp_path) - - assert any("duplicate contract id same-id" in e for e in errors) - assert any("contract in a comment" in e for e in errors) - - -def test_unknown_decision_is_an_error(tmp_path): - _write(tmp_path, "src/mod.py", ''' - def f(): - """F. - - @sc [decision:inner_choice] undotted - Sub-analysis decisions need their analysis prefix. - - @sc [decision:no_such_choice] unknown - Prose. - """ - ''') - - contracts, errors = collect(tmp_path) - - assert errors == [] - problems = decision_errors(contracts, RECORD) - assert len(problems) == 2 - assert "'inner_choice'" in problems[0] - assert "'no_such_choice'" in problems[1] - assert decision_ids(RECORD) == {"top_choice", "stage.inner_choice"} - - -def test_parser_reads_decision_markers_from_decorators_and_pytestmark( - tmp_path, -): - """Find IDs in decorators and module-level pytestmark lists.""" - _write(tmp_path, "tests/test_markers.py", ''' - import pytest - - pytestmark = [pytest.mark.decision("top_choice")] - - @pytest.mark.decision("stage.inner_choice", "top_choice") - def test_it(): - pass - ''') - - markers, errors = decision_markers(tmp_path) - - assert errors == [] - assert [(m.decision, m.path, m.line) for m in markers] == [ - ("top_choice", "tests/test_markers.py", 4), - ("stage.inner_choice", "tests/test_markers.py", 6), - ("top_choice", "tests/test_markers.py", 6), - ] - assert decision_marker_errors(markers, RECORD) == [] - - -def test_unknown_decision_marker_is_an_error(tmp_path): - """Reject decision IDs missing from the ASTRA record.""" - _write(tmp_path, "tests/test_bad_marker.py", ''' - import pytest - pytestmark = [pytest.mark.decision("missing_choice")] - ''') - - markers, errors = decision_markers(tmp_path) - - assert errors == [] - assert decision_marker_errors(markers, RECORD) == [ - "tests/test_bad_marker.py:3: decision marker cites unknown decision " - "'missing_choice'" - ] - - -def test_governs_resolves_all_refs_relative_to_the_contract_file(tmp_path): - """A multi-key coupling must not lose refs or resolve them from cwd.""" - - base = "workflow/config/cfis" - _write(tmp_path, f"{base}/default.sex", "DEBLEND_MINCONT 0.002\n") - _write(tmp_path, f"{base}/default.psfex", "PSF_SIZE 51,51\n") - _write(tmp_path, f"{base}/stamps.ini", "[STAMP]\nSIZE = 51\n") - _write(tmp_path, f"{base}/default.param", "VIGNET(51,51)\n") - _write(tmp_path, f"{base}/stars.setools", "[MASK:stars]\nFLAGS == 0\n") - _write(tmp_path, f"{base}/kernel.conv", "CONV NORM\n1 2 1\n") - # Bare-file refs also cover formats the shared resolver cannot select into. - _write(tmp_path, "workflow/config.yaml", "psf_model: psfex\n") - refs = ( - "default.sex#DEBLEND_MINCONT", - "default.psfex#PSF_SIZE", - "stamps.ini#STAMP.SIZE", - "default.param#VIGNET", - "stars.setools#MASK:stars.FLAGS", - "kernel.conv", - ) - _write(tmp_path, f"{base}/CONTRACTS", f""" - @sc [decision:top_choice,governs:{';'.join(refs)}] coupled-config - Keep the apertures coupled. - Ordinary prose is not an import rule. - """) - _write(tmp_path, "workflow/CONTRACTS", """ - @sc [decision:stage.inner_choice,governs:config.yaml] model-choice - Select matching exposure and tile models. - """) - - contracts, errors = collect(tmp_path) - by_id = {c.id: c for c in contracts} - coupled = by_id["coupled-config"] - - assert errors == [] - assert coupled.meta["governs"] == ";".join(refs) - assert coupled.scope == base - assert coupled.line == 2 - assert coupled.prose == ( - "Keep the apertures coupled. Ordinary prose is not an import rule." - ) - assert governed_refs(coupled) == tuple(f"{base}/{ref}" for ref in refs) - assert governs_errors(contracts, tmp_path) == [] - assert decision_errors(contracts, RECORD) == [] - assert forbid_rules(tmp_path / base / "CONTRACTS") == [] - - -@pytest.mark.parametrize("bad_ref", [ - "missing.sex#KEY", - "default.sex#MISSING", - "stamps.ini#STAMP.MISSING", - "stamps.ini#SIZE", # INI selectors need SECTION.KEY. - "stars.setools#MASK:missing.FLAGS", - "../default.sex#KEY", # No escape from the governing directory. - "/absolute/default.sex#KEY", -]) -@pytest.mark.parametrize("bad_first", [True, False]) -def test_every_unresolvable_governs_ref_is_an_error( - tmp_path, bad_ref, bad_first -): - """Checking only the first/last ref silently loses part of a coupling.""" - - base = "workflow/config" - _write(tmp_path, f"{base}/default.sex", "KEY 1\n") - _write(tmp_path, f"{base}/stamps.ini", "[STAMP]\nSIZE = 51\n") - _write(tmp_path, f"{base}/stars.setools", "[MASK:stars]\nFLAGS == 0\n") - refs = [bad_ref, "default.sex#KEY"] - if not bad_first: - refs.reverse() - _write(tmp_path, f"{base}/CONTRACTS", f""" - @sc [governs:{';'.join(refs)}] broken-coupling - Prose. - """) - - contracts, errors = collect(tmp_path) - problems = governs_errors(contracts, tmp_path) - - assert errors == [] - assert len(problems) == 1 - assert f"{base}/CONTRACTS:2" in problems[0] - assert "broken-coupling" in problems[0] - assert bad_ref in problems[0] - - -@pytest.mark.parametrize("metadata, message", [ - ("governs:", "malformed contract metadata"), - ("governs:a.sex#KEY b.sex#KEY", "malformed contract metadata"), - ("governs:a.sex#KEY,b.sex#KEY", "malformed contract metadata"), - ("governs:;a.sex#KEY", "empty governs ref"), - ("governs:a.sex#KEY;", "empty governs ref"), - ("governs:a.sex#KEY;;b.sex#KEY", "empty governs ref"), - ("governs:a.sex#KEY,governs:b.sex#KEY", "duplicate metadata key"), -]) -def test_malformed_governs_does_not_silently_drop_refs( - tmp_path, metadata, message -): - _write(tmp_path, "workflow/config/CONTRACTS", f""" - @sc [{metadata}] malformed-coupling - Prose. - """) - - contracts, errors = collect(tmp_path) - - assert contracts == [] - assert len(errors) == 1 - assert "workflow/config/CONTRACTS:2" in errors[0] - assert message in errors[0] - - -@pytest.mark.parametrize("assertion", ["", " = 51"]) -def test_config_coverage_uses_governed_refs_not_the_sidecar_path( - tmp_path, assertion -): - _write(tmp_path, "workflow/config/CONTRACTS", """ - @sc [decision:top_choice,governs:a.sex#UNRECORDED;b.ini#S.SIZE] config-coupling - Prose. - - @sc [decision:stage.inner_choice,governs:a.sex#OTHER] off-record-key - A different key in the same file is not the anchored key. - - @sc [governs:config.yaml] whole-file - Prose. - """) - record = { - "decisions": { - "top_choice": { - "rationale": f"Anchor: workflow/config/b.ini#S.SIZE{assertion}." - }, - "uncovered": {}, - }, - "analyses": {"stage": {"decisions": {"inner_choice": { - "rationale": "Anchor: workflow/config/a.sex#KEY." - }}}}, - "description": "Anchor: workflow/config/config.yaml.", - } - - contracts, errors = collect(tmp_path) - uncovered, unanchored = coverage_report(contracts, record) - covered, _ = coverage_report( - contracts, - record, - [DecisionMarker("uncovered", "tests/science/test_x.py", 1)], - ) - - assert errors == [] - assert uncovered == ["uncovered"] - assert covered == [] - assert [c.id for c in unanchored] == ["off-record-key"] - - -def test_repository_contracts_are_valid_and_cite_real_decisions(): - record, contracts, markers, errors = _repository() - errors = ( - errors - + decision_errors(contracts, record) - + decision_marker_errors(markers, record) - + governs_errors(contracts, REPO_ROOT) - ) - - message = "Contract problems:\n - " + "\n - ".join(errors) - assert not errors, message - - -def test_contract_coverage_report(): - """Report ASTRA decision gaps and unanchored contracts.""" - record, contracts, markers, _ = _repository() - uncovered, unanchored = coverage_report(contracts, record, markers) - - print( - f"\n{len(contracts)} contracts; {len(markers)} test decision markers; " - f"{len(uncovered)} decisions cited by neither:" - ) - for decision in uncovered: - print(f" {decision}") - print(f"{len(unanchored)} @sc contracts off the record's anchors:") - for contract in unanchored: - targets = governed_refs(contract) - where = "; ".join(targets) if targets else ( - f"{contract.path}::{contract.scope}" - ) - print(f" {contract.id} at {where}") - - def test_forbidden_import_is_found(tmp_path): _write(tmp_path, "pkg/utilities/CONTRACTS", """ @cc no-up-imports @@ -402,7 +29,8 @@ def test_forbidden_import_is_found(tmp_path): assert rules == [("no-up-imports", "pkg.utilities.*", "pkg.modules.*")] assert len(violations) == 2 - assert all("bad.py:1" in v and "no-up-imports" in v for v in violations) + assert all("bad.py:1" in problem and "no-up-imports" in problem + for problem in violations) def test_utilities_do_not_import_modules(): @@ -411,6 +39,5 @@ def test_utilities_do_not_import_modules(): assert [rule[0] for rule in rules] == ["utilities-do-not-import-modules"] violations = import_violations(REPO_ROOT / "src", rules) - message = "Forbidden imports:\n - " + "\n - ".join(violations) assert not violations, message diff --git a/tests/unit/test_decisions.py b/tests/unit/test_decisions.py new file mode 100644 index 000000000..42e3a7428 --- /dev/null +++ b/tests/unit/test_decisions.py @@ -0,0 +1,366 @@ +"""Decision-tag parsing and the static Values resolver.""" + +from pathlib import Path + +import pytest + +from tests.helpers.decisions import ( + decision_ids, + decision_marker_errors, + decision_markers, + forbid_rules, + import_violations, + load_yaml, + main, + scan_tags, + tag_errors, + value_errors, +) + + +REPO_ROOT = Path(__file__).resolve().parents[2] + + +def _record(rationale="Values: THRESH = 1.", decision="choice"): + return {"decisions": {decision: {"label": "Choice", "rationale": rationale}}} + + +def _write(path, text): + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(text, encoding="utf-8") + return path + + +def _config_tag(path, decision="choice", content="THRESH 1\n"): + return _write(path, f"# @sc [decision:{decision}]\n{content}") + + +def test_config_tags_govern_paragraph_and_section(tmp_path): + config = _write( + tmp_path / "settings.ini", + "# @sc [decision:choice]\n[SCIENCE]\nKEY = 1\n# comment\nOTHER = 2\n\n" + "# @sc [decision:choice]\n[OUTPUT]\nSAVE = True\n", + ) + tags, errors = scan_tags(tmp_path) + + assert errors == [] + assert len(tags) == 2 + science, output = tags + assert science.site.path == "settings.ini" + assert (science.site.start, science.site.end) == (2, 6) + assert science.site.section == "SCIENCE" + assert output.site.section == "OUTPUT" + assert output.site.scope == "section" + + +def test_python_declaration_statement_and_snakemake_tags(tmp_path): + _write( + tmp_path / "src" / "mod.py", + '"""Module."""\n\n' + "# @sc [decision:choice]\nWIDTH = 51\n\n" + "def fit():\n" + ' """Fit.\n\n' + " @sc [decision:choice,label:coupling] fit-coupling\n" + " The local fit constraint is preserved.\n" + ' """\n' + " SCALE = 1.0\n", + ) + _write( + tmp_path / "workflow" / "rules.smk", + "# @sc [decision:choice]\nrule detect:\n output: 'catalogue'\n\n", + ) + tags, errors = scan_tags(tmp_path) + + assert errors == [] + assert {(tag.site.kind, tag.site.symbol) for tag in tags} == { + ("python_statement", "WIDTH"), + ("python_declaration", "fit"), + ("snakemake", ""), + } + contract = next(tag for tag in tags if tag.ident) + assert contract.ident == "fit-coupling" + assert contract.prose == "The local fit constraint is preserved." + + +def test_tag_grammar_rejects_bad_metadata_missing_sites_and_duplicate_contract_ids( + tmp_path, +): + _write( + tmp_path / "a.sex", + "# @sc [decision choice] malformed\nKEY 1\n\n" + "# @sc [decision:missing,label:x] same-id\n# Local prose.\nKEY 2\n\n" + "# @sc [decision:choice,label:x] same-id\n# Local prose.\nKEY 3\n\n" + "# @sc [decision:choice]", + ) + tags, errors = scan_tags(tmp_path) + + assert len(tags) == 2 + assert any("malformed @sc metadata" in error for error in errors) + assert any("duplicate local-contract id same-id" in error for error in errors) + assert any("same-id" in error for error in errors) + assert any("@sc tag governs no site" in error for error in errors) + + +def test_decision_citations_are_checked_in_both_directions(tmp_path): + record = { + "decisions": {"top": {}, "orphan": {}}, + "analyses": {"stage": {"decisions": {"inner": {}}}}, + } + _config_tag(tmp_path / "a.sex", "top") + _config_tag(tmp_path / "b.sex", "stage.inner") + _config_tag(tmp_path / "c.sex", "not-real") + tags, parse_errors = scan_tags(tmp_path) + + assert parse_errors == [] + assert decision_ids(record) == {"top", "orphan", "stage.inner"} + errors = tag_errors(tags, record) + assert any("unknown decision 'not-real'" in error for error in errors) + assert any("orphan: decision has no tagged site" in error for error in errors) + + +def test_repeated_decision_metadata_cites_multiple_real_decisions(tmp_path): + _write( + tmp_path / "shared.sex", + "# @sc [decision:first,decision:second]\nTHRESH 1\n", + ) + record = {"decisions": {"first": {}, "second": {}}} + tags, errors = scan_tags(tmp_path) + + assert errors == [] + assert tags[0].decisions == ("first", "second") + assert tag_errors(tags, record) == [] + + +def test_value_change_is_reported_with_decision_ref_expected_and_actual(tmp_path): + _config_tag(tmp_path / "detect.sex", content="THRESH 1.5\n") + record = _record("Detection. Values: THRESH = 1.") + tags, errors = scan_tags(tmp_path) + + assert errors == [] + problems = value_errors(tmp_path, record, tags) + assert len(problems) == 1 + for detail in ("choice", "THRESH", "expected", "1", "actual", "1.5"): + assert detail in problems[0] + + +def test_key_movement_inside_tagged_paragraph_preserves_value(tmp_path): + config = _config_tag( + tmp_path / "detect.sex", + content="# explanatory comment\nOTHER 2\nTHRESH 1\n", + ) + record = _record("Detection. Values: THRESH = 1.") + tags, errors = scan_tags(tmp_path) + + assert errors == [] + assert value_errors(tmp_path, record, tags) == [] + config.write_text( + "# @sc [decision:choice]\nTHRESH 1\nOTHER 2\n", encoding="utf-8" + ) + tags, errors = scan_tags(tmp_path) + assert errors == [] + assert value_errors(tmp_path, record, tags) == [] + + +def test_removed_tag_orphans_decision(tmp_path): + config = _config_tag(tmp_path / "detect.sex") + record = _record() + tags, _ = scan_tags(tmp_path) + assert tag_errors(tags, record) == [] + + config.write_text("THRESH 1\n", encoding="utf-8") + tags, _ = scan_tags(tmp_path) + assert tag_errors(tags, record) == ["choice: decision has no tagged site"] + + +def test_equal_values_at_multiple_sites_still_require_qualified_refs(tmp_path): + _config_tag(tmp_path / "default.param", content="VIGNET(51,51)\n") + _config_tag(tmp_path / "default_noimaflags.param", content="VIGNET(51,51)\n") + record = _record("Stamps. Values: VIGNET = 51.") + tags, errors = scan_tags(tmp_path) + + assert errors == [] + problem, = value_errors(tmp_path, record, tags) + assert "exactly one tagged site" in problem + assert "2" in problem + + +def test_deleting_one_of_two_vignet_tagged_sites_fails_its_assertion(tmp_path): + one = _config_tag(tmp_path / "default.param", content="VIGNET(51,51)\n") + two = _config_tag(tmp_path / "default_noimaflags.param", content="VIGNET(51,51)\n") + record = _record( + "Stamps. Values: default.param#VIGNET = 51; " + "default_noimaflags.param#VIGNET = 51." + ) + tags, errors = scan_tags(tmp_path) + assert errors == [] + assert value_errors(tmp_path, record, tags) == [] + + two.unlink() + tags, errors = scan_tags(tmp_path) + assert not errors + problems = value_errors(tmp_path, record, tags) + assert len(problems) == 1 + assert "default_noimaflags.param#VIGNET" in problems[0] + assert "found 0" in problems[0] + assert one.exists() + + +def test_absent_key_requires_tagged_scope_and_detects_added_key(tmp_path): + config = _write( + tmp_path / "settings.sex", + "# @sc [decision:choice,scope:file]\nOTHER 2\n", + ) + record = _record("No fixed threshold. Values: settings.sex#THRESH = absent.") + tags, errors = scan_tags(tmp_path) + + assert errors == [] + assert value_errors(tmp_path, record, tags) == [] + + config.write_text( + "# @sc [decision:choice,scope:file]\nOTHER 2\nTHRESH 1\n", + encoding="utf-8", + ) + tags, errors = scan_tags(tmp_path) + assert errors == [] + assert "expected no active setting" in value_errors(tmp_path, record, tags)[0] + + +def test_absent_assertion_fails_when_its_scope_is_removed(tmp_path): + config = _write( + tmp_path / "settings.ini", + "# @sc [decision:choice]\n[SCIENCE]\nOTHER = 2\n\n", + ) + record = _record("No fixed threshold. Values: SCIENCE.THRESH = absent.") + tags, errors = scan_tags(tmp_path) + assert errors == [] + assert value_errors(tmp_path, record, tags) == [] + + # The tag no longer governs the named section; it cannot assert absence. + config.write_text( + "# @sc [decision:choice]\n[OTHER]\nKEY = 2\n", encoding="utf-8" + ) + tags, errors = scan_tags(tmp_path) + assert errors == [] + problem, = value_errors(tmp_path, record, tags) + assert "SCIENCE.THRESH" in problem + assert "found 0" in problem + + +def test_numeric_lists_and_astromatic_boolean_words_keep_reader_semantics(tmp_path): + _config_tag( + tmp_path / "values.sex", + content="THRESH 5e-4\nAPERTURE 2.5, 3.5\nFLAG Y\n", + ) + record = _record( + "Values. Values: THRESH = 0.0005; APERTURE = [2.50, 3.500]; FLAG = True." + ) + tags, errors = scan_tags(tmp_path) + + assert errors == [] + assert value_errors(tmp_path, record, tags) == [] + + +def test_config_boolean_semantics_and_setools_predicates(tmp_path): + ini = _write(tmp_path / "config.ini", "# @sc [decision:choice]\n[SCIENCE]\nENABLED = 1\n") + record = _record("Toggle. Values: SCIENCE.ENABLED = True.") + tags, errors = scan_tags(tmp_path) + assert errors == [] + assert value_errors(tmp_path, record, tags) == [] + + ini.write_text("# @sc [decision:choice]\n[SCIENCE]\nENABLED = Y\n", encoding="utf-8") + tags, _ = scan_tags(tmp_path) + assert value_errors(tmp_path, record, tags) + ini.write_text("# @sc [decision:choice]\n[SCIENCE]\nENABLED = False\n", encoding="utf-8") + tags, _ = scan_tags(tmp_path) + assert value_errors(tmp_path, record, tags) + + setools = _write( + tmp_path / "stars.setools", + "# @sc [decision:choice]\n[MASK:stars]\nMAG_AUTO > 18.\nMAG_AUTO < 22.\n", + ) + cuts = _record('Star cut. Values: stars.setools#MASK:stars.MAG_AUTO = ["> 18.", "< 22."].') + tags, errors = scan_tags(tmp_path) + assert errors == [] + assert value_errors(tmp_path, cuts, tags) == [] + setools.write_text( + "# @sc [decision:choice]\n[MASK:stars]\nMAG_AUTO >= 18.\nMAG_AUTO < 22.\n", + encoding="utf-8", + ) + tags, _ = scan_tags(tmp_path) + assert value_errors(tmp_path, cuts, tags) + + +def test_python_value_reassignment_fails_closed(tmp_path): + _write( + tmp_path / "constants.py", + 'def fit():\n """Fit.\n\n @sc [decision:choice]\n """\n' + " WIDTH = 51\n WIDTH = 53\n", + ) + record = _record("Stamp. Values: WIDTH = 51.") + tags, errors = scan_tags(tmp_path) + assert errors == [] + problem, = value_errors(tmp_path, record, tags) + assert "exactly one tagged site" in problem + + +def test_python_literal_and_dict_selector_is_static(tmp_path): + _write( + tmp_path / "constants.py", + "raise RuntimeError('must not execute')\n" + "# @sc [decision:choice]\nCOMPLETENESS = {'run': {'expect': 40, 'warn': True}}\n", + ) + record = _record("Counts. Values: COMPLETENESS[run.expect] = 40.") + tags, errors = scan_tags(tmp_path) + assert errors == [] + assert value_errors(tmp_path, record, tags) == [] + + +def test_decision_markers_keep_unknown_id_check(tmp_path): + _write( + tmp_path / "tests" / "test_markers.py", + "import pytest\npytestmark = [pytest.mark.decision('choice')]\n" + "@pytest.mark.decision('missing')\ndef test_it():\n pass\n", + ) + markers, errors = decision_markers(tmp_path) + + assert errors == [] + assert [(marker.decision, marker.line) for marker in markers] == [ + ("choice", 2), ("missing", 3) + ] + assert decision_marker_errors(markers, _record()) == [ + "tests/test_markers.py:3: decision marker cites unknown decision 'missing'" + ] + + +def test_cli_reports_decision_and_local_contract_for_a_location(tmp_path, capsys): + _write(tmp_path / "astra.yaml", 'decisions:\n choice:\n label: Choice\n rationale: "First sentence. Values: THRESH = 1."\n') + _write( + tmp_path / "detect.sex", + "# @sc [decision:choice,label:coupling] threshold-coupling\n" + "# The threshold must remain coupled.\nTHRESH 1\n", + ) + + assert main(["detect.sex:3", "--root", str(tmp_path)]) == 0 + output = capsys.readouterr().out + assert "choice — Choice" in output + assert "First sentence." in output + assert "Values: THRESH = 1." in output + assert "threshold-coupling" in output + assert "The threshold must remain coupled." in output + + +def test_preserved_utilities_import_rule(tmp_path): + contracts = _write( + tmp_path / "pkg" / "utilities" / "CONTRACTS", + "@cc no-up-imports\nforbid: pkg.utilities.* -> pkg.modules.*\n", + ) + _write(tmp_path / "pkg" / "utilities" / "bad.py", "from ..modules import runner\n") + rules = forbid_rules(contracts) + assert rules == [("no-up-imports", "pkg.utilities.*", "pkg.modules.*")] + assert len(import_violations(tmp_path, rules)) == 2 + + +def test_decision_ids_cover_nested_analysis_paths(): + assert decision_ids( + {"decisions": {"outer": {}}, "analyses": {"child": {"decisions": {"inner": {}}}}} + ) == {"outer", "child.inner"} From b57139fcb6322c815e9c0b0690226225e28551c8 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 04:42:43 +0200 Subject: [PATCH 25/40] fix(decisions): tighten tag discovery and section scopes --- tests/helpers/decisions.py | 16 +++++++++------- tests/unit/test_decisions.py | 28 ++++++++++++++++++++++++++++ 2 files changed, 37 insertions(+), 7 deletions(-) diff --git a/tests/helpers/decisions.py b/tests/helpers/decisions.py index f55fcf2f8..50d3cc15d 100644 --- a/tests/helpers/decisions.py +++ b/tests/helpers/decisions.py @@ -174,7 +174,7 @@ def _python_docstring_tags(path, source, tree): continue doc_node = body[0].value for offset, doc_line in enumerate(doc_node.value.splitlines()): - if "@sc" not in doc_line: + if not doc_line.strip().startswith("@sc"): continue line_no = doc_node.lineno + offset consumed.add(line_no) @@ -258,12 +258,14 @@ def _comment_site(path, lines, index, meta, tree=None, *, snakemake=False): if header: end = len(lines) for j in range(n + 1, len(lines)): - comment = _comment_body(lines[j]) - if comment is not None and comment.startswith("@sc"): - end = j - break if re.match(r"^\s*\[[^]]+\]\s*(?:[#;].*)?$", lines[j]): end = j + k = j - 1 + while k > n and _comment_body(lines[k]) is not None: + if _comment_body(lines[k]).startswith("@sc"): + end = k + break + k -= 1 break return Site(path, start, end, "config", section=header.group(1).strip(), scope="section") end = len(lines) @@ -310,13 +312,13 @@ def _parse_file_tags(path, root): tag_comments = [ (line_no, column, comment[1:].lstrip()) for line_no, (column, comment) in comment_tokens.items() - if "@sc" in comment + if comment[1:].lstrip().startswith("@sc") ] else: tag_comments = [ (index + 1, len(line) - len(line.lstrip()), body) for index, line in enumerate(lines) - if (body := _comment_body(line)) is not None and "@sc" in body + if (body := _comment_body(line)) is not None and body.startswith("@sc") ] for line_no, column, body in tag_comments: index = line_no - 1 diff --git a/tests/unit/test_decisions.py b/tests/unit/test_decisions.py index 42e3a7428..3e48ecc1a 100644 --- a/tests/unit/test_decisions.py +++ b/tests/unit/test_decisions.py @@ -53,6 +53,34 @@ def test_config_tags_govern_paragraph_and_section(tmp_path): assert output.site.scope == "section" +def test_section_tag_covers_the_entire_section_across_nested_key_tags(tmp_path): + _write( + tmp_path / "settings.ini", + "# @sc [decision:section_choice]\n[S]\nA = 1\n" + "# @sc [decision:key_choice]\nB = 2\n\n" + "# @sc [decision:next_choice]\n[N]\nC = 3\n", + ) + tags, errors = scan_tags(tmp_path) + + assert errors == [] + section_tag = next(tag for tag in tags if tag.decisions == ("section_choice",)) + assert section_tag.site.section == "S" + assert section_tag.site.start <= 6 <= section_tag.site.end + + +def test_prose_mentions_of_sc_are_not_tags(tmp_path): + _write( + tmp_path / "mod.py", + '"""An @sc citation points to a decision.\n\n' + "A paragraph that explains the @sc syntax.\n\n" + "@sc [decision:choice]\n\n\"\"\"\n", + ) + tags, errors = scan_tags(tmp_path) + assert errors == [] + assert len(tags) == 1 + assert tags[0].decisions == ("choice",) + + def test_python_declaration_statement_and_snakemake_tags(tmp_path): _write( tmp_path / "src" / "mod.py", From 9641fc4cbdd62c6a78c349bb2d8a5628dbb82e3b Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 04:47:24 +0200 Subject: [PATCH 26/40] fix(decisions): resolve multiline INI values within sites --- tests/helpers/decisions.py | 10 ++++++++++ tests/unit/test_decisions.py | 13 +++++++++++++ 2 files changed, 23 insertions(+) diff --git a/tests/helpers/decisions.py b/tests/helpers/decisions.py index 50d3cc15d..272c0f2dd 100644 --- a/tests/helpers/decisions.py +++ b/tests/helpers/decisions.py @@ -840,6 +840,16 @@ def _config_values_in_site(root, site, selector): if not match or match.group(1).strip() != key: continue value = match.group(2).strip() + base_indent = len(raw) - len(raw.lstrip()) + for continuation in lines[number:site.end]: + if not continuation.strip(): + break + if continuation.lstrip().startswith(("#", ";")): + continue + indent = len(continuation) - len(continuation.lstrip()) + if indent <= base_indent: + break + value += " " + continuation.strip() predicate = False elif suffix == ".setools": match = re.match(rf"^{re.escape(key)}(?=$|\s|=|<|>)(.*)$", stripped) diff --git a/tests/unit/test_decisions.py b/tests/unit/test_decisions.py index 3e48ecc1a..af42433b1 100644 --- a/tests/unit/test_decisions.py +++ b/tests/unit/test_decisions.py @@ -288,6 +288,19 @@ def test_numeric_lists_and_astromatic_boolean_words_keep_reader_semantics(tmp_pa assert value_errors(tmp_path, record, tags) == [] +def test_ini_multiline_value_keeps_continuation_lines(tmp_path): + _write( + tmp_path / "config.ini", + "# @sc [decision:choice]\n[SCIENCE]\nMODULE = alpha, beta,\n" + " gamma, delta\n", + ) + record = _record("Runner chain. Values: SCIENCE.MODULE = alpha,beta,gamma,delta.") + tags, errors = scan_tags(tmp_path) + + assert errors == [] + assert value_errors(tmp_path, record, tags) == [] + + def test_config_boolean_semantics_and_setools_predicates(tmp_path): ini = _write(tmp_path / "config.ini", "# @sc [decision:choice]\n[SCIENCE]\nENABLED = 1\n") record = _record("Toggle. Values: SCIENCE.ENABLED = True.") From f3a8dbf13ac88c0f80a8839ae93c1dbdda1a283c Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 04:48:57 +0200 Subject: [PATCH 27/40] fix(decisions): scope Python selectors to tagged declarations --- tests/helpers/decisions.py | 19 ++++++++++--------- tests/unit/test_decisions.py | 14 ++++++++++++++ 2 files changed, 24 insertions(+), 9 deletions(-) diff --git a/tests/helpers/decisions.py b/tests/helpers/decisions.py index 272c0f2dd..f234932f7 100644 --- a/tests/helpers/decisions.py +++ b/tests/helpers/decisions.py @@ -921,18 +921,19 @@ def _python_value_in_site(root, site, selector): source = (Path(root) / site.path).read_text(encoding="utf-8") tree = _python_tree(source) symbol, _ = _code_selector(selector) - if site.symbol and (symbol == site.symbol or symbol.startswith(site.symbol + ".")): - full_selector = selector - elif site.symbol: - full_selector = f"{site.symbol}.{selector}" - else: - full_selector = selector - full_symbol, _ = _code_selector(full_selector) - if site.symbol and full_symbol != site.symbol and not full_symbol.startswith(site.symbol + "."): - return None + direct = bool( + site.symbol + and (symbol == site.symbol or symbol.startswith(site.symbol + ".")) + ) + full_selector = ( + selector if direct else f"{site.symbol}.{selector}" + if site.symbol else selector + ) try: node = _selected_python_node(tree, full_selector) except ValueError as error: + if not direct: + return None if "needs one binding" in str(error) or "missing" in str(error): return None raise diff --git a/tests/unit/test_decisions.py b/tests/unit/test_decisions.py index af42433b1..f056f52ac 100644 --- a/tests/unit/test_decisions.py +++ b/tests/unit/test_decisions.py @@ -344,6 +344,20 @@ def test_python_value_reassignment_fails_closed(tmp_path): assert "exactly one tagged site" in problem +def test_python_qualified_selector_ignores_other_tagged_scopes(tmp_path): + _write( + tmp_path / "constants.py", + "# @sc [decision:choice]\nOTHER = 5\n\n" + 'def fit():\n """Fit.\n\n @sc [decision:choice]\n """\n' + " PARAMS = {'limits': {'T': 1}}\n", + ) + record = _record("Prior. Values: fit.PARAMS[limits.T] = 1.") + tags, errors = scan_tags(tmp_path) + + assert errors == [] + assert value_errors(tmp_path, record, tags) == [] + + def test_python_literal_and_dict_selector_is_static(tmp_path): _write( tmp_path / "constants.py", From 3b6664369189ee16f920a7613d57655ca9a105b7 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 04:52:08 +0200 Subject: [PATCH 28/40] fix(decisions): ignore values outside Python site grammar --- tests/helpers/decisions.py | 5 ++++- tests/unit/test_decisions.py | 16 ++++++++++++++++ 2 files changed, 20 insertions(+), 1 deletion(-) diff --git a/tests/helpers/decisions.py b/tests/helpers/decisions.py index f234932f7..21d149989 100644 --- a/tests/helpers/decisions.py +++ b/tests/helpers/decisions.py @@ -920,7 +920,10 @@ def _absent_scope_matches(root, site, selector): def _python_value_in_site(root, site, selector): source = (Path(root) / site.path).read_text(encoding="utf-8") tree = _python_tree(source) - symbol, _ = _code_selector(selector) + try: + symbol, _ = _code_selector(selector) + except ValueError: + return None direct = bool( site.symbol and (symbol == site.symbol or symbol.startswith(site.symbol + ".")) diff --git a/tests/unit/test_decisions.py b/tests/unit/test_decisions.py index f056f52ac..096ef7f7e 100644 --- a/tests/unit/test_decisions.py +++ b/tests/unit/test_decisions.py @@ -331,6 +331,22 @@ def test_config_boolean_semantics_and_setools_predicates(tmp_path): assert value_errors(tmp_path, cuts, tags) +def test_config_ref_ignores_unrelated_python_tagged_site(tmp_path): + _write( + tmp_path / "stars.setools", + '# @sc [decision:choice]\n[MASK:stars]\nFLAGS == 0\n', + ) + _write( + tmp_path / "other.py", + "# @sc [decision:choice]\nOTHER = 1\n", + ) + record = _record('Star flags. Values: MASK:stars.FLAGS = "== 0".') + tags, errors = scan_tags(tmp_path) + + assert errors == [] + assert value_errors(tmp_path, record, tags) == [] + + def test_python_value_reassignment_fails_closed(tmp_path): _write( tmp_path / "constants.py", From e95839126f0930771f0bb76531372e3969cc85c9 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 04:54:21 +0200 Subject: [PATCH 29/40] migrate record anchors to site tags --- astra.yaml | 242 +++--------------- .../modules/make_cat_package/make_cat.py | 2 + src/shapepipe/modules/make_cat_runner.py | 1 + .../modules/mccd_package/__init__.py | 1 + .../merge_headers_package/merge_headers.py | 1 + src/shapepipe/modules/ngmix_package/ngmix.py | 16 ++ src/shapepipe/modules/ngmix_runner.py | 4 +- .../sextractor_package/sextractor_script.py | 2 + .../vignetmaker_package/vignetmaker.py | 1 + src/shapepipe/utilities/mask_query.py | 1 + tests/unit/test_decisions.py | 15 ++ workflow/config.yaml | 1 + workflow/config/cfis/config_MCCD.ini | 11 + workflow/config/cfis/config_exp_Sp.ini | 2 + workflow/config/cfis/config_exp_mccd.ini | 8 + workflow/config/cfis/config_exp_psfex.ini | 22 ++ workflow/config/cfis/config_tile_Fe.ini | 4 + workflow/config/cfis/config_tile_Mc.ini | 1 + .../config/cfis/config_tile_Ng_template.ini | 6 + .../config/cfis/config_tile_PiViVi_psfex.ini | 22 ++ workflow/config/cfis/config_tile_Sx.ini | 14 + workflow/config/cfis/default.conv | 1 + workflow/config/cfis/default.param | 1 + workflow/config/cfis/default.psfex | 22 ++ workflow/config/cfis/default_exp.sex | 31 +++ workflow/config/cfis/default_noimaflags.param | 2 + workflow/config/cfis/default_tile.sex | 33 +++ workflow/config/cfis/final_cat.param | 4 + workflow/config/cfis/gauss_3.0_7x7.conv | 1 + workflow/config/cfis/star_selection.setools | 25 ++ 30 files changed, 284 insertions(+), 213 deletions(-) diff --git a/astra.yaml b/astra.yaml index ad56721c2..54ce9416a 100644 --- a/astra.yaml +++ b/astra.yaml @@ -75,10 +75,7 @@ decisions: per CCD; preprocessing merges an exposure's stars into one train and one test catalogue). Science-path PSF rejection bypasses this table: psfex_interp drops the epoch per object inside the tile run. - Anchor: workflow/scripts/completeness.py::COMPLETENESS; - workflow/scripts/completeness.py::COMPLETENESS[exp_psf.psfex.psfex_interp_runner.warn] = True; - workflow/scripts/completeness.py::COMPLETENESS[exp_psf.mccd.mask_query_runner.expect] = 40; - workflow/scripts/completeness.py::COMPLETENESS[exp_psf.mccd.mccd_preprocessing_runner.expect] = 2. + Values: COMPLETENESS[exp_psf.psfex.psfex_interp_runner.warn] = True; COMPLETENESS[exp_psf.mccd.mask_query_runner.expect] = 40; COMPLETENESS[exp_psf.mccd.mccd_preprocessing_runner.expect] = 2. default: exact_counts options: exact_counts: @@ -106,13 +103,7 @@ decisions: stamp. The stamp bounds the measurable galaxy size and truncates the wings of large galaxies. No rationale for 51 px (about 9.5 arcsec) is recorded. - Anchor: workflow/config/cfis/default_noimaflags.param#VIGNET = 51; - workflow/config/cfis/default.param#VIGNET = 51; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.STAMP_SIZE = 51; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.STAMP_SIZE = 51; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.MASKING = False; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.MASKING = False; - workflow/config/cfis/default.psfex#PSF_SIZE = 51. + Values: workflow/config/cfis/default_noimaflags.param#VIGNET = 51; workflow/config/cfis/default.param#VIGNET = 51; VIGNETMAKER_RUNNER_RUN_1.STAMP_SIZE = 51; VIGNETMAKER_RUNNER_RUN_2.STAMP_SIZE = 51; VIGNETMAKER_RUNNER_RUN_1.MASKING = False; VIGNETMAKER_RUNNER_RUN_2.MASKING = False; PSF_SIZE = 51. default: px_51 options: px_51: @@ -131,12 +122,7 @@ decisions: value assumes the MegaPipe stacks are calibrated to 30; nothing in the repo checks it. Magnitude cuts (the star-selection window, downstream galaxy cuts) inherit their stage's convention. - Anchor: workflow/config/cfis/default_tile.sex#MAG_ZEROPOINT = 30.0; - workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER = False; - workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.MAG_ZP = 30.0; - workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER = True; - workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.ZP_KEY = PHOTZP; - src/shapepipe/modules/sextractor_package/sextractor_script.py::SExtractorCaller.get_zero_point. + Values: MAG_ZEROPOINT = 30.0; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER = False; NGMIX_RUNNER.MAG_ZP = 30.0; workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER = True; SEXTRACTOR_RUNNER.ZP_KEY = PHOTZP. default: fixed_30_tiles_header_exposures options: fixed_30_tiles_header_exposures: @@ -217,9 +203,7 @@ analyses: (detection.detection_source_mode). Sky-fixed masks never touch pixels: an object inside a star halo is measured from the same unmodified pixels as one outside it. - Anchor: workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE = True; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_PATTERN = flag, image, weight, background, background_rms; - src/shapepipe/modules/ngmix_package/ngmix.py::prepare_ngmix_weights. + Values: SEXTRACTOR_RUNNER.FLAG_IMAGE = True; VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_PATTERN = flag, image, weight, background, background_rms. default: instrument_flags_only options: instrument_flags_only: @@ -243,15 +227,7 @@ analyses: MASK_EXT. The intended map is the UNIONS star-body product (bit 2); halo bits 0 and 1 are left out because halos say nothing about whether a star is a good PSF sample. - Anchor: workflow/config/cfis/config_exp_psfex.ini; - workflow/config/cfis/star_selection.setools#MASK:star_selection.IMAFLAGS_ISO = "== 0"; - workflow/config/cfis/star_selection.setools#MASK:preselect.IMAFLAGS_ISO = "== 0"; - workflow/config/cfis/star_selection.setools#MASK:flag.IMAFLAGS_ISO = "== 0"; - workflow/config/cfis/star_selection.setools#MASK:star_selection.MASK_EXT = absent; - workflow/config/cfis/star_selection.setools#MASK:preselect.MASK_EXT = absent; - workflow/config/cfis/star_selection.setools#MASK:flag.MASK_EXT = absent; - src/shapepipe/modules/mask_query_runner.py::mask_query_runner; - src/shapepipe/utilities/mask_query.py::flag_positions. + Values: MASK:star_selection.IMAFLAGS_ISO = "== 0"; MASK:preselect.IMAFLAGS_ISO = "== 0"; MASK:flag.IMAFLAGS_ISO = "== 0"; MASK:star_selection.MASK_EXT = absent; MASK:preselect.MASK_EXT = absent; MASK:flag.MASK_EXT = absent. default: instrument_flags_only options: instrument_flags_only: @@ -279,9 +255,6 @@ analyses: (False for boolean maps, reading as unmasked; typically -1 for integer maps). The committed config sets no MASK_EXT_PATHS, so no mask column is written and all mask cuts happen downstream. - Anchor: src/shapepipe/modules/make_cat_runner.py::make_cat_runner; - src/shapepipe/modules/make_cat_package/make_cat.py::save_mask_ext_data; - src/shapepipe/utilities/mask_query.py::query_map. default: deferred_downstream options: deferred_downstream: @@ -341,20 +314,7 @@ analyses: of 0.65 arcsec (about 3.5 px at 0.187 arcsec/px). Exposures only feed star selection and keep stock values with the 3x3 FWHM 2 px kernel. The tiles differ from Guinot+22 in all three. - Anchor: workflow/config/cfis/default_tile.sex#DETECT_THRESH = 1.0; - workflow/config/cfis/default_tile.sex#ANALYSIS_THRESH = 1.0; - workflow/config/cfis/default_tile.sex#DETECT_MINAREA = 3; - workflow/config/cfis/default_tile.sex#FILTER = Y; - workflow/config/cfis/default_tile.sex#SEEING_FWHM = 0.6; - workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE = $SP_CONFIG/gauss_3.0_7x7.conv; - workflow/config/cfis/gauss_3.0_7x7.conv; - workflow/config/cfis/default_exp.sex#DETECT_THRESH = 1.5; - workflow/config/cfis/default_exp.sex#ANALYSIS_THRESH = 1.5; - workflow/config/cfis/default_exp.sex#DETECT_MINAREA = 5; - workflow/config/cfis/default_exp.sex#FILTER = Y; - workflow/config/cfis/default_exp.sex#SEEING_FWHM = 0.6; - workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE = $SP_CONFIG/default.conv; - workflow/config/cfis/default.conv. + Values: workflow/config/cfis/default_tile.sex#DETECT_THRESH = 1.0; workflow/config/cfis/default_tile.sex#ANALYSIS_THRESH = 1.0; workflow/config/cfis/default_tile.sex#DETECT_MINAREA = 3; workflow/config/cfis/default_tile.sex#FILTER = Y; workflow/config/cfis/default_tile.sex#SEEING_FWHM = 0.6; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE = $SP_CONFIG/gauss_3.0_7x7.conv; workflow/config/cfis/default_exp.sex#DETECT_THRESH = 1.5; workflow/config/cfis/default_exp.sex#ANALYSIS_THRESH = 1.5; workflow/config/cfis/default_exp.sex#DETECT_MINAREA = 5; workflow/config/cfis/default_exp.sex#FILTER = Y; workflow/config/cfis/default_exp.sex#SEEING_FWHM = 0.6; workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE = $SP_CONFIG/default.conv. default: megapipe_tiles options: megapipe_tiles: @@ -374,10 +334,7 @@ analyses: Tiles use the MegaPipe DEBLEND_MINCONT; exposures use a lower one, the value Guinot+22 lists. Both share DEBLEND_NTHRESH. Contrast sets object count, centroids, and blend contamination in shapes. - Anchor: workflow/config/cfis/default_tile.sex#DEBLEND_MINCONT = 0.002; - workflow/config/cfis/default_exp.sex#DEBLEND_MINCONT = 0.001; - workflow/config/cfis/default_tile.sex#DEBLEND_NTHRESH = 32; - workflow/config/cfis/default_exp.sex#DEBLEND_NTHRESH = 32. + Values: workflow/config/cfis/default_tile.sex#DEBLEND_MINCONT = 0.002; workflow/config/cfis/default_exp.sex#DEBLEND_MINCONT = 0.001; workflow/config/cfis/default_tile.sex#DEBLEND_NTHRESH = 32; workflow/config/cfis/default_exp.sex#DEBLEND_NTHRESH = 32. default: megapipe_tiles options: megapipe_tiles: @@ -399,18 +356,7 @@ analyses: BACKGROUND_RMS (shape_measurement.galaxy_pixel_weights). The header background path is off. Residual sky offsets propagate into thresholds, fluxes, completeness and shapes. - Anchor: workflow/config/cfis/default_tile.sex#BACK_TYPE = AUTO; - workflow/config/cfis/default_tile.sex#BACK_SIZE = 512; - workflow/config/cfis/default_tile.sex#BACK_FILTERSIZE = 9; - workflow/config/cfis/default_tile.sex#BACKPHOTO_TYPE = LOCAL; - workflow/config/cfis/default_tile.sex#BACKPHOTO_THICK = 30; - workflow/config/cfis/default_exp.sex#BACK_TYPE = AUTO; - workflow/config/cfis/default_exp.sex#BACK_SIZE = 64; - workflow/config/cfis/default_exp.sex#BACK_FILTERSIZE = 3; - workflow/config/cfis/default_exp.sex#BACKPHOTO_TYPE = GLOBAL; - workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.BKG_FROM_HEADER = False; - workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.BKG_FROM_HEADER = False; - src/shapepipe/modules/sextractor_package/sextractor_script.py::SExtractorCaller.get_background. + Values: workflow/config/cfis/default_tile.sex#BACK_TYPE = AUTO; workflow/config/cfis/default_tile.sex#BACK_SIZE = 512; workflow/config/cfis/default_tile.sex#BACK_FILTERSIZE = 9; workflow/config/cfis/default_tile.sex#BACKPHOTO_TYPE = LOCAL; BACKPHOTO_THICK = 30; workflow/config/cfis/default_exp.sex#BACK_TYPE = AUTO; workflow/config/cfis/default_exp.sex#BACK_SIZE = 64; workflow/config/cfis/default_exp.sex#BACK_FILTERSIZE = 3; workflow/config/cfis/default_exp.sex#BACKPHOTO_TYPE = GLOBAL; workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.BKG_FROM_HEADER = False; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.BKG_FROM_HEADER = False. default: auto_megapipe_tiles options: auto_megapipe_tiles: @@ -429,15 +375,7 @@ analyses: RESCALE_WEIGHTS and WEIGHT_GAIN keep their SExtractor defaults. Guinot+22 keeps every non-tabulated parameter at its default, which would mean no weight map. - Anchor: workflow/config/cfis/default_tile.sex#WEIGHT_TYPE = MAP_WEIGHT; - workflow/config/cfis/default_exp.sex#WEIGHT_TYPE = MAP_WEIGHT; - workflow/config/cfis/default_tile.sex#RESCALE_WEIGHTS = Y; - workflow/config/cfis/default_exp.sex#RESCALE_WEIGHTS = Y; - workflow/config/cfis/default_tile.sex#WEIGHT_GAIN = Y; - workflow/config/cfis/default_exp.sex#WEIGHT_GAIN = Y; - workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.WEIGHT_IMAGE = True; - workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.WEIGHT_IMAGE = True; - src/shapepipe/modules/sextractor_package/sextractor_script.py::SExtractorCaller.set_input_files. + Values: workflow/config/cfis/default_tile.sex#WEIGHT_TYPE = MAP_WEIGHT; workflow/config/cfis/default_exp.sex#WEIGHT_TYPE = MAP_WEIGHT; workflow/config/cfis/default_tile.sex#RESCALE_WEIGHTS = Y; workflow/config/cfis/default_exp.sex#RESCALE_WEIGHTS = Y; workflow/config/cfis/default_tile.sex#WEIGHT_GAIN = Y; workflow/config/cfis/default_exp.sex#WEIGHT_GAIN = Y; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.WEIGHT_IMAGE = True; workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.WEIGHT_IMAGE = True. default: map_weight options: map_weight: @@ -451,12 +389,7 @@ analyses: INTERP_TYPE ALL on both passes (SExtractor default NONE): SExtractor invents flux across zero-weight pixels, changing detections and photometry near masked regions. No rationale recorded. - Anchor: workflow/config/cfis/default_tile.sex#INTERP_TYPE = ALL; - workflow/config/cfis/default_exp.sex#INTERP_TYPE = ALL; - workflow/config/cfis/default_tile.sex#INTERP_MAXXLAG = 16; - workflow/config/cfis/default_tile.sex#INTERP_MAXYLAG = 16; - workflow/config/cfis/default_exp.sex#INTERP_MAXXLAG = 16; - workflow/config/cfis/default_exp.sex#INTERP_MAXYLAG = 16. + Values: workflow/config/cfis/default_tile.sex#INTERP_TYPE = ALL; workflow/config/cfis/default_exp.sex#INTERP_TYPE = ALL; workflow/config/cfis/default_tile.sex#INTERP_MAXXLAG = 16; workflow/config/cfis/default_tile.sex#INTERP_MAXYLAG = 16; workflow/config/cfis/default_exp.sex#INTERP_MAXXLAG = 16; workflow/config/cfis/default_exp.sex#INTERP_MAXYLAG = 16. default: interp_all options: interp_all: @@ -469,10 +402,7 @@ analyses: CLEAN on both passes deletes detections consistent with being wings of a brighter neighbour, a post-deblend change to the object list. Stock value; no rationale recorded. - Anchor: workflow/config/cfis/default_tile.sex#CLEAN_PARAM = 1.0; - workflow/config/cfis/default_exp.sex#CLEAN_PARAM = 1.0; - workflow/config/cfis/default_tile.sex#CLEAN = Y; - workflow/config/cfis/default_exp.sex#CLEAN = Y. + Values: workflow/config/cfis/default_tile.sex#CLEAN_PARAM = 1.0; workflow/config/cfis/default_exp.sex#CLEAN_PARAM = 1.0; workflow/config/cfis/default_tile.sex#CLEAN = Y; workflow/config/cfis/default_exp.sex#CLEAN = Y. default: clean_1 options: clean_1: @@ -483,8 +413,7 @@ analyses: MASK_TYPE CORRECT on both passes replaces neighbour pixels by their mirror across the object centre during photometry, changing blend fluxes and windowed moments. Stock value; no rationale recorded. - Anchor: workflow/config/cfis/default_tile.sex#MASK_TYPE = CORRECT; - workflow/config/cfis/default_exp.sex#MASK_TYPE = CORRECT. + Values: workflow/config/cfis/default_tile.sex#MASK_TYPE = CORRECT; workflow/config/cfis/default_exp.sex#MASK_TYPE = CORRECT. default: correct options: correct: @@ -503,13 +432,7 @@ analyses: documentation), since neither .sex file sets one. Whether the delivered exposure CCDs and MegaPipe tiles carry SATURATE is unverified. - Anchor: workflow/config/cfis/default_exp.sex#SATUR_KEY = SATURATE; - workflow/config/cfis/default_tile.sex#SATUR_KEY = SATURATE; - workflow/config/cfis/star_selection.setools#MASK:star_selection.FLAGS = "== 0"; - workflow/config/cfis/star_selection.setools#MASK:preselect.FLAGS = "== 0"; - workflow/config/cfis/star_selection.setools#MASK:flag.FLAGS = "== 0"; - workflow/config/cfis/default_tile.sex#SATUR_LEVEL = absent; - workflow/config/cfis/default_exp.sex#SATUR_LEVEL = absent. + Values: workflow/config/cfis/default_exp.sex#SATUR_KEY = SATURATE; workflow/config/cfis/default_tile.sex#SATUR_KEY = SATURATE; MASK:star_selection.FLAGS = "== 0"; MASK:preselect.FLAGS = "== 0"; MASK:flag.FLAGS = "== 0"; workflow/config/cfis/default_tile.sex#SATUR_LEVEL = absent; workflow/config/cfis/default_exp.sex#SATUR_LEVEL = absent. default: header_saturate options: header_saturate: @@ -524,13 +447,7 @@ analyses: magnitude window and the catalogue magnitude; FLUX_AUTO is PSFEx's photometric normalisation. A different Kron factor shifts magnitudes and so every magnitude-based cut. - Anchor: workflow/config/cfis/default_tile.sex#PHOT_AUTOPARAMS = 2.5,3.5; - workflow/config/cfis/default_exp.sex#PHOT_AUTOPARAMS = 2.5,3.5; - workflow/config/cfis/default_tile.sex#PHOT_APERTURES = 5; - workflow/config/cfis/default_exp.sex#PHOT_APERTURES = 5; - workflow/config/cfis/default_tile.sex#PHOT_FLUXFRAC = 0.5; - workflow/config/cfis/default_exp.sex#PHOT_FLUXFRAC = 0.5; - workflow/config/cfis/default.psfex#PHOTFLUX_KEY = FLUX_AUTO. + Values: workflow/config/cfis/default_tile.sex#PHOT_AUTOPARAMS = 2.5,3.5; workflow/config/cfis/default_exp.sex#PHOT_AUTOPARAMS = 2.5,3.5; workflow/config/cfis/default_tile.sex#PHOT_APERTURES = 5; workflow/config/cfis/default_exp.sex#PHOT_APERTURES = 5; workflow/config/cfis/default_tile.sex#PHOT_FLUXFRAC = 0.5; workflow/config/cfis/default_exp.sex#PHOT_FLUXFRAC = 0.5; PHOTFLUX_KEY = FLUX_AUTO. default: kron_25_35 options: kron_25_35: @@ -544,10 +461,7 @@ analyses: catalogue carries no IMAFLAGS_ISO. [LINT] final_cat.param, read by the post-processing merge, requests IMAFLAGS_ISO, which the tile chain never produces (issue #912). - Anchor: workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.DETECTION_IMAGE = False; - workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE = False; - workflow/config/cfis/default_noimaflags.param; - workflow/config/cfis/final_cat.param#IMAFLAGS_ISO. + Values: SEXTRACTOR_RUNNER.DETECTION_IMAGE = False; SEXTRACTOR_RUNNER.FLAG_IMAGE = False. default: sx_nomask_single_image options: sx_nomask_single_image: @@ -570,10 +484,7 @@ analyses: endpoints match MegaCam's raw DATASEC (2048 pixel indices inclusively), but the strict cut excludes both. The excluded strip is likely prescan (unverified). - Anchor: workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.CCD_SIZE = 33,2080,1,4612; - workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.MAKE_POST_PROCESS = True; - src/shapepipe/modules/sextractor_package/sextractor_script.py::make_post_process; - src/shapepipe/modules/sextractor_package/sextractor_script.py::ccd_candidate_mask. + Values: SEXTRACTOR_RUNNER.CCD_SIZE = 33,2080,1,4612; SEXTRACTOR_RUNNER.MAKE_POST_PROCESS = True. default: trimmed_bounds_33_2080 options: trimmed_bounds_33_2080: @@ -654,8 +565,6 @@ analyses: (stamp placement, epoch membership, position seeding) uses that solution. [HARDCODED] no astrometric re-derivation or refinement exists; a joint re-fit would move every stamp centre and position seed. - Anchor: src/shapepipe/modules/split_exp_package/split_exp.py::SplitExposures.create_hdus; - src/shapepipe/modules/merge_headers_package/merge_headers.py::merge_headers. default: delivered_headers options: delivered_headers: @@ -669,8 +578,7 @@ analyses: rationale: >- Any HDU count other than N_HDU raises; every CCD is a candidate epoch wherever the WCS lands it. - Anchor: workflow/config/cfis/config_exp_Sp.ini#SPLIT_EXP_RUNNER.N_HDU = 40; - src/shapepipe/modules/split_exp_package/split_exp.py::SplitExposures.create_hdus. + Values: SPLIT_EXP_RUNNER.N_HDU = 40. default: all_40_hdus options: all_40_hdus: @@ -689,9 +597,7 @@ analyses: an epoch letter rather than a prefix, so EXP_PREFIX is blank; downstream code drops the p when it needs the bare exposure ID. A mis-parse changes N_EPOCH and which exposures are fit. - Anchor: workflow/config/cfis/config_tile_Fe.ini#FIND_EXPOSURES_RUNNER.COLNUM = 3; - workflow/config/cfis/config_tile_Fe.ini#FIND_EXPOSURES_RUNNER.EXP_PREFIX; - src/shapepipe/modules/find_exposures_package/find_exposures.py::FindExposures.get_exposure_list. + Values: FIND_EXPOSURES_RUNNER.COLNUM = 3. default: history_parse options: history_parse: @@ -706,12 +612,7 @@ analyses: Windowed, isophotal and model centroids differ systematically for blends and asymmetric galaxies, and the centroid feeds the position seed and the centroid prior. - Anchor: workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.POSITION_PARAMS = XWIN_IMAGE,YWIN_IMAGE; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.COORD = PIX; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS = XWIN_WORLD,YWIN_WORLD; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.POSITION_PARAMS = XWIN_WORLD,YWIN_WORLD; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.COORD = SPHE; - workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS = XWIN_IMAGE,YWIN_IMAGE. + Values: VIGNETMAKER_RUNNER_RUN_1.POSITION_PARAMS = XWIN_IMAGE,YWIN_IMAGE; VIGNETMAKER_RUNNER_RUN_1.COORD = PIX; workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS = XWIN_WORLD,YWIN_WORLD; VIGNETMAKER_RUNNER_RUN_2.POSITION_PARAMS = XWIN_WORLD,YWIN_WORLD; VIGNETMAKER_RUNNER_RUN_2.COORD = SPHE; workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS = XWIN_IMAGE,YWIN_IMAGE. default: xwin_windowed options: xwin_windowed: @@ -728,8 +629,6 @@ analyses: an image edge are kept, zero-filled outside the image; a stamp centre that rounds outside the image raises and fails the vignet run for the whole tile. - Anchor: src/shapepipe/modules/vignetmaker_package/vignetmaker.py::get_stamps; - src/shapepipe/modules/vignetmaker_package/vignetmaker.py::VignetMaker._get_stamp_me. default: round_and_zero_pad options: round_and_zero_pad: @@ -791,17 +690,7 @@ analyses: objects, so small-N behaviour changes selection on sparse CCDs. PSFEx's own selection is off; see psfex_candidate_vetting for what PSFEx may still apply. - Anchor: workflow/config/cfis/star_selection.setools#MASK:star_selection.FLAGS = "== 0"; - workflow/config/cfis/star_selection.setools#MASK:preselect.FLAGS = "== 0"; - workflow/config/cfis/star_selection.setools#MASK:star_selection.IMAFLAGS_ISO = "== 0"; - workflow/config/cfis/star_selection.setools#MASK:preselect.IMAFLAGS_ISO = "== 0"; - workflow/config/cfis/star_selection.setools#MASK:star_selection.MAG_AUTO = ["> 18.", "< 22."]; - workflow/config/cfis/star_selection.setools#MASK:star_selection.FWHM_IMAGE = ["<= mode(FWHM_IMAGE{preselect}) + 0.2", ">= mode(FWHM_IMAGE{preselect}) - 0.2"]; - workflow/config/cfis/star_selection.setools#MASK:preselect.MAG_AUTO = ["> 0", "< 21"]; - workflow/config/cfis/star_selection.setools#MASK:preselect.FWHM_IMAGE = ["> 0.3 / 0.187", "< 1.5 / 0.187"]; - workflow/config/cfis/star_selection.setools#PLOT:fwhm_field.SCATTER = "FWHM_IMAGE{star_selection}*0.187"; - workflow/config/cfis/default.psfex#SAMPLE_AUTOSELECT = N; - src/shapepipe/pipeline/str_handler.py::StrInterpreter._mode. + Values: MASK:star_selection.FLAGS = "== 0"; MASK:preselect.FLAGS = "== 0"; MASK:star_selection.IMAFLAGS_ISO = "== 0"; MASK:preselect.IMAFLAGS_ISO = "== 0"; MASK:star_selection.MAG_AUTO = ["> 18.", "< 22."]; MASK:star_selection.FWHM_IMAGE = ["<= mode(FWHM_IMAGE{preselect}) + 0.2", ">= mode(FWHM_IMAGE{preselect}) - 0.2"]; MASK:preselect.MAG_AUTO = ["> 0", "< 21"]; MASK:preselect.FWHM_IMAGE = ["> 0.3 / 0.187", "< 1.5 / 0.187"]; PLOT:fwhm_field.SCATTER = "FWHM_IMAGE{star_selection}*0.187"; SAMPLE_AUTOSELECT = N. default: mode_centred_box options: mode_centred_box: @@ -827,11 +716,7 @@ analyses: against an independent residual test. The permutation is seeded from the digits of the unit's file number, so a CCD gets the same split on every run. - Anchor: workflow/config/cfis/star_selection.setools#RAND_SPLIT:star_split.RATIO = 20; - src/shapepipe/modules/setools_package/setools.py::SETools._make_rand_split; - workflow/config/cfis/config_exp_psfex.ini#PSFEX_RUNNER.FILE_PATTERN = star_split_ratio_80; - workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.FILE_PATTERN = star_split_ratio_80,star_split_ratio_20,psfex_cat; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.ME_DOT_PSF_PATTERN = star_split_ratio_80. + Values: RAND_SPLIT:star_split.RATIO = 20; PSFEX_RUNNER.FILE_PATTERN = star_split_ratio_80; PSFEX_INTERP_RUNNER.FILE_PATTERN = star_split_ratio_80,star_split_ratio_20,psfex_cat; PSFEX_INTERP_RUNNER.ME_DOT_PSF_PATTERN = star_split_ratio_80. default: split_80_20_seeded options: split_80_20_seeded: @@ -865,16 +750,7 @@ analyses: Compiled-in values admit no `= value` assertion, so the anchors assert only that each SAMPLE_* key is absent; open issue #919 would pin them in default.psfex. - Anchor: workflow/config/cfis/default.psfex#SAMPLE_AUTOSELECT = N; - workflow/config/cfis/default.psfex#BADPIXEL_FILTER = N; - workflow/config/cfis/default.psfex#PSF_RECENTER = N; - workflow/config/cfis/default.psfex#SAMPLE_FWHMRANGE = absent; - workflow/config/cfis/default.psfex#SAMPLE_VARIABILITY = absent; - workflow/config/cfis/default.psfex#SAMPLE_MINSN = absent; - workflow/config/cfis/default.psfex#SAMPLE_MAXELLIP = absent; - workflow/config/cfis/default.psfex#SAMPLE_FLAGMASK = absent; - workflow/config/cfis/default.psfex#SAMPLE_WFLAGMASK = absent; - workflow/config/cfis/default.psfex#SAMPLE_IMAFLAGMASK = absent. + Values: SAMPLE_AUTOSELECT = N; BADPIXEL_FILTER = N; PSF_RECENTER = N; SAMPLE_FWHMRANGE = absent; SAMPLE_VARIABILITY = absent; SAMPLE_MINSN = absent; SAMPLE_MAXELLIP = absent; SAMPLE_FLAGMASK = absent; SAMPLE_WFLAGMASK = absent; SAMPLE_IMAFLAGMASK = absent. default: builtin_defaults options: builtin_defaults: @@ -892,17 +768,7 @@ analyses: its exposure chain matches PSFEx's: SExtractor reads the split image, weight and instrument flag directly, and mask_query sits between SExtractor and setools. - Anchor: workflow/config.yaml; - workflow/config/cfis/config_MCCD.ini#INSTANCE.N_COMP_LOC = 8; - workflow/config/cfis/config_MCCD.ini#INSTANCE.D_COMP_GLOB = 8; - workflow/config/cfis/config_MCCD.ini#INSTANCE.FP_GEOMETRY = CFIS; - workflow/config/cfis/config_MCCD.ini#INSTANCE.RMSE_THRESH = 1.25; - workflow/config/cfis/config_MCCD.ini#INPUTS.MIN_N_STARS = 20; - workflow/config/cfis/config_exp_mccd.ini#EXECUTION.MODULE = sextractor_runner,mask_query_runner,setools_runner,mccd_preprocessing_runner,mccd_fit_val_runner,merge_starcat_runner,mccd_plots_runner; - workflow/config/cfis/config_exp_mccd.ini#SEXTRACTOR_RUNNER.INPUT_DIR = $SP_RUN/output/run_sp_exp_Sp/split_exp_runner/output; - workflow/config/cfis/config_exp_mccd.ini#SEXTRACTOR_RUNNER.FILE_PATTERN = image,weight,flag; - workflow/config/cfis/config_exp_mccd.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE = True; - src/shapepipe/modules/mccd_package. + Values: INSTANCE.N_COMP_LOC = 8; INSTANCE.D_COMP_GLOB = 8; INSTANCE.FP_GEOMETRY = CFIS; INSTANCE.RMSE_THRESH = 1.25; INPUTS.MIN_N_STARS = 20; EXECUTION.MODULE = sextractor_runner,mask_query_runner,setools_runner,mccd_preprocessing_runner,mccd_fit_val_runner,merge_starcat_runner,mccd_plots_runner; SEXTRACTOR_RUNNER.INPUT_DIR = $SP_RUN/output/run_sp_exp_Sp/split_exp_runner/output; SEXTRACTOR_RUNNER.FILE_PATTERN = image,weight,flag; SEXTRACTOR_RUNNER.FLAG_IMAGE = True. default: psfex options: psfex: @@ -921,13 +787,7 @@ analyses: Model flexibility trades overfitting against PSF leakage, the dominant additive systematic in cosmic shear. Stock values; no rationale recorded. - Anchor: workflow/config/cfis/default.psfex#BASIS_TYPE = PIXEL; - workflow/config/cfis/default.psfex#PSF_ACCURACY = 0.01; - workflow/config/cfis/default.psfex#BASIS_NUMBER = 20; - workflow/config/cfis/default.psfex#PSFVAR_DEGREES = 2; - workflow/config/cfis/default.psfex#PSF_SAMPLING = 1; - workflow/config/cfis/default.psfex#PSFVAR_KEYS = XWIN_IMAGE,YWIN_IMAGE; - workflow/config/cfis/default.psfex#MEF_TYPE = INDEPENDENT. + Values: BASIS_TYPE = PIXEL; PSF_ACCURACY = 0.01; BASIS_NUMBER = 20; PSFVAR_DEGREES = 2; PSF_SAMPLING = 1; PSFVAR_KEYS = XWIN_IMAGE,YWIN_IMAGE; MEF_TYPE = INDEPENDENT. default: pixel_basis_deg2_per_ccd options: pixel_basis_deg2_per_ccd: @@ -947,11 +807,7 @@ analyses: The pipeline has no minimum-epoch floor: NGMIX_N_EPOCH records what survived and epoch-count cuts happen downstream. Guinot+22 discards the CCD from PSF estimation rather than gating at interpolation. - Anchor: src/shapepipe/modules/psfex_interp_package/psfex_interp.py::PSFExInterpolator.interpsfex; - workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH = 22; - workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.CHI2_THRESH = 2; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH = 22; - workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.CHI2_THRESH = 2. + Values: workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH = 22; workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.CHI2_THRESH = 2; workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH = 22; workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.CHI2_THRESH = 2. default: stars22_chi2_2 options: stars22_chi2_2: @@ -1092,8 +948,6 @@ analyses: therefore do not depend on how a tile is chunked, and metacal's fixnoise counter-noise cancels across image-simulation branches that share an object's box. - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::position_seed; - src/shapepipe/modules/ngmix_package/ngmix.py::Ngmix.process. default: position_seed options: position_seed: @@ -1110,7 +964,6 @@ analyses: [HARDCODED] ngmix Fitter(model='gauss') for galaxy and PSF. The standard defence, that model bias largely cancels in the metacal response, is not stated in the code. - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::make_runners. default: gauss options: gauss: @@ -1129,8 +982,6 @@ analyses: retries decide which objects converge. A failed fit is flagged; an exception during an object's fit drops it with no ngmix row, leaving it to the catalogue sentinels (catalogue_assembly.failure_sentinels). - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::make_runners; - src/shapepipe/modules/ngmix_package/ngmix.py::Ngmix.process. default: prior_guess_t025_ntry5_2 options: prior_guess_t025_ntry5_2: @@ -1152,11 +1003,7 @@ analyses: exposure pixels, and star selection uses 0.187 arcsec/px; the ngmix_runner comment says pixel scale also sets a noise window, but get_noise is never called. - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::get_prior; - src/shapepipe/modules/ngmix_package/ngmix.py::get_prior.T_range = -1,1e3; - src/shapepipe/modules/ngmix_package/ngmix.py::get_prior.F_range = -100,1e9; - src/shapepipe/modules/ngmix_package/ngmix.py::make_runners; - workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.PIXEL_SCALE = 0.186. + Values: get_prior.T_range = -1,1e3; get_prior.F_range = -100,1e9; NGMIX_RUNNER.PIXEL_SCALE = 0.186. default: gpriorba04_flat options: gpriorba04_flat: @@ -1177,13 +1024,7 @@ analyses: fitgauss and is unset in the committed config; it moves the response directly. No sheared-PSF types run, so the catalogue has no PSF response term. - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal; - src/shapepipe/modules/ngmix_package/ngmix.py::METACAL_TYPES = noshear,1p,1m,2p,2m; - src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal.metacal_pars[types] = noshear,1p,1m,2p,2m; - src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal.metacal_pars[step] = 0.01; - src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal.metacal_pars[fixnoise] = True; - src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal.metacal_pars[use_noise_image] = True; - src/shapepipe/modules/ngmix_runner.py::ngmix_runner. + Values: METACAL_TYPES = noshear,1p,1m,2p,2m; do_ngmix_metacal.metacal_pars[types] = noshear,1p,1m,2p,2m; do_ngmix_metacal.metacal_pars[step] = 0.01; do_ngmix_metacal.metacal_pars[fixnoise] = True; do_ngmix_metacal.metacal_pars[use_noise_image] = True. default: five_types_step001_fitgauss options: five_types_step001_fitgauss: @@ -1201,8 +1042,6 @@ analyses: CENTROID_SOURCE). Extraction and prior share one projection and one rounding, so they cannot disagree near a rounding tie. "hsm" re-centres on adaptive moments measured from the stamp. - Anchor: src/shapepipe/modules/ngmix_runner.py::ngmix_runner; - src/shapepipe/modules/ngmix_package/ngmix.py::make_ngmix_observation. default: wcs options: wcs: @@ -1220,7 +1059,6 @@ analyses: its weight divided by FSCALE squared (background RMS scaled with the image) before the joint fit, so the epochs share the tile's zero-point. - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::rescale_epoch_fluxes. default: fscale options: fscale: @@ -1232,8 +1070,6 @@ analyses: and metacal reconvolution kernel) are epoch averages weighted by each epoch's summed galaxy inverse variance; epochs whose PSF fit failed are left out. These columns feed downstream PSF-leakage estimates. - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::_average_psf_fits; - src/shapepipe/modules/ngmix_package/ngmix.py::average_original_psf. default: galaxy_weight_sum options: galaxy_weight_sum: @@ -1247,9 +1083,7 @@ analyses: fallback is a scalar 1/sigma_mad^2. Each epoch is background-subtracted with the SExtractor background vignet (BKG_SUB, on unless set False). - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_ngmix_weights; - src/shapepipe/modules/ngmix_package/ngmix.py::background_subtract; - workflow/config/cfis/config_tile_Ng_template.ini#NGMIX_RUNNER.BKG_RMS_VIGNET_PATH = $NGMIX_VIGNET_DIR/vignetmaker_runner_run_2/output/background_rms_vignet{file_number_string}.sqlite. + Values: NGMIX_RUNNER.BKG_RMS_VIGNET_PATH = $NGMIX_VIGNET_DIR/vignetmaker_runner_run_2/output/background_rms_vignet{file_number_string}.sqlite. default: rms_vignet_weights options: rms_vignet_weights: @@ -1265,8 +1099,7 @@ analyses: without it the g-prior swamps the PSF likelihood. The recovered PSF shape and size are flat for PSF_NOISE from 1e-4 to 1e-6 on the digital twin. - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::PSF_NOISE = 1e-5; - src/shapepipe/modules/ngmix_package/ngmix.py::make_ngmix_observation. + Values: PSF_NOISE = 1e-5. default: psf_noise_1em5 options: psf_noise_1em5: @@ -1279,7 +1112,6 @@ analyses: A wrong flip mis-registers the tile coverage flag against the epoch, changing flagged pixels and the masked-fraction cut. The docstring warns it is wrong for THELI CCDs. - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::Ngmix.MegaCamFlip. default: megapipe_flip options: megapipe_flip: @@ -1309,9 +1141,6 @@ analyses: (symmetrized_4fold_noise). The cost of not symmetrizing, a hole in the galaxy light, is bounded by central_defect_veto. Noise stays the default until a survey A/B against interpolation. - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_ngmix_weights; - src/shapepipe/modules/ngmix_package/ngmix.py::do_ngmix_metacal; - src/shapepipe/modules/ngmix_runner.py::ngmix_runner. default: noise options: noise: @@ -1385,9 +1214,6 @@ analyses: instead cut the target's own light along an unsheared boundary, a sharp edge that rings in the FFTs. The recommended comparison arm is uberseg (weight-only), with defect_fill held equal across arms. - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::uberseg_weight; - src/shapepipe/modules/ngmix_package/ngmix.py::prepare_ngmix_weights; - src/shapepipe/modules/ngmix_runner.py::ngmix_runner. default: none options: none: @@ -1443,7 +1269,6 @@ analyses: -0.98%; -0.24% at 11 px), and noise fill needs 14 px for 0.7 and 0.9 arcsec galaxies. Measured on feat/defect-fill-veto and feat/defect-interpolation. - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_postage_stamps. default: disabled options: disabled: @@ -1479,7 +1304,6 @@ analyses: DES, so the cost in effective number density has to be measured, not assumed. A defect near the centre is handled separately (central_defect_veto). - Anchor: src/shapepipe/modules/ngmix_package/ngmix.py::prepare_postage_stamps. default: one_third options: one_third: @@ -1890,10 +1714,7 @@ analyses: and 0.01, are bypassed); the committed config sets neither key, so enabling classification alone raises. Unlike Guinot+22 (guinot22_spread_model_cut), it applies no s > 0 or magnitude cut. - Anchor: workflow/config/cfis/config_tile_Mc.ini#MAKE_CAT_RUNNER.SM_DO_CLASSIFICATION = False; - src/shapepipe/modules/make_cat_runner.py::make_cat_runner; - src/shapepipe/modules/make_cat_package/make_cat.py::save_sm_data; - workflow/config/cfis/final_cat.param#SPREAD_CLASS. + Values: MAKE_CAT_RUNNER.SM_DO_CLASSIFICATION = False. default: deferred_downstream options: deferred_downstream: @@ -1912,8 +1733,6 @@ analyses: both. [HARDCODED] make_cat attaches only TILE_ID; with no unique-object rule or overlap flag, duplicates are left to downstream selection. - Anchor: src/shapepipe/modules/make_cat_package/make_cat.py::save_sextractor_data; - src/shapepipe/modules/make_cat_package/__init__.py. default: no_dedup_in_pipeline options: no_dedup_in_pipeline: @@ -1937,7 +1756,6 @@ analyses: NGMIX_N_EPOCH 0. A failed object's NGMIX_MCAL_FLAGS reads 0, the success value, so a cut on flags alone keeps it; NGMIX_N_EPOCH > 0 removes it. - Anchor: src/shapepipe/modules/make_cat_package/make_cat.py::SaveCatalogue._save_ngmix_data. default: sentinel_values options: sentinel_values: diff --git a/src/shapepipe/modules/make_cat_package/make_cat.py b/src/shapepipe/modules/make_cat_package/make_cat.py index eafba894a..1a4f6cb8c 100644 --- a/src/shapepipe/modules/make_cat_package/make_cat.py +++ b/src/shapepipe/modules/make_cat_package/make_cat.py @@ -111,6 +111,7 @@ def save_sextractor_data(final_cat_file, sexcat_path, remove_vignet=True): int Number of objects saved + @sc [decision:catalogue_assembly.tile_overlap_handling] """ sexcat_file = file_io.FITSCatalogue(sexcat_path, SEx_catalogue=True) sexcat_file.open() @@ -175,6 +176,7 @@ def save_sm_data( ------- int Number of objects saved + @sc [decision:catalogue_assembly.star_galaxy_classification] """ final_cat_file.open() diff --git a/src/shapepipe/modules/make_cat_runner.py b/src/shapepipe/modules/make_cat_runner.py index 405f7def4..32ee5b018 100644 --- a/src/shapepipe/modules/make_cat_runner.py +++ b/src/shapepipe/modules/make_cat_runner.py @@ -45,6 +45,7 @@ def make_cat_runner( thresholds must come from SM_STAR_THRESH and SM_GAL_THRESH; the function defaults of :func:`make_cat.save_sm_data` are never used. + @sc [decision:masking.sky_mask_application] """ # Set input file paths if len(input_file_list) == 3: diff --git a/src/shapepipe/modules/mccd_package/__init__.py b/src/shapepipe/modules/mccd_package/__init__.py index d0ccf038c..facac4d97 100644 --- a/src/shapepipe/modules/mccd_package/__init__.py +++ b/src/shapepipe/modules/mccd_package/__init__.py @@ -175,6 +175,7 @@ Option to remove validated stars that are outliers in terms of shape before drawing the plots +@sc [decision:star_selection_psf.psf_modelling_software] """ __all__ = [ diff --git a/src/shapepipe/modules/merge_headers_package/merge_headers.py b/src/shapepipe/modules/merge_headers_package/merge_headers.py index 1b15a1a5f..37cc393e3 100644 --- a/src/shapepipe/modules/merge_headers_package/merge_headers.py +++ b/src/shapepipe/modules/merge_headers_package/merge_headers.py @@ -35,6 +35,7 @@ def merge_headers(input_file_list, output_dir, tile_number=None): TypeError For invalid ``output_dir`` type + @sc [decision:preparation.astrometric_solution_source] """ if not isinstance(output_dir, str): raise TypeError( diff --git a/src/shapepipe/modules/ngmix_package/ngmix.py b/src/shapepipe/modules/ngmix_package/ngmix.py index b2e092240..4c150d789 100644 --- a/src/shapepipe/modules/ngmix_package/ngmix.py +++ b/src/shapepipe/modules/ngmix_package/ngmix.py @@ -27,6 +27,7 @@ # Neighbour treatments selectable with the BLEND_HANDLING option. BLEND_HANDLINGS = ("noisefill", "uberseg") +# @sc [decision:shape_measurement.metacal_scheme] METACAL_TYPES = ('noshear', '1p', '1m', '2p', '2m') # Noise budget for the PSF observation's flat weight map (psf_wt = @@ -36,6 +37,7 @@ # value is non-critical once it is finite (validated on the digital twin: the # recovered PSF shape/size are flat across 1e-4..1e-6). See # make_ngmix_observation. +# @sc [decision:shape_measurement.psf_likelihood_noise] PSF_NOISE = 1e-5 @@ -102,6 +104,7 @@ def get_prior(pixel_scale, rng, T_range=None, F_range=None): Returns ------- ngmix.joint_prior.PriorSimpleSep + @sc [decision:shape_measurement.fit_priors] """ if T_range is None: T_range = [-1.0, 1.0e3] @@ -165,6 +168,7 @@ def position_seed(ra, dec, ccd): ------- int Seed in ``[0, 2**32)`` for ``numpy.random.RandomState``. + @sc [decision:shape_measurement.ngmix_seed_mode] """ box_x = int(np.floor((ra * 3600) / 3) + (ccd + 1)) box_y = int(np.floor((dec * 3600) / 3) + (ccd + 2)) @@ -586,6 +590,7 @@ def MegaCamFlip(self, vign, ccd_nb): numpy.ndarray The flipped postage stamp + @sc [decision:shape_measurement.megacam_ccd_flip] """ if ccd_nb < 18 or ccd_nb in [36, 37]: # swap x axis so origin is on top-right @@ -971,6 +976,7 @@ def process(self): dict Dictionary containing the NGMIX metacal results + @sc [decision:shape_measurement.fit_initialisation,decision:shape_measurement.ngmix_seed_mode] """ tile_cat = Tile_cat(self._tile_cat_path, self._seg_cat_path) vignet_cat = self._vignet_cat @@ -1152,6 +1158,7 @@ def prepare_postage_stamps( gal_obj=None, ): # define per-object lists of individual exposures to go into ngmix + """@sc [decision:shape_measurement.central_defect_veto,decision:shape_measurement.epoch_masked_fraction_cut]""" stamp = Postage_stamp(bkg_sub=bkg_sub) # Read each store's per-object dict ONCE: every sqlitedict access # unpickles the object's whole all-epoch dict, so keeping these out of @@ -1300,6 +1307,7 @@ def background_subtract(gal,bkg): ------- numpy.ndarray background subtracted galaxy + @sc [decision:shape_measurement.galaxy_pixel_weights] """ # background subtraction @@ -1329,6 +1337,7 @@ def rescale_epoch_fluxes(gal, weight, header, bkg_rms=None): rescaled weight image numpy.ndarray or None rescaled background RMS image + @sc [decision:shape_measurement.epoch_flux_rescaling] """ Fscale = header['FSCALE'] @@ -1533,6 +1542,7 @@ def uberseg_weight(weight, seg, object_number, dilate_neighbour=0): ------- numpy.ndarray Copy of ``weight`` with neighbour-side pixels zeroed. + @sc [decision:shape_measurement.blend_handling] """ weight = np.copy(weight) @@ -1612,6 +1622,7 @@ def prepare_ngmix_weights( Variance map for NGMIX. numpy.ndarray Noise image. + @sc [decision:masking.pixel_mask_source,decision:shape_measurement.blend_handling,decision:shape_measurement.defect_fill,decision:shape_measurement.galaxy_pixel_weights] """ if blend_handling not in BLEND_HANDLINGS: raise ValueError( @@ -1741,6 +1752,7 @@ def make_ngmix_observation( Returns ------- ngmix.observation.Observation + @sc [decision:shape_measurement.centroid_source,decision:shape_measurement.psf_likelihood_noise] """ psf_jacob = ngmix.Jacobian( row=(psf.shape[0] - 1) / 2, @@ -1836,6 +1848,7 @@ def _average_psf_fits(results_and_weights): dict Keys ``g_psf``, ``g_psf_err``, ``T_psf``, ``T_psf_err`` (weighted averages over the surviving epochs) and ``n_epoch`` (their count). + @sc [decision:shape_measurement.psf_epoch_averaging] """ n_epoch_used = 0 wsum = 0 @@ -1941,6 +1954,7 @@ def average_original_psf(gal_obs_list, psf_runner): ------- dict Same keys as :func:`average_multiepoch_psf`. + @sc [decision:shape_measurement.psf_epoch_averaging] """ def fit(gal_obs): # Fit a COPY so gal_obs.psf stays pristine for metacal — see docstring. @@ -1973,6 +1987,7 @@ def make_runners(prior, flux_guess, rng): ------- tuple (runner, psf_runner) : ngmix.runners.Runner, ngmix.runners.PSFRunner + @sc [decision:shape_measurement.fit_initialisation,decision:shape_measurement.fit_priors,decision:shape_measurement.galaxy_model] """ fitter = ngmix.fitting.Fitter(model='gauss', prior=prior) guesser = ngmix.guessers.TPSFFluxAndPriorGuesser(rng=rng, T=0.25, prior=prior) @@ -2046,6 +2061,7 @@ def do_ngmix_metacal( dict (:func:`average_original_psf`). The two PSF dicts share keys but describe different PSFs; the named fields guard against transposing them. Unpacks positionally as ``resdict, psf_res, psf_orig_res``. + @sc [decision:shape_measurement.defect_fill,decision:shape_measurement.metacal_scheme] """ n_epoch = len(stamp.gals) if n_epoch == 0: diff --git a/src/shapepipe/modules/ngmix_runner.py b/src/shapepipe/modules/ngmix_runner.py index 27648947f..63f03872b 100644 --- a/src/shapepipe/modules/ngmix_runner.py +++ b/src/shapepipe/modules/ngmix_runner.py @@ -42,7 +42,9 @@ def ngmix_runner( module_config_sec, w_log, ): - """Define The Ngmix Runner.""" + """Define The Ngmix Runner. + @sc [decision:shape_measurement.blend_handling,decision:shape_measurement.centroid_source,decision:shape_measurement.defect_fill,decision:shape_measurement.metacal_scheme] + """ # Read config file entries # Photometric zero point diff --git a/src/shapepipe/modules/sextractor_package/sextractor_script.py b/src/shapepipe/modules/sextractor_package/sextractor_script.py index a1075822d..8faf240ab 100644 --- a/src/shapepipe/modules/sextractor_package/sextractor_script.py +++ b/src/shapepipe/modules/sextractor_package/sextractor_script.py @@ -478,6 +478,7 @@ def get_zero_point(self, use_zp, zp_key=None): zp_key: str Header key corresponding to the zero point + @sc [decision:photometric_zeropoint] """ if use_zp and not isinstance(zp_key, type(None)): zp_value = get_header_value(self._meas_img_path, zp_key) @@ -495,6 +496,7 @@ def get_background(self, use_bkg, bkg_key=None): bkg_key: str Header key corresponding to the background value + @sc [decision:detection.background_model] """ if use_bkg and not isinstance(bkg_key, type(None)): bkg_value = get_header_value(self._meas_img_path, bkg_key) diff --git a/src/shapepipe/modules/vignetmaker_package/vignetmaker.py b/src/shapepipe/modules/vignetmaker_package/vignetmaker.py index b03712bb1..9f119be20 100644 --- a/src/shapepipe/modules/vignetmaker_package/vignetmaker.py +++ b/src/shapepipe/modules/vignetmaker_package/vignetmaker.py @@ -313,6 +313,7 @@ def _get_stamp_me(self, image_dirs, image_pattern): dict Dictionary containing object id and vignets for each epoch + @sc [decision:preparation.stamp_positioning_and_padding] """ cat = file_io.FITSCatalogue(self._galcat_path, SEx_catalogue=True) cat.open() diff --git a/src/shapepipe/utilities/mask_query.py b/src/shapepipe/utilities/mask_query.py index fe40b3806..d0920f27e 100644 --- a/src/shapepipe/utilities/mask_query.py +++ b/src/shapepipe/utilities/mask_query.py @@ -210,6 +210,7 @@ def query_map(path, ra, dec): Map value at each position; positions outside the map's coverage carry the map's sentinel value + @sc [decision:masking.sky_mask_application] """ values, _ = query_map_coverage(path, ra, dec) diff --git a/tests/unit/test_decisions.py b/tests/unit/test_decisions.py index 096ef7f7e..85978845d 100644 --- a/tests/unit/test_decisions.py +++ b/tests/unit/test_decisions.py @@ -12,6 +12,9 @@ import_violations, load_yaml, main, + repository_errors, + _iter_decision_defs, + _parse_values, scan_tags, tag_errors, value_errors, @@ -420,6 +423,18 @@ def test_cli_reports_decision_and_local_contract_for_a_location(tmp_path, capsys assert "The threshold must remain coupled." in output +def test_repository_decisions_have_sites_and_all_values_match(): + errors = repository_errors(REPO_ROOT) + assert not errors, errors + + record = load_yaml(REPO_ROOT / "astra.yaml") + values_count = sum( + len(_parse_values(definition.get("rationale", ""))[0]) + for _, definition in _iter_decision_defs(record) + ) + assert values_count == 151 + + def test_preserved_utilities_import_rule(tmp_path): contracts = _write( tmp_path / "pkg" / "utilities" / "CONTRACTS", diff --git a/workflow/config.yaml b/workflow/config.yaml index 63b29413c..29ed5189f 100644 --- a/workflow/config.yaml +++ b/workflow/config.yaml @@ -1,3 +1,4 @@ +# @sc [decision:star_selection_psf.psf_modelling_software,scope:file] # Run configuration for the ShapePipe Snakemake workflow. # # A "run" is declared by a tile list plus the paths below. Everything here is diff --git a/workflow/config/cfis/config_MCCD.ini b/workflow/config/cfis/config_MCCD.ini index bf5e6852e..be2f7f968 100644 --- a/workflow/config/cfis/config_MCCD.ini +++ b/workflow/config/cfis/config_MCCD.ini @@ -6,20 +6,31 @@ PREPROCESSED_OUTPUT_DIR = ./output OUTPUT_DIR = ./output INPUT_REGEX_FILE_PATTERN = star_split_ratio_80-*-*.fits INPUT_SEPARATOR = - + +# @sc [decision:star_selection_psf.psf_modelling_software] MIN_N_STARS = 20 + OUTLIER_STD_MAX = 100. USE_SNR_WEIGHTS = False [INSTANCE] + +# @sc [decision:star_selection_psf.psf_modelling_software] N_COMP_LOC = 8 D_COMP_GLOB = 8 + KSIG_LOC = 0.00 KSIG_GLOB = 0.00 FILTER_PATH = None D_HYB_LOC = 2 MIN_D_COMP_GLOB = None + +# @sc [decision:star_selection_psf.psf_modelling_software] RMSE_THRESH = 1.25 + CCD_STAR_THRESH = 0.15 + +# @sc [decision:star_selection_psf.psf_modelling_software] FP_GEOMETRY = CFIS [FIT] diff --git a/workflow/config/cfis/config_exp_Sp.ini b/workflow/config/cfis/config_exp_Sp.ini index dd27d6ccd..58d03c001 100644 --- a/workflow/config/cfis/config_exp_Sp.ini +++ b/workflow/config/cfis/config_exp_Sp.ini @@ -75,4 +75,6 @@ NUMBERING_SCHEME = -0000000 OUTPUT_SUFFIX = image, weight, flag # Number of HDUs/CCDs of mosaic + +# @sc [decision:preparation.ccd_split_extent] N_HDU = 40 diff --git a/workflow/config/cfis/config_exp_mccd.ini b/workflow/config/cfis/config_exp_mccd.ini index a7d6a3500..ae1479803 100644 --- a/workflow/config/cfis/config_exp_mccd.ini +++ b/workflow/config/cfis/config_exp_mccd.ini @@ -22,6 +22,8 @@ RUN_DATETIME = False [EXECUTION] # Module name, single string or comma-separated list of valid module runner names + +# @sc [decision:star_selection_psf.psf_modelling_software] MODULE = sextractor_runner, mask_query_runner, setools_runner, mccd_preprocessing_runner, mccd_fit_val_runner, merge_starcat_runner, mccd_plots_runner @@ -61,9 +63,13 @@ TIMEOUT = 96:00:00 [SEXTRACTOR_RUNNER] # The split CCDs, and nothing else: ShapePipe generates no masks + +# @sc [decision:star_selection_psf.psf_modelling_software] INPUT_DIR = $SP_RUN/output/run_sp_exp_Sp/split_exp_runner/output # Read the instrument flag image split_exp wrote per CCD + +# @sc [decision:star_selection_psf.psf_modelling_software] FILE_PATTERN = image, weight, flag # Explicit extensions: a 3-entry FILE_PATTERN override must not fall back on @@ -84,6 +90,8 @@ DOT_CONV_FILE = $SP_CONFIG/default.conv WEIGHT_IMAGE = True # Use input flag image if True + +# @sc [decision:star_selection_psf.psf_modelling_software] FLAG_IMAGE = True # Use input PSF file if True diff --git a/workflow/config/cfis/config_exp_psfex.ini b/workflow/config/cfis/config_exp_psfex.ini index 495d9b78a..0bbd46b64 100644 --- a/workflow/config/cfis/config_exp_psfex.ini +++ b/workflow/config/cfis/config_exp_psfex.ini @@ -1,3 +1,4 @@ +# @sc [decision:masking.psf_star_mask_veto,scope:file] # ShapePipe configuration file for single-exposures. PSFex PSF model. # Process exposures after splitting, from star detection to PSF model. # ShapePipe generates no masks: SExtractor reads the instrument flag image @@ -56,6 +57,7 @@ TIMEOUT = 96:00:00 ## Module options +# @sc [decision:detection.background_model] [SEXTRACTOR_RUNNER] # The split CCDs, and nothing else: ShapePipe generates no masks @@ -76,12 +78,18 @@ EXEC_PATH = source-extractor # SExtractor configuration files DOT_SEX_FILE = $SP_CONFIG/default_exp.sex DOT_PARAM_FILE = $SP_CONFIG//default.param + +# @sc [decision:detection.detection_threshold_policy] DOT_CONV_FILE = $SP_CONFIG/default.conv # Use input weight image if True + +# @sc [decision:detection.weight_map_usage] WEIGHT_IMAGE = True # Use input flag image if True + +# @sc [decision:masking.pixel_mask_source] FLAG_IMAGE = True # Use input PSF file if True @@ -96,9 +104,13 @@ DETECTION_IMAGE = False DETECTION_WEIGHT = False # True if photometry zero-point is to be read from exposure image header + +# @sc [decision:photometric_zeropoint] ZP_FROM_HEADER = True # If ZP_FROM_HEADER is True, zero-point key name + +# @sc [decision:photometric_zeropoint] ZP_KEY = PHOTZP # Background information from image header. @@ -178,6 +190,8 @@ SETOOLS_CONFIG_PATH = $SP_CONFIG/star_selection.setools [PSFEX_RUNNER] # Use 80% sample for PSF model + +# @sc [decision:star_selection_psf.psf_train_validation_split] FILE_PATTERN = star_split_ratio_80 NUMBERING_SCHEME = -0000000-0 @@ -191,6 +205,8 @@ DOT_PSFEX_FILE = $SP_CONFIG/default.psfex [PSFEX_INTERP_RUNNER] # Use 20% sample for PSF validation + +# @sc [decision:star_selection_psf.psf_train_validation_split] FILE_PATTERN = star_split_ratio_80, star_split_ratio_20, psfex_cat FILE_EXT = .psf, .fits, .cat @@ -204,13 +220,19 @@ NUMBERING_SCHEME = -0000000-0 MODE = VALIDATION # Column names of position parameters + +# @sc [decision:preparation.object_position_columns] POSITION_PARAMS = XWIN_IMAGE,YWIN_IMAGE # If True, measure and store ellipticity of the PSF (using moments) GET_SHAPES = True # Minimum number of stars per CCD for PSF model to be computed + +# @sc [decision:star_selection_psf.psf_acceptance_thresholds] STAR_THRESH = 22 # Maximum chi^2 for PSF model to be computed on CCD + +# @sc [decision:star_selection_psf.psf_acceptance_thresholds] CHI2_THRESH = 2 diff --git a/workflow/config/cfis/config_tile_Fe.ini b/workflow/config/cfis/config_tile_Fe.ini index 5fa45bead..573113506 100644 --- a/workflow/config/cfis/config_tile_Fe.ini +++ b/workflow/config/cfis/config_tile_Fe.ini @@ -69,10 +69,14 @@ FILE_EXT = .fits NUMBERING_SCHEME = -000-000 # Column number of exposure name in FITS header + +# @sc [decision:preparation.epoch_provenance_from_tile_history] COLNUM = 3 # Prefix to remove from exposure name. CFIS exposure names carry no # prefix -- the trailing "p" is a suffix (epoch letter), kept as part of # the exposure name, not stripped by this key. + +# @sc [decision:preparation.epoch_provenance_from_tile_history] EXP_PREFIX = diff --git a/workflow/config/cfis/config_tile_Mc.ini b/workflow/config/cfis/config_tile_Mc.ini index 5920d0a91..e45564b21 100644 --- a/workflow/config/cfis/config_tile_Mc.ini +++ b/workflow/config/cfis/config_tile_Mc.ini @@ -79,6 +79,7 @@ FILE_EXT = .fits, .sqlite, .fits # sections below NUMBERING_SCHEME = -000-000 +# @sc [decision:catalogue_assembly.star_galaxy_classification] SM_DO_CLASSIFICATION = False SHAPE_MEASUREMENT_TYPE = ngmix diff --git a/workflow/config/cfis/config_tile_Ng_template.ini b/workflow/config/cfis/config_tile_Ng_template.ini index 04606636d..7e5df62e4 100644 --- a/workflow/config/cfis/config_tile_Ng_template.ini +++ b/workflow/config/cfis/config_tile_Ng_template.ini @@ -101,6 +101,8 @@ NUMBERING_SCHEME = -000-000 # 1/RMS^2 inverse-variance ngmix weights. When set, the file must exist for # every tile (missing file -> error, no per-tile fallback); omit the option # entirely to fall back to the scalar sigma_mad noise estimate. + +# @sc [decision:shape_measurement.galaxy_pixel_weights] BKG_RMS_VIGNET_PATH = $NGMIX_VIGNET_DIR/vignetmaker_runner_run_2/output/background_rms_vignet{file_number_string}.sqlite # Number of objects to batch save during processing, optional. Omit or set @@ -112,9 +114,13 @@ BKG_RMS_VIGNET_PATH = $NGMIX_VIGNET_DIR/vignetmaker_runner_run_2/output/backgrou SAVE_BATCH = 250 # Magnitude zero-point + +# @sc [decision:photometric_zeropoint] MAG_ZP = 30.0 # Pixel scale in arcsec + +# @sc [decision:shape_measurement.fit_priors] PIXEL_SCALE = 0.186 # ID_OBJ_MIN/MAX: this chunk's closed SExtractor NUMBER-column range, diff --git a/workflow/config/cfis/config_tile_PiViVi_psfex.ini b/workflow/config/cfis/config_tile_PiViVi_psfex.ini index 27a759528..0427c86a1 100644 --- a/workflow/config/cfis/config_tile_PiViVi_psfex.ini +++ b/workflow/config/cfis/config_tile_PiViVi_psfex.ini @@ -93,15 +93,21 @@ NUMBERING_SCHEME = -000-000 MODE = MULTI-EPOCH # Column names of position parameters + +# @sc [decision:preparation.object_position_columns] POSITION_PARAMS = XWIN_WORLD,YWIN_WORLD # If True, measure and store ellipticity of the PSF GET_SHAPES = True # Number of stars threshold + +# @sc [decision:star_selection_psf.psf_acceptance_thresholds] STAR_THRESH = 22 # chi^2 threshold + +# @sc [decision:star_selection_psf.psf_acceptance_thresholds] CHI2_THRESH = 2 # Multi-epoch mode parameters @@ -112,6 +118,8 @@ CHI2_THRESH = 2 ME_DOT_PSF_EXP_DIR = $SP_EXP # Input psf file pattern + +# @sc [decision:star_selection_psf.psf_train_validation_split] ME_DOT_PSF_PATTERN = star_split_ratio_80 @@ -127,7 +135,9 @@ FILE_EXT = .fits, .fits # NUMBERING_SCHEME (optional) string with numbering pattern for input files NUMBERING_SCHEME = -000-000 +# @sc [decision:postage_stamp_size] MASKING = False + MASK_VALUE = 0 # Run mode for psfex interpolation: @@ -137,10 +147,14 @@ MASK_VALUE = 0 MODE = CLASSIC # Coordinate frame type, one in PIX (pixel frame), SPHE (spherical coordinates) + +# @sc [decision:preparation.object_position_columns] COORD = PIX POSITION_PARAMS = XWIN_IMAGE,YWIN_IMAGE # Vignet size in pixels + +# @sc [decision:postage_stamp_size] STAMP_SIZE = 51 # Output file name prefix, file name is _vignet.fits @@ -161,7 +175,9 @@ FILE_EXT = .fits, .sqlite, .txt # NUMBERING_SCHEME (optional) string with numbering pattern for input files NUMBERING_SCHEME = -000-000 +# @sc [decision:postage_stamp_size] MASKING = False + MASK_VALUE = 0 # Run mode for psfex interpolation: @@ -171,10 +187,14 @@ MASK_VALUE = 0 MODE = MULTI-EPOCH # Coordinate frame type, one in PIX (pixel frame), SPHE (spherical coordinates) + +# @sc [decision:preparation.object_position_columns] COORD = SPHE POSITION_PARAMS = XWIN_WORLD,YWIN_WORLD # Vignet size in pixels + +# @sc [decision:postage_stamp_size] STAMP_SIZE = 51 # Output file name prefix, file name is vignet.fits @@ -185,4 +205,6 @@ PREFIX = # the v2.0 per-exposure pipeline; output dirs are discovered by scanning $SP_EXP. ME_IMAGE_EXP_DIR = $SP_EXP ME_IMAGE_EXP_RUNNERS = split_exp_runner, split_exp_runner, split_exp_runner, sextractor_runner, sextractor_runner + +# @sc [decision:masking.pixel_mask_source] ME_IMAGE_PATTERN = flag, image, weight, background, background_rms diff --git a/workflow/config/cfis/config_tile_Sx.ini b/workflow/config/cfis/config_tile_Sx.ini index 4dccaf448..1e32dfd44 100644 --- a/workflow/config/cfis/config_tile_Sx.ini +++ b/workflow/config/cfis/config_tile_Sx.ini @@ -74,12 +74,18 @@ EXEC_PATH = source-extractor # SExtractor configuration files DOT_SEX_FILE = $SP_CONFIG/default_tile.sex DOT_PARAM_FILE = $SP_CONFIG/default_noimaflags.param + +# @sc [decision:detection.detection_threshold_policy] DOT_CONV_FILE = $SP_CONFIG/gauss_3.0_7x7.conv # Use input weight image if True + +# @sc [decision:detection.weight_map_usage] WEIGHT_IMAGE = True # Use input flag image if True + +# @sc [decision:detection.detection_source_mode] FLAG_IMAGE = False # Use input PSF file if True @@ -87,14 +93,18 @@ PSF_FILE = False # Use distinct image for detection (SExtractor in # dual-image mode) if True + +# @sc [decision:detection.detection_source_mode] DETECTION_IMAGE = False # Distinct weight image for detection (SExtractor # in dual-image mode) DETECTION_WEIGHT = False +# @sc [decision:photometric_zeropoint] ZP_FROM_HEADER = False +# @sc [decision:detection.background_model] BKG_FROM_HEADER = False # Type of image check (optional), default not used, can be a list of @@ -109,10 +119,14 @@ SUFFIX = sexcat ## Post-processing # Necessary for tiles, to enable multi-exposure processing + +# @sc [decision:detection.epoch_membership_ccd_bounds] MAKE_POST_PROCESS = True # World coordinate keywords, SExtractor output. Format: KEY_X,KEY_Y WORLD_POSITION = XWIN_WORLD,YWIN_WORLD # Number of pixels in x,y of a CCD. Format: Nx,Ny + +# @sc [decision:detection.epoch_membership_ccd_bounds] CCD_SIZE = 33,2080,1,4612 diff --git a/workflow/config/cfis/default.conv b/workflow/config/cfis/default.conv index 2590b9cba..2c1a71ac6 100644 --- a/workflow/config/cfis/default.conv +++ b/workflow/config/cfis/default.conv @@ -1,3 +1,4 @@ +# @sc [decision:detection.detection_threshold_policy,scope:file] CONV NORM # 3x3 ``all-ground'' convolution mask with FWHM = 2 pixels. 1 2 1 diff --git a/workflow/config/cfis/default.param b/workflow/config/cfis/default.param index 09ad8405e..3fe6978bc 100644 --- a/workflow/config/cfis/default.param +++ b/workflow/config/cfis/default.param @@ -59,6 +59,7 @@ FWHM_WORLD #FWHM assuming a gaussian core ELONGATION #A_IMAGE/B_IMAGE ELLIPTICITY #1 - B_IMAGE/A_IMAGE +# @sc [decision:postage_stamp_size] VIGNET(51,51) #Pixel data around detection [count] # For GaaP photometry diff --git a/workflow/config/cfis/default.psfex b/workflow/config/cfis/default.psfex index a9d1a906c..f9634da07 100644 --- a/workflow/config/cfis/default.psfex +++ b/workflow/config/cfis/default.psfex @@ -1,40 +1,62 @@ +# @sc [decision:star_selection_psf.psfex_candidate_vetting,scope:file] # Default configuration file for PSFEx 3.17.1 # EB 2017-11-30 # #-------------------------------- PSF model ---------------------------------- +# @sc [decision:star_selection_psf.psf_model_complexity] BASIS_TYPE PIXEL # NONE, PIXEL, GAUSS-LAGUERRE or FILE BASIS_NUMBER 20 # Basis number or parameter + BASIS_NAME basis.fits # Basis filename (FITS data-cube) BASIS_SCALE 1.0 # Gauss-Laguerre beta parameter NEWBASIS_TYPE NONE # Create new basis: NONE, PCA_INDEPENDENT # or PCA_COMMON NEWBASIS_NUMBER 8 # Number of new basis vectors + +# @sc [decision:star_selection_psf.psf_model_complexity] PSF_SAMPLING 1. # Sampling step in pixel units (0.0 = auto) + PSF_PIXELSIZE 1.0 # Effective pixel size in pixel step units + +# @sc [decision:star_selection_psf.psf_model_complexity] PSF_ACCURACY 0.01 # Accuracy to expect from PSF "pixel" values + +# @sc [decision:postage_stamp_size] PSF_SIZE 51,51 # Image size of the PSF model + PSF_RECENTER N # Allow recentering of PSF-candidates Y/N ? + +# @sc [decision:star_selection_psf.psf_model_complexity] MEF_TYPE INDEPENDENT # INDEPENDENT or COMMON #------------------------- Point source measurements ------------------------- CENTER_KEYS XWIN_IMAGE,YWIN_IMAGE # Catalogue parameters for source pre-centering + +# @sc [decision:detection.photometry_parameters] PHOTFLUX_KEY FLUX_AUTO # Catalogue parameter for photometric norm. + PHOTFLUXERR_KEY FLUXERR_AUTO # Catalogue parameter for photometric error #----------------------------- PSF variability ------------------------------- +# @sc [decision:star_selection_psf.psf_model_complexity] PSFVAR_KEYS XWIN_IMAGE,YWIN_IMAGE # Catalogue or FITS (preceded by :) params + PSFVAR_GROUPS 1,1 # Group tag for each context key + +# @sc [decision:star_selection_psf.psf_model_complexity] PSFVAR_DEGREES 2 # Polynom degree for each group + PSFVAR_NSNAP 9 # Number of PSF snapshots per axis HIDDENMEF_TYPE COMMON # INDEPENDENT or COMMON STABILITY_TYPE EXPOSURE # EXPOSURE or SEQUENCE #----------------------------- Sample selection ------------------------------ +# @sc [decision:star_selection_psf.star_selection_box] SAMPLE_AUTOSELECT N # Automatically select the FWHM (Y/N) ? BADPIXEL_FILTER N # Filter bad-pixels in samples (Y/N) ? diff --git a/workflow/config/cfis/default_exp.sex b/workflow/config/cfis/default_exp.sex index b87275ecb..e8203ba8f 100644 --- a/workflow/config/cfis/default_exp.sex +++ b/workflow/config/cfis/default_exp.sex @@ -1,3 +1,4 @@ +# @sc [decision:detection.saturation_level,scope:file] # Default configuration file for SExtractor 2.19.5 # EB 2017-11-30 # @@ -11,33 +12,49 @@ PARAMETERS_NAME default.param #------------------------------- Extraction ---------------------------------- DETECT_TYPE CCD # CCD (linear) or PHOTO (with gamma correction) + +# @sc [decision:detection.detection_threshold_policy] DETECT_MINAREA 5 # min. # of pixels above threshold + DETECT_MAXAREA 0 # max. # of pixels above threshold (0=unlimited) THRESH_TYPE RELATIVE # threshold type: RELATIVE (in sigmas) # or ABSOLUTE (in ADUs) + +# @sc [decision:detection.detection_threshold_policy] DETECT_THRESH 1.5 # or , in mag.arcsec-2 ANALYSIS_THRESH 1.5 # or , in mag.arcsec-2 +# @sc [decision:detection.detection_threshold_policy] FILTER Y # apply filter for detection (Y or N)? + FILTER_NAME default.conv FILTER_THRESH # Threshold[s] for retina filtering +# @sc [decision:detection.deblending_policy] DEBLEND_NTHRESH 32 # Number of deblending sub-thresholds DEBLEND_MINCONT 0.001 # Minimum contrast parameter for deblending +# @sc [decision:detection.spurious_detection_cleaning] CLEAN Y # Clean spurious detections? (Y or N)? CLEAN_PARAM 1.0 # Cleaning efficiency +# @sc [decision:detection.blend_photometry_mask_type] MASK_TYPE CORRECT # type of detection MASKing: can be one of + # NONE, BLANK or CORRECT #-------------------------------- WEIGHTing ---------------------------------- +# @sc [decision:detection.weight_map_usage] WEIGHT_TYPE MAP_WEIGHT # type of WEIGHTing: NONE, BACKGROUND, # MAP_RMS, MAP_VAR or MAP_WEIGHT RESCALE_WEIGHTS Y # Rescale input weights/variances (Y/N)? + WEIGHT_IMAGE weight.fits # weight-map filename + +# @sc [decision:detection.weight_map_usage] WEIGHT_GAIN Y # modulate gain (E/ADU) with weights? (Y/N) + WEIGHT_THRESH # weight threshold[s] for bad pixels #-------------------------------- FLAGging ----------------------------------- @@ -48,12 +65,16 @@ FLAG_TYPE OR # flag pixel combination: OR, AND, MIN, MAX #------------------------------ Photometry ----------------------------------- +# @sc [decision:detection.photometry_parameters] PHOT_APERTURES 5 # MAG_APER aperture diameter(s) in pixels PHOT_AUTOPARAMS 2.5, 3.5 # MAG_AUTO parameters: , + PHOT_PETROPARAMS 2.0, 3.5 # MAG_PETRO parameters: , # PHOT_AUTOAPERS 0.0,0.0 # , minimum apertures # for MAG_AUTO and MAG_PETRO + +# @sc [decision:detection.photometry_parameters] PHOT_FLUXFRAC 0.5 # flux fraction[s] used for FLUX_RADIUS SATUR_KEY SATURATE # keyword for saturation level (in ADUs) @@ -66,17 +87,25 @@ PIXEL_SCALE 0. # size of pixel in arcsec (0=use FITS WCS info) #------------------------- Star/Galaxy Separation ---------------------------- +# @sc [decision:detection.detection_threshold_policy] SEEING_FWHM 0.6 # stellar FWHM in arcsec + STARNNW_NAME default.nnw #------------------------------ Background ----------------------------------- +# @sc [decision:detection.background_model] BACK_TYPE AUTO # AUTO or MANUAL + BACK_VALUE 0.0 # Default background value in MANUAL mode + +# @sc [decision:detection.background_model] BACK_SIZE 64 # Background mesh: or , BACK_FILTERSIZE 3 # Background filter: or , +# @sc [decision:detection.background_model] BACKPHOTO_TYPE GLOBAL # can be GLOBAL or LOCAL + BACKPHOTO_THICK 24 # thickness of the background LOCAL annulus BACK_FILTTHRESH 0.0 # Threshold above which the background- # map filter operates @@ -120,6 +149,8 @@ WRITE_XML N # Write XML file (Y/N)? NTHREADS 1 # 1 single thread FITS_UNSIGNED N # Treat FITS integer values as unsigned (Y/N)? + +# @sc [decision:detection.zero_weight_interpolation] INTERP_MAXXLAG 16 # Max. lag along X for 0-weight interpolation INTERP_MAXYLAG 16 # Max. lag along Y for 0-weight interpolation INTERP_TYPE ALL # Interpolation type: NONE, VAR_ONLY or ALL diff --git a/workflow/config/cfis/default_noimaflags.param b/workflow/config/cfis/default_noimaflags.param index b251f5b74..fff22641d 100644 --- a/workflow/config/cfis/default_noimaflags.param +++ b/workflow/config/cfis/default_noimaflags.param @@ -1,3 +1,4 @@ +# @sc [decision:detection.detection_source_mode,scope:file] NUMBER #Running object number EXT_NUMBER #FITS extension number @@ -56,6 +57,7 @@ FWHM_WORLD #FWHM assuming a gaussian core ELONGATION #A_IMAGE/B_IMAGE ELLIPTICITY #1 - B_IMAGE/A_IMAGE +# @sc [decision:postage_stamp_size] VIGNET(51,51) #Pixel data around detection [count] # For GaaP photometry diff --git a/workflow/config/cfis/default_tile.sex b/workflow/config/cfis/default_tile.sex index e706e81e4..7c0f61b84 100644 --- a/workflow/config/cfis/default_tile.sex +++ b/workflow/config/cfis/default_tile.sex @@ -1,3 +1,4 @@ +# @sc [decision:detection.saturation_level,scope:file] # Default configuration file for SExtractor 2.19.5 # EB 2017-11-30 # @@ -11,33 +12,49 @@ PARAMETERS_NAME default.param #------------------------------- Extraction ---------------------------------- DETECT_TYPE CCD # CCD (linear) or PHOTO (with gamma correction) + +# @sc [decision:detection.detection_threshold_policy] DETECT_MINAREA 3 # min. # of pixels above threshold + DETECT_MAXAREA 0 # max. # of pixels above threshold (0=unlimited) THRESH_TYPE RELATIVE # threshold type: RELATIVE (in sigmas) # or ABSOLUTE (in ADUs) + +# @sc [decision:detection.detection_threshold_policy] DETECT_THRESH 1.0 # or , in mag.arcsec-2 ANALYSIS_THRESH 1.0 # or , in mag.arcsec-2 +# @sc [decision:detection.detection_threshold_policy] FILTER Y # apply filter for detection (Y or N)? + FILTER_NAME gauss_3.0_7x7.conv FILTER_THRESH # Threshold[s] for retina filtering +# @sc [decision:detection.deblending_policy] DEBLEND_NTHRESH 32 # Number of deblending sub-thresholds DEBLEND_MINCONT 0.002 # Minimum contrast parameter for deblending +# @sc [decision:detection.spurious_detection_cleaning] CLEAN Y # Clean spurious detections? (Y or N)? CLEAN_PARAM 1.0 # Cleaning efficiency +# @sc [decision:detection.blend_photometry_mask_type] MASK_TYPE CORRECT # type of detection MASKing: can be one of + # NONE, BLANK or CORRECT #-------------------------------- WEIGHTing ---------------------------------- +# @sc [decision:detection.weight_map_usage] WEIGHT_TYPE MAP_WEIGHT # type of WEIGHTing: NONE, BACKGROUND, # MAP_RMS, MAP_VAR or MAP_WEIGHT RESCALE_WEIGHTS Y # Rescale input weights/variances (Y/N)? + WEIGHT_IMAGE weight.fits # weight-map filename + +# @sc [decision:detection.weight_map_usage] WEIGHT_GAIN Y # modulate gain (E/ADU) with weights? (Y/N) + WEIGHT_THRESH # weight threshold[s] for bad pixels #-------------------------------- FLAGging ----------------------------------- @@ -48,17 +65,23 @@ FLAG_TYPE OR # flag pixel combination: OR, AND, MIN, MAX #------------------------------ Photometry ----------------------------------- +# @sc [decision:detection.photometry_parameters] PHOT_APERTURES 5 # MAG_APER aperture diameter(s) in pixels PHOT_AUTOPARAMS 2.5, 3.5 # MAG_AUTO parameters: , + PHOT_PETROPARAMS 2.0, 3.5 # MAG_PETRO parameters: , # PHOT_AUTOAPERS 0.0,0.0 # , minimum apertures # for MAG_AUTO and MAG_PETRO + +# @sc [decision:detection.photometry_parameters] PHOT_FLUXFRAC 0.5 # flux fraction[s] used for FLUX_RADIUS SATUR_KEY SATURATE # keyword for saturation level (in ADUs) +# @sc [decision:photometric_zeropoint] MAG_ZEROPOINT 30.0 # magnitude zero-point + MAG_GAMMA 4.0 # gamma of emulsion (for photographic scans) GAIN_KEY GAIN # keyword for detector gain in e-/ADU @@ -66,18 +89,26 @@ PIXEL_SCALE 0. # size of pixel in arcsec (0=use FITS WCS info) #------------------------- Star/Galaxy Separation ---------------------------- +# @sc [decision:detection.detection_threshold_policy] SEEING_FWHM 0.6 # stellar FWHM in arcsec + STARNNW_NAME default.nnw #------------------------------ Background ----------------------------------- +# @sc [decision:detection.background_model] BACK_TYPE AUTO # AUTO or MANUAL + BACK_VALUE 0.0 # Default background value in MANUAL mode + +# @sc [decision:detection.background_model] BACK_SIZE 512 # Background mesh: or , BACK_FILTERSIZE 9 # Background filter: or , +# @sc [decision:detection.background_model] BACKPHOTO_TYPE LOCAL # can be GLOBAL or LOCAL BACKPHOTO_THICK 30 # thickness of the background LOCAL annulus + BACK_FILTTHRESH 0.0 # Threshold above which the background- # map filter operates @@ -120,6 +151,8 @@ WRITE_XML N # Write XML file (Y/N)? NTHREADS 1 # 1 single thread FITS_UNSIGNED N # Treat FITS integer values as unsigned (Y/N)? + +# @sc [decision:detection.zero_weight_interpolation] INTERP_MAXXLAG 16 # Max. lag along X for 0-weight interpolation INTERP_MAXYLAG 16 # Max. lag along Y for 0-weight interpolation INTERP_TYPE ALL # Interpolation type: NONE, VAR_ONLY or ALL diff --git a/workflow/config/cfis/final_cat.param b/workflow/config/cfis/final_cat.param index 00bcb3f73..12debe7fb 100644 --- a/workflow/config/cfis/final_cat.param +++ b/workflow/config/cfis/final_cat.param @@ -1,3 +1,4 @@ +# @sc [decision:catalogue_assembly.star_galaxy_classification,scope:file] # coordinates XWIN_WORLD YWIN_WORLD @@ -8,7 +9,10 @@ TILE_ID # flags FLAGS + +# @sc [decision:detection.detection_source_mode] IMAFLAGS_ISO + NGMIX_MCAL_FLAGS # PSF ellipticity (original image PSF) diff --git a/workflow/config/cfis/gauss_3.0_7x7.conv b/workflow/config/cfis/gauss_3.0_7x7.conv index 527acbbb0..3f12d1eaa 100644 --- a/workflow/config/cfis/gauss_3.0_7x7.conv +++ b/workflow/config/cfis/gauss_3.0_7x7.conv @@ -1,3 +1,4 @@ +# @sc [decision:detection.detection_threshold_policy,scope:file] CONV NORM # 7x7 convolution mask of a gaussian PSF with FWHM = 3.0 pixels. 0.004963 0.021388 0.051328 0.068707 0.051328 0.021388 0.004963 diff --git a/workflow/config/cfis/star_selection.setools b/workflow/config/cfis/star_selection.setools index 10ab60fd8..5a71d8c31 100644 --- a/workflow/config/cfis/star_selection.setools +++ b/workflow/config/cfis/star_selection.setools @@ -28,28 +28,47 @@ ## config, which ships commented out; SETools has no bitwise operators, so the ## bit selection happens there and this file only ever tests for zero. +# @sc [decision:masking.psf_star_mask_veto] [MASK:preselect] + +# @sc [decision:star_selection_psf.star_selection_box] MAG_AUTO > 0 MAG_AUTO < 21 FWHM_IMAGE > 0.3 / 0.187 FWHM_IMAGE < 1.5 / 0.187 + +# @sc [decision:detection.saturation_level,decision:star_selection_psf.star_selection_box] FLAGS == 0 + +# @sc [decision:star_selection_psf.star_selection_box] IMAFLAGS_ISO == 0 + NO_SAVE +# @sc [decision:masking.psf_star_mask_veto] [MASK:flag] + +# @sc [decision:detection.saturation_level] FLAGS == 0 + IMAFLAGS_ISO == 0 NO_SAVE +# @sc [decision:masking.psf_star_mask_veto] [MASK:star_selection] # Star selection using the FWHM mode + +# @sc [decision:star_selection_psf.star_selection_box] MAG_AUTO > 18. MAG_AUTO < 22. FWHM_IMAGE <= mode(FWHM_IMAGE{preselect}) + 0.2 FWHM_IMAGE >= mode(FWHM_IMAGE{preselect}) - 0.2 + +# @sc [decision:detection.saturation_level,decision:star_selection_psf.star_selection_box] FLAGS == 0 + +# @sc [decision:star_selection_psf.star_selection_box] IMAFLAGS_ISO == 0 [MASK:fwhm_mag_cut] @@ -63,7 +82,10 @@ NO_SAVE # Split the 'star_selection' sample into # two random sub-samples with ratio 80/20 [RAND_SPLIT:star_split] + +# @sc [decision:star_selection_psf.psf_train_validation_split] RATIO = 20 + MASK = star_selection # The following selection is only used for plotting @@ -100,7 +122,10 @@ TYPE = scatter FORMAT = png X = X_IMAGE{star_selection} Y = Y_IMAGE{star_selection} + +# @sc [decision:star_selection_psf.star_selection_box] SCATTER = FWHM_IMAGE{star_selection}*0.187 + MARKER = . LABEL = "FWHM (arcsec)" TITLE = "FWHM of stars" From 3c5d72a7fcb6e5e8460ef189e06517425bb7c9d2 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 05:25:03 +0200 Subject: [PATCH 30/40] refactor: dissolve config contract sidecars into site tags --- astra.yaml | 51 +++++--- workflow/CONTRACTS | 12 -- workflow/config/cfis/CONTRACTS | 116 ------------------ workflow/config/cfis/config_exp_mccd.ini | 6 +- workflow/config/cfis/config_exp_psfex.ini | 12 +- workflow/config/cfis/config_tile_Mc.ini | 2 + .../config/cfis/config_tile_Ng_template.ini | 2 +- .../config/cfis/config_tile_PiViVi_mccd.ini | 2 + .../config/cfis/config_tile_PiViVi_psfex.ini | 7 +- workflow/config/cfis/config_tile_Sx.ini | 8 +- workflow/config/cfis/default.psfex | 4 +- workflow/config/cfis/default_exp.sex | 7 ++ workflow/config/cfis/default_tile.sex | 6 + workflow/config/cfis/final_cat.param | 5 + workflow/config/cfis/star_selection.setools | 16 +-- 15 files changed, 92 insertions(+), 164 deletions(-) delete mode 100644 workflow/CONTRACTS delete mode 100644 workflow/config/cfis/CONTRACTS diff --git a/astra.yaml b/astra.yaml index 54ce9416a..3fec5498a 100644 --- a/astra.yaml +++ b/astra.yaml @@ -120,8 +120,10 @@ decisions: Tiles use a fixed zero-point of 30 (ZP_FROM_HEADER=False), repeated in ngmix's MAG_ZP; exposures read the per-image header PHOTZP. The fixed value assumes the MegaPipe stacks are calibrated to 30; nothing in the - repo checks it. Magnitude cuts (the star-selection window, downstream - galaxy cuts) inherit their stage's convention. + repo checks it. Exposure epochs are rescaled by FSCALE, which must agree + with header PHOTZP as well as tile MAG_ZEROPOINT and ngmix MAG_ZP. + Magnitude cuts (the star-selection window, downstream galaxy cuts) + inherit their stage's convention. Values: MAG_ZEROPOINT = 30.0; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER = False; NGMIX_RUNNER.MAG_ZP = 30.0; workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER = True; SEXTRACTOR_RUNNER.ZP_KEY = PHOTZP. default: fixed_30_tiles_header_exposures options: @@ -203,7 +205,7 @@ analyses: (detection.detection_source_mode). Sky-fixed masks never touch pixels: an object inside a star halo is measured from the same unmodified pixels as one outside it. - Values: SEXTRACTOR_RUNNER.FLAG_IMAGE = True; VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_PATTERN = flag, image, weight, background, background_rms. + Values: workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE = True; VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_PATTERN = flag, image, weight, background, background_rms. default: instrument_flags_only options: instrument_flags_only: @@ -313,7 +315,8 @@ analyses: adopts; the 7x7 Gaussian of FWHM 3 px is near the CFIS average seeing of 0.65 arcsec (about 3.5 px at 0.187 arcsec/px). Exposures only feed star selection and keep stock values with the 3x3 FWHM 2 px kernel. - The tiles differ from Guinot+22 in all three. + The tiles differ from Guinot+22 in all three. Matching image + simulations must use the same prescription. Values: workflow/config/cfis/default_tile.sex#DETECT_THRESH = 1.0; workflow/config/cfis/default_tile.sex#ANALYSIS_THRESH = 1.0; workflow/config/cfis/default_tile.sex#DETECT_MINAREA = 3; workflow/config/cfis/default_tile.sex#FILTER = Y; workflow/config/cfis/default_tile.sex#SEEING_FWHM = 0.6; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE = $SP_CONFIG/gauss_3.0_7x7.conv; workflow/config/cfis/default_exp.sex#DETECT_THRESH = 1.5; workflow/config/cfis/default_exp.sex#ANALYSIS_THRESH = 1.5; workflow/config/cfis/default_exp.sex#DETECT_MINAREA = 5; workflow/config/cfis/default_exp.sex#FILTER = Y; workflow/config/cfis/default_exp.sex#SEEING_FWHM = 0.6; workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE = $SP_CONFIG/default.conv. default: megapipe_tiles options: @@ -333,7 +336,8 @@ analyses: rationale: >- Tiles use the MegaPipe DEBLEND_MINCONT; exposures use a lower one, the value Guinot+22 lists. Both share DEBLEND_NTHRESH. Contrast sets - object count, centroids, and blend contamination in shapes. + object count, centroids, and blend contamination in shapes; + matching image simulations must preserve the stage-specific settings. Values: workflow/config/cfis/default_tile.sex#DEBLEND_MINCONT = 0.002; workflow/config/cfis/default_exp.sex#DEBLEND_MINCONT = 0.001; workflow/config/cfis/default_tile.sex#DEBLEND_NTHRESH = 32; workflow/config/cfis/default_exp.sex#DEBLEND_NTHRESH = 32. default: megapipe_tiles options: @@ -388,7 +392,8 @@ analyses: rationale: >- INTERP_TYPE ALL on both passes (SExtractor default NONE): SExtractor invents flux across zero-weight pixels, changing detections and - photometry near masked regions. No rationale recorded. + photometry near masked regions. Matching image simulations must use + the same interpolation operator; this is not a calibrated pixel repair. Values: workflow/config/cfis/default_tile.sex#INTERP_TYPE = ALL; workflow/config/cfis/default_exp.sex#INTERP_TYPE = ALL; workflow/config/cfis/default_tile.sex#INTERP_MAXXLAG = 16; workflow/config/cfis/default_tile.sex#INTERP_MAXYLAG = 16; workflow/config/cfis/default_exp.sex#INTERP_MAXXLAG = 16; workflow/config/cfis/default_exp.sex#INTERP_MAXYLAG = 16. default: interp_all options: @@ -401,7 +406,8 @@ analyses: rationale: >- CLEAN on both passes deletes detections consistent with being wings of a brighter neighbour, a post-deblend - change to the object list. Stock value; no rationale recorded. + change to the object list. Keep the same cleaning prescription in + matching image simulations. Values: workflow/config/cfis/default_tile.sex#CLEAN_PARAM = 1.0; workflow/config/cfis/default_exp.sex#CLEAN_PARAM = 1.0; workflow/config/cfis/default_tile.sex#CLEAN = Y; workflow/config/cfis/default_exp.sex#CLEAN = Y. default: clean_1 options: @@ -412,7 +418,8 @@ analyses: rationale: >- MASK_TYPE CORRECT on both passes replaces neighbour pixels by their mirror across the object centre during photometry, changing blend - fluxes and windowed moments. Stock value; no rationale recorded. + fluxes and windowed moments. This SExtractor photometry choice is + separate from ngmix's neighbour-pixel weighting decision. Values: workflow/config/cfis/default_tile.sex#MASK_TYPE = CORRECT; workflow/config/cfis/default_exp.sex#MASK_TYPE = CORRECT. default: correct options: @@ -431,7 +438,8 @@ analyses: falls back to its built-in SATUR_LEVEL (50000 ADU per its documentation), since neither .sex file sets one. Whether the delivered exposure CCDs and MegaPipe tiles carry SATURATE is - unverified. + unverified. Changing the card or pinning a fixed level requires + checking the resulting bright-star selection. Values: workflow/config/cfis/default_exp.sex#SATUR_KEY = SATURATE; workflow/config/cfis/default_tile.sex#SATUR_KEY = SATURATE; MASK:star_selection.FLAGS = "== 0"; MASK:preselect.FLAGS = "== 0"; MASK:flag.FLAGS = "== 0"; workflow/config/cfis/default_tile.sex#SATUR_LEVEL = absent; workflow/config/cfis/default_exp.sex#SATUR_LEVEL = absent. default: header_saturate options: @@ -446,7 +454,8 @@ analyses: rationale recorded. MAG_AUTO is the axis of the star-selection magnitude window and the catalogue magnitude; FLUX_AUTO is PSFEx's photometric normalisation. A different Kron factor shifts magnitudes - and so every magnitude-based cut. + and so every magnitude-based cut. The reported apertures and flux- + fraction measurements must also match the simulated catalogue. Values: workflow/config/cfis/default_tile.sex#PHOT_AUTOPARAMS = 2.5,3.5; workflow/config/cfis/default_exp.sex#PHOT_AUTOPARAMS = 2.5,3.5; workflow/config/cfis/default_tile.sex#PHOT_APERTURES = 5; workflow/config/cfis/default_exp.sex#PHOT_APERTURES = 5; workflow/config/cfis/default_tile.sex#PHOT_FLUXFRAC = 0.5; workflow/config/cfis/default_exp.sex#PHOT_FLUXFRAC = 0.5; PHOTFLUX_KEY = FLUX_AUTO. default: kron_25_35 options: @@ -767,8 +776,10 @@ analyses: counts as warnings because no campaign has run it. Up to the model, its exposure chain matches PSFEx's: SExtractor reads the split image, weight and instrument flag directly, and mask_query sits - between SExtractor and setools. - Values: INSTANCE.N_COMP_LOC = 8; INSTANCE.D_COMP_GLOB = 8; INSTANCE.FP_GEOMETRY = CFIS; INSTANCE.RMSE_THRESH = 1.25; INPUTS.MIN_N_STARS = 20; EXECUTION.MODULE = sextractor_runner,mask_query_runner,setools_runner,mccd_preprocessing_runner,mccd_fit_val_runner,merge_starcat_runner,mccd_plots_runner; SEXTRACTOR_RUNNER.INPUT_DIR = $SP_RUN/output/run_sp_exp_Sp/split_exp_runner/output; SEXTRACTOR_RUNNER.FILE_PATTERN = image,weight,flag; SEXTRACTOR_RUNNER.FLAG_IMAGE = True. + between SExtractor and setools. The selected model name also flows + through SP_PSF in downstream inputs, so producer and consumer choices + must move together. + Values: INSTANCE.N_COMP_LOC = 8; INSTANCE.D_COMP_GLOB = 8; INSTANCE.FP_GEOMETRY = CFIS; INSTANCE.RMSE_THRESH = 1.25; INPUTS.MIN_N_STARS = 20; workflow/config/cfis/config_exp_mccd.ini#EXECUTION.MODULE = sextractor_runner,mask_query_runner,setools_runner,mccd_preprocessing_runner,mccd_fit_val_runner,merge_starcat_runner,mccd_plots_runner; SEXTRACTOR_RUNNER.INPUT_DIR = $SP_RUN/output/run_sp_exp_Sp/split_exp_runner/output; SEXTRACTOR_RUNNER.FILE_PATTERN = image,weight,flag; SEXTRACTOR_RUNNER.FLAG_IMAGE = True. default: psfex options: psfex: @@ -785,7 +796,10 @@ analyses: assumes for PSF pixel values; per the PSFEx documentation it enters the fit weights, and so how closely the model follows bright stars. Model flexibility trades overfitting against PSF leakage, the - dominant additive systematic in cosmic shear. + dominant additive systematic in cosmic shear. A more flexible model + must be supported by the surviving training stars and checked on + held-out residuals under the acceptance gate; successful optimisation + alone does not establish a usable PSF. Stock values; no rationale recorded. Values: BASIS_TYPE = PIXEL; PSF_ACCURACY = 0.01; BASIS_NUMBER = 20; PSFVAR_DEGREES = 2; PSF_SAMPLING = 1; PSFVAR_KEYS = XWIN_IMAGE,YWIN_IMAGE; MEF_TYPE = INDEPENDENT. default: pixel_basis_deg2_per_ccd @@ -1000,9 +1014,10 @@ analyses: fitter is built with the same joint prior. Prior width drives noise bias; no rationale recorded. Guinot+22 used a wider flux prior and a flat prior on r50 rather than T. [LINT] the epoch stamps are - exposure pixels, and star selection uses 0.187 arcsec/px; the - ngmix_runner comment says pixel scale also sets a noise window, but - get_noise is never called. + exposure pixels; star-selection cuts and diagnostics use 0.187 + arcsec/px while NGMIX_RUNNER.PIXEL_SCALE is 0.186. Reconcile these + exposure-pixel conversions; the ngmix_runner comment's noise-window + claim is stale because get_noise is never called. Values: get_prior.T_range = -1,1e3; get_prior.F_range = -100,1e9; NGMIX_RUNNER.PIXEL_SCALE = 0.186. default: gpriorba04_flat options: @@ -1732,7 +1747,9 @@ analyses: Adjacent tiles overlap, and objects in the overlap are measured in both. [HARDCODED] make_cat attaches only TILE_ID; with no unique-object rule or overlap flag, duplicates are left to - downstream selection. + downstream selection. NUMBER_LIST selects input tiles; it does not + carve disjoint sky regions, and no config key makes these catalogues + unique or flags their overlap rows. default: no_dedup_in_pipeline options: no_dedup_in_pipeline: diff --git a/workflow/CONTRACTS b/workflow/CONTRACTS deleted file mode 100644 index 0f6488cbd..000000000 --- a/workflow/CONTRACTS +++ /dev/null @@ -1,12 +0,0 @@ -Scientific contracts for workflow-level configuration -===================================================== - -This directory contains config.yaml, so its PSF-selection contract belongs here rather than only beside the CFIS stage configs. -sc-list includes this file when queried for config.yaml or any config below workflow/config/cfis/. -Refs in governs: are relative to this directory and use the shared ASTRA anchor resolver; config.yaml is a whole-file ref because the resolver has no YAML-key selector. -CFIS key-level contracts live in config/cfis/CONTRACTS. - -@sc [decision:star_selection_psf.psf_modelling_software,governs:config.yaml;config/cfis/config_exp_psfex.ini#EXECUTION.MODULE;config/cfis/config_tile_PiViVi_psfex.ini#EXECUTION.MODULE;config/cfis/config_exp_mccd.ini#EXECUTION.MODULE;config/cfis/config_tile_PiViVi_mccd.ini#EXECUTION.MODULE;config/cfis/config_MCCD.ini#INSTANCE.FP_GEOMETRY] psf-model-selects-a-matched-chain -The psf_model selector must choose a matching exposure-model producer and tile-interpolation consumer, including the model selected through SP_PSF in downstream inputs. -PSFEx models each CCD independently; MCCD couples the focal plane, so switching software changes the scientific model, not just the executable name. -The MCCD exposure chain shares PSFEx's detection inputs, mask_query and star selection, but no campaign has validated its focal-plane model; its warning-only completeness counts do not establish scientific equivalence to the PSFEx chain. diff --git a/workflow/config/cfis/CONTRACTS b/workflow/config/cfis/CONTRACTS deleted file mode 100644 index af1f1bd44..000000000 --- a/workflow/config/cfis/CONTRACTS +++ /dev/null @@ -1,116 +0,0 @@ -Scientific contracts for the committed CFIS configuration -======================================================= - -Read these before changing a governed key; astra.yaml holds the decision and alternatives. -Query sc-list with a config path (for example workflow/config/cfis/default_tile.sex) to see these contracts and the parent workflow/CONTRACTS. -The directory contracts remain visible to sc-list even though it does not scan config comments. -Sidecars are canonical rather than duplicated in comments, keeping native config inputs unchanged and each constraint in one place. - -A governs: value lists ASTRA-style locators relative to this directory, separated by semicolons without spaces; value assertions stay in astra.yaml. -Use file#KEY for sectionless configs, file#SECTION.KEY for INI/SETools, or a bare file for a whole-file constraint, including absent keys. -Keep each tag on one line and separate contracts with a blank line: sc-list reads the following nonblank lines as prose. -Use plain prose for scientific constraints; only:/forbid: lines are for the separate import-boundary checker. -The contract tests resolve every ref and decision id; they do not enforce the scientific prose or certify that a [LINT] is fixed. -Changing a scientific choice requires updating its ASTRA decision and committed universe, not just editing this file. - -Detection and photometry ------------------------- - -@sc [decision:detection.detection_threshold_policy,governs:default_tile.sex#DETECT_THRESH;default_tile.sex#ANALYSIS_THRESH;default_tile.sex#DETECT_MINAREA;default_tile.sex#THRESH_TYPE;default_tile.sex#FILTER;default_tile.sex#FILTER_NAME;config_tile_Sx.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE;gauss_3.0_7x7.conv;default_exp.sex#DETECT_THRESH;default_exp.sex#ANALYSIS_THRESH;default_exp.sex#DETECT_MINAREA;default_exp.sex#THRESH_TYPE;default_exp.sex#FILTER;default_exp.sex#FILTER_NAME;config_exp_psfex.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE;default.conv] tile-detection-matches-megapipe -Keep tile significance, minimum area and matched filter together as the MegaPipe detection prescription used by Gwyn's UNIONS tile catalogue, in both data and matching image simulations. -Exposure settings serve PSF-star detection and intentionally differ; copying them onto tiles changes the galaxy sample rather than standardising an implementation detail. -The runner's DOT_CONV_FILE overrides FILTER_NAME, so a kernel change must reach the effective command line, not just the .sex file. - -@sc [decision:detection.deblending_policy,governs:default_tile.sex#DEBLEND_MINCONT;default_exp.sex#DEBLEND_MINCONT;default_tile.sex#DEBLEND_NTHRESH;default_exp.sex#DEBLEND_NTHRESH] deblend-contrast-is-stage-specific -Preserve the deliberate tile/exposure difference in DEBLEND_MINCONT: tiles follow MegaPipe, while exposures select PSF-star candidates. -Do not harmonise the contrasts because DEBLEND_NTHRESH is shared; contrast changes object multiplicity, centroids and blend contamination. -A change must propagate to the matching simulations and the recorded detection prescription. - -@sc [decision:detection.background_model,governs:default_tile.sex#BACK_TYPE;default_tile.sex#BACK_SIZE;default_tile.sex#BACK_FILTERSIZE;default_tile.sex#BACKPHOTO_TYPE;default_tile.sex#BACKPHOTO_THICK;default_exp.sex#BACK_TYPE;default_exp.sex#BACK_SIZE;default_exp.sex#BACK_FILTERSIZE;default_exp.sex#BACKPHOTO_TYPE;config_tile_Sx.ini#SEXTRACTOR_RUNNER.BKG_FROM_HEADER;config_exp_psfex.ini#SEXTRACTOR_RUNNER.BKG_FROM_HEADER;config_exp_psfex.ini#SEXTRACTOR_RUNNER.CHECKIMAGE] background-model-follows-image-role -Keep the tile mesh, smoothing and local photometric annulus together as part of MegaPipe detection; the exposure mesh and global photometric background are a different prescription. -Exposure BACKGROUND and BACKGROUND_RMS check images also supply ngmix's epoch subtraction and noise weights, so changing their estimator changes shapes, not just detection photometry. -BKG_FROM_HEADER replaces the AUTO estimate with a manual background; do not enable that override while claiming the same background model. - -@sc [decision:detection.zero_weight_interpolation,governs:default_tile.sex#INTERP_TYPE;default_tile.sex#INTERP_MAXXLAG;default_tile.sex#INTERP_MAXYLAG;default_exp.sex#INTERP_TYPE;default_exp.sex#INTERP_MAXXLAG;default_exp.sex#INTERP_MAXYLAG] interpolated-detections-are-not-repaired-pixels -Treat INTERP_TYPE and both maximum lags as part of the detection and photometry prescription, including in matching simulations. -Flux invented across zero-weight pixels changes objects near defects; it does not make those pixels valid for ngmix or replace its separate defect-fill decision. -No scientific justification for the committed interpolation choice is recorded, so do not present it as a calibrated repair. - -@sc [decision:detection.spurious_detection_cleaning,governs:default_tile.sex#CLEAN;default_tile.sex#CLEAN_PARAM;default_exp.sex#CLEAN;default_exp.sex#CLEAN_PARAM] cleaning-changes-the-detection-sample -Keep CLEAN and CLEAN_PARAM together when reproducing the detection sample in data and simulations. -Cleaning removes detections after deblending, so switching it off or changing its strength changes catalogue membership even with identical detection thresholds and deblending. - -@sc [decision:detection.blend_photometry_mask_type,governs:default_tile.sex#MASK_TYPE;default_exp.sex#MASK_TYPE] blend-photometry-keeps-mirror-correction -The selected CORRECT treatment mirrors neighbour pixels across the target centre for SExtractor photometry; retain that convention when reproducing fluxes and windowed moments. -Replacing it with blanking changes photometry and centroids, and must not be confused with ngmix's separate neighbour-weighting decision. - -@sc [decision:detection.saturation_level,governs:default_tile.sex#SATUR_KEY;default_exp.sex#SATUR_KEY;star_selection.setools#MASK:star_selection.FLAGS;default_tile.sex;default_exp.sex] saturation-flags-follow-image-header -SATUR_KEY must refer to the delivered image's saturation card for the intended per-image saturation flags, which FLAGS-based PSF-star rejection consumes. -Do not interpret a missing card as evidence of no saturation: SExtractor then uses its built-in level because no SATUR_LEVEL is pinned here. -Header availability remains unverified; changing the key or adding a fixed level requires checking the bright-star selection, not just whether SExtractor runs. - -@sc [decision:detection.photometry_parameters,governs:default_tile.sex#PHOT_AUTOPARAMS;default_exp.sex#PHOT_AUTOPARAMS;default_tile.sex#PHOT_APERTURES;default_exp.sex#PHOT_APERTURES;default_tile.sex#PHOT_FLUXFRAC;default_exp.sex#PHOT_FLUXFRAC;default.psfex#PHOTFLUX_KEY;default.psfex#PHOTFLUXERR_KEY] kron-definition-links-selection-and-psf-normalisation -Keep the AUTO flux/error definition used for PSFEx normalisation consistent with the MAG_AUTO definition used for star selection and catalogue magnitudes. -A Kron-parameter change requires reconsidering the magnitude window and PSF normalisation together, not silently retaining their old interpretation. -Aperture diameter and flux fraction also define reported measurements; they must match the simulated catalogue rather than be treated as output formatting. - -@sc [decision:detection.detection_source_mode,governs:config_tile_Sx.ini#SEXTRACTOR_RUNNER.DETECTION_IMAGE;config_tile_Sx.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE;config_tile_Sx.ini#SEXTRACTOR_RUNNER.DOT_PARAM_FILE;default_tile.sex#PARAMETERS_NAME;default_noimaflags.param;final_cat.param#IMAFLAGS_ISO] tile-detection-has-no-instrument-flags -Tile detection uses the r-band tile itself, with no separate detection coadd or instrument flag image; keep the requested columns consistent with those inputs. -The runner's DOT_PARAM_FILE overrides PARAMETERS_NAME and deliberately selects the list without IMAFLAGS_ISO. -[LINT] final_cat.param requests IMAFLAGS_ISO although this tile chain never produces it (issue #912); resolve the export/input mismatch rather than inventing a clean mask column or assuming FLAG_IMAGE alone supplies a mask. - -Shared pixel data and calibration ---------------------------------- - -@sc [decision:masking.pixel_mask_source,governs:config_exp_psfex.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE;config_exp_psfex.ini#SEXTRACTOR_RUNNER.FILE_PATTERN;config_exp_psfex.ini#SEXTRACTOR_RUNNER.INPUT_DIR;config_exp_mccd.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE;config_exp_mccd.ini#SEXTRACTOR_RUNNER.FILE_PATTERN;config_exp_mccd.ini#SEXTRACTOR_RUNNER.INPUT_DIR;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_PATTERN;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_EXP_RUNNERS;star_selection.setools#MASK:star_selection.IMAFLAGS_ISO] instrument-flags-share-pixel-provenance -Exposure IMAFLAGS_ISO, under either PSF chain, and the multi-epoch flag stamps must come from the same delivered instrument flag image split per CCD. -Keep the ME_IMAGE_PATTERN and ME_IMAGE_EXP_RUNNERS lists aligned so a flag stamp cannot silently become an image, weight or sky-mask product. -Sky-fixed healsparse masks remain object-level catalogue information; substituting or rasterising them into this pixel path changes the masking decision. - -@sc [decision:masking.psf_star_mask_veto,governs:star_selection.setools#MASK:star_selection.IMAFLAGS_ISO;star_selection.setools#MASK:preselect.IMAFLAGS_ISO;star_selection.setools#MASK:flag.IMAFLAGS_ISO;star_selection.setools] psf-stars-vetoed-on-instrument-flags-only -PSF-star candidates are rejected on the instrument flags (IMAFLAGS_ISO == 0) and on no sky-fixed map: no mask block cuts on MASK_EXT. -A MASK_EXT cut changes the PSF training sample and is a change to this decision, not a cleanup; querying maps through MASK_PATHS without the cut records MASK_EXT and changes no star. - -@sc [decision:photometric_zeropoint,governs:default_tile.sex#MAG_ZEROPOINT;config_tile_Sx.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER;config_tile_Ng_template.ini#NGMIX_RUNNER.MAG_ZP;config_exp_psfex.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER;config_exp_psfex.ini#SEXTRACTOR_RUNNER.ZP_KEY] tile-and-epoch-photometry-share-flux-scale -Tile MAG_ZEROPOINT, ngmix MAG_ZP and the FSCALE-rescaled epochs must describe the same flux calibration. -Exposure photometry intentionally reads its own header zero-point; do not copy the fixed tile convention to exposures or change a header toggle independently of the calibration. -Equal config values do not verify that delivered MegaPipe stacks have the assumed calibration; that input assumption still needs checking. - -@sc [decision:postage_stamp_size,governs:default_noimaflags.param#VIGNET;default.param#VIGNET;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.STAMP_SIZE;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.STAMP_SIZE;default.psfex#PSF_SIZE] stamp-apertures-move-together -Keep SExtractor VIGNET for galaxies and training stars, both vignetmaker STAMP_SIZEs and the PSFEx PSF_SIZE coupled. -A resize must update the whole set and regenerate cutouts and models; changing only one desynchronises the pixel apertures used by the fit. -Larger stamps also change wing truncation and the measurable galaxy-size range, so this is a scientific change rather than only an allocation change. - -@sc [decision:preparation.object_position_columns,governs:config_tile_Sx.ini#SEXTRACTOR_RUNNER.WORLD_POSITION;config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.POSITION_PARAMS;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_1.COORD;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.POSITION_PARAMS;config_tile_PiViVi_psfex.ini#VIGNETMAKER_RUNNER_RUN_2.COORD;config_exp_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS;default.psfex#CENTER_KEYS;default.psfex#PSFVAR_KEYS;final_cat.param#XWIN_WORLD;final_cat.param#YWIN_WORLD] windowed-positions-share-coordinate-frame -Use the same windowed centroid through epoch membership, stamp placement, PSF interpolation and exported positions. -Pair IMAGE columns with pixel coordinates for tile stamps and exposure validation, and WORLD columns with sky coordinates for multi-epoch placement and interpolation. -Changing POSITION_PARAMS without its coordinate frame, or replacing windowed centroids in only one consumer, breaks the shared position underlying stamp offsets, centroid priors and position seeding. - -PSF modelling and diagnostics ----------------------------- - -@sc [decision:star_selection_psf.psf_model_complexity,governs:default.psfex#BASIS_TYPE;default.psfex#BASIS_NUMBER;default.psfex#PSF_SAMPLING;default.psfex#PSF_ACCURACY;default.psfex#PSFVAR_KEYS;default.psfex#PSFVAR_DEGREES] psf-flexibility-requires-star-support -Treat the pixel basis, spatial polynomial, sampling and assumed pixel accuracy as a joint model choice. -Greater flexibility must be supported by the surviving training stars on each CCD and checked on held-out residuals under the acceptance gate; successful optimisation alone does not establish a usable PSF. -PSF_ACCURACY changes bright-star leverage in the fit weights, not merely an optimisation stopping tolerance. - -@sc [decision:star_selection_psf.psfex_candidate_vetting,governs:default.psfex#SAMPLE_AUTOSELECT;default.psfex#BADPIXEL_FILTER;default.psfex#PSF_RECENTER;default.psfex] psfex-vetting-is-not-fully-disabled -Keep explicit stellar-locus selection under SETools control, but do not infer from SAMPLE_AUTOSELECT being off that every other PSFEx candidate cut is inactive. -The omitted SAMPLE_* keys take PSFEx's compiled defaults (psfex -dd, PSFEx 3.21.1 in the develop-runtime image): FWHMRANGE 2.0,10.0, VARIABILITY 0.2, MINSN 20, MAXELLIP 0.3, FLAGMASK 0x00fe, WFLAGMASK 0x0000, IMAFLAGMASK 0x0. -From the PSFEx source, not re-read here, MINSN, MAXELLIP, FLAGMASK, FWHMRANGE and VARIABILITY still cut candidates with SAMPLE_AUTOSELECT N; SETools applies no signal-to-noise or ellipticity cut, so these can reject stars it kept. -A PSFEx version change can move these defaults: re-run psfex -dd and compare surviving candidates before claiming unchanged selection. -Compiled-in values cannot be asserted from the config, only their omission from default.psfex; issue #919 pins them there. -BADPIXEL_FILTER and PSF_RECENTER also change the candidate pixels or positions, so enabling them is a scientific change, not a harmless cleanup. - -@sc [decision:star_selection_psf.star_selection_box,governs:config_tile_Ng_template.ini#NGMIX_RUNNER.PIXEL_SCALE;default_exp.sex#PIXEL_SCALE;star_selection.setools#MASK:preselect.FWHM_IMAGE;star_selection.setools#MASK:star_selection.FWHM_IMAGE;star_selection.setools#PLOT:fwhm_field.SCATTER;star_selection.setools] exposure-size-conventions-agree -Conversions of the same exposure pixels must use a consistent exposure scale for stellar-size selection, diagnostics and ngmix's one-pixel centroid prior; tile resampling is not a reason to assign a tile scale to epoch stamps. -Diagnostic labels and statistics must describe the cut actually applied. -[LINT] ngmix's PIXEL_SCALE is 0.186 arcsec/px while star selection's cuts, plot and statistics use 0.187; reconcile ngmix with the exposure convention rather than propagating the mismatch. - -Catalogue export ----------------- - -@sc [decision:catalogue_assembly.tile_overlap_handling,governs:config_tile_Mc.ini#MAKE_CAT_RUNNER.NUMBER_LIST;final_cat.param#TILE_ID] tile-id-is-provenance-not-deduplication -Preserve TILE_ID when exporting the per-tile catalogues: overlapping tiles can contain separate measurements of the same sky object, and TILE_ID is the supplied provenance, not a uniqueness flag. -NUMBER_LIST selects a tile to assemble, not a disjoint sky region; duplicate removal remains downstream. -No key flags or removes overlap objects, so making this catalogue unique or flagging its overlaps needs new code, not a config change. diff --git a/workflow/config/cfis/config_exp_mccd.ini b/workflow/config/cfis/config_exp_mccd.ini index ae1479803..4c4bd170e 100644 --- a/workflow/config/cfis/config_exp_mccd.ini +++ b/workflow/config/cfis/config_exp_mccd.ini @@ -64,12 +64,12 @@ TIMEOUT = 96:00:00 # The split CCDs, and nothing else: ShapePipe generates no masks -# @sc [decision:star_selection_psf.psf_modelling_software] +# @sc [decision:masking.pixel_mask_source,decision:star_selection_psf.psf_modelling_software] INPUT_DIR = $SP_RUN/output/run_sp_exp_Sp/split_exp_runner/output # Read the instrument flag image split_exp wrote per CCD -# @sc [decision:star_selection_psf.psf_modelling_software] +# @sc [decision:masking.pixel_mask_source,decision:star_selection_psf.psf_modelling_software] FILE_PATTERN = image, weight, flag # Explicit extensions: a 3-entry FILE_PATTERN override must not fall back on @@ -91,7 +91,7 @@ WEIGHT_IMAGE = True # Use input flag image if True -# @sc [decision:star_selection_psf.psf_modelling_software] +# @sc [decision:masking.pixel_mask_source,decision:star_selection_psf.psf_modelling_software] FLAG_IMAGE = True # Use input PSF file if True diff --git a/workflow/config/cfis/config_exp_psfex.ini b/workflow/config/cfis/config_exp_psfex.ini index 0bbd46b64..5551a1d1c 100644 --- a/workflow/config/cfis/config_exp_psfex.ini +++ b/workflow/config/cfis/config_exp_psfex.ini @@ -23,6 +23,8 @@ RUN_DATETIME = False [EXECUTION] # Module name, single string or comma-separated list of valid module runner names + +# @sc [decision:star_selection_psf.psf_modelling_software] MODULE = sextractor_runner, mask_query_runner, setools_runner, psfex_runner, psfex_interp_runner # Run mode, SMP or MPI @@ -57,13 +59,18 @@ TIMEOUT = 96:00:00 ## Module options -# @sc [decision:detection.background_model] +# @sc [decision:detection.background_model,label:override] header-background-replaces-auto +# BKG_FROM_HEADER replaces the AUTO background with a manual header value; enabling it changes the estimator rather than only its source. [SEXTRACTOR_RUNNER] # The split CCDs, and nothing else: ShapePipe generates no masks + +# @sc [decision:masking.pixel_mask_source] INPUT_DIR = $SP_RUN/output/run_sp_exp_Sp/split_exp_runner/output # Read the instrument flag image split_exp wrote per CCD + +# @sc [decision:masking.pixel_mask_source] FILE_PATTERN = image, weight, flag # Explicit extensions: a 3-entry FILE_PATTERN override must not fall back on @@ -79,7 +86,8 @@ EXEC_PATH = source-extractor DOT_SEX_FILE = $SP_CONFIG/default_exp.sex DOT_PARAM_FILE = $SP_CONFIG//default.param -# @sc [decision:detection.detection_threshold_policy] +# @sc [decision:detection.detection_threshold_policy,label:effective-filter] exposure-dot-conv-overrides-filter +# DOT_CONV_FILE overrides FILTER_NAME in the exposure runner; keep the effective convolution kernel aligned with the intended PSF-star detection prescription. DOT_CONV_FILE = $SP_CONFIG/default.conv # Use input weight image if True diff --git a/workflow/config/cfis/config_tile_Mc.ini b/workflow/config/cfis/config_tile_Mc.ini index e45564b21..16c0a5e8f 100644 --- a/workflow/config/cfis/config_tile_Mc.ini +++ b/workflow/config/cfis/config_tile_Mc.ini @@ -68,6 +68,8 @@ FILE_PATTERN = sexcat, galaxy_psf, ngmix # NUMBER_LIST selects this unit; the workflow sets SP_UNIT_NUM to the # dashed tile ID (e.g. -210-282) + +# @sc [decision:catalogue_assembly.tile_overlap_handling] NUMBER_LIST = $SP_UNIT_NUM # FILE_EXT (optional) list of string extensions to identify input files diff --git a/workflow/config/cfis/config_tile_Ng_template.ini b/workflow/config/cfis/config_tile_Ng_template.ini index 7e5df62e4..eef3532d0 100644 --- a/workflow/config/cfis/config_tile_Ng_template.ini +++ b/workflow/config/cfis/config_tile_Ng_template.ini @@ -120,7 +120,7 @@ MAG_ZP = 30.0 # Pixel scale in arcsec -# @sc [decision:shape_measurement.fit_priors] +# @sc [decision:shape_measurement.fit_priors,decision:star_selection_psf.star_selection_box] PIXEL_SCALE = 0.186 # ID_OBJ_MIN/MAX: this chunk's closed SExtractor NUMBER-column range, diff --git a/workflow/config/cfis/config_tile_PiViVi_mccd.ini b/workflow/config/cfis/config_tile_PiViVi_mccd.ini index d312ca494..0300c3374 100644 --- a/workflow/config/cfis/config_tile_PiViVi_mccd.ini +++ b/workflow/config/cfis/config_tile_PiViVi_mccd.ini @@ -16,6 +16,8 @@ RUN_DATETIME = False ## ShapePipe execution options + +# @sc [decision:star_selection_psf.psf_modelling_software] [EXECUTION] # Module name, single string or comma-separated list of valid module runner names diff --git a/workflow/config/cfis/config_tile_PiViVi_psfex.ini b/workflow/config/cfis/config_tile_PiViVi_psfex.ini index 0427c86a1..f06d1278c 100644 --- a/workflow/config/cfis/config_tile_PiViVi_psfex.ini +++ b/workflow/config/cfis/config_tile_PiViVi_psfex.ini @@ -16,6 +16,8 @@ RUN_DATETIME = False ## ShapePipe execution options + +# @sc [decision:star_selection_psf.psf_modelling_software] [EXECUTION] # Module name, single string or comma-separated list of valid module runner names @@ -204,7 +206,10 @@ PREFIX = # run outputs. ME_IMAGE_EXP_DIR/ME_IMAGE_EXP_RUNNERS replace ME_IMAGE_DIR for # the v2.0 per-exposure pipeline; output dirs are discovered by scanning $SP_EXP. ME_IMAGE_EXP_DIR = $SP_EXP -ME_IMAGE_EXP_RUNNERS = split_exp_runner, split_exp_runner, split_exp_runner, sextractor_runner, sextractor_runner # @sc [decision:masking.pixel_mask_source] +ME_IMAGE_EXP_RUNNERS = split_exp_runner, split_exp_runner, split_exp_runner, sextractor_runner, sextractor_runner + +# @sc [decision:masking.pixel_mask_source,label:coupling] flag-stamp-runner-order +# ME_IMAGE_PATTERN and ME_IMAGE_EXP_RUNNERS are parallel ordered lists; keep flag, image, weight, background and background-RMS entries paired. ME_IMAGE_PATTERN = flag, image, weight, background, background_rms diff --git a/workflow/config/cfis/config_tile_Sx.ini b/workflow/config/cfis/config_tile_Sx.ini index 1e32dfd44..4cf14ce8d 100644 --- a/workflow/config/cfis/config_tile_Sx.ini +++ b/workflow/config/cfis/config_tile_Sx.ini @@ -73,9 +73,13 @@ EXEC_PATH = source-extractor # SExtractor configuration files DOT_SEX_FILE = $SP_CONFIG/default_tile.sex + +# @sc [decision:detection.detection_source_mode,label:override] tile-parameter-file-overrides-column-list +# DOT_PARAM_FILE overrides PARAMETERS_NAME; keep it pointed at default_noimaflags.param so tile detections do not request IMAFLAGS_ISO. DOT_PARAM_FILE = $SP_CONFIG/default_noimaflags.param -# @sc [decision:detection.detection_threshold_policy] +# @sc [decision:detection.detection_threshold_policy,label:effective-filter] tile-dot-conv-overrides-filter +# DOT_CONV_FILE overrides FILTER_NAME in the tile runner; keep the effective convolution kernel aligned with the intended detection prescription. DOT_CONV_FILE = $SP_CONFIG/gauss_3.0_7x7.conv # Use input weight image if True @@ -124,6 +128,8 @@ SUFFIX = sexcat MAKE_POST_PROCESS = True # World coordinate keywords, SExtractor output. Format: KEY_X,KEY_Y + +# @sc [decision:preparation.object_position_columns] WORLD_POSITION = XWIN_WORLD,YWIN_WORLD # Number of pixels in x,y of a CCD. Format: Nx,Ny diff --git a/workflow/config/cfis/default.psfex b/workflow/config/cfis/default.psfex index f9634da07..2ecb65438 100644 --- a/workflow/config/cfis/default.psfex +++ b/workflow/config/cfis/default.psfex @@ -33,16 +33,18 @@ MEF_TYPE INDEPENDENT # INDEPENDENT or COMMON #------------------------- Point source measurements ------------------------- +# @sc [decision:preparation.object_position_columns] CENTER_KEYS XWIN_IMAGE,YWIN_IMAGE # Catalogue parameters for source pre-centering # @sc [decision:detection.photometry_parameters] PHOTFLUX_KEY FLUX_AUTO # Catalogue parameter for photometric norm. +# @sc [decision:detection.photometry_parameters] PHOTFLUXERR_KEY FLUXERR_AUTO # Catalogue parameter for photometric error #----------------------------- PSF variability ------------------------------- -# @sc [decision:star_selection_psf.psf_model_complexity] +# @sc [decision:preparation.object_position_columns,decision:star_selection_psf.psf_model_complexity] PSFVAR_KEYS XWIN_IMAGE,YWIN_IMAGE # Catalogue or FITS (preceded by :) params PSFVAR_GROUPS 1,1 # Group tag for each context key diff --git a/workflow/config/cfis/default_exp.sex b/workflow/config/cfis/default_exp.sex index e8203ba8f..5e69b450d 100644 --- a/workflow/config/cfis/default_exp.sex +++ b/workflow/config/cfis/default_exp.sex @@ -17,7 +17,10 @@ DETECT_TYPE CCD # CCD (linear) or PHOTO (with gamma correction) DETECT_MINAREA 5 # min. # of pixels above threshold DETECT_MAXAREA 0 # max. # of pixels above threshold (0=unlimited) + +# @sc [decision:detection.detection_threshold_policy] THRESH_TYPE RELATIVE # threshold type: RELATIVE (in sigmas) + # or ABSOLUTE (in ADUs) # @sc [decision:detection.detection_threshold_policy] @@ -27,7 +30,9 @@ ANALYSIS_THRESH 1.5 # or , in mag.arcsec-2 # @sc [decision:detection.detection_threshold_policy] FILTER Y # apply filter for detection (Y or N)? +# @sc [decision:detection.detection_threshold_policy] FILTER_NAME default.conv + FILTER_THRESH # Threshold[s] for retina filtering # @sc [decision:detection.deblending_policy] @@ -83,6 +88,8 @@ MAG_ZEROPOINT 30.0 # magnitude zero-point MAG_GAMMA 4.0 # gamma of emulsion (for photographic scans) GAIN_KEY GAIN # keyword for detector gain in e-/ADU + +# @sc [decision:star_selection_psf.star_selection_box] PIXEL_SCALE 0. # size of pixel in arcsec (0=use FITS WCS info) #------------------------- Star/Galaxy Separation ---------------------------- diff --git a/workflow/config/cfis/default_tile.sex b/workflow/config/cfis/default_tile.sex index 7c0f61b84..508aae389 100644 --- a/workflow/config/cfis/default_tile.sex +++ b/workflow/config/cfis/default_tile.sex @@ -7,6 +7,7 @@ CATALOG_TYPE FITS_LDAC +# @sc [decision:detection.detection_source_mode] PARAMETERS_NAME default.param #------------------------------- Extraction ---------------------------------- @@ -17,7 +18,10 @@ DETECT_TYPE CCD # CCD (linear) or PHOTO (with gamma correction) DETECT_MINAREA 3 # min. # of pixels above threshold DETECT_MAXAREA 0 # max. # of pixels above threshold (0=unlimited) + +# @sc [decision:detection.detection_threshold_policy] THRESH_TYPE RELATIVE # threshold type: RELATIVE (in sigmas) + # or ABSOLUTE (in ADUs) # @sc [decision:detection.detection_threshold_policy] @@ -27,7 +31,9 @@ ANALYSIS_THRESH 1.0 # or , in mag.arcsec-2 # @sc [decision:detection.detection_threshold_policy] FILTER Y # apply filter for detection (Y or N)? +# @sc [decision:detection.detection_threshold_policy] FILTER_NAME gauss_3.0_7x7.conv + FILTER_THRESH # Threshold[s] for retina filtering # @sc [decision:detection.deblending_policy] diff --git a/workflow/config/cfis/final_cat.param b/workflow/config/cfis/final_cat.param index 12debe7fb..f3b56c465 100644 --- a/workflow/config/cfis/final_cat.param +++ b/workflow/config/cfis/final_cat.param @@ -1,10 +1,15 @@ # @sc [decision:catalogue_assembly.star_galaxy_classification,scope:file] # coordinates + +# @sc [decision:preparation.object_position_columns] XWIN_WORLD YWIN_WORLD # tile ID, for plot of tile-dependent additive bias. # Can maybe be removed. + +# @sc [decision:catalogue_assembly.tile_overlap_handling,label:provenance] tile-id-is-not-unique-object-id +# TILE_ID is tile provenance, not an object-uniqueness or overlap flag; duplicate removal or overlap flagging requires explicit downstream code. TILE_ID # flags diff --git a/workflow/config/cfis/star_selection.setools b/workflow/config/cfis/star_selection.setools index 5a71d8c31..97fcdb4a9 100644 --- a/workflow/config/cfis/star_selection.setools +++ b/workflow/config/cfis/star_selection.setools @@ -1,3 +1,4 @@ +# @sc [decision:masking.psf_star_mask_veto,decision:star_selection_psf.star_selection_box,scope:file] ## SETools configuration file for star/galaxy separation based on size/mag properties ## ## ONE mask cut, and it is the instrument flags: @@ -28,24 +29,20 @@ ## config, which ships commented out; SETools has no bitwise operators, so the ## bit selection happens there and this file only ever tests for zero. -# @sc [decision:masking.psf_star_mask_veto] [MASK:preselect] -# @sc [decision:star_selection_psf.star_selection_box] MAG_AUTO > 0 MAG_AUTO < 21 FWHM_IMAGE > 0.3 / 0.187 FWHM_IMAGE < 1.5 / 0.187 -# @sc [decision:detection.saturation_level,decision:star_selection_psf.star_selection_box] +# @sc [decision:detection.saturation_level] FLAGS == 0 -# @sc [decision:star_selection_psf.star_selection_box] IMAFLAGS_ISO == 0 NO_SAVE -# @sc [decision:masking.psf_star_mask_veto] [MASK:flag] # @sc [decision:detection.saturation_level] @@ -55,20 +52,18 @@ IMAFLAGS_ISO == 0 NO_SAVE -# @sc [decision:masking.psf_star_mask_veto] [MASK:star_selection] # Star selection using the FWHM mode -# @sc [decision:star_selection_psf.star_selection_box] MAG_AUTO > 18. MAG_AUTO < 22. FWHM_IMAGE <= mode(FWHM_IMAGE{preselect}) + 0.2 FWHM_IMAGE >= mode(FWHM_IMAGE{preselect}) - 0.2 -# @sc [decision:detection.saturation_level,decision:star_selection_psf.star_selection_box] +# @sc [decision:detection.saturation_level] FLAGS == 0 -# @sc [decision:star_selection_psf.star_selection_box] +# @sc [decision:masking.pixel_mask_source] IMAFLAGS_ISO == 0 [MASK:fwhm_mag_cut] @@ -123,7 +118,8 @@ FORMAT = png X = X_IMAGE{star_selection} Y = Y_IMAGE{star_selection} -# @sc [decision:star_selection_psf.star_selection_box] +# @sc [label:diagnostic] stellar-diagnostic-matches-cut +# FWHM_FIELD plot labels and summary statistics must describe the size cut actually applied. SCATTER = FWHM_IMAGE{star_selection}*0.187 MARKER = . From cb0f4ac952cba180643f2d1b516f6bb61f76bedc Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 05:26:44 +0200 Subject: [PATCH 31/40] docs: document decision tags and Values checks --- CLAUDE.md | 32 ++++++++++++++++++-------------- astra.yaml | 31 ++++++++++++++++++++----------- tests/README.md | 4 +++- 3 files changed, 41 insertions(+), 26 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index 2ed249e6f..4722a455b 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -123,16 +123,20 @@ where the change lives — and *scientific* decisions in `astra.yaml`, below. ## Scientific decisions live in `astra.yaml` `astra.yaml` at the repo root is the pipeline's decision record: every -consequential scientific choice embedded in the code and the committed configs, -with its rationale, its alternatives, and an anchor to the code or config that -implements it. `universes/committed.yaml` pins the option the committed -configuration selects for every decision. The format is ASTRA. -The `decision` pytest marker links each `tests/science/` guardrail to the -ASTRA decisions it protects, and `tests/unit/test_contracts.py` validates -those IDs alongside `@sc` references. -`uvx astra-tools@0.2.17 guide` is the briefing and `uvx astra-tools@0.2.17 spec` -the field reference. The file's header states its conventions (anchor grammar, -`[HARDCODED]`, `[LINT]`). +consequential scientific choice embedded in the code and committed configs, +with its rationale, alternatives and selected default. `universes/committed.yaml` +pins the option the committed configuration selects for every decision. +`@sc [decision:]` tags live at implementing code/config sites; optional +`Values:` sentences assert committed values within those tagged sites. The +rationale stays in ASTRA, while local `@sc` contracts hold site-specific +constraints. `tests/unit/test_decisions.py` checks tag syntax, bidirectional +decision/site coverage, values, and test `decision` markers. To inspect tags at +an implementation site, run `python -m tests.helpers.decisions [:]`; +`--decision ` lists its sites. The format is ASTRA. + +`uvx astra-tools@0.2.17 guide` is the briefing and +`uvx astra-tools@0.2.17 spec` the field reference. The file's header states +its conventions, including `[HARDCODED]` and `[LINT]`. Membership test: a different defensible choice would change which objects enter the shear catalogue, or the numbers attached to them. Detection thresholds, @@ -143,10 +147,10 @@ PRD, CosmoStat/shapepipe#848. **A scientific change is not finished until the record is.** When a change moves what the pipeline measures, amend `astra.yaml` in the same PR (add the decision, -or edit its rationale, options and anchors), pin the selected option in -`universes/committed.yaml`, and say so in the PR description. The anchor test -`tests/unit/test_astra_anchors.py` runs in CI; a scientific change that breaks it -or leaves the record stale is unfinished. Before committing: +or edit its rationale, options, Values and site tags), pin the selected option +in `universes/committed.yaml`, and say so in the PR description. +`tests/unit/test_decisions.py` runs in CI; a scientific change that breaks its +site/value checks or leaves the record stale is unfinished. Before committing: ```bash uvx astra-tools@0.2.17 validate diff --git a/astra.yaml b/astra.yaml index 3fec5498a..aa9962b36 100644 --- a/astra.yaml +++ b/astra.yaml @@ -2,17 +2,26 @@ # the code and the committed workflow configs (workflow/config/cfis/), why # they stand, and the alternatives. CLAUDE.md says when to amend it. # -# Conventions (tests/helpers/astra_record.py parses them and documents the -# full value grammar; tests/unit/test_astra_{anchors,values}.py enforce it): -# * Every rationale ends with one sentence "Anchor: ; ." Refs are -# repo-relative: CODE `path::Symbol` (a def, class or assignment target, -# dotted for nesting, `[key]` into a dict literal), CONFIG -# `path#SECTION.KEY` (`path#KEY` for sectionless .sex/.psfex/.param -# files), or FILE `path`. No line numbers. -# * `ref = value` asserts the value at that location; a bare ref asserts -# only that it exists. `= absent` asserts a config key has no active -# line. The anchor is the canonical expectation; prose may repeat a value -# for readability, and the code and configs remain execution truth. +# Conventions (`tests/helpers/decisions.py` implements them and documents the +# full value grammar; `tests/unit/test_decisions.py` enforces them): +# * Put `@sc [decision:]` at every code/config site that implements a +# decision. Decision citations have no id or prose; the record is the source +# of rationale. IDs are bare for top-level decisions and dotted for +# sub-analysis decisions; repeat `decision:` metadata to cite several. +# A local constraint adds a stable id and prose, with optional decision +# metadata: `@sc [decision:,label:] `. +# * In Python, put tags in a def/class docstring or as a comment immediately +# above a module statement. In Snakemake, put a comment immediately above +# the rule/statement. In config, a comment tag governs the following +# paragraph; a section header governs the whole section, and +# `scope:file` governs the whole file. +# * A rationale may end with `Values: = ; ... .`. Every ref must +# resolve to exactly one site tagged for that decision and match its value. +# Use a bare key when it selects one site; qualify it with a path suffix and +# `#` or `::` when needed. `= absent` requires a tagged section or file scope. +# * The checker validates citations in both directions, unique local-contract +# ids, literal values and test `decision` markers. The value grammar is +# documented in the helper module's docstring. # * A decision's `default` is the option the committed code and configs # select; universes/committed.yaml pins it. # * `excluded: true` means considered and rejected. An option the code diff --git a/tests/README.md b/tests/README.md index 50439f247..ea3e460e7 100644 --- a/tests/README.md +++ b/tests/README.md @@ -33,7 +33,9 @@ decision(*ids) ASTRA decisions protected by a test; ids checked against astra.ya `--strict-markers` is on, so a typo'd marker is an error, not a silent no-op. Science guardrails use `decision(*ids)` at module or test level; AST parsing checks every ID against `astra.yaml` without importing the tests. -Use `pytest -m "not unions"` to run the survey-generic tests. +`tests/unit/test_decisions.py` also checks that `@sc` site tags and rationale +`Values:` refs agree with the decision record, and that test markers name real +decisions. Use `pytest -m "not unions"` to run the survey-generic tests. A `candide`-marked test is **collected everywhere** (so `--collect-only` shows it exists) but **skipped off-cluster** with a clear reason. Candide is detected From 2bb3c84bf12e5ab9a28912475dc02eab86a9cfb5 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 06:04:19 +0200 Subject: [PATCH 32/40] fix(decisions): tighten absence and duplicate checks --- astra.yaml | 5 +- tests/helpers/decisions.py | 227 +++++++++++++++++++++++++++++++---- tests/unit/test_decisions.py | 188 +++++++++++++++++++++++++++-- 3 files changed, 383 insertions(+), 37 deletions(-) diff --git a/astra.yaml b/astra.yaml index aa9962b36..569e368b4 100644 --- a/astra.yaml +++ b/astra.yaml @@ -18,7 +18,10 @@ # * A rationale may end with `Values: = ; ... .`. Every ref must # resolve to exactly one site tagged for that decision and match its value. # Use a bare key when it selects one site; qualify it with a path suffix and -# `#` or `::` when needed. `= absent` requires a tagged section or file scope. +# `#` or `::` when needed. `= absent` requires a same-decision tag somewhere +# in the file (in that section or file-wide for INI/SETools) and asserts that +# no active setting exists anywhere in the target scope. INI absence checks +# include keys inherited from `[DEFAULT]`. # * The checker validates citations in both directions, unique local-contract # ids, literal values and test `decision` markers. The value grammar is # documented in the helper module's docstring. diff --git a/tests/helpers/decisions.py b/tests/helpers/decisions.py index 21d149989..56f61a355 100644 --- a/tests/helpers/decisions.py +++ b/tests/helpers/decisions.py @@ -11,8 +11,11 @@ 1/0 are numbers. Outer whitespace is trimmed; other strings are case-sensitive. Lists preserve order and length. Only .param VIGNET and .psfex PSF_SIZE accept square-size shorthand: 51 = 51,51. -* ``= absent`` asserts a config key has no active line; its tagged file or - section scope must still exist. Quote it ("absent") to mean the text. +* ``= absent`` asserts that a config key has no active line anywhere in the + file or named section. The same decision must tag at least one site in that + file (and, for INI/SETools, in the named section or at file scope). INI + absence checks include values inherited from ``[DEFAULT]``. Quote it + ("absent") to mean the text. * .setools refs use ``SECTION.KEY``. Predicates keep their operators as quoted text; repeated cuts on one key are an ordered list, e.g. ``MAG_AUTO = ["> 18.", "< 22."]``. Expressions compare as text. @@ -277,7 +280,17 @@ def _comment_site(path, lines, index, meta, tree=None, *, snakemake=False): if comment is not None and comment.startswith("@sc"): end = j break - return Site(path, start, end, "config", scope="paragraph") + suffix = Path(path).suffix.lower() + section = "" + if suffix in {".ini", ".setools"}: + for prior in lines[:index]: + header = re.match(r"^\s*\[([^]]+)\]\s*(?:[#;].*)?$", prior) + if header: + section = header.group(1).strip() + if suffix == ".ini" and not section: + section = "DEFAULT" + return Site(path, start, end, "config", section=section, + scope="paragraph") return None @@ -785,7 +798,7 @@ def _path_matches(site_path, qualifier): def _split_config_key(selector, suffix, site): if suffix in {".ini", ".setools"} and "." in selector: return selector.rsplit(".", 1) - section = site.section if site.scope == "section" else "" + section = site.section if suffix in {".ini", ".setools"} else "" return section, selector.rsplit(".", 1)[-1] @@ -891,30 +904,167 @@ def _section_exists(text, suffix, section): except configparser.Error as error: raise ValueError(f"cannot parse INI file: {error}") from error return section == parser.default_section or parser.has_section(section) - return any( - line.strip() == f"[{section}]" for line in text.splitlines() - ) + return any(line.strip() == f"[{section}]" for line in text.splitlines()) -def _absent_scope_matches(root, site, selector): - target = Path(root) / site.path +def _yaml_key_lines(lines, selector): + """Return lines for a simple nested YAML mapping selector.""" + + wanted = selector.split(".") + stack, found = [], [] + for number, raw in enumerate(lines, 1): + match = re.match(r"^(\s*)([^:#][^:]*?)\s*:\s*(?:.*)?$", raw) + if not match: + continue + indent = len(match.group(1)) + while stack and stack[-1][0] >= indent: + stack.pop() + key = match.group(2).strip().strip("'\"") + stack.append((indent, key)) + if [part for _, part in stack] == wanted: + found.append(number) + return found + + +def _active_config_lines(root, path, selector, site=None): + """Find active line settings for a key throughout one config file.""" + + target = Path(root) / path + suffix = target.suffix.lower() + lines = target.read_text(encoding="utf-8").splitlines() + if suffix in {".yaml", ".yml"}: + return _yaml_key_lines(lines, selector) + + file_site = site or Site(path, 1, len(lines), "config", scope="file") + section, key = _split_config_key(selector, suffix, file_site) + current_section = "DEFAULT" if suffix == ".ini" else "" + found = [] + for number, raw in enumerate(lines, 1): + stripped = _strip_config_comment(raw, suffix) + if not stripped: + continue + header = re.match(r"^\[([^]]+)\]$", stripped) + if header: + current_section = header.group(1).strip() + continue + if suffix in {".ini", ".setools"} and section: + if current_section != section: + continue + if suffix == ".ini": + match = re.match(r"^([^:=\s][^:=]*?)\s*(?:[:=]\s*(.*))?$", stripped) + active_key = match.group(1).strip() if match else None + elif suffix == ".setools": + match = re.match(rf"^{re.escape(key)}(?=$|\s|=|<|>)(.*)$", stripped) + active_key = key if match else None + elif suffix == ".param": + match = re.match( + rf"^{re.escape(key)}(?:\s*\(([^)]*)\)|\s+(.*))?$", stripped + ) + active_key = key if match else None + else: + match = re.match(rf"^{re.escape(key)}(?=$|\s|=|\()(.*)$", stripped) + active_key = key if match else None + if active_key == key: + found.append(number) + return found + + +def _config_absent_actual(root, path, selector): + """Return whether a config setting is active anywhere in its target scope.""" + + target = Path(root) / path suffix = target.suffix.lower() - if site.scope not in {"file", "section"}: - return None text = target.read_text(encoding="utf-8") - section = selector.rsplit(".", 1)[0] if suffix in {".ini", ".setools"} and "." in selector else "" - if site.scope == "section": - if section and section != site.section: + file_site = Site(path, 1, len(text.splitlines()), "config", scope="file") + section, key = _split_config_key(selector, suffix, file_site) + if suffix == ".ini": + try: + parser = _ini_parser(text, strict=False) + except configparser.Error as error: + raise ValueError(f"cannot parse INI file: {error}") from error + if section and not _section_exists(text, suffix, section): return None - section = site.section - if section and not _section_exists(text, suffix, section): + sections = [section] if section else [parser.default_section, *parser.sections()] + active_sections = [ + name for name in sections if parser.has_option(name, key) + ] + if section and section != parser.default_section: + explicit = parser._sections.get(section, {}) + if key not in explicit and parser.has_option(parser.default_section, key): + return f"active setting inherited from [{parser.default_section}]" + return ( + f"active setting in section(s) {active_sections!r}" + if active_sections else _NO_SETTING + ) + if suffix == ".setools" and section and not _section_exists( + text, suffix, section + ): return None + active = _active_config_lines(root, path, selector) + return f"active setting on line(s) {active!r}" if active else _NO_SETTING + + +def _check_absent_entry(root, decision, ref, kind, qualifier, selector, tags): + """Check absence against a same-decision tag in the file or section.""" + + if kind == "python": + return ( + f"{decision}: ref {ref!r}: expected no active setting, actual " + " (absence refs apply only to config files)" + ) + matching_paths = set() + for tag in tags: + if decision not in tag.decisions or tag.site is None: + continue + site = tag.site + if qualifier and not _path_matches(site.path, qualifier): + continue + suffix = Path(site.path).suffix.lower() + if suffix not in _CONFIG_SUFFIXES: + continue + section, _ = _split_config_key(selector, suffix, site) + if suffix in {".ini", ".setools"} and section: + if site.scope != "file" and site.section != section: + continue + try: + text = (Path(root) / site.path).read_text(encoding="utf-8") + except OSError: + continue + if not _section_exists(text, suffix, section): + continue + matching_paths.add(site.path) + + if len(matching_paths) != 1: + actual = ( + "" if not matching_paths + else f"" + ) + return ( + f"{decision}: ref {ref!r}: expected no active setting, actual " + f"{actual} (absence requires a same-decision tag in the file or " + f"named section; found {len(matching_paths)})" + ) + + path, = matching_paths try: - actual = _config_values_in_site(root, site, selector) - except ValueError as error: - # A no-value option is still an active setting for an absence check. - return str(error) - return _NO_SETTING if actual is None else actual + actual = _config_absent_actual(root, path, selector) + except (ValueError, OSError, configparser.Error, yaml.YAMLError) as error: + return ( + f"{decision}: ref {ref!r}: expected no active setting, " + f"actual : {error}" + ) + if actual is None: + section, _ = _split_config_key( + selector, Path(path).suffix.lower(), + Site(path, 1, 1, "config", scope="file"), + ) + return ( + f"{decision}: ref {ref!r}: expected no active setting, actual " + f" (section {section!r} does not exist)" + ) + if actual is _NO_SETTING: + return None + return f"{decision}: ref {ref!r}: expected no active setting, actual {actual}" def _python_value_in_site(root, site, selector): @@ -937,7 +1087,13 @@ def _python_value_in_site(root, site, selector): except ValueError as error: if not direct: return None - if "needs one binding" in str(error) or "missing" in str(error): + message = str(error) + if "needs one binding" in message: + found = re.search(r"found (\d+)", message) + if direct and found and int(found.group(1)) > 1: + raise ValueError(f"ambiguous binding: {message}") from error + return None + if "missing" in message: return None raise try: @@ -982,6 +1138,11 @@ def _check_value_entry(root, decision, reference, tags): except ValueError as error: return f"{decision}: ref {reference!r}: expected , actual : {error}" + if expected == ABSENT: + return _check_absent_entry( + root, decision, ref, kind, qualifier, selector, tags + ) + candidates = [] for site in _sites_for(tags, decision): suffix = Path(site.path).suffix.lower() @@ -991,13 +1152,27 @@ def _check_value_entry(root, decision, reference, tags): if qualifier and not _path_matches(site.path, qualifier): continue try: - if expected == ABSENT: - actual = _absent_scope_matches(root, site, selector) - else: - actual = _site_actual(root, site, expected_kind, selector) + actual = _site_actual(root, site, expected_kind, selector) except (ValueError, OSError, SyntaxError, configparser.Error, yaml.YAMLError) as error: return f"{decision}: ref {ref!r}: expected {expected!r}, actual : {error}" if actual is not None: + if expected_kind == "config": + try: + active_lines = _active_config_lines( + root, site.path, selector, site + ) + except (ValueError, OSError, configparser.Error, yaml.YAMLError) as error: + return f"{decision}: ref {ref!r}: expected {expected!r}, actual : {error}" + outside = [ + number for number in active_lines + if not site.start <= number <= site.end + ] + if outside: + return ( + f"{decision}: ref {ref!r}: expected {expected!r}, " + f"actual {actual!r}; duplicate active setting outside " + f"the governed paragraph on line(s) {outside!r}" + ) candidates.append((site, actual)) if len(candidates) != 1: diff --git a/tests/unit/test_decisions.py b/tests/unit/test_decisions.py index 85978845d..cb549c447 100644 --- a/tests/unit/test_decisions.py +++ b/tests/unit/test_decisions.py @@ -13,8 +13,6 @@ load_yaml, main, repository_errors, - _iter_decision_defs, - _parse_values, scan_tags, tag_errors, value_errors, @@ -236,10 +234,10 @@ def test_deleting_one_of_two_vignet_tagged_sites_fails_its_assertion(tmp_path): assert one.exists() -def test_absent_key_requires_tagged_scope_and_detects_added_key(tmp_path): +def test_absent_key_needs_only_a_same_decision_tag_in_the_file(tmp_path): config = _write( tmp_path / "settings.sex", - "# @sc [decision:choice,scope:file]\nOTHER 2\n", + "# @sc [decision:choice]\nOTHER 2\n\nUNRELATED 3\n", ) record = _record("No fixed threshold. Values: settings.sex#THRESH = absent.") tags, errors = scan_tags(tmp_path) @@ -248,7 +246,7 @@ def test_absent_key_requires_tagged_scope_and_detects_added_key(tmp_path): assert value_errors(tmp_path, record, tags) == [] config.write_text( - "# @sc [decision:choice,scope:file]\nOTHER 2\nTHRESH 1\n", + "# @sc [decision:choice]\nOTHER 2\n\nTHRESH 1\n", encoding="utf-8", ) tags, errors = scan_tags(tmp_path) @@ -256,6 +254,59 @@ def test_absent_key_requires_tagged_scope_and_detects_added_key(tmp_path): assert "expected no active setting" in value_errors(tmp_path, record, tags)[0] +def test_absent_key_fails_without_a_same_decision_tag_in_the_file(tmp_path): + _write(tmp_path / "settings.sex", "# @sc [decision:other]\nOTHER 2\n") + record = _record("No fixed threshold. Values: settings.sex#THRESH = absent.") + tags, errors = scan_tags(tmp_path) + + assert errors == [] + problem, = value_errors(tmp_path, record, tags) + assert "same-decision tag" in problem + assert "found 0" in problem + + +def test_ini_absence_sees_a_key_inherited_from_default(tmp_path): + _write( + tmp_path / "settings.ini", + "[DEFAULT]\nTHRESH = 1\n" + "# @sc [decision:choice]\n[SCIENCE]\nOTHER = 2\n", + ) + record = _record("No fixed threshold. Values: SCIENCE.THRESH = absent.") + tags, errors = scan_tags(tmp_path) + + assert errors == [] + problem, = value_errors(tmp_path, record, tags) + assert "expected no active setting" in problem + assert "DEFAULT" in problem + + +def test_duplicate_active_setting_outside_governed_paragraph_fails(tmp_path): + _write( + tmp_path / "settings.sex", + "# @sc [decision:choice]\nTHRESH 1\n\nTHRESH 1\n", + ) + record = _record("Threshold. Values: settings.sex#THRESH = 1.") + tags, errors = scan_tags(tmp_path) + + assert errors == [] + problem, = value_errors(tmp_path, record, tags) + assert "duplicate active setting outside the governed paragraph" in problem + + +def test_ini_duplicate_active_setting_outside_governed_paragraph_fails(tmp_path): + _write( + tmp_path / "settings.ini", + "[SCIENCE]\n# @sc [decision:choice]\nTHRESH = 1\n\n" + "OTHER = 2\nTHRESH = 1\n", + ) + record = _record("Threshold. Values: SCIENCE.THRESH = 1.") + tags, errors = scan_tags(tmp_path) + + assert errors == [] + problem, = value_errors(tmp_path, record, tags) + assert "duplicate active setting outside the governed paragraph" in problem + + def test_absent_assertion_fails_when_its_scope_is_removed(tmp_path): config = _write( tmp_path / "settings.ini", @@ -291,6 +342,64 @@ def test_numeric_lists_and_astromatic_boolean_words_keep_reader_semantics(tmp_pa assert value_errors(tmp_path, record, tags) == [] +@pytest.mark.parametrize( + "actual, expected", + [ + ("1.000001", "1"), + ("1", "True"), + ("0", "False"), + ("51,51", "51"), + ("51,52", "51,51"), + ("1,2", "2,1"), + ("1,1,1", "1,1"), + ("1", "[1]"), + ("map_weight", "MAP_WEIGHT"), + ], +) +def test_normalization_does_not_hide_drift(tmp_path, actual, expected): + _config_tag(tmp_path / "config.sex", content=f"KEY {actual}\n") + record = _record(f"Value. Values: KEY = {expected}.") + tags, errors = scan_tags(tmp_path) + + assert errors == [] + problem, = value_errors(tmp_path, record, tags) + assert "expected" in problem and "actual" in problem + + +@pytest.mark.parametrize( + "contents, selector", + [ + ("X = get_value()", "X"), + ("X = 1 / 3", "X"), + ("X = 1\nX = 2", "X"), + ("X = 1\nX += 1", "X"), + ("X, Y = 1, 2", "X"), + ("def X():\n return 1", "X"), + ("X = {'a': 1, 'a': 2}", "X[a]"), + ("X = dict(**other)", "X[a]"), + ("X = {'a': 1, **other}", "X[a]"), + ("X = {variable: 1}", "X[a]"), + ("X = {'a': 1}\nX['a'] = 2", "X[a]"), + ("X = {'a': 1}\nX['b']['c'] = 2", "X[a]"), + ("X = {'a': 1}\nX['a'] += 1", "X[a]"), + ("X = {'a': 1}\ndel X['a']", "X[a]"), + ("X = 1\nX.attr = 2", "X"), + ("def f():\n X = {'a': 1}\n X['a'] = 2", "f.X[a]"), + ], +) +def test_python_nonliteral_or_ambiguous_values_fail_closed( + tmp_path, contents, selector +): + _write( + tmp_path / "constants.py", f"# @sc [decision:choice]\n{contents}\n" + ) + record = _record(f"Static value. Values: {selector} = 1.") + tags, errors = scan_tags(tmp_path) + + assert errors == [] + assert value_errors(tmp_path, record, tags) + + def test_ini_multiline_value_keeps_continuation_lines(tmp_path): _write( tmp_path / "config.ini", @@ -363,6 +472,21 @@ def test_python_value_reassignment_fails_closed(tmp_path): assert "exactly one tagged site" in problem +def test_reassigned_module_binding_reports_ambiguity_not_zero_sites(tmp_path): + _write( + tmp_path / "constants.py", + "# @sc [decision:choice]\nWIDTH = 51\nWIDTH = 53\n", + ) + record = _record("Stamp. Values: WIDTH = 51.") + tags, errors = scan_tags(tmp_path) + + assert errors == [] + problem, = value_errors(tmp_path, record, tags) + assert "ambiguous binding" in problem + assert "found 2" in problem + assert "found 0" not in problem + + def test_python_qualified_selector_ignores_other_tagged_scopes(tmp_path): _write( tmp_path / "constants.py", @@ -427,12 +551,56 @@ def test_repository_decisions_have_sites_and_all_values_match(): errors = repository_errors(REPO_ROOT) assert not errors, errors + +@pytest.mark.parametrize( + "path, pattern, replacement", + [ + ( + "workflow/config/cfis/config_tile_Sx.ini", + "WEIGHT_IMAGE = True", + "WEIGHT_IMAGE = False", + ), + ( + "workflow/config/cfis/config_tile_Sx.ini", + "MAKE_POST_PROCESS = True", + "MAKE_POST_PROCESS = False", + ), + ( + "workflow/config/cfis/star_selection.setools", + "[MASK:star_selection]\n", + "[MASK:star_selection]\nMASK_EXT == 0\n", + ), + ( + "workflow/config/cfis/default_tile.sex", + "SATUR_KEY", + "SATUR_LEVEL 50000\nSATUR_KEY", + ), + ( + "src/shapepipe/modules/ngmix_package/ngmix.py", + "\n boot = ngmix.metacal.", + "\n metacal_pars['step'] = 0.02\n boot = ngmix.metacal.", + ), + ], +) +def test_real_record_catches_gate_and_absence_drift( + tmp_path, path, pattern, replacement +): + """Real-record mutations of gates and absences must break Values checks.""" + record = load_yaml(REPO_ROOT / "astra.yaml") - values_count = sum( - len(_parse_values(definition.get("rationale", ""))[0]) - for _, definition in _iter_decision_defs(record) - ) - assert values_count == 151 + tags, parse_errors = scan_tags(REPO_ROOT) + assert parse_errors == [] + for relative in {tag.path for tag in tags}: + source = REPO_ROOT / relative + if source.is_file(): + _write(tmp_path / relative, source.read_text(encoding="utf-8")) + assert value_errors(tmp_path, record) == [] + + target = tmp_path / path + text = target.read_text(encoding="utf-8") + assert text.count(pattern) == 1 + target.write_text(text.replace(pattern, replacement), encoding="utf-8") + assert value_errors(tmp_path, record) def test_preserved_utilities_import_rule(tmp_path): From bc1d02a0db4be58e731ca66383785b8f4b0ce386 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 06:23:24 +0200 Subject: [PATCH 33/40] fix(config): narrow decision tag placements --- astra.yaml | 10 ++++++---- workflow/config.yaml | 2 +- workflow/config/cfis/config_MCCD.ini | 1 + workflow/config/cfis/config_exp_psfex.ini | 6 +++--- workflow/config/cfis/config_tile_Mc.ini | 1 - workflow/config/cfis/config_tile_Ng_template.ini | 2 +- workflow/config/cfis/config_tile_PiViVi_mccd.ini | 2 +- workflow/config/cfis/config_tile_PiViVi_psfex.ini | 6 +++--- workflow/config/cfis/default.psfex | 7 +++++-- workflow/config/cfis/default_exp.sex | 5 +---- workflow/config/cfis/default_tile.sex | 5 +---- workflow/config/cfis/final_cat.param | 1 - workflow/config/cfis/star_selection.setools | 6 ++++-- workflow/scripts/completeness.py | 3 ++- 14 files changed, 29 insertions(+), 28 deletions(-) diff --git a/astra.yaml b/astra.yaml index 569e368b4..48f8edcef 100644 --- a/astra.yaml +++ b/astra.yaml @@ -115,7 +115,7 @@ decisions: stamp. The stamp bounds the measurable galaxy size and truncates the wings of large galaxies. No rationale for 51 px (about 9.5 arcsec) is recorded. - Values: workflow/config/cfis/default_noimaflags.param#VIGNET = 51; workflow/config/cfis/default.param#VIGNET = 51; VIGNETMAKER_RUNNER_RUN_1.STAMP_SIZE = 51; VIGNETMAKER_RUNNER_RUN_2.STAMP_SIZE = 51; VIGNETMAKER_RUNNER_RUN_1.MASKING = False; VIGNETMAKER_RUNNER_RUN_2.MASKING = False; PSF_SIZE = 51. + Values: workflow/config/cfis/default_noimaflags.param#VIGNET = 51; workflow/config/cfis/default.param#VIGNET = 51; VIGNETMAKER_RUNNER_RUN_1.STAMP_SIZE = 51; VIGNETMAKER_RUNNER_RUN_2.STAMP_SIZE = 51; PSF_SIZE = 51. default: px_51 options: px_51: @@ -329,7 +329,7 @@ analyses: star selection and keep stock values with the 3x3 FWHM 2 px kernel. The tiles differ from Guinot+22 in all three. Matching image simulations must use the same prescription. - Values: workflow/config/cfis/default_tile.sex#DETECT_THRESH = 1.0; workflow/config/cfis/default_tile.sex#ANALYSIS_THRESH = 1.0; workflow/config/cfis/default_tile.sex#DETECT_MINAREA = 3; workflow/config/cfis/default_tile.sex#FILTER = Y; workflow/config/cfis/default_tile.sex#SEEING_FWHM = 0.6; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE = $SP_CONFIG/gauss_3.0_7x7.conv; workflow/config/cfis/default_exp.sex#DETECT_THRESH = 1.5; workflow/config/cfis/default_exp.sex#ANALYSIS_THRESH = 1.5; workflow/config/cfis/default_exp.sex#DETECT_MINAREA = 5; workflow/config/cfis/default_exp.sex#FILTER = Y; workflow/config/cfis/default_exp.sex#SEEING_FWHM = 0.6; workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE = $SP_CONFIG/default.conv. + Values: workflow/config/cfis/default_tile.sex#DETECT_THRESH = 1.0; workflow/config/cfis/default_tile.sex#ANALYSIS_THRESH = 1.0; workflow/config/cfis/default_tile.sex#DETECT_MINAREA = 3; workflow/config/cfis/default_tile.sex#FILTER = Y; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE = $SP_CONFIG/gauss_3.0_7x7.conv; workflow/config/cfis/default_exp.sex#DETECT_THRESH = 1.5; workflow/config/cfis/default_exp.sex#ANALYSIS_THRESH = 1.5; workflow/config/cfis/default_exp.sex#DETECT_MINAREA = 5; workflow/config/cfis/default_exp.sex#FILTER = Y; workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE = $SP_CONFIG/default.conv. default: megapipe_tiles options: megapipe_tiles: @@ -791,7 +791,7 @@ analyses: between SExtractor and setools. The selected model name also flows through SP_PSF in downstream inputs, so producer and consumer choices must move together. - Values: INSTANCE.N_COMP_LOC = 8; INSTANCE.D_COMP_GLOB = 8; INSTANCE.FP_GEOMETRY = CFIS; INSTANCE.RMSE_THRESH = 1.25; INPUTS.MIN_N_STARS = 20; workflow/config/cfis/config_exp_mccd.ini#EXECUTION.MODULE = sextractor_runner,mask_query_runner,setools_runner,mccd_preprocessing_runner,mccd_fit_val_runner,merge_starcat_runner,mccd_plots_runner; SEXTRACTOR_RUNNER.INPUT_DIR = $SP_RUN/output/run_sp_exp_Sp/split_exp_runner/output; SEXTRACTOR_RUNNER.FILE_PATTERN = image,weight,flag; SEXTRACTOR_RUNNER.FLAG_IMAGE = True. + Values: INSTANCE.N_COMP_LOC = 8; INSTANCE.D_COMP_GLOB = 8; INSTANCE.FP_GEOMETRY = CFIS; INSTANCE.RMSE_THRESH = 1.25; INPUTS.MIN_N_STARS = 20; config_MCCD.ini#FIT.LOC_MODEL = hybrid; workflow/config/cfis/config_exp_mccd.ini#EXECUTION.MODULE = sextractor_runner,mask_query_runner,setools_runner,mccd_preprocessing_runner,mccd_fit_val_runner,merge_starcat_runner,mccd_plots_runner; SEXTRACTOR_RUNNER.INPUT_DIR = $SP_RUN/output/run_sp_exp_Sp/split_exp_runner/output; SEXTRACTOR_RUNNER.FILE_PATTERN = image,weight,flag; SEXTRACTOR_RUNNER.FLAG_IMAGE = True. default: psfex options: psfex: @@ -813,7 +813,7 @@ analyses: held-out residuals under the acceptance gate; successful optimisation alone does not establish a usable PSF. Stock values; no rationale recorded. - Values: BASIS_TYPE = PIXEL; PSF_ACCURACY = 0.01; BASIS_NUMBER = 20; PSFVAR_DEGREES = 2; PSF_SAMPLING = 1; PSFVAR_KEYS = XWIN_IMAGE,YWIN_IMAGE; MEF_TYPE = INDEPENDENT. + Values: BASIS_TYPE = PIXEL; PSF_ACCURACY = 0.01; BASIS_NUMBER = 20; PSFVAR_DEGREES = 2; PSF_SAMPLING = 1; PSFVAR_KEYS = XWIN_IMAGE,YWIN_IMAGE; PSFVAR_GROUPS = 1,1; MEF_TYPE = INDEPENDENT. default: pixel_basis_deg2_per_ccd options: pixel_basis_deg2_per_ccd: @@ -1168,6 +1168,8 @@ analyses: (symmetrized_4fold_noise). The cost of not symmetrizing, a hole in the galaxy light, is bounded by central_defect_veto. Noise stays the default until a survey A/B against interpolation. + Values: VIGNETMAKER_RUNNER_RUN_1.MASKING = False; + VIGNETMAKER_RUNNER_RUN_2.MASKING = False. default: noise options: noise: diff --git a/workflow/config.yaml b/workflow/config.yaml index 29ed5189f..495e8d08d 100644 --- a/workflow/config.yaml +++ b/workflow/config.yaml @@ -1,4 +1,3 @@ -# @sc [decision:star_selection_psf.psf_modelling_software,scope:file] # Run configuration for the ShapePipe Snakemake workflow. # # A "run" is declared by a tile list plus the paths below. Everything here is @@ -30,6 +29,7 @@ inputs: container: /project/def-mjhudson/cdaley/containers/shapepipe-develop-runtime.sif # PSF model used by the exposure and tile interpolation stages. +# @sc [decision:star_selection_psf.psf_modelling_software] psf_model: psfex # THE TWO ROOTS (D5). diff --git a/workflow/config/cfis/config_MCCD.ini b/workflow/config/cfis/config_MCCD.ini index be2f7f968..8560ea34b 100644 --- a/workflow/config/cfis/config_MCCD.ini +++ b/workflow/config/cfis/config_MCCD.ini @@ -34,6 +34,7 @@ CCD_STAR_THRESH = 0.15 FP_GEOMETRY = CFIS [FIT] +# @sc [decision:star_selection_psf.psf_modelling_software] LOC_MODEL = hybrid PSF_SIZE = 6.2 PSF_SIZE_TYPE = R2 diff --git a/workflow/config/cfis/config_exp_psfex.ini b/workflow/config/cfis/config_exp_psfex.ini index 5551a1d1c..2ea73b604 100644 --- a/workflow/config/cfis/config_exp_psfex.ini +++ b/workflow/config/cfis/config_exp_psfex.ini @@ -1,4 +1,3 @@ -# @sc [decision:masking.psf_star_mask_veto,scope:file] # ShapePipe configuration file for single-exposures. PSFex PSF model. # Process exposures after splitting, from star detection to PSF model. # ShapePipe generates no masks: SExtractor reads the instrument flag image @@ -59,8 +58,6 @@ TIMEOUT = 96:00:00 ## Module options -# @sc [decision:detection.background_model,label:override] header-background-replaces-auto -# BKG_FROM_HEADER replaces the AUTO background with a manual header value; enabling it changes the estimator rather than only its source. [SEXTRACTOR_RUNNER] # The split CCDs, and nothing else: ShapePipe generates no masks @@ -125,6 +122,8 @@ ZP_KEY = PHOTZP # If BKG_FROM_HEADER is True, background value will be read from header. # In that case, the value of BACK_TYPE will be set atomatically to MANUAL. # This is used e.g. for the LSB images. +# @sc [decision:detection.background_model,label:override] header-background-replaces-auto +# BKG_FROM_HEADER replaces the AUTO background with a manual header value; enabling it changes the estimator rather than only its source. BKG_FROM_HEADER = False # LSB images: # BKG_FROM_HEADER = True @@ -147,6 +146,7 @@ SUFFIX = sexcat MAKE_POST_PROCESS = FALSE +# @sc [decision:masking.psf_star_mask_veto] [MASK_QUERY_RUNNER] INPUT_DIR = $SP_RUN/output/run_sp_exp_SxSePsf/sextractor_runner/output diff --git a/workflow/config/cfis/config_tile_Mc.ini b/workflow/config/cfis/config_tile_Mc.ini index 16c0a5e8f..8e0085129 100644 --- a/workflow/config/cfis/config_tile_Mc.ini +++ b/workflow/config/cfis/config_tile_Mc.ini @@ -69,7 +69,6 @@ FILE_PATTERN = sexcat, galaxy_psf, ngmix # NUMBER_LIST selects this unit; the workflow sets SP_UNIT_NUM to the # dashed tile ID (e.g. -210-282) -# @sc [decision:catalogue_assembly.tile_overlap_handling] NUMBER_LIST = $SP_UNIT_NUM # FILE_EXT (optional) list of string extensions to identify input files diff --git a/workflow/config/cfis/config_tile_Ng_template.ini b/workflow/config/cfis/config_tile_Ng_template.ini index eef3532d0..7e5df62e4 100644 --- a/workflow/config/cfis/config_tile_Ng_template.ini +++ b/workflow/config/cfis/config_tile_Ng_template.ini @@ -120,7 +120,7 @@ MAG_ZP = 30.0 # Pixel scale in arcsec -# @sc [decision:shape_measurement.fit_priors,decision:star_selection_psf.star_selection_box] +# @sc [decision:shape_measurement.fit_priors] PIXEL_SCALE = 0.186 # ID_OBJ_MIN/MAX: this chunk's closed SExtractor NUMBER-column range, diff --git a/workflow/config/cfis/config_tile_PiViVi_mccd.ini b/workflow/config/cfis/config_tile_PiViVi_mccd.ini index 0300c3374..26144df47 100644 --- a/workflow/config/cfis/config_tile_PiViVi_mccd.ini +++ b/workflow/config/cfis/config_tile_PiViVi_mccd.ini @@ -17,12 +17,12 @@ RUN_DATETIME = False ## ShapePipe execution options -# @sc [decision:star_selection_psf.psf_modelling_software] [EXECUTION] # Module name, single string or comma-separated list of valid module runner names #MODULE = mccd_interp_runner, +# @sc [decision:star_selection_psf.psf_modelling_software] MODULE = ${SP_PSF}_interp_runner, vignetmaker_runner, vignetmaker_runner # Parallel processing mode, SMP or MPI diff --git a/workflow/config/cfis/config_tile_PiViVi_psfex.ini b/workflow/config/cfis/config_tile_PiViVi_psfex.ini index f06d1278c..5811aa18d 100644 --- a/workflow/config/cfis/config_tile_PiViVi_psfex.ini +++ b/workflow/config/cfis/config_tile_PiViVi_psfex.ini @@ -17,12 +17,12 @@ RUN_DATETIME = False ## ShapePipe execution options -# @sc [decision:star_selection_psf.psf_modelling_software] [EXECUTION] # Module name, single string or comma-separated list of valid module runner names #MODULE = psfex_interp_runner, +# @sc [decision:star_selection_psf.psf_modelling_software] MODULE = ${SP_PSF}_interp_runner, vignetmaker_runner, vignetmaker_runner # Parallel processing mode, SMP or MPI @@ -137,7 +137,7 @@ FILE_EXT = .fits, .fits # NUMBERING_SCHEME (optional) string with numbering pattern for input files NUMBERING_SCHEME = -000-000 -# @sc [decision:postage_stamp_size] +# @sc [decision:shape_measurement.defect_fill] MASKING = False MASK_VALUE = 0 @@ -177,7 +177,7 @@ FILE_EXT = .fits, .sqlite, .txt # NUMBERING_SCHEME (optional) string with numbering pattern for input files NUMBERING_SCHEME = -000-000 -# @sc [decision:postage_stamp_size] +# @sc [decision:shape_measurement.defect_fill] MASKING = False MASK_VALUE = 0 diff --git a/workflow/config/cfis/default.psfex b/workflow/config/cfis/default.psfex index 2ecb65438..9797709b6 100644 --- a/workflow/config/cfis/default.psfex +++ b/workflow/config/cfis/default.psfex @@ -1,4 +1,3 @@ -# @sc [decision:star_selection_psf.psfex_candidate_vetting,scope:file] # Default configuration file for PSFEx 3.17.1 # EB 2017-11-30 # @@ -26,6 +25,7 @@ PSF_ACCURACY 0.01 # Accuracy to expect from PSF "pixel" values # @sc [decision:postage_stamp_size] PSF_SIZE 51,51 # Image size of the PSF model +# @sc [decision:star_selection_psf.psfex_candidate_vetting] PSF_RECENTER N # Allow recentering of PSF-candidates Y/N ? # @sc [decision:star_selection_psf.psf_model_complexity] @@ -47,6 +47,7 @@ PHOTFLUXERR_KEY FLUXERR_AUTO # Catalogue parameter for photometric error # @sc [decision:preparation.object_position_columns,decision:star_selection_psf.psf_model_complexity] PSFVAR_KEYS XWIN_IMAGE,YWIN_IMAGE # Catalogue or FITS (preceded by :) params +# @sc [decision:star_selection_psf.psf_model_complexity] PSFVAR_GROUPS 1,1 # Group tag for each context key # @sc [decision:star_selection_psf.psf_model_complexity] @@ -58,10 +59,12 @@ STABILITY_TYPE EXPOSURE # EXPOSURE or SEQUENCE #----------------------------- Sample selection ------------------------------ -# @sc [decision:star_selection_psf.star_selection_box] +# @sc [decision:star_selection_psf.star_selection_box,decision:star_selection_psf.psfex_candidate_vetting] SAMPLE_AUTOSELECT N # Automatically select the FWHM (Y/N) ? +# @sc [decision:star_selection_psf.psfex_candidate_vetting] BADPIXEL_FILTER N # Filter bad-pixels in samples (Y/N) ? + BADPIXEL_NMAX 0 # Maximum number of bad pixels allowed #----------------------- PSF homogeneisation kernel -------------------------- diff --git a/workflow/config/cfis/default_exp.sex b/workflow/config/cfis/default_exp.sex index 5e69b450d..ed8b8b928 100644 --- a/workflow/config/cfis/default_exp.sex +++ b/workflow/config/cfis/default_exp.sex @@ -1,4 +1,3 @@ -# @sc [decision:detection.saturation_level,scope:file] # Default configuration file for SExtractor 2.19.5 # EB 2017-11-30 # @@ -30,7 +29,6 @@ ANALYSIS_THRESH 1.5 # or , in mag.arcsec-2 # @sc [decision:detection.detection_threshold_policy] FILTER Y # apply filter for detection (Y or N)? -# @sc [decision:detection.detection_threshold_policy] FILTER_NAME default.conv FILTER_THRESH # Threshold[s] for retina filtering @@ -82,6 +80,7 @@ PHOT_AUTOAPERS 0.0,0.0 # , minimum apertures # @sc [decision:detection.photometry_parameters] PHOT_FLUXFRAC 0.5 # flux fraction[s] used for FLUX_RADIUS +# @sc [decision:detection.saturation_level] SATUR_KEY SATURATE # keyword for saturation level (in ADUs) MAG_ZEROPOINT 30.0 # magnitude zero-point @@ -89,12 +88,10 @@ MAG_GAMMA 4.0 # gamma of emulsion (for photographic scans) GAIN_KEY GAIN # keyword for detector gain in e-/ADU -# @sc [decision:star_selection_psf.star_selection_box] PIXEL_SCALE 0. # size of pixel in arcsec (0=use FITS WCS info) #------------------------- Star/Galaxy Separation ---------------------------- -# @sc [decision:detection.detection_threshold_policy] SEEING_FWHM 0.6 # stellar FWHM in arcsec STARNNW_NAME default.nnw diff --git a/workflow/config/cfis/default_tile.sex b/workflow/config/cfis/default_tile.sex index 508aae389..1fccf449a 100644 --- a/workflow/config/cfis/default_tile.sex +++ b/workflow/config/cfis/default_tile.sex @@ -1,4 +1,3 @@ -# @sc [decision:detection.saturation_level,scope:file] # Default configuration file for SExtractor 2.19.5 # EB 2017-11-30 # @@ -7,7 +6,6 @@ CATALOG_TYPE FITS_LDAC -# @sc [decision:detection.detection_source_mode] PARAMETERS_NAME default.param #------------------------------- Extraction ---------------------------------- @@ -31,7 +29,6 @@ ANALYSIS_THRESH 1.0 # or , in mag.arcsec-2 # @sc [decision:detection.detection_threshold_policy] FILTER Y # apply filter for detection (Y or N)? -# @sc [decision:detection.detection_threshold_policy] FILTER_NAME gauss_3.0_7x7.conv FILTER_THRESH # Threshold[s] for retina filtering @@ -83,6 +80,7 @@ PHOT_AUTOAPERS 0.0,0.0 # , minimum apertures # @sc [decision:detection.photometry_parameters] PHOT_FLUXFRAC 0.5 # flux fraction[s] used for FLUX_RADIUS +# @sc [decision:detection.saturation_level] SATUR_KEY SATURATE # keyword for saturation level (in ADUs) # @sc [decision:photometric_zeropoint] @@ -95,7 +93,6 @@ PIXEL_SCALE 0. # size of pixel in arcsec (0=use FITS WCS info) #------------------------- Star/Galaxy Separation ---------------------------- -# @sc [decision:detection.detection_threshold_policy] SEEING_FWHM 0.6 # stellar FWHM in arcsec STARNNW_NAME default.nnw diff --git a/workflow/config/cfis/final_cat.param b/workflow/config/cfis/final_cat.param index f3b56c465..231fbec36 100644 --- a/workflow/config/cfis/final_cat.param +++ b/workflow/config/cfis/final_cat.param @@ -1,4 +1,3 @@ -# @sc [decision:catalogue_assembly.star_galaxy_classification,scope:file] # coordinates # @sc [decision:preparation.object_position_columns] diff --git a/workflow/config/cfis/star_selection.setools b/workflow/config/cfis/star_selection.setools index 97fcdb4a9..2ab3bff8e 100644 --- a/workflow/config/cfis/star_selection.setools +++ b/workflow/config/cfis/star_selection.setools @@ -1,4 +1,3 @@ -# @sc [decision:masking.psf_star_mask_veto,decision:star_selection_psf.star_selection_box,scope:file] ## SETools configuration file for star/galaxy separation based on size/mag properties ## ## ONE mask cut, and it is the instrument flags: @@ -29,6 +28,7 @@ ## config, which ships commented out; SETools has no bitwise operators, so the ## bit selection happens there and this file only ever tests for zero. +# @sc [decision:masking.psf_star_mask_veto,decision:star_selection_psf.star_selection_box] [MASK:preselect] MAG_AUTO > 0 @@ -43,6 +43,7 @@ IMAFLAGS_ISO == 0 NO_SAVE +# @sc [decision:masking.psf_star_mask_veto] [MASK:flag] # @sc [decision:detection.saturation_level] @@ -52,6 +53,7 @@ IMAFLAGS_ISO == 0 NO_SAVE +# @sc [decision:masking.psf_star_mask_veto,decision:star_selection_psf.star_selection_box] [MASK:star_selection] # Star selection using the FWHM mode @@ -118,7 +120,7 @@ FORMAT = png X = X_IMAGE{star_selection} Y = Y_IMAGE{star_selection} -# @sc [label:diagnostic] stellar-diagnostic-matches-cut +# @sc [decision:star_selection_psf.star_selection_box,label:diagnostic] stellar-diagnostic-matches-cut # FWHM_FIELD plot labels and summary statistics must describe the size cut actually applied. SCATTER = FWHM_IMAGE{star_selection}*0.187 diff --git a/workflow/scripts/completeness.py b/workflow/scripts/completeness.py index 41fbc87d2..c29f4479b 100644 --- a/workflow/scripts/completeness.py +++ b/workflow/scripts/completeness.py @@ -1,7 +1,7 @@ #!/usr/bin/env python3 """The count-based completeness table — the single failure policy. -@sc [decision:per_unit_completeness,label:policy] exact-counts-fail-the-unit +@sc [label:policy] exact-counts-fail-the-unit A mandatory runner below its ``expect`` count fails its whole unit (an exposure, a tile or an ngmix chunk), so a partial unit never reaches the catalogue. On the psfex path only exposure-side psfex_interp may fall short (a @@ -81,6 +81,7 @@ # stage -> {runner_subdir: {expect, [warn], [subpath]}} # exp_psf and tile_vignets are selected by $SP_PSF at check time. +# @sc [decision:per_unit_completeness] COMPLETENESS = { # --- tile prepare (phase A) --- # get_images counts are CONFIG-FLAVOR-DEPENDENT: the v2.0 bash table said 4/6 From 6149db21e550917f1d7bc11d853843cbf3e519e8 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 06:29:41 +0200 Subject: [PATCH 34/40] fix(science): refine local decision-site contracts --- .../find_exposures_package/find_exposures.py | 7 ++----- .../modules/make_cat_package/make_cat.py | 19 ++++++++----------- src/shapepipe/modules/make_cat_runner.py | 6 +----- .../modules/mccd_package/__init__.py | 1 - .../mccd_package/shapepipe_auxiliary_mccd.py | 2 ++ src/shapepipe/modules/ngmix_package/ngmix.py | 2 ++ .../psfex_interp_package/psfex_interp.py | 9 ++++----- .../vignetmaker_package/vignetmaker.py | 7 +------ src/shapepipe/pipeline/str_handler.py | 6 +----- workflow/config/cfis/final_cat.param | 3 +-- workflow/scripts/completeness.py | 13 ++++++------- 11 files changed, 28 insertions(+), 47 deletions(-) diff --git a/src/shapepipe/modules/find_exposures_package/find_exposures.py b/src/shapepipe/modules/find_exposures_package/find_exposures.py index 0f3815a93..2ee704ada 100644 --- a/src/shapepipe/modules/find_exposures_package/find_exposures.py +++ b/src/shapepipe/modules/find_exposures_package/find_exposures.py @@ -65,11 +65,8 @@ def get_exposure_list(self): FITS header. @sc [decision:preparation.epoch_provenance_from_tile_history,label:convention] epochs-from-tile-history - The epoch list is the deduplicated file names in HISTORY column COLNUM - with the extension stripped and the trailing ``p`` kept; a tile whose - header cannot be read must fail, never yield an empty or partial list. - EXP_PREFIX strips only a leading prefix; CFIS names carry none, so the - CFIS config leaves it blank and must not set it to the ``p`` suffix. + Apply ``EXP_PREFIX`` only at the start of the parsed name; do not strip + matching text from the middle or remove the trailing ``p`` here. Returns ------- diff --git a/src/shapepipe/modules/make_cat_package/make_cat.py b/src/shapepipe/modules/make_cat_package/make_cat.py index 1a4f6cb8c..c45d00e18 100644 --- a/src/shapepipe/modules/make_cat_package/make_cat.py +++ b/src/shapepipe/modules/make_cat_package/make_cat.py @@ -253,10 +253,8 @@ def save_mask_ext_data(final_cat_file, band_paths, w_log): shared with the ``mask_query`` module: one primitive, two consumers. @sc [decision:masking.sky_mask_application,label:scope] mask-columns-verbatim - Each ``MASK_`` holds the map value at the object's windowed position - verbatim, off-coverage sentinel included; nothing here thresholds, - interprets or removes an object. Without MASK_EXT_PATHS no column is - written and every mask cut happens downstream. + Query ``XWIN_WORLD`` and ``YWIN_WORLD`` in catalogue order and pass the + ``query_map`` result to ``MASK_`` unchanged. Parameters ---------- @@ -420,13 +418,12 @@ def _save_ngmix_data(self, ngmix_cat_path, moments=False): moments : bool, optional If True, write the parallel ``NGMIXm_*`` (moments-branch) columns. - @sc [decision:catalogue_assembly.failure_sentinels,label:coupling] never-fit-sentinels-out-of-range - A detection with no ngmix row keeps NGMIX_N_EPOCH 0 and shape sentinels - outside any measured range: ellipticities and their errors -10, flux - and magnitude errors -1, size errors 1e30. A cut on NGMIX_N_EPOCH > 0 - or on these values removes such a row independently of the flag - columns; the size and flux sentinels (0) lie inside the physical range - and do not. Keep every sentinel out of range when changing one. + @sc [decision:catalogue_assembly.failure_sentinels,label:coupling] failure-sentinel-cut-semantics + Missing-row ``T``, ``SNR``, flux, magnitude, PSF-size and flag values + initialize to 0, which is in range and cannot identify failure. Use + ``NGMIX_N_EPOCH == 0`` for that cut; the -10 ellipticity, -1 + flux/magnitude-error and 1e30 size-error sentinels are out of range and + can also identify missing fits. """ self._key_ends = ["1M", "1P", "2M", "2P", "NOSHEAR"] diff --git a/src/shapepipe/modules/make_cat_runner.py b/src/shapepipe/modules/make_cat_runner.py index 32ee5b018..5cf80981a 100644 --- a/src/shapepipe/modules/make_cat_runner.py +++ b/src/shapepipe/modules/make_cat_runner.py @@ -39,11 +39,7 @@ def make_cat_runner( ): """Define The Make Catalogue Runner. - @sc [decision:catalogue_assembly.star_galaxy_classification,label:scope] classification-deferred-downstream - The final catalogue carries every detection: no star/galaxy cut is made - here, and separation happens downstream. With SM_DO_CLASSIFICATION on, the - thresholds must come from SM_STAR_THRESH and SM_GAL_THRESH; the function - defaults of :func:`make_cat.save_sm_data` are never used. + @sc [decision:catalogue_assembly.star_galaxy_classification] @sc [decision:masking.sky_mask_application] """ diff --git a/src/shapepipe/modules/mccd_package/__init__.py b/src/shapepipe/modules/mccd_package/__init__.py index facac4d97..d0ccf038c 100644 --- a/src/shapepipe/modules/mccd_package/__init__.py +++ b/src/shapepipe/modules/mccd_package/__init__.py @@ -175,7 +175,6 @@ Option to remove validated stars that are outliers in terms of shape before drawing the plots -@sc [decision:star_selection_psf.psf_modelling_software] """ __all__ = [ diff --git a/src/shapepipe/modules/mccd_package/shapepipe_auxiliary_mccd.py b/src/shapepipe/modules/mccd_package/shapepipe_auxiliary_mccd.py index f0aea4f65..170e746b9 100644 --- a/src/shapepipe/modules/mccd_package/shapepipe_auxiliary_mccd.py +++ b/src/shapepipe/modules/mccd_package/shapepipe_auxiliary_mccd.py @@ -249,6 +249,8 @@ def mccd_fit_pipeline( Fit the MCCD model to the Observations. + @sc [decision:star_selection_psf.psf_modelling_software] + Parameters ---------- trainstar_path : str diff --git a/src/shapepipe/modules/ngmix_package/ngmix.py b/src/shapepipe/modules/ngmix_package/ngmix.py index 4c150d789..efcde1415 100644 --- a/src/shapepipe/modules/ngmix_package/ngmix.py +++ b/src/shapepipe/modules/ngmix_package/ngmix.py @@ -1900,6 +1900,8 @@ def average_multiepoch_psf(obsdict): Keys: 'g_psf', 'g_psf_err', 'T_psf', 'T_psf_err' (weighted averages over the epochs whose PSF fit succeeded) and 'n_epoch' (the number of those surviving epochs). + + @sc [decision:shape_measurement.psf_epoch_averaging] """ # ignore_failed_psf=True drops failed-PSF epochs from the galaxy fit but # keeps them in obsdict; _average_psf_fits skips them on flags != 0. diff --git a/src/shapepipe/modules/psfex_interp_package/psfex_interp.py b/src/shapepipe/modules/psfex_interp_package/psfex_interp.py index b136e44a2..82e43f0ad 100644 --- a/src/shapepipe/modules/psfex_interp_package/psfex_interp.py +++ b/src/shapepipe/modules/psfex_interp_package/psfex_interp.py @@ -240,11 +240,10 @@ def interpsfex(self, dotpsfpath, pos): Use PSFEx generated model to perform spatial PSF interpolation. - @sc [decision:star_selection_psf.psf_acceptance_thresholds,label:gate] psf-gate-drops-epoch - A model with ACCEPTED below STAR_THRESH or CHI2 above CHI2_THRESH - yields a failure sentinel, not PSFs, and every caller drops that CCD's - epoch; the gate is the same for the validation and multi-epoch passes. - The thresholds come from config, never from literals here. + @sc [decision:star_selection_psf.psf_acceptance_thresholds,label:gate] psf-gate-path-specific-output + Validation logs a rejected gate and skips writing validation output; + the multi-epoch science path logs the same sentinel and skips that CCD + so affected objects lose its epoch. Parameters ---------- diff --git a/src/shapepipe/modules/vignetmaker_package/vignetmaker.py b/src/shapepipe/modules/vignetmaker_package/vignetmaker.py index 9f119be20..586dae28e 100644 --- a/src/shapepipe/modules/vignetmaker_package/vignetmaker.py +++ b/src/shapepipe/modules/vignetmaker_package/vignetmaker.py @@ -21,12 +21,7 @@ def get_stamps(image, positions, rad): Extract postage stamps and record their sub-pixel centring. - @sc [decision:preparation.stamp_positioning_and_padding,label:convention] stamps-off-image-raise - Stamps are cut around the rounded pixel with no interpolation, and OFFSET - is the remainder of that same rounding, the Jacobian origin ngmix uses; two - roundings would put the stamp and its centroid prior out of step. Edge - overruns are zero-padded and kept; a centre rounding outside the image - raises, never wraps. + @sc [decision:preparation.stamp_positioning_and_padding] The image is zero-padded by ``rad`` on every side and a ``(2 * rad + 1, 2 * rad + 1)`` stamp is sliced around the integer pixel diff --git a/src/shapepipe/pipeline/str_handler.py b/src/shapepipe/pipeline/str_handler.py index 7e6f5d5f3..06d6e3cbb 100644 --- a/src/shapepipe/pipeline/str_handler.py +++ b/src/shapepipe/pipeline/str_handler.py @@ -258,11 +258,7 @@ def _mode(self, input, eps=0.001, iter_max=1000): Compute the mode, the most frequent value of a continuous distribution. - @sc [decision:star_selection_psf.star_selection_box,label:estimator] mode-median-fallback-at-20 - Below 20 objects the result is the median; from 20 up it is the - iterative histogram-zoom mode. The star-selection FWHM box is centred - on this value, so moving the threshold or the binning changes which - stars train the PSF on sparse CCDs. + @sc [decision:star_selection_psf.star_selection_box] Parameters ---------- diff --git a/workflow/config/cfis/final_cat.param b/workflow/config/cfis/final_cat.param index 231fbec36..550daaf2f 100644 --- a/workflow/config/cfis/final_cat.param +++ b/workflow/config/cfis/final_cat.param @@ -7,8 +7,7 @@ YWIN_WORLD # tile ID, for plot of tile-dependent additive bias. # Can maybe be removed. -# @sc [decision:catalogue_assembly.tile_overlap_handling,label:provenance] tile-id-is-not-unique-object-id -# TILE_ID is tile provenance, not an object-uniqueness or overlap flag; duplicate removal or overlap flagging requires explicit downstream code. +# @sc [decision:catalogue_assembly.tile_overlap_handling] TILE_ID # flags diff --git a/workflow/scripts/completeness.py b/workflow/scripts/completeness.py index c29f4479b..4e3f5c329 100644 --- a/workflow/scripts/completeness.py +++ b/workflow/scripts/completeness.py @@ -1,13 +1,6 @@ #!/usr/bin/env python3 """The count-based completeness table — the single failure policy. -@sc [label:policy] exact-counts-fail-the-unit -A mandatory runner below its ``expect`` count fails its whole unit (an -exposure, a tile or an ngmix chunk), so a partial unit never reaches the -catalogue. On the psfex path only exposure-side psfex_interp may fall short (a -CCD rejected by the acceptance gate writes nothing); the never-run MCCD chain -warns throughout. - This is the ported ``complete_check`` count table from the v2.0 bash layer (``run_job_sp_canfar_v2.0.bash`` job dispatch, survey §4): every non-warning runner is expected to produce exactly its ``expect`` count. Across smk-g6 @@ -371,6 +364,12 @@ def _unit_from_run_dir(run_dir): def main(argv=None) -> int: + """Run the CLI and persist the per-unit verdict. + + @sc [label:policy] exact-counts-fail-the-unit + ``--job-rc`` can fail a stage even when product counts pass; the log records + every verdict, while the manifest is emitted only for success. + """ p = argparse.ArgumentParser(description="ShapePipe per-unit completeness check") sub = p.add_subparsers(dest="cmd", required=True) c = sub.add_parser("check", help="count products, write the manifest") From 7fc2e698f1ad98ec265d5199db6f222284b10ebe Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 06:38:27 +0200 Subject: [PATCH 35/40] docs: simplify and wrap Values assertions --- astra.yaml | 222 +++++++++++++++++++++++++++++++++++++++++++++-------- 1 file changed, 191 insertions(+), 31 deletions(-) diff --git a/astra.yaml b/astra.yaml index 48f8edcef..d7f00100c 100644 --- a/astra.yaml +++ b/astra.yaml @@ -87,7 +87,10 @@ decisions: per CCD; preprocessing merges an exposure's stars into one train and one test catalogue). Science-path PSF rejection bypasses this table: psfex_interp drops the epoch per object inside the tile run. - Values: COMPLETENESS[exp_psf.psfex.psfex_interp_runner.warn] = True; COMPLETENESS[exp_psf.mccd.mask_query_runner.expect] = 40; COMPLETENESS[exp_psf.mccd.mccd_preprocessing_runner.expect] = 2. + Values: + COMPLETENESS[exp_psf.psfex.psfex_interp_runner.warn] = True; + COMPLETENESS[exp_psf.mccd.mask_query_runner.expect] = 40; + COMPLETENESS[exp_psf.mccd.mccd_preprocessing_runner.expect] = 2. default: exact_counts options: exact_counts: @@ -115,7 +118,12 @@ decisions: stamp. The stamp bounds the measurable galaxy size and truncates the wings of large galaxies. No rationale for 51 px (about 9.5 arcsec) is recorded. - Values: workflow/config/cfis/default_noimaflags.param#VIGNET = 51; workflow/config/cfis/default.param#VIGNET = 51; VIGNETMAKER_RUNNER_RUN_1.STAMP_SIZE = 51; VIGNETMAKER_RUNNER_RUN_2.STAMP_SIZE = 51; PSF_SIZE = 51. + Values: + default_noimaflags.param#VIGNET = 51; + default.param#VIGNET = 51; + VIGNETMAKER_RUNNER_RUN_1.STAMP_SIZE = 51; + VIGNETMAKER_RUNNER_RUN_2.STAMP_SIZE = 51; + PSF_SIZE = 51. default: px_51 options: px_51: @@ -136,7 +144,12 @@ decisions: with header PHOTZP as well as tile MAG_ZEROPOINT and ngmix MAG_ZP. Magnitude cuts (the star-selection window, downstream galaxy cuts) inherit their stage's convention. - Values: MAG_ZEROPOINT = 30.0; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER = False; NGMIX_RUNNER.MAG_ZP = 30.0; workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER = True; SEXTRACTOR_RUNNER.ZP_KEY = PHOTZP. + Values: + MAG_ZEROPOINT = 30.0; + config_tile_Sx.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER = False; + NGMIX_RUNNER.MAG_ZP = 30.0; + config_exp_psfex.ini#SEXTRACTOR_RUNNER.ZP_FROM_HEADER = True; + SEXTRACTOR_RUNNER.ZP_KEY = PHOTZP. default: fixed_30_tiles_header_exposures options: fixed_30_tiles_header_exposures: @@ -217,7 +230,10 @@ analyses: (detection.detection_source_mode). Sky-fixed masks never touch pixels: an object inside a star halo is measured from the same unmodified pixels as one outside it. - Values: workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE = True; VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_PATTERN = flag, image, weight, background, background_rms. + Values: + config_exp_psfex.ini#SEXTRACTOR_RUNNER.FLAG_IMAGE = True; + VIGNETMAKER_RUNNER_RUN_2.ME_IMAGE_PATTERN = flag, image, weight, + background, background_rms. default: instrument_flags_only options: instrument_flags_only: @@ -241,7 +257,13 @@ analyses: MASK_EXT. The intended map is the UNIONS star-body product (bit 2); halo bits 0 and 1 are left out because halos say nothing about whether a star is a good PSF sample. - Values: MASK:star_selection.IMAFLAGS_ISO = "== 0"; MASK:preselect.IMAFLAGS_ISO = "== 0"; MASK:flag.IMAFLAGS_ISO = "== 0"; MASK:star_selection.MASK_EXT = absent; MASK:preselect.MASK_EXT = absent; MASK:flag.MASK_EXT = absent. + Values: + MASK:star_selection.IMAFLAGS_ISO = "== 0"; + MASK:preselect.IMAFLAGS_ISO = "== 0"; + MASK:flag.IMAFLAGS_ISO = "== 0"; + MASK:star_selection.MASK_EXT = absent; + MASK:preselect.MASK_EXT = absent; + MASK:flag.MASK_EXT = absent. default: instrument_flags_only options: instrument_flags_only: @@ -329,7 +351,19 @@ analyses: star selection and keep stock values with the 3x3 FWHM 2 px kernel. The tiles differ from Guinot+22 in all three. Matching image simulations must use the same prescription. - Values: workflow/config/cfis/default_tile.sex#DETECT_THRESH = 1.0; workflow/config/cfis/default_tile.sex#ANALYSIS_THRESH = 1.0; workflow/config/cfis/default_tile.sex#DETECT_MINAREA = 3; workflow/config/cfis/default_tile.sex#FILTER = Y; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE = $SP_CONFIG/gauss_3.0_7x7.conv; workflow/config/cfis/default_exp.sex#DETECT_THRESH = 1.5; workflow/config/cfis/default_exp.sex#ANALYSIS_THRESH = 1.5; workflow/config/cfis/default_exp.sex#DETECT_MINAREA = 5; workflow/config/cfis/default_exp.sex#FILTER = Y; workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE = $SP_CONFIG/default.conv. + Values: + default_tile.sex#DETECT_THRESH = 1.0; + default_tile.sex#ANALYSIS_THRESH = 1.0; + default_tile.sex#DETECT_MINAREA = 3; + default_tile.sex#FILTER = Y; + config_tile_Sx.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE = + $SP_CONFIG/gauss_3.0_7x7.conv; + default_exp.sex#DETECT_THRESH = 1.5; + default_exp.sex#ANALYSIS_THRESH = 1.5; + default_exp.sex#DETECT_MINAREA = 5; + default_exp.sex#FILTER = Y; + config_exp_psfex.ini#SEXTRACTOR_RUNNER.DOT_CONV_FILE = + $SP_CONFIG/default.conv. default: megapipe_tiles options: megapipe_tiles: @@ -350,7 +384,11 @@ analyses: the value Guinot+22 lists. Both share DEBLEND_NTHRESH. Contrast sets object count, centroids, and blend contamination in shapes; matching image simulations must preserve the stage-specific settings. - Values: workflow/config/cfis/default_tile.sex#DEBLEND_MINCONT = 0.002; workflow/config/cfis/default_exp.sex#DEBLEND_MINCONT = 0.001; workflow/config/cfis/default_tile.sex#DEBLEND_NTHRESH = 32; workflow/config/cfis/default_exp.sex#DEBLEND_NTHRESH = 32. + Values: + default_tile.sex#DEBLEND_MINCONT = 0.002; + default_exp.sex#DEBLEND_MINCONT = 0.001; + default_tile.sex#DEBLEND_NTHRESH = 32; + default_exp.sex#DEBLEND_NTHRESH = 32. default: megapipe_tiles options: megapipe_tiles: @@ -372,7 +410,18 @@ analyses: BACKGROUND_RMS (shape_measurement.galaxy_pixel_weights). The header background path is off. Residual sky offsets propagate into thresholds, fluxes, completeness and shapes. - Values: workflow/config/cfis/default_tile.sex#BACK_TYPE = AUTO; workflow/config/cfis/default_tile.sex#BACK_SIZE = 512; workflow/config/cfis/default_tile.sex#BACK_FILTERSIZE = 9; workflow/config/cfis/default_tile.sex#BACKPHOTO_TYPE = LOCAL; BACKPHOTO_THICK = 30; workflow/config/cfis/default_exp.sex#BACK_TYPE = AUTO; workflow/config/cfis/default_exp.sex#BACK_SIZE = 64; workflow/config/cfis/default_exp.sex#BACK_FILTERSIZE = 3; workflow/config/cfis/default_exp.sex#BACKPHOTO_TYPE = GLOBAL; workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.BKG_FROM_HEADER = False; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.BKG_FROM_HEADER = False. + Values: + default_tile.sex#BACK_TYPE = AUTO; + default_tile.sex#BACK_SIZE = 512; + default_tile.sex#BACK_FILTERSIZE = 9; + default_tile.sex#BACKPHOTO_TYPE = LOCAL; + BACKPHOTO_THICK = 30; + default_exp.sex#BACK_TYPE = AUTO; + default_exp.sex#BACK_SIZE = 64; + default_exp.sex#BACK_FILTERSIZE = 3; + default_exp.sex#BACKPHOTO_TYPE = GLOBAL; + config_exp_psfex.ini#SEXTRACTOR_RUNNER.BKG_FROM_HEADER = False; + config_tile_Sx.ini#SEXTRACTOR_RUNNER.BKG_FROM_HEADER = False. default: auto_megapipe_tiles options: auto_megapipe_tiles: @@ -391,7 +440,15 @@ analyses: RESCALE_WEIGHTS and WEIGHT_GAIN keep their SExtractor defaults. Guinot+22 keeps every non-tabulated parameter at its default, which would mean no weight map. - Values: workflow/config/cfis/default_tile.sex#WEIGHT_TYPE = MAP_WEIGHT; workflow/config/cfis/default_exp.sex#WEIGHT_TYPE = MAP_WEIGHT; workflow/config/cfis/default_tile.sex#RESCALE_WEIGHTS = Y; workflow/config/cfis/default_exp.sex#RESCALE_WEIGHTS = Y; workflow/config/cfis/default_tile.sex#WEIGHT_GAIN = Y; workflow/config/cfis/default_exp.sex#WEIGHT_GAIN = Y; workflow/config/cfis/config_tile_Sx.ini#SEXTRACTOR_RUNNER.WEIGHT_IMAGE = True; workflow/config/cfis/config_exp_psfex.ini#SEXTRACTOR_RUNNER.WEIGHT_IMAGE = True. + Values: + default_tile.sex#WEIGHT_TYPE = MAP_WEIGHT; + default_exp.sex#WEIGHT_TYPE = MAP_WEIGHT; + default_tile.sex#RESCALE_WEIGHTS = Y; + default_exp.sex#RESCALE_WEIGHTS = Y; + default_tile.sex#WEIGHT_GAIN = Y; + default_exp.sex#WEIGHT_GAIN = Y; + config_tile_Sx.ini#SEXTRACTOR_RUNNER.WEIGHT_IMAGE = True; + config_exp_psfex.ini#SEXTRACTOR_RUNNER.WEIGHT_IMAGE = True. default: map_weight options: map_weight: @@ -406,7 +463,13 @@ analyses: invents flux across zero-weight pixels, changing detections and photometry near masked regions. Matching image simulations must use the same interpolation operator; this is not a calibrated pixel repair. - Values: workflow/config/cfis/default_tile.sex#INTERP_TYPE = ALL; workflow/config/cfis/default_exp.sex#INTERP_TYPE = ALL; workflow/config/cfis/default_tile.sex#INTERP_MAXXLAG = 16; workflow/config/cfis/default_tile.sex#INTERP_MAXYLAG = 16; workflow/config/cfis/default_exp.sex#INTERP_MAXXLAG = 16; workflow/config/cfis/default_exp.sex#INTERP_MAXYLAG = 16. + Values: + default_tile.sex#INTERP_TYPE = ALL; + default_exp.sex#INTERP_TYPE = ALL; + default_tile.sex#INTERP_MAXXLAG = 16; + default_tile.sex#INTERP_MAXYLAG = 16; + default_exp.sex#INTERP_MAXXLAG = 16; + default_exp.sex#INTERP_MAXYLAG = 16. default: interp_all options: interp_all: @@ -420,7 +483,11 @@ analyses: consistent with being wings of a brighter neighbour, a post-deblend change to the object list. Keep the same cleaning prescription in matching image simulations. - Values: workflow/config/cfis/default_tile.sex#CLEAN_PARAM = 1.0; workflow/config/cfis/default_exp.sex#CLEAN_PARAM = 1.0; workflow/config/cfis/default_tile.sex#CLEAN = Y; workflow/config/cfis/default_exp.sex#CLEAN = Y. + Values: + default_tile.sex#CLEAN_PARAM = 1.0; + default_exp.sex#CLEAN_PARAM = 1.0; + default_tile.sex#CLEAN = Y; + default_exp.sex#CLEAN = Y. default: clean_1 options: clean_1: @@ -432,7 +499,9 @@ analyses: mirror across the object centre during photometry, changing blend fluxes and windowed moments. This SExtractor photometry choice is separate from ngmix's neighbour-pixel weighting decision. - Values: workflow/config/cfis/default_tile.sex#MASK_TYPE = CORRECT; workflow/config/cfis/default_exp.sex#MASK_TYPE = CORRECT. + Values: + default_tile.sex#MASK_TYPE = CORRECT; + default_exp.sex#MASK_TYPE = CORRECT. default: correct options: correct: @@ -452,7 +521,14 @@ analyses: delivered exposure CCDs and MegaPipe tiles carry SATURATE is unverified. Changing the card or pinning a fixed level requires checking the resulting bright-star selection. - Values: workflow/config/cfis/default_exp.sex#SATUR_KEY = SATURATE; workflow/config/cfis/default_tile.sex#SATUR_KEY = SATURATE; MASK:star_selection.FLAGS = "== 0"; MASK:preselect.FLAGS = "== 0"; MASK:flag.FLAGS = "== 0"; workflow/config/cfis/default_tile.sex#SATUR_LEVEL = absent; workflow/config/cfis/default_exp.sex#SATUR_LEVEL = absent. + Values: + default_exp.sex#SATUR_KEY = SATURATE; + default_tile.sex#SATUR_KEY = SATURATE; + MASK:star_selection.FLAGS = "== 0"; + MASK:preselect.FLAGS = "== 0"; + MASK:flag.FLAGS = "== 0"; + default_tile.sex#SATUR_LEVEL = absent; + default_exp.sex#SATUR_LEVEL = absent. default: header_saturate options: header_saturate: @@ -468,7 +544,14 @@ analyses: photometric normalisation. A different Kron factor shifts magnitudes and so every magnitude-based cut. The reported apertures and flux- fraction measurements must also match the simulated catalogue. - Values: workflow/config/cfis/default_tile.sex#PHOT_AUTOPARAMS = 2.5,3.5; workflow/config/cfis/default_exp.sex#PHOT_AUTOPARAMS = 2.5,3.5; workflow/config/cfis/default_tile.sex#PHOT_APERTURES = 5; workflow/config/cfis/default_exp.sex#PHOT_APERTURES = 5; workflow/config/cfis/default_tile.sex#PHOT_FLUXFRAC = 0.5; workflow/config/cfis/default_exp.sex#PHOT_FLUXFRAC = 0.5; PHOTFLUX_KEY = FLUX_AUTO. + Values: + default_tile.sex#PHOT_AUTOPARAMS = 2.5,3.5; + default_exp.sex#PHOT_AUTOPARAMS = 2.5,3.5; + default_tile.sex#PHOT_APERTURES = 5; + default_exp.sex#PHOT_APERTURES = 5; + default_tile.sex#PHOT_FLUXFRAC = 0.5; + default_exp.sex#PHOT_FLUXFRAC = 0.5; + PHOTFLUX_KEY = FLUX_AUTO. default: kron_25_35 options: kron_25_35: @@ -482,7 +565,9 @@ analyses: catalogue carries no IMAFLAGS_ISO. [LINT] final_cat.param, read by the post-processing merge, requests IMAFLAGS_ISO, which the tile chain never produces (issue #912). - Values: SEXTRACTOR_RUNNER.DETECTION_IMAGE = False; SEXTRACTOR_RUNNER.FLAG_IMAGE = False. + Values: + SEXTRACTOR_RUNNER.DETECTION_IMAGE = False; + SEXTRACTOR_RUNNER.FLAG_IMAGE = False. default: sx_nomask_single_image options: sx_nomask_single_image: @@ -505,7 +590,9 @@ analyses: endpoints match MegaCam's raw DATASEC (2048 pixel indices inclusively), but the strict cut excludes both. The excluded strip is likely prescan (unverified). - Values: SEXTRACTOR_RUNNER.CCD_SIZE = 33,2080,1,4612; SEXTRACTOR_RUNNER.MAKE_POST_PROCESS = True. + Values: + SEXTRACTOR_RUNNER.CCD_SIZE = 33,2080,1,4612; + SEXTRACTOR_RUNNER.MAKE_POST_PROCESS = True. default: trimmed_bounds_33_2080 options: trimmed_bounds_33_2080: @@ -599,7 +686,8 @@ analyses: rationale: >- Any HDU count other than N_HDU raises; every CCD is a candidate epoch wherever the WCS lands it. - Values: SPLIT_EXP_RUNNER.N_HDU = 40. + Values: + SPLIT_EXP_RUNNER.N_HDU = 40. default: all_40_hdus options: all_40_hdus: @@ -618,7 +706,8 @@ analyses: an epoch letter rather than a prefix, so EXP_PREFIX is blank; downstream code drops the p when it needs the bare exposure ID. A mis-parse changes N_EPOCH and which exposures are fit. - Values: FIND_EXPOSURES_RUNNER.COLNUM = 3. + Values: + FIND_EXPOSURES_RUNNER.COLNUM = 3. default: history_parse options: history_parse: @@ -633,7 +722,15 @@ analyses: Windowed, isophotal and model centroids differ systematically for blends and asymmetric galaxies, and the centroid feeds the position seed and the centroid prior. - Values: VIGNETMAKER_RUNNER_RUN_1.POSITION_PARAMS = XWIN_IMAGE,YWIN_IMAGE; VIGNETMAKER_RUNNER_RUN_1.COORD = PIX; workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS = XWIN_WORLD,YWIN_WORLD; VIGNETMAKER_RUNNER_RUN_2.POSITION_PARAMS = XWIN_WORLD,YWIN_WORLD; VIGNETMAKER_RUNNER_RUN_2.COORD = SPHE; workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS = XWIN_IMAGE,YWIN_IMAGE. + Values: + VIGNETMAKER_RUNNER_RUN_1.POSITION_PARAMS = XWIN_IMAGE,YWIN_IMAGE; + VIGNETMAKER_RUNNER_RUN_1.COORD = PIX; + config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS = + XWIN_WORLD,YWIN_WORLD; + VIGNETMAKER_RUNNER_RUN_2.POSITION_PARAMS = XWIN_WORLD,YWIN_WORLD; + VIGNETMAKER_RUNNER_RUN_2.COORD = SPHE; + config_exp_psfex.ini#PSFEX_INTERP_RUNNER.POSITION_PARAMS = + XWIN_IMAGE,YWIN_IMAGE. default: xwin_windowed options: xwin_windowed: @@ -711,7 +808,18 @@ analyses: objects, so small-N behaviour changes selection on sparse CCDs. PSFEx's own selection is off; see psfex_candidate_vetting for what PSFEx may still apply. - Values: MASK:star_selection.FLAGS = "== 0"; MASK:preselect.FLAGS = "== 0"; MASK:star_selection.IMAFLAGS_ISO = "== 0"; MASK:preselect.IMAFLAGS_ISO = "== 0"; MASK:star_selection.MAG_AUTO = ["> 18.", "< 22."]; MASK:star_selection.FWHM_IMAGE = ["<= mode(FWHM_IMAGE{preselect}) + 0.2", ">= mode(FWHM_IMAGE{preselect}) - 0.2"]; MASK:preselect.MAG_AUTO = ["> 0", "< 21"]; MASK:preselect.FWHM_IMAGE = ["> 0.3 / 0.187", "< 1.5 / 0.187"]; PLOT:fwhm_field.SCATTER = "FWHM_IMAGE{star_selection}*0.187"; SAMPLE_AUTOSELECT = N. + Values: + MASK:star_selection.FLAGS = "== 0"; + MASK:preselect.FLAGS = "== 0"; + MASK:star_selection.IMAFLAGS_ISO = "== 0"; + MASK:preselect.IMAFLAGS_ISO = "== 0"; + MASK:star_selection.MAG_AUTO = ["> 18.", "< 22."]; + MASK:star_selection.FWHM_IMAGE = ["<= mode(FWHM_IMAGE{preselect}) + + 0.2", ">= mode(FWHM_IMAGE{preselect}) - 0.2"]; + MASK:preselect.MAG_AUTO = ["> 0", "< 21"]; + MASK:preselect.FWHM_IMAGE = ["> 0.3 / 0.187", "< 1.5 / 0.187"]; + PLOT:fwhm_field.SCATTER = "FWHM_IMAGE{star_selection}*0.187"; + SAMPLE_AUTOSELECT = N. default: mode_centred_box options: mode_centred_box: @@ -737,7 +845,12 @@ analyses: against an independent residual test. The permutation is seeded from the digits of the unit's file number, so a CCD gets the same split on every run. - Values: RAND_SPLIT:star_split.RATIO = 20; PSFEX_RUNNER.FILE_PATTERN = star_split_ratio_80; PSFEX_INTERP_RUNNER.FILE_PATTERN = star_split_ratio_80,star_split_ratio_20,psfex_cat; PSFEX_INTERP_RUNNER.ME_DOT_PSF_PATTERN = star_split_ratio_80. + Values: + RAND_SPLIT:star_split.RATIO = 20; + PSFEX_RUNNER.FILE_PATTERN = star_split_ratio_80; + PSFEX_INTERP_RUNNER.FILE_PATTERN = + star_split_ratio_80,star_split_ratio_20,psfex_cat; + PSFEX_INTERP_RUNNER.ME_DOT_PSF_PATTERN = star_split_ratio_80. default: split_80_20_seeded options: split_80_20_seeded: @@ -771,7 +884,17 @@ analyses: Compiled-in values admit no `= value` assertion, so the anchors assert only that each SAMPLE_* key is absent; open issue #919 would pin them in default.psfex. - Values: SAMPLE_AUTOSELECT = N; BADPIXEL_FILTER = N; PSF_RECENTER = N; SAMPLE_FWHMRANGE = absent; SAMPLE_VARIABILITY = absent; SAMPLE_MINSN = absent; SAMPLE_MAXELLIP = absent; SAMPLE_FLAGMASK = absent; SAMPLE_WFLAGMASK = absent; SAMPLE_IMAFLAGMASK = absent. + Values: + SAMPLE_AUTOSELECT = N; + BADPIXEL_FILTER = N; + PSF_RECENTER = N; + SAMPLE_FWHMRANGE = absent; + SAMPLE_VARIABILITY = absent; + SAMPLE_MINSN = absent; + SAMPLE_MAXELLIP = absent; + SAMPLE_FLAGMASK = absent; + SAMPLE_WFLAGMASK = absent; + SAMPLE_IMAFLAGMASK = absent. default: builtin_defaults options: builtin_defaults: @@ -791,7 +914,19 @@ analyses: between SExtractor and setools. The selected model name also flows through SP_PSF in downstream inputs, so producer and consumer choices must move together. - Values: INSTANCE.N_COMP_LOC = 8; INSTANCE.D_COMP_GLOB = 8; INSTANCE.FP_GEOMETRY = CFIS; INSTANCE.RMSE_THRESH = 1.25; INPUTS.MIN_N_STARS = 20; config_MCCD.ini#FIT.LOC_MODEL = hybrid; workflow/config/cfis/config_exp_mccd.ini#EXECUTION.MODULE = sextractor_runner,mask_query_runner,setools_runner,mccd_preprocessing_runner,mccd_fit_val_runner,merge_starcat_runner,mccd_plots_runner; SEXTRACTOR_RUNNER.INPUT_DIR = $SP_RUN/output/run_sp_exp_Sp/split_exp_runner/output; SEXTRACTOR_RUNNER.FILE_PATTERN = image,weight,flag; SEXTRACTOR_RUNNER.FLAG_IMAGE = True. + Values: + INSTANCE.N_COMP_LOC = 8; + INSTANCE.D_COMP_GLOB = 8; + INSTANCE.FP_GEOMETRY = CFIS; + INSTANCE.RMSE_THRESH = 1.25; + INPUTS.MIN_N_STARS = 20; + FIT.LOC_MODEL = hybrid; + config_exp_mccd.ini#EXECUTION.MODULE = + sextractor_runner,mask_query_runner,setools_runner,mccd_preprocessing_runner,mccd_fit_val_runner,merge_starcat_runner,mccd_plots_runner; + SEXTRACTOR_RUNNER.INPUT_DIR = + $SP_RUN/output/run_sp_exp_Sp/split_exp_runner/output; + SEXTRACTOR_RUNNER.FILE_PATTERN = image,weight,flag; + SEXTRACTOR_RUNNER.FLAG_IMAGE = True. default: psfex options: psfex: @@ -813,7 +948,15 @@ analyses: held-out residuals under the acceptance gate; successful optimisation alone does not establish a usable PSF. Stock values; no rationale recorded. - Values: BASIS_TYPE = PIXEL; PSF_ACCURACY = 0.01; BASIS_NUMBER = 20; PSFVAR_DEGREES = 2; PSF_SAMPLING = 1; PSFVAR_KEYS = XWIN_IMAGE,YWIN_IMAGE; PSFVAR_GROUPS = 1,1; MEF_TYPE = INDEPENDENT. + Values: + BASIS_TYPE = PIXEL; + PSF_ACCURACY = 0.01; + BASIS_NUMBER = 20; + PSFVAR_DEGREES = 2; + PSF_SAMPLING = 1; + PSFVAR_KEYS = XWIN_IMAGE,YWIN_IMAGE; + PSFVAR_GROUPS = 1,1; + MEF_TYPE = INDEPENDENT. default: pixel_basis_deg2_per_ccd options: pixel_basis_deg2_per_ccd: @@ -833,7 +976,11 @@ analyses: The pipeline has no minimum-epoch floor: NGMIX_N_EPOCH records what survived and epoch-count cuts happen downstream. Guinot+22 discards the CCD from PSF estimation rather than gating at interpolation. - Values: workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH = 22; workflow/config/cfis/config_exp_psfex.ini#PSFEX_INTERP_RUNNER.CHI2_THRESH = 2; workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH = 22; workflow/config/cfis/config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.CHI2_THRESH = 2. + Values: + config_exp_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH = 22; + config_exp_psfex.ini#PSFEX_INTERP_RUNNER.CHI2_THRESH = 2; + config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.STAR_THRESH = 22; + config_tile_PiViVi_psfex.ini#PSFEX_INTERP_RUNNER.CHI2_THRESH = 2. default: stars22_chi2_2 options: stars22_chi2_2: @@ -1030,7 +1177,10 @@ analyses: arcsec/px while NGMIX_RUNNER.PIXEL_SCALE is 0.186. Reconcile these exposure-pixel conversions; the ngmix_runner comment's noise-window claim is stale because get_noise is never called. - Values: get_prior.T_range = -1,1e3; get_prior.F_range = -100,1e9; NGMIX_RUNNER.PIXEL_SCALE = 0.186. + Values: + get_prior.T_range = -1,1e3; + get_prior.F_range = -100,1e9; + NGMIX_RUNNER.PIXEL_SCALE = 0.186. default: gpriorba04_flat options: gpriorba04_flat: @@ -1051,7 +1201,12 @@ analyses: fitgauss and is unset in the committed config; it moves the response directly. No sheared-PSF types run, so the catalogue has no PSF response term. - Values: METACAL_TYPES = noshear,1p,1m,2p,2m; do_ngmix_metacal.metacal_pars[types] = noshear,1p,1m,2p,2m; do_ngmix_metacal.metacal_pars[step] = 0.01; do_ngmix_metacal.metacal_pars[fixnoise] = True; do_ngmix_metacal.metacal_pars[use_noise_image] = True. + Values: + METACAL_TYPES = noshear,1p,1m,2p,2m; + do_ngmix_metacal.metacal_pars[types] = noshear,1p,1m,2p,2m; + do_ngmix_metacal.metacal_pars[step] = 0.01; + do_ngmix_metacal.metacal_pars[fixnoise] = True; + do_ngmix_metacal.metacal_pars[use_noise_image] = True. default: five_types_step001_fitgauss options: five_types_step001_fitgauss: @@ -1110,7 +1265,9 @@ analyses: fallback is a scalar 1/sigma_mad^2. Each epoch is background-subtracted with the SExtractor background vignet (BKG_SUB, on unless set False). - Values: NGMIX_RUNNER.BKG_RMS_VIGNET_PATH = $NGMIX_VIGNET_DIR/vignetmaker_runner_run_2/output/background_rms_vignet{file_number_string}.sqlite. + Values: + NGMIX_RUNNER.BKG_RMS_VIGNET_PATH = + $NGMIX_VIGNET_DIR/vignetmaker_runner_run_2/output/background_rms_vignet{file_number_string}.sqlite. default: rms_vignet_weights options: rms_vignet_weights: @@ -1126,7 +1283,8 @@ analyses: without it the g-prior swamps the PSF likelihood. The recovered PSF shape and size are flat for PSF_NOISE from 1e-4 to 1e-6 on the digital twin. - Values: PSF_NOISE = 1e-5. + Values: + PSF_NOISE = 1e-5. default: psf_noise_1em5 options: psf_noise_1em5: @@ -1168,7 +1326,8 @@ analyses: (symmetrized_4fold_noise). The cost of not symmetrizing, a hole in the galaxy light, is bounded by central_defect_veto. Noise stays the default until a survey A/B against interpolation. - Values: VIGNETMAKER_RUNNER_RUN_1.MASKING = False; + Values: + VIGNETMAKER_RUNNER_RUN_1.MASKING = False; VIGNETMAKER_RUNNER_RUN_2.MASKING = False. default: noise options: @@ -1743,7 +1902,8 @@ analyses: and 0.01, are bypassed); the committed config sets neither key, so enabling classification alone raises. Unlike Guinot+22 (guinot22_spread_model_cut), it applies no s > 0 or magnitude cut. - Values: MAKE_CAT_RUNNER.SM_DO_CLASSIFICATION = False. + Values: + MAKE_CAT_RUNNER.SM_DO_CLASSIFICATION = False. default: deferred_downstream options: deferred_downstream: From f293ccd46686ee1d8099e748c57ecba773585394 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 06:44:45 +0200 Subject: [PATCH 36/40] docs(astra): record tile/exposure header evidence and the pixel-scale history A sampled tile and exposure carry SATURATE, so the SExtractor fallback level never applies; the exposure's FSCALE matches its PHOTZP against the tiles' zero-point 30. The fit_priors lint now names what #858 settled (the WCS is the source of truth) and what still overrides it. Co-Authored-By: Claude Opus 5.5 --- astra.yaml | 21 +++++++++++++-------- 1 file changed, 13 insertions(+), 8 deletions(-) diff --git a/astra.yaml b/astra.yaml index d7f00100c..7f0e15761 100644 --- a/astra.yaml +++ b/astra.yaml @@ -140,7 +140,8 @@ decisions: Tiles use a fixed zero-point of 30 (ZP_FROM_HEADER=False), repeated in ngmix's MAG_ZP; exposures read the per-image header PHOTZP. The fixed value assumes the MegaPipe stacks are calibrated to 30; nothing in the - repo checks it. Exposure epochs are rescaled by FSCALE, which must agree + repo checks it, but on a sampled exposure (2114045p) FSCALE equals + 10^(-0.4 (PHOTZP - 30)) to 0.02%. Exposure epochs are rescaled by FSCALE, which must agree with header PHOTZP as well as tile MAG_ZEROPOINT and ngmix MAG_ZP. Magnitude cuts (the star-selection window, downstream galaxy cuts) inherit their stage's convention. @@ -517,9 +518,10 @@ analyses: which bright stars leave the PSF sample; tile FLAGS reach the catalogue for downstream cuts. Where the card is absent, SExtractor falls back to its built-in SATUR_LEVEL (50000 ADU per its - documentation), since neither .sex file sets one. Whether the - delivered exposure CCDs and MegaPipe tiles carry SATURATE is - unverified. Changing the card or pinning a fixed level requires + documentation), since neither .sex file sets one. Delivered data + carry the card: a sampled tile (CFIS.186.307) has SATURATE 9558.7 + and every CCD of a sampled exposure (2114045p) 65535, and split_exp + keeps each CCD's header. Changing the card or pinning a fixed level requires checking the resulting bright-star selection. Values: default_exp.sex#SATUR_KEY = SATURATE; @@ -1173,10 +1175,13 @@ analyses: fitter is built with the same joint prior. Prior width drives noise bias; no rationale recorded. Guinot+22 used a wider flux prior and a flat prior on r50 rather than T. [LINT] the epoch stamps are - exposure pixels; star-selection cuts and diagnostics use 0.187 - arcsec/px while NGMIX_RUNNER.PIXEL_SCALE is 0.186. Reconcile these - exposure-pixel conversions; the ngmix_runner comment's noise-window - claim is stale because get_noise is never called. + exposure pixels and star selection uses 0.187 arcsec/px (MegaCam + native), but NGMIX_RUNNER.PIXEL_SCALE pins 0.186 (near the coadd + scale), overriding the WCS-derived value #858 made the default; + without the key, pixel_scale_from_wcs reads the first exposure + CCD's WCS (its docstring says the tile's). The ngmix_runner + comment's noise-window claim is stale: get_noise is never + called. Values: get_prior.T_range = -1,1e3; get_prior.F_range = -100,1e9; From 1fb32631f736ecca420c4e9ab3591c02de467c3b Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 07:05:44 +0200 Subject: [PATCH 37/40] test: run committed SExtractor configs with real tools --- tests/unit/test_committed_configs_run.py | 120 +++++++++++++++++++++++ tests/unit/test_decisions.py | 14 +++ workflow/config/cfis/default.conv | 2 +- workflow/config/cfis/gauss_3.0_7x7.conv | 2 +- 4 files changed, 136 insertions(+), 2 deletions(-) create mode 100644 tests/unit/test_committed_configs_run.py diff --git a/tests/unit/test_committed_configs_run.py b/tests/unit/test_committed_configs_run.py new file mode 100644 index 000000000..a20fbfb1f --- /dev/null +++ b/tests/unit/test_committed_configs_run.py @@ -0,0 +1,120 @@ +"""Exercise committed SExtractor/PSFEx configs against the real tools.""" + +from pathlib import Path +import shutil +import subprocess + +import numpy as np +import pytest +from astropy.io import fits +from astropy.wcs import WCS + + +ROOT = Path(__file__).resolve().parents[2] +CONFIG_DIR = ROOT / "workflow" / "config" / "cfis" +SEX = shutil.which("source-extractor") +PSFEX = shutil.which("psfex") + + +def _synthetic_images(directory): + """Write a WCS image with isolated stars and the maps used by both runners.""" + size = 512 + wcs = WCS(naxis=2) + wcs.wcs.crpix = [size / 2, size / 2] + wcs.wcs.cdelt = [-5.16e-5, 5.16e-5] + wcs.wcs.crval = [180.0, 30.0] + wcs.wcs.ctype = ["RA---TAN", "DEC--TAN"] + header = wcs.to_header() + header["GAIN"] = 1.0 + header["SATURATE"] = 60000.0 + header["PHOTZP"] = 30.0 + + rng = np.random.default_rng(42) + yy, xx = np.mgrid[:size, :size] + image = rng.normal(100.0, 3.0, (size, size)).astype(np.float32) + for x in np.linspace(55, size - 55, 4): + for y in np.linspace(55, size - 55, 4): + image += 5000.0 * np.exp( + -((xx - x) ** 2 + (yy - y) ** 2) / 8.0 + ) + + fits.PrimaryHDU(image, header).writeto(directory / "image.fits") + fits.PrimaryHDU(np.ones_like(image)).writeto(directory / "weight.fits") + fits.PrimaryHDU(np.zeros_like(image, dtype=np.int16)).writeto( + directory / "flag.fits" + ) + + +def _run_sextractor(directory, *, tile): + config = CONFIG_DIR / ("default_tile.sex" if tile else "default_exp.sex") + parameters = CONFIG_DIR / ( + "default_noimaflags.param" if tile else "default.param" + ) + convolution = CONFIG_DIR / ( + "gauss_3.0_7x7.conv" if tile else "default.conv" + ) + catalogue = directory / ("tile.cat" if tile else "exposure.cat") + command = [ + SEX, + "image.fits", + "-c", str(config), + "-PARAMETERS_NAME", str(parameters), + "-FILTER_NAME", str(convolution), + "-CATALOG_NAME", str(catalogue), + "-WEIGHT_IMAGE", "weight.fits", + "-FLAG_IMAGE", "NONE" if tile else "flag.fits", + "-VERBOSE_TYPE", "QUIET", + "-WRITE_XML", "N", + ] + result = subprocess.run( + command, + cwd=directory, + check=False, + capture_output=True, + text=True, + timeout=8, + ) + assert result.returncode == 0, result.stdout + result.stderr + with fits.open(catalogue) as hdus: + objects = next(hdu for hdu in hdus if hdu.name == "LDAC_OBJECTS") + assert objects.data is not None and len(objects.data) > 0 + return catalogue + + +def test_committed_convolution_filters_start_with_sextractor_directive(): + """SExtractor requires CONV on line 1 of every convolution file.""" + conv_files = sorted(CONFIG_DIR.rglob("*.conv")) + assert conv_files + for path in conv_files: + assert path.read_text(encoding="utf-8").splitlines()[0].startswith("CONV "), path + + +@pytest.mark.skipif(not (SEX and PSFEX), reason="source-extractor and psfex are required") +def test_committed_sextractor_and_psfex_configs_run(tmp_path): + """Both detection configs emit catalogues and the exposure catalogue fits a PSF.""" + _synthetic_images(tmp_path) + _run_sextractor(tmp_path, tile=True) + exposure_catalogue = _run_sextractor(tmp_path, tile=False) + + subprocess.run( + [ + PSFEX, + str(exposure_catalogue), + "-c", str(CONFIG_DIR / "default.psfex"), + # Smaller fixture stamp keeps this committed-config smoke test fast. + "-PSF_SIZE", "21,21", + "-PSF_SUFFIX", ".psf", + "-WRITE_XML", "N", + "-CHECKIMAGE_TYPE", "NONE", + "-CHECKPLOT_TYPE", "NONE", + "-VERBOSE_TYPE", "QUIET", + ], + cwd=tmp_path, + check=True, + capture_output=True, + text=True, + timeout=8, + ) + psf_file = exposure_catalogue.with_suffix(".psf") + assert psf_file.is_file() + assert psf_file.stat().st_size > 0 diff --git a/tests/unit/test_decisions.py b/tests/unit/test_decisions.py index cb549c447..ad28b5081 100644 --- a/tests/unit/test_decisions.py +++ b/tests/unit/test_decisions.py @@ -36,6 +36,20 @@ def _config_tag(path, decision="choice", content="THRESH 1\n"): return _write(path, f"# @sc [decision:{decision}]\n{content}") +def test_file_scope_tag_can_follow_a_required_config_header(tmp_path): + config = _write( + tmp_path / "filter.conv", + "CONV NORM\n# @sc [decision:choice,scope:file]\n1\n", + ) + tags, errors = scan_tags(tmp_path) + + assert errors == [] + assert len(tags) == 1 + assert tags[0].site.path == "filter.conv" + assert tags[0].site.scope == "file" + assert (tags[0].site.start, tags[0].site.end) == (1, 3) + + def test_config_tags_govern_paragraph_and_section(tmp_path): config = _write( tmp_path / "settings.ini", diff --git a/workflow/config/cfis/default.conv b/workflow/config/cfis/default.conv index 2c1a71ac6..c4d9756f0 100644 --- a/workflow/config/cfis/default.conv +++ b/workflow/config/cfis/default.conv @@ -1,5 +1,5 @@ -# @sc [decision:detection.detection_threshold_policy,scope:file] CONV NORM +# @sc [decision:detection.detection_threshold_policy,scope:file] # 3x3 ``all-ground'' convolution mask with FWHM = 2 pixels. 1 2 1 2 4 2 diff --git a/workflow/config/cfis/gauss_3.0_7x7.conv b/workflow/config/cfis/gauss_3.0_7x7.conv index 3f12d1eaa..f184ebca5 100644 --- a/workflow/config/cfis/gauss_3.0_7x7.conv +++ b/workflow/config/cfis/gauss_3.0_7x7.conv @@ -1,5 +1,5 @@ -# @sc [decision:detection.detection_threshold_policy,scope:file] CONV NORM +# @sc [decision:detection.detection_threshold_policy,scope:file] # 7x7 convolution mask of a gaussian PSF with FWHM = 3.0 pixels. 0.004963 0.021388 0.051328 0.068707 0.051328 0.021388 0.004963 0.021388 0.092163 0.221178 0.296069 0.221178 0.092163 0.021388 From 6f6a2bcc4a1fe156f63e6154242f5fbc97ae1125 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 07:17:28 +0200 Subject: [PATCH 38/40] docs(astra): correct pipeline decision rationale --- astra.yaml | 123 +++++++++++++++++++++++++++-------------------------- 1 file changed, 63 insertions(+), 60 deletions(-) diff --git a/astra.yaml b/astra.yaml index 7f0e15761..605eb1f17 100644 --- a/astra.yaml +++ b/astra.yaml @@ -12,8 +12,12 @@ # metadata: `@sc [decision:,label:] `. # * In Python, put tags in a def/class docstring or as a comment immediately # above a module statement. In Snakemake, put a comment immediately above -# the rule/statement. In config, a comment tag governs the following -# paragraph; a section header governs the whole section, and +# the rule/statement. In config, a tag governs the active settings in the +# following paragraph, which ends at the next blank line or `@sc` line. +# Every active key before that boundary is governed. A blank line between a +# tag and its key is an error. To govern a whole section, put the tag +# directly above its `[SECTION]` header; it governs active keys through the +# next section header. A description comment may sit above the tag, and # `scope:file` governs the whole file. # * A rationale may end with `Values: = ; ... .`. Every ref must # resolve to exactly one site tagged for that decision and match its value. @@ -586,12 +590,12 @@ analyses: epoch_membership_ccd_bounds: label: Which exposure CCDs an object belongs to (N_EPOCH) rationale: >- - CCD_SIZE bounds each CCD's usable x range with strict inequalities, - and a WCS inversion failure skips the CCD, lowering N_EPOCH; this - sets how many exposures enter each galaxy's multi-epoch fit. The x - endpoints match MegaCam's raw DATASEC (2048 pixel indices - inclusively), but the strict cut excludes both. The excluded strip is - likely prescan (unverified). + `all_world2pix(..., 0)` returns 0-based pixels, and the strict test + accepts `33 < x < 2080`. In 1-based FITS coordinates this is x=35 + through x=2080 inclusive: it trims DATASEC columns 33 and 34 but + admits its upper endpoint, column 2080. A WCS inversion failure skips + that CCD, lowering N_EPOCH; this sets how many exposures enter each + galaxy's multi-epoch fit. Values: SEXTRACTOR_RUNNER.CCD_SIZE = 33,2080,1,4612; SEXTRACTOR_RUNNER.MAKE_POST_PROCESS = True. @@ -971,10 +975,12 @@ analyses: label: Per-CCD PSF-model quality gate rationale: >- A CCD whose model has ACCEPTED < STAR_THRESH or CHI2 > CHI2_THRESH - is not interpolated, on both the validation and the multi-epoch - pass; STAR_THRESH applies the published 22-star floor to the 80% - training sample. In the science path the CCD's epoch is dropped for - every object on it, and an object left with no epoch has no shape. + is not interpolated on either the validation or multi-epoch pass. + ACCEPTED counts stars that survive PSFEx's own sample cuts; PSFEx + fits the 80% training split, so STAR_THRESH applies the published + 22-star floor to that accepted count. In the science path the CCD's + epoch is dropped for every object on it, and an object left with no + epoch has no shape. The pipeline has no minimum-epoch floor: NGMIX_N_EPOCH records what survived and epoch-count cuts happen downstream. Guinot+22 discards the CCD from PSF estimation rather than gating at interpolation. @@ -992,8 +998,9 @@ analyses: label: ">= 20 stars on the science path" excluded: true excluded_reason: >- - 20 is the pre-split floor; with 80% of stars in the model it gates - below both the published floor and the validation pass. + A 20-star science-path threshold would pass models with 20 or 21 + PSFEx-accepted training stars, below the 22-star floor retained + on the validation path. des_25: label: DES Y3 threshold (25 stars) excluded: true @@ -1171,17 +1178,18 @@ analyses: [HARDCODED] ellipticity GPriorBA with sigma 0.4; flat T and F priors with negative support (the bounds decide which noisy fits survive and which rail); a centroid prior one pixel scale wide - (PIXEL_SCALE, derived from the WCS when the key is absent). The PSF - fitter is built with the same joint prior. Prior width drives noise - bias; no rationale recorded. Guinot+22 used a wider flux prior and a - flat prior on r50 rather than T. [LINT] the epoch stamps are - exposure pixels and star selection uses 0.187 arcsec/px (MegaCam - native), but NGMIX_RUNNER.PIXEL_SCALE pins 0.186 (near the coadd - scale), overriding the WCS-derived value #858 made the default; - without the key, pixel_scale_from_wcs reads the first exposure - CCD's WCS (its docstring says the tile's). The ngmix_runner - comment's noise-window claim is stale: get_noise is never - called. + (PIXEL_SCALE). The PSF fitter is built with the same joint prior. + Prior width drives noise bias; no rationale recorded. Guinot+22 used + a wider flux prior and a flat prior on r50 rather than T. [LINT] star + selection uses 0.187 arcsec/px, while NGMIX_RUNNER.PIXEL_SCALE pins + 0.186. If the key is absent, pixel_scale_from_wcs raises + AttributeError: it calls `.values()` on the first merged-header + value, but merge_headers inserts TILE_ID as a string before the + exposure entries, which split_exp stores as object ndarrays. The + fallback expects nested mappings; its docstring says it reads the + tile WCS. The key cannot be dropped until that fallback is fixed. + The ngmix_runner comment's noise-window claim is stale: get_noise is + never called. Values: get_prior.T_range = -1,1e3; get_prior.F_range = -100,1e9; @@ -1316,18 +1324,17 @@ analyses: weights through, so a zero-weight pixel's content spreads into the weighted pixels within about a PSF width. DES's ngmixer fills for that reason: "it may be important for codes that take moments or use - FFTs". On develop the fill rides on BLEND_HANDLING, which the - committed config leaves at noisefill: defect pixels get independent - noise at the per-pixel RMS; under uberseg the fill is skipped and raw - defects enter metacal. [LINT] the prepare_ngmix_weights docstring - says noisefill keeps the weight of filled pixels (the code zeroes - it), and the ngmix_runner comment says noisefill fills neighbour - pixels (it fills flagged pixels and leaves neighbours untouched). On - feat/defect-fill-veto every defect pixel is zero-weighted and filled - before metacal under every BLEND_HANDLING; uberseg only removes - weight from neighbour-side pixels. The fill uses the unsymmetrized - defect set: DES symmetrized its masks, but four-fold symmetrization - quadruples m and still leaves an additive c1 + FFTs". In the committed default (`BLEND_HANDLING = noisefill`), + masked pixels are replaced with independent noise at the per-pixel + background RMS when supplied, or the stamp's robust noise scale + otherwise. With `BLEND_HANDLING = uberseg`, the image is left + untouched and weights are zeroed on defects and neighbour-side + pixels. [LINT] the prepare_ngmix_weights docstring says noisefill + keeps the weight of filled pixels (the code zeroes it), and the + ngmix_runner comment says noisefill fills neighbour pixels (it fills + flagged pixels and leaves neighbours untouched). The committed fill + uses the unsymmetrized defect set: DES symmetrized its masks, but + four-fold symmetrization quadruples m and still leaves an additive c1 (symmetrized_4fold_noise). The cost of not symmetrizing, a hole in the galaxy light, is bounded by central_defect_veto. Noise stays the default until a survey A/B against interpolation. @@ -1339,41 +1346,37 @@ analyses: noise: label: Independent noise on the unsymmetrized defect set description: >- - Defect pixels get independent noise at the per-pixel background - RMS and keep weight 0, consistent with metacal's fixnoise noise - image, which covers every pixel. On the fill branches this is - DEFECT_FILL = noise. + Masked pixels get independent noise using the per-pixel + background RMS when supplied, or the stamp's robust noise scale + otherwise; their inverse-variance weights stay zero. Metacal's + fixnoise noise image covers every pixel. insights: [mask_metacal_acts_on_whole_stamp, mask_bad_column_symmetrize] interpolate: label: Interpolate short bounded runs; noise-fill the rest description: >- - On feat/defect-interpolation (a stacked draft), not develop; - DEFECT_FILL = interpolate. Only short bounded runs (at most 3 px - across, clean on both sides) are interpolated, Clough-Tocher from - clean pixels within 4 px, averaged over the four quarter turns - and shared with the fixnoise image; everything else is - noise-filled. Weights are also zeroed on the quarter-turn orbit - of each interpolated pixel while those pixels keep their light, - which cancels the weight term: c1 goes from -1.3e-3 to 2e-6 for a - column 8 px from a 0.5 arcsec galaxy. The fill itself is never - symmetrized: that gives m = +0.89% against +0.19% for a 3-px - bleed at 6 px. Measured on feat/defect-interpolation. + Not implemented. PR #916 measures c1 = -1.3e-3 for a column + 8 px from a 0.5 arcsec galaxy and +2e-6 when the weight is also + zeroed on its quarter-turn orbit. For a 3-px bleed 6 px from that + galaxy, PR #916 measures m11 = +0.89% when the fill is + symmetrized, against +0.19% otherwise. insights: [mask_interpolate_with_noise, mask_sharp_edges_ring] symmetrized_4fold_noise: label: Four-fold-symmetrized defect set (M | rot90 | rot180 | rot270), then noise fill + description: Not implemented. excluded: true excluded_reason: >- - Measured on feat/defect-fill-veto, it quadruples m (-2.7% against - -0.64% for a 3-px bleed at 10 px, half-light radius 0.5 arcsec, - PSF 0.7 arcsec) and still leaves c1 = 7.3e-4 through an - elliptical PSF. + PR #915 measures m11 = -2.7% with four-fold symmetrization + against -0.64% without it for a 3-px bleed 10 px from a galaxy + with 0.5 arcsec half-light radius and a 0.7 arcsec PSF. Through + a PSF with ellipticity (0.05, 0.02), it measures c1 = 7.3e-4 + with symmetrization against 0.7e-4 without it. insights: [mask_bad_column_symmetrize, mask_des_defect_practice, mask_fixed_orientation] raw: - label: "No fill: raw defect values (develop under BLEND_HANDLING = uberseg)" + label: "No fill: raw defect values (BLEND_HANDLING = uberseg)" description: >- - Defect pixels keep weight 0, but their raw values (bad columns, - saturation, bleeds, cosmic rays, bright-star light) stay in the - image that metacal deconvolves, shears and reconvolves. + With BLEND_HANDLING = uberseg, defect pixels keep weight 0 but + their image values stay untouched in the image that metacal + deconvolves, shears and reconvolves. excluded: true excluded_reason: >- Metacal acts on every pixel regardless of weight, so raw defects From c82a8d727aa3d7a1daa2f3ed5223410f2e07a7f2 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 07:32:19 +0200 Subject: [PATCH 39/40] style(config): remove redundant migration blank lines --- workflow/config/cfis/config_MCCD.ini | 4 ---- workflow/config/cfis/config_exp_Sp.ini | 1 - workflow/config/cfis/config_exp_mccd.ini | 4 ---- workflow/config/cfis/config_exp_psfex.ini | 13 ------------- workflow/config/cfis/config_tile_Fe.ini | 2 -- workflow/config/cfis/config_tile_Mc.ini | 1 - workflow/config/cfis/config_tile_Ng_template.ini | 3 --- workflow/config/cfis/config_tile_PiViVi_mccd.ini | 1 - workflow/config/cfis/config_tile_PiViVi_psfex.ini | 11 ----------- workflow/config/cfis/config_tile_Sx.ini | 9 --------- workflow/config/cfis/default.psfex | 9 --------- workflow/config/cfis/default_exp.sex | 12 ------------ workflow/config/cfis/default_tile.sex | 11 ----------- workflow/config/cfis/final_cat.param | 3 --- workflow/config/cfis/star_selection.setools | 6 ------ 15 files changed, 90 deletions(-) diff --git a/workflow/config/cfis/config_MCCD.ini b/workflow/config/cfis/config_MCCD.ini index 8560ea34b..a195d5241 100644 --- a/workflow/config/cfis/config_MCCD.ini +++ b/workflow/config/cfis/config_MCCD.ini @@ -6,7 +6,6 @@ PREPROCESSED_OUTPUT_DIR = ./output OUTPUT_DIR = ./output INPUT_REGEX_FILE_PATTERN = star_split_ratio_80-*-*.fits INPUT_SEPARATOR = - - # @sc [decision:star_selection_psf.psf_modelling_software] MIN_N_STARS = 20 @@ -14,7 +13,6 @@ OUTLIER_STD_MAX = 100. USE_SNR_WEIGHTS = False [INSTANCE] - # @sc [decision:star_selection_psf.psf_modelling_software] N_COMP_LOC = 8 D_COMP_GLOB = 8 @@ -24,12 +22,10 @@ KSIG_GLOB = 0.00 FILTER_PATH = None D_HYB_LOC = 2 MIN_D_COMP_GLOB = None - # @sc [decision:star_selection_psf.psf_modelling_software] RMSE_THRESH = 1.25 CCD_STAR_THRESH = 0.15 - # @sc [decision:star_selection_psf.psf_modelling_software] FP_GEOMETRY = CFIS diff --git a/workflow/config/cfis/config_exp_Sp.ini b/workflow/config/cfis/config_exp_Sp.ini index 58d03c001..0b24a9562 100644 --- a/workflow/config/cfis/config_exp_Sp.ini +++ b/workflow/config/cfis/config_exp_Sp.ini @@ -75,6 +75,5 @@ NUMBERING_SCHEME = -0000000 OUTPUT_SUFFIX = image, weight, flag # Number of HDUs/CCDs of mosaic - # @sc [decision:preparation.ccd_split_extent] N_HDU = 40 diff --git a/workflow/config/cfis/config_exp_mccd.ini b/workflow/config/cfis/config_exp_mccd.ini index 4c4bd170e..b0aecb8ae 100644 --- a/workflow/config/cfis/config_exp_mccd.ini +++ b/workflow/config/cfis/config_exp_mccd.ini @@ -22,7 +22,6 @@ RUN_DATETIME = False [EXECUTION] # Module name, single string or comma-separated list of valid module runner names - # @sc [decision:star_selection_psf.psf_modelling_software] MODULE = sextractor_runner, mask_query_runner, setools_runner, mccd_preprocessing_runner, mccd_fit_val_runner, @@ -63,12 +62,10 @@ TIMEOUT = 96:00:00 [SEXTRACTOR_RUNNER] # The split CCDs, and nothing else: ShapePipe generates no masks - # @sc [decision:masking.pixel_mask_source,decision:star_selection_psf.psf_modelling_software] INPUT_DIR = $SP_RUN/output/run_sp_exp_Sp/split_exp_runner/output # Read the instrument flag image split_exp wrote per CCD - # @sc [decision:masking.pixel_mask_source,decision:star_selection_psf.psf_modelling_software] FILE_PATTERN = image, weight, flag @@ -90,7 +87,6 @@ DOT_CONV_FILE = $SP_CONFIG/default.conv WEIGHT_IMAGE = True # Use input flag image if True - # @sc [decision:masking.pixel_mask_source,decision:star_selection_psf.psf_modelling_software] FLAG_IMAGE = True diff --git a/workflow/config/cfis/config_exp_psfex.ini b/workflow/config/cfis/config_exp_psfex.ini index 2ea73b604..2c573e652 100644 --- a/workflow/config/cfis/config_exp_psfex.ini +++ b/workflow/config/cfis/config_exp_psfex.ini @@ -22,7 +22,6 @@ RUN_DATETIME = False [EXECUTION] # Module name, single string or comma-separated list of valid module runner names - # @sc [decision:star_selection_psf.psf_modelling_software] MODULE = sextractor_runner, mask_query_runner, setools_runner, psfex_runner, psfex_interp_runner @@ -61,12 +60,10 @@ TIMEOUT = 96:00:00 [SEXTRACTOR_RUNNER] # The split CCDs, and nothing else: ShapePipe generates no masks - # @sc [decision:masking.pixel_mask_source] INPUT_DIR = $SP_RUN/output/run_sp_exp_Sp/split_exp_runner/output # Read the instrument flag image split_exp wrote per CCD - # @sc [decision:masking.pixel_mask_source] FILE_PATTERN = image, weight, flag @@ -82,18 +79,15 @@ EXEC_PATH = source-extractor # SExtractor configuration files DOT_SEX_FILE = $SP_CONFIG/default_exp.sex DOT_PARAM_FILE = $SP_CONFIG//default.param - # @sc [decision:detection.detection_threshold_policy,label:effective-filter] exposure-dot-conv-overrides-filter # DOT_CONV_FILE overrides FILTER_NAME in the exposure runner; keep the effective convolution kernel aligned with the intended PSF-star detection prescription. DOT_CONV_FILE = $SP_CONFIG/default.conv # Use input weight image if True - # @sc [decision:detection.weight_map_usage] WEIGHT_IMAGE = True # Use input flag image if True - # @sc [decision:masking.pixel_mask_source] FLAG_IMAGE = True @@ -109,12 +103,10 @@ DETECTION_IMAGE = False DETECTION_WEIGHT = False # True if photometry zero-point is to be read from exposure image header - # @sc [decision:photometric_zeropoint] ZP_FROM_HEADER = True # If ZP_FROM_HEADER is True, zero-point key name - # @sc [decision:photometric_zeropoint] ZP_KEY = PHOTZP @@ -198,7 +190,6 @@ SETOOLS_CONFIG_PATH = $SP_CONFIG/star_selection.setools [PSFEX_RUNNER] # Use 80% sample for PSF model - # @sc [decision:star_selection_psf.psf_train_validation_split] FILE_PATTERN = star_split_ratio_80 @@ -213,7 +204,6 @@ DOT_PSFEX_FILE = $SP_CONFIG/default.psfex [PSFEX_INTERP_RUNNER] # Use 20% sample for PSF validation - # @sc [decision:star_selection_psf.psf_train_validation_split] FILE_PATTERN = star_split_ratio_80, star_split_ratio_20, psfex_cat @@ -228,7 +218,6 @@ NUMBERING_SCHEME = -0000000-0 MODE = VALIDATION # Column names of position parameters - # @sc [decision:preparation.object_position_columns] POSITION_PARAMS = XWIN_IMAGE,YWIN_IMAGE @@ -236,11 +225,9 @@ POSITION_PARAMS = XWIN_IMAGE,YWIN_IMAGE GET_SHAPES = True # Minimum number of stars per CCD for PSF model to be computed - # @sc [decision:star_selection_psf.psf_acceptance_thresholds] STAR_THRESH = 22 # Maximum chi^2 for PSF model to be computed on CCD - # @sc [decision:star_selection_psf.psf_acceptance_thresholds] CHI2_THRESH = 2 diff --git a/workflow/config/cfis/config_tile_Fe.ini b/workflow/config/cfis/config_tile_Fe.ini index 573113506..e66f407ea 100644 --- a/workflow/config/cfis/config_tile_Fe.ini +++ b/workflow/config/cfis/config_tile_Fe.ini @@ -69,14 +69,12 @@ FILE_EXT = .fits NUMBERING_SCHEME = -000-000 # Column number of exposure name in FITS header - # @sc [decision:preparation.epoch_provenance_from_tile_history] COLNUM = 3 # Prefix to remove from exposure name. CFIS exposure names carry no # prefix -- the trailing "p" is a suffix (epoch letter), kept as part of # the exposure name, not stripped by this key. - # @sc [decision:preparation.epoch_provenance_from_tile_history] EXP_PREFIX = diff --git a/workflow/config/cfis/config_tile_Mc.ini b/workflow/config/cfis/config_tile_Mc.ini index 8e0085129..e45564b21 100644 --- a/workflow/config/cfis/config_tile_Mc.ini +++ b/workflow/config/cfis/config_tile_Mc.ini @@ -68,7 +68,6 @@ FILE_PATTERN = sexcat, galaxy_psf, ngmix # NUMBER_LIST selects this unit; the workflow sets SP_UNIT_NUM to the # dashed tile ID (e.g. -210-282) - NUMBER_LIST = $SP_UNIT_NUM # FILE_EXT (optional) list of string extensions to identify input files diff --git a/workflow/config/cfis/config_tile_Ng_template.ini b/workflow/config/cfis/config_tile_Ng_template.ini index 7e5df62e4..8ec39f5f1 100644 --- a/workflow/config/cfis/config_tile_Ng_template.ini +++ b/workflow/config/cfis/config_tile_Ng_template.ini @@ -101,7 +101,6 @@ NUMBERING_SCHEME = -000-000 # 1/RMS^2 inverse-variance ngmix weights. When set, the file must exist for # every tile (missing file -> error, no per-tile fallback); omit the option # entirely to fall back to the scalar sigma_mad noise estimate. - # @sc [decision:shape_measurement.galaxy_pixel_weights] BKG_RMS_VIGNET_PATH = $NGMIX_VIGNET_DIR/vignetmaker_runner_run_2/output/background_rms_vignet{file_number_string}.sqlite @@ -114,12 +113,10 @@ BKG_RMS_VIGNET_PATH = $NGMIX_VIGNET_DIR/vignetmaker_runner_run_2/output/backgrou SAVE_BATCH = 250 # Magnitude zero-point - # @sc [decision:photometric_zeropoint] MAG_ZP = 30.0 # Pixel scale in arcsec - # @sc [decision:shape_measurement.fit_priors] PIXEL_SCALE = 0.186 diff --git a/workflow/config/cfis/config_tile_PiViVi_mccd.ini b/workflow/config/cfis/config_tile_PiViVi_mccd.ini index 26144df47..864f790f9 100644 --- a/workflow/config/cfis/config_tile_PiViVi_mccd.ini +++ b/workflow/config/cfis/config_tile_PiViVi_mccd.ini @@ -16,7 +16,6 @@ RUN_DATETIME = False ## ShapePipe execution options - [EXECUTION] # Module name, single string or comma-separated list of valid module runner names diff --git a/workflow/config/cfis/config_tile_PiViVi_psfex.ini b/workflow/config/cfis/config_tile_PiViVi_psfex.ini index 5811aa18d..62073d482 100644 --- a/workflow/config/cfis/config_tile_PiViVi_psfex.ini +++ b/workflow/config/cfis/config_tile_PiViVi_psfex.ini @@ -16,7 +16,6 @@ RUN_DATETIME = False ## ShapePipe execution options - [EXECUTION] # Module name, single string or comma-separated list of valid module runner names @@ -95,7 +94,6 @@ NUMBERING_SCHEME = -000-000 MODE = MULTI-EPOCH # Column names of position parameters - # @sc [decision:preparation.object_position_columns] POSITION_PARAMS = XWIN_WORLD,YWIN_WORLD @@ -103,12 +101,10 @@ POSITION_PARAMS = XWIN_WORLD,YWIN_WORLD GET_SHAPES = True # Number of stars threshold - # @sc [decision:star_selection_psf.psf_acceptance_thresholds] STAR_THRESH = 22 # chi^2 threshold - # @sc [decision:star_selection_psf.psf_acceptance_thresholds] CHI2_THRESH = 2 @@ -120,7 +116,6 @@ CHI2_THRESH = 2 ME_DOT_PSF_EXP_DIR = $SP_EXP # Input psf file pattern - # @sc [decision:star_selection_psf.psf_train_validation_split] ME_DOT_PSF_PATTERN = star_split_ratio_80 @@ -149,13 +144,11 @@ MASK_VALUE = 0 MODE = CLASSIC # Coordinate frame type, one in PIX (pixel frame), SPHE (spherical coordinates) - # @sc [decision:preparation.object_position_columns] COORD = PIX POSITION_PARAMS = XWIN_IMAGE,YWIN_IMAGE # Vignet size in pixels - # @sc [decision:postage_stamp_size] STAMP_SIZE = 51 @@ -189,13 +182,11 @@ MASK_VALUE = 0 MODE = MULTI-EPOCH # Coordinate frame type, one in PIX (pixel frame), SPHE (spherical coordinates) - # @sc [decision:preparation.object_position_columns] COORD = SPHE POSITION_PARAMS = XWIN_WORLD,YWIN_WORLD # Vignet size in pixels - # @sc [decision:postage_stamp_size] STAMP_SIZE = 51 @@ -206,10 +197,8 @@ PREFIX = # run outputs. ME_IMAGE_EXP_DIR/ME_IMAGE_EXP_RUNNERS replace ME_IMAGE_DIR for # the v2.0 per-exposure pipeline; output dirs are discovered by scanning $SP_EXP. ME_IMAGE_EXP_DIR = $SP_EXP - # @sc [decision:masking.pixel_mask_source] ME_IMAGE_EXP_RUNNERS = split_exp_runner, split_exp_runner, split_exp_runner, sextractor_runner, sextractor_runner - # @sc [decision:masking.pixel_mask_source,label:coupling] flag-stamp-runner-order # ME_IMAGE_PATTERN and ME_IMAGE_EXP_RUNNERS are parallel ordered lists; keep flag, image, weight, background and background-RMS entries paired. ME_IMAGE_PATTERN = flag, image, weight, background, background_rms diff --git a/workflow/config/cfis/config_tile_Sx.ini b/workflow/config/cfis/config_tile_Sx.ini index 4cf14ce8d..0c05c38b5 100644 --- a/workflow/config/cfis/config_tile_Sx.ini +++ b/workflow/config/cfis/config_tile_Sx.ini @@ -73,22 +73,18 @@ EXEC_PATH = source-extractor # SExtractor configuration files DOT_SEX_FILE = $SP_CONFIG/default_tile.sex - # @sc [decision:detection.detection_source_mode,label:override] tile-parameter-file-overrides-column-list # DOT_PARAM_FILE overrides PARAMETERS_NAME; keep it pointed at default_noimaflags.param so tile detections do not request IMAFLAGS_ISO. DOT_PARAM_FILE = $SP_CONFIG/default_noimaflags.param - # @sc [decision:detection.detection_threshold_policy,label:effective-filter] tile-dot-conv-overrides-filter # DOT_CONV_FILE overrides FILTER_NAME in the tile runner; keep the effective convolution kernel aligned with the intended detection prescription. DOT_CONV_FILE = $SP_CONFIG/gauss_3.0_7x7.conv # Use input weight image if True - # @sc [decision:detection.weight_map_usage] WEIGHT_IMAGE = True # Use input flag image if True - # @sc [decision:detection.detection_source_mode] FLAG_IMAGE = False @@ -97,7 +93,6 @@ PSF_FILE = False # Use distinct image for detection (SExtractor in # dual-image mode) if True - # @sc [decision:detection.detection_source_mode] DETECTION_IMAGE = False @@ -107,7 +102,6 @@ DETECTION_WEIGHT = False # @sc [decision:photometric_zeropoint] ZP_FROM_HEADER = False - # @sc [decision:detection.background_model] BKG_FROM_HEADER = False @@ -123,16 +117,13 @@ SUFFIX = sexcat ## Post-processing # Necessary for tiles, to enable multi-exposure processing - # @sc [decision:detection.epoch_membership_ccd_bounds] MAKE_POST_PROCESS = True # World coordinate keywords, SExtractor output. Format: KEY_X,KEY_Y - # @sc [decision:preparation.object_position_columns] WORLD_POSITION = XWIN_WORLD,YWIN_WORLD # Number of pixels in x,y of a CCD. Format: Nx,Ny - # @sc [decision:detection.epoch_membership_ccd_bounds] CCD_SIZE = 33,2080,1,4612 diff --git a/workflow/config/cfis/default.psfex b/workflow/config/cfis/default.psfex index 9797709b6..840f1ddfd 100644 --- a/workflow/config/cfis/default.psfex +++ b/workflow/config/cfis/default.psfex @@ -13,21 +13,16 @@ BASIS_SCALE 1.0 # Gauss-Laguerre beta parameter NEWBASIS_TYPE NONE # Create new basis: NONE, PCA_INDEPENDENT # or PCA_COMMON NEWBASIS_NUMBER 8 # Number of new basis vectors - # @sc [decision:star_selection_psf.psf_model_complexity] PSF_SAMPLING 1. # Sampling step in pixel units (0.0 = auto) PSF_PIXELSIZE 1.0 # Effective pixel size in pixel step units - # @sc [decision:star_selection_psf.psf_model_complexity] PSF_ACCURACY 0.01 # Accuracy to expect from PSF "pixel" values - # @sc [decision:postage_stamp_size] PSF_SIZE 51,51 # Image size of the PSF model - # @sc [decision:star_selection_psf.psfex_candidate_vetting] PSF_RECENTER N # Allow recentering of PSF-candidates Y/N ? - # @sc [decision:star_selection_psf.psf_model_complexity] MEF_TYPE INDEPENDENT # INDEPENDENT or COMMON @@ -35,10 +30,8 @@ MEF_TYPE INDEPENDENT # INDEPENDENT or COMMON # @sc [decision:preparation.object_position_columns] CENTER_KEYS XWIN_IMAGE,YWIN_IMAGE # Catalogue parameters for source pre-centering - # @sc [decision:detection.photometry_parameters] PHOTFLUX_KEY FLUX_AUTO # Catalogue parameter for photometric norm. - # @sc [decision:detection.photometry_parameters] PHOTFLUXERR_KEY FLUXERR_AUTO # Catalogue parameter for photometric error @@ -46,10 +39,8 @@ PHOTFLUXERR_KEY FLUXERR_AUTO # Catalogue parameter for photometric error # @sc [decision:preparation.object_position_columns,decision:star_selection_psf.psf_model_complexity] PSFVAR_KEYS XWIN_IMAGE,YWIN_IMAGE # Catalogue or FITS (preceded by :) params - # @sc [decision:star_selection_psf.psf_model_complexity] PSFVAR_GROUPS 1,1 # Group tag for each context key - # @sc [decision:star_selection_psf.psf_model_complexity] PSFVAR_DEGREES 2 # Polynom degree for each group diff --git a/workflow/config/cfis/default_exp.sex b/workflow/config/cfis/default_exp.sex index ed8b8b928..e68343b30 100644 --- a/workflow/config/cfis/default_exp.sex +++ b/workflow/config/cfis/default_exp.sex @@ -11,17 +11,13 @@ PARAMETERS_NAME default.param #------------------------------- Extraction ---------------------------------- DETECT_TYPE CCD # CCD (linear) or PHOTO (with gamma correction) - # @sc [decision:detection.detection_threshold_policy] DETECT_MINAREA 5 # min. # of pixels above threshold DETECT_MAXAREA 0 # max. # of pixels above threshold (0=unlimited) - # @sc [decision:detection.detection_threshold_policy] THRESH_TYPE RELATIVE # threshold type: RELATIVE (in sigmas) - # or ABSOLUTE (in ADUs) - # @sc [decision:detection.detection_threshold_policy] DETECT_THRESH 1.5 # or , in mag.arcsec-2 ANALYSIS_THRESH 1.5 # or , in mag.arcsec-2 @@ -30,7 +26,6 @@ ANALYSIS_THRESH 1.5 # or , in mag.arcsec-2 FILTER Y # apply filter for detection (Y or N)? FILTER_NAME default.conv - FILTER_THRESH # Threshold[s] for retina filtering # @sc [decision:detection.deblending_policy] @@ -43,7 +38,6 @@ CLEAN_PARAM 1.0 # Cleaning efficiency # @sc [decision:detection.blend_photometry_mask_type] MASK_TYPE CORRECT # type of detection MASKing: can be one of - # NONE, BLANK or CORRECT #-------------------------------- WEIGHTing ---------------------------------- @@ -54,7 +48,6 @@ WEIGHT_TYPE MAP_WEIGHT # type of WEIGHTing: NONE, BACKGROUND, RESCALE_WEIGHTS Y # Rescale input weights/variances (Y/N)? WEIGHT_IMAGE weight.fits # weight-map filename - # @sc [decision:detection.weight_map_usage] WEIGHT_GAIN Y # modulate gain (E/ADU) with weights? (Y/N) @@ -76,7 +69,6 @@ PHOT_PETROPARAMS 2.0, 3.5 # MAG_PETRO parameters: , # PHOT_AUTOAPERS 0.0,0.0 # , minimum apertures # for MAG_AUTO and MAG_PETRO - # @sc [decision:detection.photometry_parameters] PHOT_FLUXFRAC 0.5 # flux fraction[s] used for FLUX_RADIUS @@ -87,13 +79,11 @@ MAG_ZEROPOINT 30.0 # magnitude zero-point MAG_GAMMA 4.0 # gamma of emulsion (for photographic scans) GAIN_KEY GAIN # keyword for detector gain in e-/ADU - PIXEL_SCALE 0. # size of pixel in arcsec (0=use FITS WCS info) #------------------------- Star/Galaxy Separation ---------------------------- SEEING_FWHM 0.6 # stellar FWHM in arcsec - STARNNW_NAME default.nnw #------------------------------ Background ----------------------------------- @@ -102,7 +92,6 @@ STARNNW_NAME default.nnw BACK_TYPE AUTO # AUTO or MANUAL BACK_VALUE 0.0 # Default background value in MANUAL mode - # @sc [decision:detection.background_model] BACK_SIZE 64 # Background mesh: or , BACK_FILTERSIZE 3 # Background filter: or , @@ -153,7 +142,6 @@ WRITE_XML N # Write XML file (Y/N)? NTHREADS 1 # 1 single thread FITS_UNSIGNED N # Treat FITS integer values as unsigned (Y/N)? - # @sc [decision:detection.zero_weight_interpolation] INTERP_MAXXLAG 16 # Max. lag along X for 0-weight interpolation INTERP_MAXYLAG 16 # Max. lag along Y for 0-weight interpolation diff --git a/workflow/config/cfis/default_tile.sex b/workflow/config/cfis/default_tile.sex index 1fccf449a..ca7367c3e 100644 --- a/workflow/config/cfis/default_tile.sex +++ b/workflow/config/cfis/default_tile.sex @@ -11,17 +11,13 @@ PARAMETERS_NAME default.param #------------------------------- Extraction ---------------------------------- DETECT_TYPE CCD # CCD (linear) or PHOTO (with gamma correction) - # @sc [decision:detection.detection_threshold_policy] DETECT_MINAREA 3 # min. # of pixels above threshold DETECT_MAXAREA 0 # max. # of pixels above threshold (0=unlimited) - # @sc [decision:detection.detection_threshold_policy] THRESH_TYPE RELATIVE # threshold type: RELATIVE (in sigmas) - # or ABSOLUTE (in ADUs) - # @sc [decision:detection.detection_threshold_policy] DETECT_THRESH 1.0 # or , in mag.arcsec-2 ANALYSIS_THRESH 1.0 # or , in mag.arcsec-2 @@ -30,7 +26,6 @@ ANALYSIS_THRESH 1.0 # or , in mag.arcsec-2 FILTER Y # apply filter for detection (Y or N)? FILTER_NAME gauss_3.0_7x7.conv - FILTER_THRESH # Threshold[s] for retina filtering # @sc [decision:detection.deblending_policy] @@ -43,7 +38,6 @@ CLEAN_PARAM 1.0 # Cleaning efficiency # @sc [decision:detection.blend_photometry_mask_type] MASK_TYPE CORRECT # type of detection MASKing: can be one of - # NONE, BLANK or CORRECT #-------------------------------- WEIGHTing ---------------------------------- @@ -54,7 +48,6 @@ WEIGHT_TYPE MAP_WEIGHT # type of WEIGHTing: NONE, BACKGROUND, RESCALE_WEIGHTS Y # Rescale input weights/variances (Y/N)? WEIGHT_IMAGE weight.fits # weight-map filename - # @sc [decision:detection.weight_map_usage] WEIGHT_GAIN Y # modulate gain (E/ADU) with weights? (Y/N) @@ -76,7 +69,6 @@ PHOT_PETROPARAMS 2.0, 3.5 # MAG_PETRO parameters: , # PHOT_AUTOAPERS 0.0,0.0 # , minimum apertures # for MAG_AUTO and MAG_PETRO - # @sc [decision:detection.photometry_parameters] PHOT_FLUXFRAC 0.5 # flux fraction[s] used for FLUX_RADIUS @@ -94,7 +86,6 @@ PIXEL_SCALE 0. # size of pixel in arcsec (0=use FITS WCS info) #------------------------- Star/Galaxy Separation ---------------------------- SEEING_FWHM 0.6 # stellar FWHM in arcsec - STARNNW_NAME default.nnw #------------------------------ Background ----------------------------------- @@ -103,7 +94,6 @@ STARNNW_NAME default.nnw BACK_TYPE AUTO # AUTO or MANUAL BACK_VALUE 0.0 # Default background value in MANUAL mode - # @sc [decision:detection.background_model] BACK_SIZE 512 # Background mesh: or , BACK_FILTERSIZE 9 # Background filter: or , @@ -154,7 +144,6 @@ WRITE_XML N # Write XML file (Y/N)? NTHREADS 1 # 1 single thread FITS_UNSIGNED N # Treat FITS integer values as unsigned (Y/N)? - # @sc [decision:detection.zero_weight_interpolation] INTERP_MAXXLAG 16 # Max. lag along X for 0-weight interpolation INTERP_MAXYLAG 16 # Max. lag along Y for 0-weight interpolation diff --git a/workflow/config/cfis/final_cat.param b/workflow/config/cfis/final_cat.param index 550daaf2f..6ebd1aae9 100644 --- a/workflow/config/cfis/final_cat.param +++ b/workflow/config/cfis/final_cat.param @@ -1,18 +1,15 @@ # coordinates - # @sc [decision:preparation.object_position_columns] XWIN_WORLD YWIN_WORLD # tile ID, for plot of tile-dependent additive bias. # Can maybe be removed. - # @sc [decision:catalogue_assembly.tile_overlap_handling] TILE_ID # flags FLAGS - # @sc [decision:detection.detection_source_mode] IMAFLAGS_ISO diff --git a/workflow/config/cfis/star_selection.setools b/workflow/config/cfis/star_selection.setools index 2ab3bff8e..a237ca89a 100644 --- a/workflow/config/cfis/star_selection.setools +++ b/workflow/config/cfis/star_selection.setools @@ -35,12 +35,10 @@ MAG_AUTO > 0 MAG_AUTO < 21 FWHM_IMAGE > 0.3 / 0.187 FWHM_IMAGE < 1.5 / 0.187 - # @sc [decision:detection.saturation_level] FLAGS == 0 IMAFLAGS_ISO == 0 - NO_SAVE # @sc [decision:masking.psf_star_mask_veto] @@ -61,10 +59,8 @@ MAG_AUTO > 18. MAG_AUTO < 22. FWHM_IMAGE <= mode(FWHM_IMAGE{preselect}) + 0.2 FWHM_IMAGE >= mode(FWHM_IMAGE{preselect}) - 0.2 - # @sc [decision:detection.saturation_level] FLAGS == 0 - # @sc [decision:masking.pixel_mask_source] IMAFLAGS_ISO == 0 @@ -79,7 +75,6 @@ NO_SAVE # Split the 'star_selection' sample into # two random sub-samples with ratio 80/20 [RAND_SPLIT:star_split] - # @sc [decision:star_selection_psf.psf_train_validation_split] RATIO = 20 @@ -119,7 +114,6 @@ TYPE = scatter FORMAT = png X = X_IMAGE{star_selection} Y = Y_IMAGE{star_selection} - # @sc [decision:star_selection_psf.star_selection_box,label:diagnostic] stellar-diagnostic-matches-cut # FWHM_FIELD plot labels and summary statistics must describe the size cut actually applied. SCATTER = FWHM_IMAGE{star_selection}*0.187 From 675b60f9e2b5207f36f435da81e88d4ebdcfb8a8 Mon Sep 17 00:00:00 2001 From: Cail Daley Date: Mon, 28 Sep 2026 07:44:29 +0200 Subject: [PATCH 40/40] fix(decisions): match pipeline config semantics --- src/shapepipe/modules/ngmix_package/ngmix.py | 8 ++- tests/helpers/decisions.py | 24 ++++--- tests/unit/test_decisions.py | 69 ++++++++++++++++++-- 3 files changed, 84 insertions(+), 17 deletions(-) diff --git a/src/shapepipe/modules/ngmix_package/ngmix.py b/src/shapepipe/modules/ngmix_package/ngmix.py index efcde1415..62502c4cd 100644 --- a/src/shapepipe/modules/ngmix_package/ngmix.py +++ b/src/shapepipe/modules/ngmix_package/ngmix.py @@ -104,6 +104,7 @@ def get_prior(pixel_scale, rng, T_range=None, F_range=None): Returns ------- ngmix.joint_prior.PriorSimpleSep + @sc [decision:shape_measurement.fit_priors] """ if T_range is None: @@ -168,6 +169,7 @@ def position_seed(ra, dec, ccd): ------- int Seed in ``[0, 2**32)`` for ``numpy.random.RandomState``. + @sc [decision:shape_measurement.ngmix_seed_mode] """ box_x = int(np.floor((ra * 3600) / 3) + (ccd + 1)) @@ -1157,8 +1159,10 @@ def prepare_postage_stamps( psf_obj=None, gal_obj=None, ): - # define per-object lists of individual exposures to go into ngmix - """@sc [decision:shape_measurement.central_defect_veto,decision:shape_measurement.epoch_masked_fraction_cut]""" + """Prepare the per-object lists of exposures passed to ngmix. + + @sc [decision:shape_measurement.central_defect_veto,decision:shape_measurement.epoch_masked_fraction_cut] + """ stamp = Postage_stamp(bkg_sub=bkg_sub) # Read each store's per-object dict ONCE: every sqlitedict access # unpickles the object's whole all-epoch dict, so keeping these out of diff --git a/tests/helpers/decisions.py b/tests/helpers/decisions.py index 56f61a355..9cd7c34a8 100644 --- a/tests/helpers/decisions.py +++ b/tests/helpers/decisions.py @@ -306,7 +306,9 @@ def _parse_file_tags(path, root): comment_tokens = {} if suffix == ".py": try: - tree = ast.parse(source, filename=relative) + with warnings.catch_warnings(): + warnings.simplefilter("ignore", SyntaxWarning) + tree = ast.parse(source, filename=relative) except SyntaxError as error: return [], [f"{relative}: cannot parse tagged Python: {error}"] found, issues, consumed = _python_docstring_tags(relative, source, tree) @@ -559,7 +561,7 @@ def _ini_parser(text, *, strict=True): parser = configparser.ConfigParser( interpolation=None, strict=strict, allow_no_value=True ) - parser.optionxform = str + parser.optionxform = str.lower parser.read_string(text) return parser @@ -834,6 +836,8 @@ def _config_values_in_site(root, site, selector): lines = text.splitlines() section, key = _split_config_key(selector, suffix, site) + if suffix == ".ini": + key = key.lower() current_section = "DEFAULT" if suffix == ".ini" else "" values, predicates = [], [] for number, raw in enumerate(lines, 1): @@ -850,7 +854,7 @@ def _config_values_in_site(root, site, selector): continue if suffix == ".ini": match = re.match(r"^([^:=\s][^:=]*?)\s*[:=]\s*(.*)$", stripped) - if not match or match.group(1).strip() != key: + if not match or match.group(1).strip().lower() != key: continue value = match.group(2).strip() base_indent = len(raw) - len(raw.lstrip()) @@ -937,6 +941,8 @@ def _active_config_lines(root, path, selector, site=None): file_site = site or Site(path, 1, len(lines), "config", scope="file") section, key = _split_config_key(selector, suffix, file_site) + if suffix == ".ini": + key = key.lower() current_section = "DEFAULT" if suffix == ".ini" else "" found = [] for number, raw in enumerate(lines, 1): @@ -952,7 +958,7 @@ def _active_config_lines(root, path, selector, site=None): continue if suffix == ".ini": match = re.match(r"^([^:=\s][^:=]*?)\s*(?:[:=]\s*(.*))?$", stripped) - active_key = match.group(1).strip() if match else None + active_key = match.group(1).strip().lower() if match else None elif suffix == ".setools": match = re.match(rf"^{re.escape(key)}(?=$|\s|=|<|>)(.*)$", stripped) active_key = key if match else None @@ -978,6 +984,7 @@ def _config_absent_actual(root, path, selector): file_site = Site(path, 1, len(text.splitlines()), "config", scope="file") section, key = _split_config_key(selector, suffix, file_site) if suffix == ".ini": + key = key.lower() try: parser = _ini_parser(text, strict=False) except configparser.Error as error: @@ -1085,14 +1092,14 @@ def _python_value_in_site(root, site, selector): try: node = _selected_python_node(tree, full_selector) except ValueError as error: - if not direct: - return None message = str(error) if "needs one binding" in message: found = re.search(r"found (\d+)", message) - if direct and found and int(found.group(1)) > 1: + if found and int(found.group(1)) > 1: raise ValueError(f"ambiguous binding: {message}") from error return None + if not direct: + return None if "missing" in message: return None raise @@ -1204,9 +1211,6 @@ def value_errors(root, record, tags=None): rationale = definition.get("rationale") if isinstance(definition, dict) else None if not isinstance(rationale, str): continue - if "Anchor:" in rationale: - errors.append(f"{decision}: legacy Anchor: sentence remains in rationale") - continue entries, problem = _parse_values(rationale) if problem: errors.append(f"{decision}: {problem}") diff --git a/tests/unit/test_decisions.py b/tests/unit/test_decisions.py index ad28b5081..8c0f39766 100644 --- a/tests/unit/test_decisions.py +++ b/tests/unit/test_decisions.py @@ -321,6 +321,48 @@ def test_ini_duplicate_active_setting_outside_governed_paragraph_fails(tmp_path) assert "duplicate active setting outside the governed paragraph" in problem +def test_ini_value_key_matching_is_case_insensitive(tmp_path): + _write( + tmp_path / "settings.ini", + "[SCIENCE]\n# @sc [decision:choice]\nmask_paths = stars.hsp\n", + ) + record = _record("Mask map. Values: SCIENCE.MASK_PATHS = stars.hsp.") + tags, errors = scan_tags(tmp_path) + + assert errors == [] + assert value_errors(tmp_path, record, tags) == [] + + +def test_ini_absence_detects_lowercase_pipeline_key(tmp_path): + _write( + tmp_path / "settings.ini", + "# @sc [decision:choice]\n[SCIENCE]\nmask_paths = stars.hsp\n", + ) + record = _record("No map. Values: SCIENCE.MASK_PATHS = absent.") + tags, errors = scan_tags(tmp_path) + + assert errors == [] + problem, = value_errors(tmp_path, record, tags) + assert "choice" in problem + assert "MASK_PATHS" in problem + assert "active setting" in problem + + +def test_ini_duplicate_matching_is_case_insensitive(tmp_path): + _write( + tmp_path / "settings.ini", + "[SCIENCE]\n# @sc [decision:choice]\nMASK_PATHS = stars.hsp\n" + "OTHER = 2\n\nmask_paths = other.hsp\n", + ) + record = _record("Mask map. Values: SCIENCE.MASK_PATHS = stars.hsp.") + tags, errors = scan_tags(tmp_path) + + assert errors == [] + problem, = value_errors(tmp_path, record, tags) + assert "choice" in problem + assert "duplicate active setting outside the governed paragraph" in problem + + def test_absent_assertion_fails_when_its_scope_is_removed(tmp_path): config = _write( tmp_path / "settings.ini", @@ -473,7 +515,7 @@ def test_config_ref_ignores_unrelated_python_tagged_site(tmp_path): assert value_errors(tmp_path, record, tags) == [] -def test_python_value_reassignment_fails_closed(tmp_path): +def test_relative_python_reassignment_reports_ambiguous_binding(tmp_path): _write( tmp_path / "constants.py", 'def fit():\n """Fit.\n\n @sc [decision:choice]\n """\n' @@ -483,7 +525,10 @@ def test_python_value_reassignment_fails_closed(tmp_path): tags, errors = scan_tags(tmp_path) assert errors == [] problem, = value_errors(tmp_path, record, tags) - assert "exactly one tagged site" in problem + assert "choice" in problem + assert "ambiguous binding" in problem + assert "found 2" in problem + assert "found 0" not in problem def test_reassigned_module_binding_reports_ambiguity_not_zero_sites(tmp_path): @@ -567,37 +612,47 @@ def test_repository_decisions_have_sites_and_all_values_match(): @pytest.mark.parametrize( - "path, pattern, replacement", + "path, pattern, replacement, decision, ref_fragment", [ ( "workflow/config/cfis/config_tile_Sx.ini", "WEIGHT_IMAGE = True", "WEIGHT_IMAGE = False", + "detection.weight_map_usage", + "WEIGHT_IMAGE", ), ( "workflow/config/cfis/config_tile_Sx.ini", "MAKE_POST_PROCESS = True", "MAKE_POST_PROCESS = False", + "detection.epoch_membership_ccd_bounds", + "MAKE_POST_PROCESS", ), ( "workflow/config/cfis/star_selection.setools", "[MASK:star_selection]\n", "[MASK:star_selection]\nMASK_EXT == 0\n", + "masking.psf_star_mask_veto", + "MASK_EXT", ), ( "workflow/config/cfis/default_tile.sex", "SATUR_KEY", "SATUR_LEVEL 50000\nSATUR_KEY", + "detection.saturation_level", + "SATUR_LEVEL", ), ( "src/shapepipe/modules/ngmix_package/ngmix.py", "\n boot = ngmix.metacal.", "\n metacal_pars['step'] = 0.02\n boot = ngmix.metacal.", + "shape_measurement.metacal_scheme", + "metacal_pars[step]", ), ], ) def test_real_record_catches_gate_and_absence_drift( - tmp_path, path, pattern, replacement + tmp_path, path, pattern, replacement, decision, ref_fragment ): """Real-record mutations of gates and absences must break Values checks.""" @@ -614,7 +669,11 @@ def test_real_record_catches_gate_and_absence_drift( text = target.read_text(encoding="utf-8") assert text.count(pattern) == 1 target.write_text(text.replace(pattern, replacement), encoding="utf-8") - assert value_errors(tmp_path, record) + problems = value_errors(tmp_path, record) + assert any( + decision in problem and ref_fragment in problem + for problem in problems + ), problems def test_preserved_utilities_import_rule(tmp_path):