diff --git a/.devcontainer/Dockerfile b/.devcontainer/Dockerfile index b41a0348df..bcf228f2c3 100644 --- a/.devcontainer/Dockerfile +++ b/.devcontainer/Dockerfile @@ -5,6 +5,9 @@ FROM mcr.microsoft.com/devcontainers/python:3.11 RUN apt-get update && export DEBIAN_FRONTEND=noninteractive && apt-get install -y libgtk-3-dev # Installs poetry and ipython -RUN pip install "poetry>=1.0.0,<2.0" ipython +RUN pip install "poetry>=2.0,<3.0" ipython + +# Create the poetry virtualenv at /workspaces/imap_processing/.venv so VS Code can find it +ENV POETRY_VIRTUALENVS_IN_PROJECT=true WORKDIR /workspaces/imap_processing diff --git a/.devcontainer/devcontainer.json b/.devcontainer/devcontainer.json index 7ac118477f..33e88d9aee 100644 --- a/.devcontainer/devcontainer.json +++ b/.devcontainer/devcontainer.json @@ -6,25 +6,23 @@ "features": { "ghcr.io/devcontainers/features/desktop-lite:1": { "version": "latest" - }, "ghcr.io/devcontainers/features/docker-in-docker:2": { - "version": "latest" + }, "ghcr.io/devcontainers/features/docker-in-docker:4.0.0": { + "version": "latest", + "moby": false } }, "forwardPorts": [6080], - "postCreateCommand": "poetry install", - "postStartCommand": ". $(poetry env info --path)/bin/activate", + "postCreateCommand": "poetry install --all-extras", "customizations": { "vscode": { "extensions": [ - "ms-python.python" + "ms-python.python", + "Anthropic.claude-code" ], // Set *default* container specific settings.json values on container create. "settings": { - // not using venvs? uncomment this - "python.defaultInterpreterPath": "/usr/local/bin/python" - - // use the active venv - // "python.defaultInterpreterPath": ".venv/bin/python3" + // poetry installs into the in-project .venv (see Dockerfile) + "python.defaultInterpreterPath": "${workspaceFolder}/.venv/bin/python" } } } diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000000..a30e15e094 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,144 @@ +# AGENTS.md + +`imap-processing` is the science data processing pipeline for NASA's IMAP mission, +run by the Science Operations Center (SOC) at LASP. It turns raw CCSDS telemetry +packets into ISTP-compliant CDF science products, one instrument and one data level +at a time. + +## Big picture + +- A high level overview of the mission lives in [docs/source/mission-overview.rst](docs/source/mission-overview.rst). This document is not needed for + software development, but provides a helpful reference for connecting together instrument-spanning questions or ideas. +- Every product is produced by a single invocation of `imap_cli` for one + `(instrument, data-level, descriptor, start-date)` combination. See [imap_processing/cli.py](imap_processing/cli.py). +- [imap_processing/cli.py](imap_processing/cli.py) is the only entry point. It defines an abstract + `ProcessInstrument` base class and one subclass per instrument (`Codice`, `Glows`, + `Hi`, `Hit`, `Idex`, `Lo`, `Mag`, `Spacecraft`, `Swapi`, `Swe`, `Ultra`). Each + subclass implements `do_processing(dependencies) -> list[xr.Dataset]`; the base + class handles downloading dependencies and writing the returned datasets to CDF. + **Adding a new data level means updating both `PROCESSING_LEVELS` in + [imap_processing/__init__.py](imap_processing/__init__.py) and the relevant `do_processing` branch.** +- Instrument subpackages live at `imap_processing//`. Two layouts are in + use — flat modules (`hi/hi_l1a.py`, `codice/codice_l1a.py`) and level subpackages + (`mag/l1a/mag_l1a.py`, `swe/l1b/swe_l1b.py`). Follow whichever the instrument + already uses; do not restructure. +- Data flows as `xarray.Dataset` objects throughout. Processing functions take + dependencies (an `imap_data_access.ProcessingInputCollection` or file paths) and + return `xr.Dataset` / `list[xr.Dataset]`. They should not write files themselves — + the CLI does that. + +## Algorithms +- Algorithm behavior is documented per-instrument in + [docs/source/algorithm-code-documentation/](docs/source/algorithm-code-documentation/). Check there before changing science logic. + + Some instruments have a full working reference distilled from their algorithm document — + a product inventory, the algorithms with equations, and an implementation-status + page listing deviations and gaps. **Read the instrument's page set before proposing or + estimating work on it.** + - CoDICE: [docs/source/algorithm-code-documentation/codice/index.rst](docs/source/algorithm-code-documentation/codice/index.rst) + - GLOWS: [docs/source/algorithm-code-documentation/glows/index.rst](docs/source/algorithm-code-documentation/glows/index.rst) + - HIT: [docs/source/algorithm-code-documentation/hit/index.rst](docs/source/algorithm-code-documentation/hit/index.rst) + - IDEX: [docs/source/algorithm-code-documentation/idex/index.rst](docs/source/algorithm-code-documentation/idex/index.rst) + - MAG: [docs/source/algorithm-code-documentation/mag/index.rst](docs/source/algorithm-code-documentation/mag/index.rst) + - SWAPI: [docs/source/algorithm-code-documentation/swapi/index.rst](docs/source/algorithm-code-documentation/swapi/index.rst) + - SWE: [docs/source/algorithm-code-documentation/swe/index.rst](docs/source/algorithm-code-documentation/swe/index.rst) + + **Many of these pages are unreviewed AI-generated drafts.** They contain the line + `.. include:: /algorithm-code-documentation/_ai_generated_notice.inc` just below the + page title (this renders as a warning banner). Treat a page with that line as a + strong starting point, not as an authority: + - When a marked page disagrees with the code, do not assume the code is wrong. The + algorithm document or CMAD may be out of date, or the deviation may be intentional. + Point out the discrepancy and let a human decide; don't "fix" code to match the page. + - When you cite a marked page in a PR, review, or answer, say that it is unverified. + - Only a person who has verified the page against the code or with the instrument + team removes the include line. Never remove it yourself unless that person asks you + to. When you edit a marked page, keep the marker. + - As the pages are reviewed by humans to correct discrepancies, they will be added into + sphinx docs. Currently, the sphinx documentation only points to autogenerated docs from the code. + - The ENAs (Lo, Hi, and Ultra) will be included in more detail at a later date to this documentation, when their algorithm documents are finalized. + + +## Use the shared infrastructure, don't reinvent it + +| Need | Use | +|---|---| +| Decommutate CCSDS packets | `packet_file_to_datasets()` in [imap_processing/utils.py](imap_processing/utils.py) | +| Raw DN → engineering units | `convert_raw_to_eu()` in [imap_processing/utils.py](imap_processing/utils.py) | +| CDF read/write | `load_cdf()` / `write_cdf()` in [imap_processing/cdf/utils.py](imap_processing/cdf/utils.py) | +| CDF global/variable attributes | `ImapCdfAttributes` in [imap_processing/cdf/imap_cdf_manager.py](imap_processing/cdf/imap_cdf_manager.py) | +| Time conversion (MET / ET / TT-J2000 ns) | [imap_processing/spice/time.py](imap_processing/spice/time.py) — never hand-roll epoch math | +| Spin, repointing, pointing frames, geometry | [imap_processing/spice/](imap_processing/spice/) | +| Data quality bitflags | [imap_processing/quality_flags.py](imap_processing/quality_flags.py) | +| ENA sky maps / pointing sets | [imap_processing/ena_maps/ena_maps.py](imap_processing/ena_maps/ena_maps.py) | +| Combining time-varying ancillary files | [imap_processing/ancillary/ancillary_dataset_combiner.py](imap_processing/ancillary/ancillary_dataset_combiner.py) | + +## Packet definitions (XTCE) + +Packet structures are XTCE XML files in `imap_processing//packet_definitions/`. +They are generated as a first pass from the instrument team's telemetry spreadsheet via +`imap_xtce --output ` ([imap_processing/ccsds/excel_to_xtce.py](imap_processing/ccsds/excel_to_xtce.py)) +and then hand-refined. Decommutation is always done through `packet_file_to_datasets()`, +which returns `dict[apid, xr.Dataset]`; instrument code dispatches on APID enums +defined in the instrument's `constants.py`. + +## CDF products (required for every new data product) + +1. Add/extend the YAML attribute configs in [imap_processing/cdf/config/](imap_processing/cdf/config/): + `imap__global_cdf_attrs.yaml` and + `imap___variable_attrs.yaml`. +2. Load them with `ImapCdfAttributes.add_instrument_global_attrs()` and + `.add_instrument_variable_attrs()` and apply them to `dataset.attrs` and each + variable's `.attrs`. +3. Set `Logical_source = "imap___"` and `Data_version`. + `write_cdf()` derives the output filename from these, so a typo here produces a + wrongly-named file rather than an error. +4. Missing ISTP attributes surface as `cdflib` `ISTPError` at write time — test by + actually calling `write_cdf()` in a test, as [imap_processing/tests/hi/test_hi_l1a.py](imap_processing/tests/hi/test_hi_l1a.py) does. + +## Environment and commands + +Poetry 2.x with the dynamic-versioning plugin. Full setup instructions, including the +required system libraries (`libnetcdf`, `libhdf5`, and openblas/gfortran on 3.14+), are +in [docs/source/development/getting-started.rst](docs/source/development/getting-started.rst). + +```bash +poetry install --extras "test,dev,doc" # or --all-extras +pre-commit install # do this once + +poetry run pytest -vvv -n auto -m "not external_kernel and not external_test_data" +poetry run pytest imap_processing/tests/hi/test_hi_l1a.py::test_sci_de_decom -vvv + +pre-commit run --all-files # ruff check+format, mypy, numpydoc, codespell +make -C docs html SPHINXOPTS="-W --keep-going" +``` + +- Default to the `-m "not external_kernel and not external_test_data"` selection. + Those markers trigger multi-hundred-MB downloads of SPICE kernels from NAIF and + test data from the SDC; assume the sandbox has no network access unless told otherwise. +- Tests live in [imap_processing/tests/](imap_processing/tests/), mirroring the instrument packages. + Global fixtures (SPICE kernels, metakernels, fake spin/repoint data, temp `DATA_DIR`) + are in [imap_processing/tests/conftest.py](imap_processing/tests/conftest.py); instrument-specific fixtures in each + subdirectory's `conftest.py`. Prefer existing fixtures over new ad-hoc setup. +- Codecov requires ~90% patch coverage, so new code needs tests. + +## Conventions + +- Ruff (`ruff check` / `ruff format`) and mypy (`strict = true`, tests excluded) are + enforced in pre-commit and CI. Configuration lives in [pyproject.toml](pyproject.toml). +- **numpydoc docstrings are validated** (`numpydoc-validation` hook) on all non-test + code: summary, `Parameters`, `Returns` with types are effectively mandatory. + Tests are exempt from docstring rules (`D` is ignored under `*/tests/*`). +- Modern type hints everywhere: `str | Path`, `list[xr.Dataset]`, `npt.NDArray`. + Target version is `py310`. +- `logger = logging.getLogger(__name__)` — never `print()`. +- Style details (naming, imports, PR checklist, review standards) are documented in + [docs/source/development/git-workflow-and-style-guide/index.rst](docs/source/development/git-workflow-and-style-guide/index.rst) — follow it rather than + inventing conventions. + +## Gotchas + +- Pre-commit blocks direct commits to `main` and `dev`, and files over 1000 KB. Work on + a feature branch; test data is downloaded at runtime, never committed (no git-lfs). +- CI runs the suite on Linux/macOS/Windows × Python 3.10–3.14, so avoid + platform-specific paths and version-specific syntax. diff --git a/docs/source/algorithm-code-documentation/_ai_generated_notice.inc b/docs/source/algorithm-code-documentation/_ai_generated_notice.inc new file mode 100644 index 0000000000..0fa3a9da0d --- /dev/null +++ b/docs/source/algorithm-code-documentation/_ai_generated_notice.inc @@ -0,0 +1,20 @@ +.. + Shared "unreviewed AI-generated draft" banner, pulled into each algorithm page + with ``.. include:: /algorithm-code-documentation/_ai_generated_notice.inc``. + Once a person familiar with the instrument has verified a page, delete that + include line from the page. To list the pages still awaiting review, run: + grep -rl "_ai_generated_notice.inc" docs/source/algorithm-code-documentation + +.. admonition:: AI-generated draft — not yet reviewed + :class: warning + + This page was generated by an AI assistant from the instrument's algorithm + document, the IMAP Calibration and Measurement Algorithms Document (CMAD), and + the source code in this repository. No one familiar with the instrument has + checked it yet, so it may contain errors or leave out context that only the + instrument team knows. + + If this page and the code disagree, the code is not necessarily wrong. The + source document may be out of date, or the difference may be intentional. + Check against the code, and with the instrument team if you need to, before + relying on anything here. diff --git a/docs/source/algorithm-code-documentation/codice.rst b/docs/source/algorithm-code-documentation/codice.rst index fdcd733adb..03e77a872b 100644 --- a/docs/source/algorithm-code-documentation/codice.rst +++ b/docs/source/algorithm-code-documentation/codice.rst @@ -28,4 +28,4 @@ L2 processing: :recursive: utils - decompress + decompress \ No newline at end of file diff --git a/docs/source/algorithm-code-documentation/codice/ancillary.rst b/docs/source/algorithm-code-documentation/codice/ancillary.rst new file mode 100644 index 0000000000..d276feff52 --- /dev/null +++ b/docs/source/algorithm-code-documentation/codice/ancillary.rst @@ -0,0 +1,268 @@ +.. _codice-ancillary: + +Ancillary and Calibration Files +=============================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +CoDICE depends more heavily on ancillary files than any other IMAP instrument in +this repository: **it cannot even unpack its own telemetry without one.** This +page lists every file the pipeline asks for, what it contains, and which level +consumes it. + +All of these arrive through ``ProcessingInputCollection.get_file_paths( +descriptor=...)``. **None of them are vendored in this repository** - the two +CSVs under ``imap_processing/codice/data/`` are historical snapshots that no +pipeline code reads. + +Summary +------- + +.. list-table:: + :header-rows: 1 + :widths: 30 10 22 38 + + * - Descriptor + - Level + - Format + - Purpose + * - ``l1a-sci-lut`` + - L1A + - JSON + - **The SCI-LUT.** Plan, ESA sweep, Lo stepping, views and collapse tables. + Required for every product except direct events. + * - ``l2-lo-gfactor`` + - L2 Lo + - CSV + - Geometric factors per (mode, ESA step, position). + * - ``l2-lo-efficiency`` + - L2 Lo + - CSV + - Efficiencies per (species, product, ESA step, position). + * - ``l2-lo-onboard-energy-table`` + - L2 Lo DE + - CSV + - APD energy channel -> energy bin index, per (APD ID, gain). + * - ``l2-lo-onboard-energy-bins`` + - L2 Lo DE + - CSV + - Energy bin index -> keV. + * - ``l2-lo-onboard-mpq-cal`` + - L2 Lo DE + - CSV + - k-factor, ESA step voltages, and the TOF channel -> ns quadratic. + * - ``l2-hi-omni-efficiency`` + - L2 Hi + - CSV + - Average efficiency per (species, energy bin) plus a ``GF`` row. + * - ``l2-hi-sectored-efficiency`` + - L2 Hi + - CSV + - Efficiency per (species, energy bin, inst_az) plus a ``GF`` row. + * - ``l2-hi-energy-table`` + - L2 Hi DE + - CSV + - SSD energy channel -> MeV, per (SSD ID, gain). + * - ``l2-hi-tof-table`` + - L2 Hi DE + - CSV + - TOF index -> (ns, MeV/nuc). + +The SCI-LUT +----------- + +**[DOC]** Sections 5.1, 7 and 8. Originates as ``26850.03-SCI-LUT-01.xls`` with +tabs ``Plan``, ``ESA Sweep``, ``Lo Stepping``, ``Views``, ``Collapse_Lo``, +``Collapse_Hi``, ``Data Products - Lo`` and ``Data Products - Hi``. The +``Table_ID`` in each science packet says which spreadsheet version to use. + +**[CODE]** The SDC receives a JSON rendering. Structure, as read by ``utils.py``: + +.. code-block:: text + + { + "": { + "view_tab": { + "(, 0x)": { + "sensor": 0|1, # 0 = Lo, 1 = Hi + "collapse_table": , # index into collapse_lo / collapse_hi + "3d_collapse": , # spins per matrix (Hi) + "compression": 0..6 # CoDICECompression + }, ... + }, + "collapse_lo": { "": { "matrix": [[...]], "variables": {name: [...]} } }, + "collapse_hi": { "": { "matrix": [[...]], "variables": {name: [...]} } }, + "plan_tab": { "(, )": { "lo_stepping": , ... } }, + "esa_sweep_tab": { "": [128 voltages] }, + "lo_stepping_tab": { + "row_number": {"data": [half-spin index per ESA step]}, + "num_steps": {"data": [ESA steps sampled in that half-spin]}, + "tunable_values": { + "spin_time_ms", "num_sectors_ms", "sector_margin_ms", + "dwell_fraction_percentage", "min_hv_settle_ms", "max_hv_settle_ms" + } + }, + "data_product_lo_tab": { + "0": { + "species": {"sw": {"species_names": [...], + "desired_species_names": [...]}}, + "ialirt": {"sw": {"species_names": [...], + "desired_species_names": [...]}} + } + }, + "data_product_hi_tab": { ... } + } + } + +Things to know: + +* **The view key is a literal string** ``"(3, 0x484)"``. If the APID hex casing + or the spacing changes, the lookup returns ``None`` and processing fails with + a ``TypeError`` rather than a helpful message. +* ``collapse_*[id]["matrix"]`` is the raw collapse pattern. + ``get_collapse_pattern_shape`` derives the *reduced* array shape from it; + ``index_to_position`` recovers which physical positions survived. +* ``collapse_*[id]["variables"]`` is used only by the aggregated-counters + products, where each row is one counter and a row of zeros means "counter + turned off in flight". +* ``species_names`` is what the instrument actually sent; + ``desired_species_names`` is what the team wants written. L1A takes the union + of ``desired_species_names`` and the hard-coded ``*_VARIABLE_NAMES``, and + NaN-fills anything that is not in ``species_names``. This is how the P3 Fe + highQ/lowQ label swap is absorbed. +* ``lo_stepping_tab["row_number"]["data"]`` is the half-spin index per ESA step + and can be **shorter than 128** (it is under the post-2025-12-18 scheme). L1A + pads with ``HALF_SPIN_FILLVAL = 63`` and masks those steps to NaN. + +Test copies live at +``imap_processing/tests/codice/data/l1a_lut/imap_codice_l1a-sci-lut_20251007_v005.json`` +(pre-FSW-change) and ``..._20260129_v002.json`` (post). + +Lo geometric factors (``l2-lo-gfactor``) +---------------------------------------- + +**[DOC]** :math:`G_m` in cm2 sr eV/eV for one angular bin, provided for **each +APD ID and ESA step**, i.e. a (128 x 24) array **per mode**. Two modes: ``full`` +and ``reduced``. + +**[CODE]** ``get_geometric_factor_lut`` reads a CSV with columns ``mode``, +``esa_step``, ``position_1`` ... ``position_24``, filters ``mode == "full"`` and +``mode == "reduced"``, sorts by ``esa_step``, sorts the position columns +numerically, and returns ``{"full": (128, 24), "reduced": (128, 24)}``. + +Test file: ``imap_codice_l2-lo-gfactor_20251212_v003.csv``. + +Lo efficiencies (``l2-lo-efficiency``) +-------------------------------------- + +**[DOC]** :math:`\varepsilon_{jlk}` - efficiency by species, ESA step and +position. + +**[CODE]** ``get_efficiency_lut`` returns the whole DataFrame; columns are +``species``, ``product``, ``esa_step``, ``position_1`` ... ``position_24``. The +``product`` column selects ``"sw"`` or (for the unimplemented NSW path) +``"nsw"``. ``get_species_efficiency(species, df)`` filters by species, sorts by +ESA step, and returns an ``xr.DataArray`` with dims ``("esa_step", "inst_az")``. + +A species with no rows produces a warning and is **skipped** - its intensity is +left as a rate. Watch for that. + +Test file: ``imap_codice_l2-lo-efficiency_20251212_v003.csv``. + +Lo direct-event calibration +--------------------------- + +Three files, all consumed by ``process_lo_direct_events``. + +``l2-lo-onboard-energy-table`` + **[CODE]** ``pd.read_csv(header=None, skiprows=1)``. Rows are APD energy + channels, columns are ordered ``APD-1-LG, APD-1-HG, APD-2-LG, ...`` so the + column index is ``apd_id * 2 + gain``. The value is an **energy bin index**, + not an energy. + +``l2-lo-onboard-energy-bins`` + **[CODE]** Column 1 is the energy in keV for each bin index. Chained after the + table above. + +``l2-lo-onboard-mpq-cal`` + **[CODE]** Read positionally, which makes it fragile: + + * ``df.loc[0, 10]`` - the k-factor. + * ``df.loc[4, 4:]`` - the 128 ESA step voltages; + ``esa_kev = esa_v * k_factor / 1000``. + * ``df.loc[2, 1]``, ``df.loc[3, 1]``, ``df.loc[4, 1]`` - the quadratic + coefficients :math:`a`, :math:`b`, :math:`c` for + :math:`\tau_{ns} = a\,\tau_{ch}^2 + b\,\tau_{ch} + c`. + * ``df.loc[6:, 0]`` - the TOF channel numbers. + + **Any change to the row/column layout of this file silently produces wrong + numbers.** There is no header validation. + +Hi efficiencies +--------------- + +``l2-hi-omni-efficiency`` + **[CODE]** Columns include ``species`` and ``average_efficiency``, one row per + (species, energy bin). Species names use hyphens where the code uses + underscores (``ne-mg-si`` vs ``ne_mg_si``); the code translates. A special row + with ``species == "GF"`` carries the geometric factor, read as + ``.values[0][-1]`` (**last column of the first GF row**). See the warning on + :ref:`codice-l2` about whether this is :math:`G_k` or :math:`\sum_k G_k`. + +``l2-hi-sectored-efficiency`` + **[CODE]** Columns ``species``, ``energy_bin``, then 12 ``inst_az`` columns. + The ``GF`` row is likewise 12 values, one per SSD, which is used as a + per-``inst_az`` ``DataArray``. + +For I-ALiRT, ``convert_to_intensities`` reads a *different* layout from the same +descriptor slot: columns ``group_0`` ... ``group_3`` plus a ``GF`` row, sorted by +``energy_bin``. + +Test files: ``imap_codice_l2-hi-omni-efficiency_20251212_v003.csv``, +``imap_codice_l2-hi-sectored-efficiency_20251212_v003.csv``. + +Hi direct-event calibration +--------------------------- + +``l2-hi-energy-table`` + **[CODE]** First column is an index and is dropped. Rows are SSD energy + channels; columns are ordered ``ssd0-LG, ssd0-MG, ssd0-HG, ssd1-LG, ...`` so + the column index is ``ssd_id * 3 + (gain - 1)``. Values are **MeV** directly + (no second-stage bin lookup, unlike Lo). + +``l2-hi-tof-table`` + **[CODE]** First column dropped. Column 0 of the remainder is **TOF in ns**; + column 1 is **energy-per-nucleon in MeV/nuc**. Both are indexed by the raw + 10-bit TOF value. + +Files in the repository that are *not* used +------------------------------------------- + +.. list-table:: + :header-rows: 1 + :widths: 44 56 + + * - Path + - Status + * - ``imap_processing/codice/data/esa_sweep_values.csv`` + - Historical ESA sweep values. **Not read by any pipeline code** - the + sweep now comes from ``esa_sweep_tab`` in the SCI-LUT. + * - ``imap_processing/codice/data/lo_stepping_values.csv`` + - Historical Lo stepping table. **Not read by any pipeline code** - the + stepping now comes from ``lo_stepping_tab`` in the SCI-LUT. + +Both predate the SCI-LUT-driven design. Treat them as documentation of what a +nominal table looks like, not as inputs. + +CoDICE SPICE Usage +------------------ + +**[CODE]** CoDICE processing uses ``imap_processing.spice.time.met_to_ttj2000ns`` +to convert acquisition times to CDF epochs. It does **not** use SPICE kernels, +pointing frames or spin data - all CoDICE L2 angles are in the instrument frame +and the +46 deg rotation to the spacecraft frame (+316 deg in the January 2026 +draft) is not applied here (see :ref:`codice-frames`). Rev 3 Chg 1 section 4.2 +says spin angles "SHOULD" be computed with SPICE from the look-direction unit +vectors and the instrument kernel. The pipeline uses tabulated constants +instead, which the document also provides. CoDICE L2 jobs therefore do not need a metakernel beyond +what the leapsecond/SCLK conversion requires. diff --git a/docs/source/algorithm-code-documentation/codice/data-products.rst b/docs/source/algorithm-code-documentation/codice/data-products.rst new file mode 100644 index 0000000000..697eb55bb4 --- /dev/null +++ b/docs/source/algorithm-code-documentation/codice/data-products.rst @@ -0,0 +1,511 @@ +.. _codice-data-products: + +Data Products and Pipeline +========================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This page is the map of **what exists, what feeds what, and what it is called**. +Use it to find the right module and the right ``Logical_source`` before diving +into an algorithm page. + +Level definitions +----------------- + +**[DOC]** Section 2.1 and section 5.2: + +.. list-table:: + :header-rows: 1 + :widths: 10 90 + + * - Level + - Meaning for CoDICE + * - L0 + - Raw CCSDS packets. A ``.pkts`` file, not produced by this repository. + * - L1A + - Raw data decommutated into formatted variables. Decompressed and + un-collapsed per the SCI-LUT, de-spun where applicable, NSO-masked, but + **counts, not rates**. Electrons are written as Flashed minus Unflashed. + * - L1B + - Counts converted to rates using accumulation times and spin rates. For + housekeeping, raw DN converted to engineering units. + * - L2 + - Geometric factors, efficiencies and other conversion factors applied to + give physical units. Species separated into distinct CDF variables. + **The lowest level usable for science analysis.** + * - L3 + - Higher-level science: partial densities, ratios, VDFs, pitch angles, + Hi+Lo combinations. **Not produced by this repository** - see + :ref:`codice-l3-scope`. + +Source packets (APIDs) +---------------------- + +**[CODE]** ``CODICEAPID`` in ``imap_processing/codice/constants.py``. Fields for +the science and housekeeping APIDs are defined in the two XTCE files in +``imap_processing/codice/packet_definitions/``. + +.. list-table:: + :header-rows: 1 + :widths: 8 9 33 12 38 + + * - Dec + - Hex + - Name + - L1A? + - Contents / cadence + * - 1136 + - 0x470 + - ``COD_NHK`` + - yes + - Nominal housekeeping: voltages, currents, digital settings. + * - 1152 + - 0x480 + - ``COD_LO_IAL`` + - I-ALiRT only + - Lo I-ALiRT trickle: a subset of SW species counts, 15 data bytes per + packet, 232 packets per set. + * - 1153 + - 0x481 + - ``COD_LO_PHA`` + - yes + - Lo direct events, 8 priorities, up to 11520 events/cycle (P3). + Segmented across CCSDS packets. + * - 1155 + - 0x483 + - ``COD_LO_SW_PRIORITY_COUNTS`` + - yes + - Sunward priority counts, priorities 0-4. + * - 1156 + - 0x484 + - ``COD_LO_SW_SPECIES_COUNTS`` + - yes + - Sunward species counts, 16 species, 128 ESA steps. + * - 1157 + - 0x485 + - ``COD_LO_NSW_SPECIES_COUNTS`` + - **no** + - Non-sunward species counts, 8 species. + * - 1158 + - 0x486 + - ``COD_LO_SW_ANGULAR_COUNTS`` + - **no** + - Sunward angular counts, 128 x 12 spin sectors x 5 positions. + * - 1159 + - 0x487 + - ``COD_LO_NSW_ANGULAR_COUNTS`` + - **no** + - Non-sunward angular counts, 128 x 12 spin sectors x 19 positions. + * - 1160 + - 0x488 + - ``COD_LO_NSW_PRIORITY_COUNTS`` + - yes + - Non-sunward priority counts, priorities 5-6. + * - 1161 + - 0x489 + - ``COD_LO_INST_COUNTS_AGGREGATED`` + - yes + - Lo engineering rates (TCR, DCR, STA, STB, SP, total position). + * - 1162 + - 0x48A + - ``COD_LO_INST_COUNTS_SINGLES`` + - yes + - Lo per-APD singles, 24 rates. + * - 1168 + - 0x490 + - ``COD_HI_IAL`` + - I-ALiRT only + - Hi I-ALiRT trickle: H only, 5 data bytes per packet, 199-239 packets per + set. + * - 1169 + - 0x491 + - ``COD_HI_PHA`` + - yes + - Hi direct events, 6 priorities, up to 10000 events/cycle (P3). + * - 1170 + - 0x492 + - ``COD_HI_INST_COUNTS_AGGREGATED`` + - yes + - Hi engineering counters summed over all spin sectors and SSDs. + * - 1171 + - 0x493 + - ``COD_HI_INST_COUNTS_SINGLES`` + - yes + - Hi per-SSD counters (TCR, SSDO, STSSD). + * - 1172 + - 0x494 + - ``COD_HI_OMNI_SPECIES_COUNTS`` + - yes + - Hi omni-directional species counts, 9 species, sqrt(2)-spaced E/n bins, + summed over 4 spins, 1 min cadence. + * - 1173 + - 0x495 + - ``COD_HI_SECT_SPECIES_COUNTS`` + - yes + - Hi sectored species counts, 4 species, x2-spaced E/n bins, 12 spin + sectors x 12 SSDs, 16 spins, 4 min cadence. + * - 1174 + - 0x496 + - ``COD_HI_INST_COUNTS_PRIORITIES`` + - yes + - Hi priority counts, 6 priorities. + +Other APIDs are defined in ``CODICEAPID`` (``COD_AUT``, ``COD_BOOT_HK``, +``COD_MEMDMP``, ``COD_SHK``, the ``COD_DIAG_*`` family, ``COD_CSTOL_CONFIG``, +etc.) but are **not processed** by ``process_l1a``. + +.. warning:: + + **[CODE]** APIDs 1157, 1158 and 1159 have ``Logical_source`` entries, CDF + variable attributes and species name lists in the repository, but **there is + no processing module for them and ``process_l1a`` has no branch for them**. + The Lo species, angular and NSW products described in sections 10.3.3 and + 10.3.4 of the document are therefore **only half built**. See + :ref:`codice-implementation-status`. + + **[DOC]** Rev 3 Chg 1 section 9.2 now says the instrument team is **not + currently producing** ``lo-nsw-species``, ``lo-sw-angular`` or + ``lo-nsw-angular`` at any level, "due to issues identified post-launch". + The missing code therefore matches current operations. The algorithms are + still fully specified in sections 10-12, and "not currently" suggests the + products may return. Ask the team before building them. See + :ref:`codice-data-caveats`. + +Products this repository can emit +--------------------------------- + +**[CODE]** These are the exact ``Logical_source`` strings, all defined in +``imap_processing/cdf/config/imap_codice_global_cdf_attrs.yaml``. A string that +is not in that file cannot be written - ``get_global_attributes`` raises. + +Level 1A +^^^^^^^^ + +Produced by ``codice_l1a.process_l1a(dependencies)``. One call processes every +APID in the L0 file and returns a list of datasets. + +.. list-table:: + :header-rows: 1 + :widths: 36 10 54 + + * - ``Logical_source`` + - APID + - Status / notes + * - ``imap_codice_l1a_hskp`` + - 1136 + - Raw DN. Written straight from ``packet_file_to_datasets``. + * - ``imap_codice_l1a_lo-counters-aggregated`` + - 1161 + - Implemented. + * - ``imap_codice_l1a_lo-counters-singles`` + - 1162 + - Implemented. + * - ``imap_codice_l1a_lo-sw-priority`` + - 1155 + - Implemented. + * - ``imap_codice_l1a_lo-nsw-priority`` + - 1160 + - Implemented. + * - ``imap_codice_l1a_lo-sw-species`` + - 1156 + - Implemented. + * - ``imap_codice_l1a_lo-nsw-species`` + - 1157 + - **Declared but not produced.** + * - ``imap_codice_l1a_lo-sw-angular`` + - 1158 + - **Declared but not produced.** + * - ``imap_codice_l1a_lo-nsw-angular`` + - 1159 + - **Declared but not produced.** + * - ``imap_codice_l1a_lo-direct-events`` + - 1153 + - Implemented. + * - ``imap_codice_l1a_lo-ialirt`` + - 1152 + - Declared, but the I-ALiRT path deliberately writes **no L1A CDF**; the + intermediate dataset borrows the ``lo-sw-species`` global attributes. + * - ``imap_codice_l1a_hi-counters-aggregated`` + - 1170 + - Implemented. + * - ``imap_codice_l1a_hi-counters-singles`` + - 1171 + - Implemented. + * - ``imap_codice_l1a_hi-priority`` + - 1174 + - Implemented. + * - ``imap_codice_l1a_hi-omni`` + - 1172 + - Implemented. + * - ``imap_codice_l1a_hi-sectored`` + - 1173 + - Implemented. + * - ``imap_codice_l1a_hi-direct-events`` + - 1169 + - Implemented. + * - ``imap_codice_l1a_hi-ialirt`` + - 1168 + - Declared; produced only inside the I-ALiRT pipeline, no CDF written. + +Level 1B +^^^^^^^^ + +Produced by ``codice_l1b.process_codice_l1b(l1a_file)``, **one file in, one file +out**. The descriptor is derived from the input file's ``Logical_source``. + +.. list-table:: + :header-rows: 1 + :widths: 40 60 + + * - ``Logical_source`` + - Status / notes + * - ``imap_codice_l1b_hskp`` + - **Produced during the L1A run**, not by ``process_codice_l1b``. See below. + * - ``imap_codice_l1b_lo-counters-aggregated`` + - Implemented. + * - ``imap_codice_l1b_lo-counters-singles`` + - Implemented. + * - ``imap_codice_l1b_lo-sw-priority`` + - Implemented. + * - ``imap_codice_l1b_lo-nsw-priority`` + - Implemented. + * - ``imap_codice_l1b_lo-sw-species`` + - Implemented. + * - ``imap_codice_l1b_lo-nsw-species`` + - **Declared, no L1A input, no rate branch.** + * - ``imap_codice_l1b_lo-sw-angular`` + - **Declared, no L1A input, no rate branch.** + * - ``imap_codice_l1b_lo-nsw-angular`` + - **Declared, no L1A input, no rate branch.** + * - ``imap_codice_l1b_lo-ialirt`` + - Computed in memory by the I-ALiRT pipeline; no CDF written. + * - ``imap_codice_l1b_hi-counters-aggregated`` + - Implemented. + * - ``imap_codice_l1b_hi-counters-singles`` + - Implemented. + * - ``imap_codice_l1b_hi-priority`` + - Implemented. + * - ``imap_codice_l1b_hi-omni`` + - Implemented. + * - ``imap_codice_l1b_hi-sectored`` + - Implemented. + * - ``imap_codice_l1b_hi-ialirt`` + - Computed in memory by the I-ALiRT pipeline; no CDF written. + +.. note:: + + **There is no L1B direct-event product**, by design. Direct events go + straight from L1A to L2, where the bit values are converted to physical + units. The document agrees (section 12.1.1 / 12.2.1 convert "L1A event data" + to L2). + +.. important:: + + **[CODE]** ``imap_codice_l1b_hskp`` is written by ``process_l1a``, not by the + L1B CLI branch. When the housekeeping APID is present, ``process_l1a`` calls + ``packet_file_to_datasets`` a **second** time with ``use_derived_value=True`` + so that the XTCE polynomial calibrators produce engineering units, and + appends that as a separate dataset. This is why a single ``l1a`` invocation + returns both ``imap_codice_l1a_hskp`` and ``imap_codice_l1b_hskp``. + +Level 2 +^^^^^^^ + +Produced by ``codice_l2.process_codice_l2(descriptor, dependencies)``. + +.. list-table:: + :header-rows: 1 + :widths: 36 64 + + * - ``Logical_source`` + - Status / notes + * - ``imap_codice_l2_lo-sw-species`` + - **Implemented.** Solar-wind and pickup-ion intensities with geometric + factors and efficiencies. + * - ``imap_codice_l2_lo-nsw-species`` + - **Declared only.** No branch in ``process_codice_l2``. + * - ``imap_codice_l2_lo-sw-angular`` + - **Declared only.** No branch. + * - ``imap_codice_l2_lo-nsw-angular`` + - **Declared only.** No branch. + * - ``imap_codice_l2_lo-direct-events`` + - **Implemented.** Physical-unit conversion. + * - ``imap_codice_l2_hi-omni`` + - **Implemented.** Omni-directional intensities. + * - ``imap_codice_l2_hi-sectored`` + - **Implemented.** Sectored intensities plus spin/elevation angles. + * - ``imap_codice_l2_hi-direct-events`` + - **Implemented.** Physical-unit conversion. + +``process_codice_l2`` also has a **pass-through branch** listing +``hi-counters-singles``, ``hi-counters-aggregated``, ``lo-counters-singles``, +``lo-counters-aggregated``, ``lo-sw-priority`` and ``lo-nsw-priority`` with the +comment "No changes needed. Just save to an L2 CDF file. TODO: May not even need +L2 files for these products". **None of those six descriptors has a +``Logical_source`` entry**, and the branch as written raises - see +:ref:`codice-implementation-status`. + +Product flow +------------ + +.. code-block:: text + + L0 .pkts + | + +-- 1136 --> l1a_hskp ------------------> l1b_hskp (derived values, same run) + | + +-- 1153 --> l1a_lo-direct-events -------------------------> l2_lo-direct-events + +-- 1169 --> l1a_hi-direct-events -------------------------> l2_hi-direct-events + | + +-- 1156 --> l1a_lo-sw-species --> l1b_lo-sw-species ------> l2_lo-sw-species + +-- 1155 --> l1a_lo-sw-priority --> l1b_lo-sw-priority ----> (none) + +-- 1160 --> l1a_lo-nsw-priority --> l1b_lo-nsw-priority --> (none) + +-- 1161 --> l1a_lo-counters-aggregated --> l1b_... -------> (none) + +-- 1162 --> l1a_lo-counters-singles --> l1b_... ----------> (none) + | + +-- 1172 --> l1a_hi-omni --> l1b_hi-omni ------------------> l2_hi-omni + +-- 1173 --> l1a_hi-sectored --> l1b_hi-sectored ----------> l2_hi-sectored + +-- 1174 --> l1a_hi-priority --> l1b_hi-priority ----------> (none) + +-- 1170 --> l1a_hi-counters-aggregated --> l1b_... -------> (none) + +-- 1171 --> l1a_hi-counters-singles --> l1b_... ----------> (none) + | + +-- 1157/1158/1159 --> NOT PROCESSED + + I-ALiRT stream (separate entry point, imap_processing/ialirt/l0/process_codice.py) + +-- 1152 --> l1a_lo_species --> convert_to_rates --> ratios (DynamoDB items) + +-- 1168 --> l1a_ialirt_hi --> convert_to_rates --> intensities + +CLI wiring +---------- + +**[CODE]** ``imap_processing/cli.py``, class ``Codice``: + +.. code-block:: python + + if self.data_level == "l1a": + datasets = codice_l1a.process_l1a(dependencies) + for i, ds in enumerate(datasets): + datasets[i] = filter_day_boundary_data(ds, self.start_date) + + elif self.data_level == "l1b": + science_files = dependencies.get_file_paths(source="codice") + if len(science_files) != 1: + raise ValueError(...) + datasets = [codice_l1b.process_codice_l1b(science_files[0])] + + elif self.data_level == "l2": + datasets = [codice_l2.process_codice_l2(self.descriptor, dependencies)] + +Things worth knowing about this: + +* **L1A is one-to-many.** A single ``--level l1a`` run emits every product + present in the L0 file. ``--descriptor`` is not used to select a product. +* **``filter_day_boundary_data`` is applied to every L1A dataset**, trimming + records that fall outside the requested UTC day. It is applied only at L1A. +* **L1B is one-to-one** and requires exactly one CoDICE science file. +* **L2 dispatches on ``self.descriptor``**, which is used both to fetch the + input file (``dependencies.get_file_paths(descriptor=descriptor)``) and to + build the output ``Logical_source``. It then re-parses the descriptor out of + the returned filename with ``ScienceFilePath``. + +Input dependencies per level +---------------------------- + +**[CODE]** What the SDC must supply in the ``ProcessingInputCollection``: + +.. list-table:: + :header-rows: 1 + :widths: 12 30 58 + + * - Level + - Descriptor / type + - Purpose + * - L1A + - ``data_type="l0"`` + - The ``.pkts`` file. + * - L1A + - ``descriptor="l1a-sci-lut"`` + - The SCI-LUT JSON. **Required for every APID except the two PHA + (direct-event) APIDs**, which are unpacked without it. + * - L1B + - ``source="codice"`` + - Exactly one L1A CDF. + * - L2 + - ``descriptor=`` + - The L1B (or, for direct events, L1A) CDF. + * - L2 Lo + - ``l2-lo-gfactor``, ``l2-lo-efficiency`` + - Geometric factors and efficiencies for species/angular intensities. + * - L2 Lo DE + - ``l2-lo-onboard-energy-table``, ``l2-lo-onboard-energy-bins``, + ``l2-lo-onboard-mpq-cal`` + - APD energy, ESA step and TOF conversions. + * - L2 Hi + - ``l2-hi-omni-efficiency``, ``l2-hi-sectored-efficiency`` + - Efficiencies and the geometric factor (row ``GF``). + * - L2 Hi DE + - ``l2-hi-energy-table``, ``l2-hi-tof-table`` + - SSD energy, TOF and energy-per-nucleon conversions. + +See :ref:`codice-ancillary` for the contents of each of these. + +Species inventories +------------------- + +**[CODE]** ``constants.py``. These lists drive which CDF variables exist; the +**actual** species ordering in the telemetry comes from the SCI-LUT and is +cross-checked at L1A. + +.. list-table:: + :header-rows: 1 + :widths: 34 66 + + * - Constant + - Members + * - ``LO_SW_SPECIES_VARIABLE_NAMES`` (16) + - ``hplus``, ``heplusplus``, ``cplus4``, ``cplus5``, ``cplus6``, + ``oplus5``, ``oplus6``, ``oplus7``, ``oplus8``, ``ne``, ``mg``, ``si``, + ``fe_loq``, ``fe_hiq``, ``heplus``, ``cnoplus`` + * - ``LO_SW_SOLAR_WIND_SPECIES_VARIABLE_NAMES`` (14) + - the above minus ``heplus`` and ``cnoplus`` + * - ``LO_SW_PICKUP_ION_SPECIES_VARIABLE_NAMES`` (2) + - ``heplus``, ``cnoplus`` + * - ``LO_SW_ANGULAR_VARIABLE_NAMES`` (5) + - ``hplus``, ``heplusplus``, ``oplus6``, ``fe_loq``, ``heplus`` + * - ``LO_NSW_ANGULAR_VARIABLE_NAMES`` (2) + - ``heplusplus``, ``heplus`` + * - ``LO_SW_PRIORITY_VARIABLE_NAMES`` (5) + - ``p0_tcrs``, ``p1_hplus``, ``p2_heplusplus``, ``p3_heavies``, ``p4_dcrs`` + * - ``LO_NSW_PRIORITY_VARIABLE_NAMES`` (2) + - ``p5_heavies``, ``p6_hplus_heplusplus`` + * - ``LO_COUNTERS_AGGREGATED_VARIABLE_NAMES`` (6) + - ``tcr``, ``dcr``, ``sta``, ``stb``, ``sp``, ``total_position_count`` + * - ``LO_COUNTERS_SINGLES_VARIABLE_NAMES`` + - ``apd_singles`` (one variable, 24 APDs on an axis) + * - ``HI_OMNI_VARIABLE_NAMES`` (9) + - ``h``, ``he3``, ``he4``, ``c``, ``o``, ``ne_mg_si``, ``fe``, ``uh``, + ``junk`` + * - ``HI_SECTORED_VARIABLE_NAMES`` (4) + - ``h``, ``he3he4``, ``cno``, ``fe`` + * - ``HI_PRIORITY_VARIABLE_NAMES`` (6) + - ``priority0`` ... ``priority5`` + * - ``HI_COUNTERS_AGGREGATED_VARIABLE_NAMES`` (7) + - ``dcr``, ``mst``, ``starts_only``, ``stops_only``, ``singles_starts``, + ``singles_stops``, ``low_tof_cutoff`` + * - ``HI_COUNTERS_SINGLES_VARIABLE_NAMES`` (3) + - ``tcr``, ``ssdo``, ``stssd`` + * - ``LO_IALIRT_VARIABLE_NAMES`` (9) + - ``heplusplus``, ``cplus5``, ``cplus6``, ``oplus6``, ``oplus7``, + ``oplus8``, ``mg``, ``fe_hiq``, ``fe_loq`` + * - ``HI_IALIRT_VARIABLE_NAMES`` + - ``h`` + +**[DOC]** The document's non-sunward Lo species list is H+, He++, O5-8, C4-6, +Ne+Mg+Si, Fe, He+ and CNO+ (8 species). There is no corresponding constant in +the code, consistent with the NSW products not being implemented (and not +currently produced by the team, per section 9.2). + +**[DOC]** Table 2 of the document lists the Lo angular species as He++, O+6, +C+5, Fe+10 and PUI He+. The code's ``LO_SW_ANGULAR_VARIABLE_NAMES`` is +``hplus``, ``heplusplus``, ``oplus6``, ``fe_loq``, ``heplus``. Resolve this +against the SCI-LUT, not either list, if the angular products are ever built. diff --git a/docs/source/algorithm-code-documentation/codice/ialirt.rst b/docs/source/algorithm-code-documentation/codice/ialirt.rst new file mode 100644 index 0000000000..167f151d9d --- /dev/null +++ b/docs/source/algorithm-code-documentation/codice/ialirt.rst @@ -0,0 +1,311 @@ +.. _codice-ialirt: + +I-ALiRT +======= + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**[DOC]** Sections 10.4 and 14. **[CODE]** +``imap_processing/ialirt/l0/process_codice.py`` - the entire CoDICE I-ALiRT +algorithm lives in this one module, which reuses the ordinary L1A/L1B/L2 +functions. + +I-ALiRT (IMAP Active Link for Real-Time) is the < 5 minute latency space-weather +stream. It is **not** part of the ``imap_cli --instrument codice`` chain: it has +its own entry point and writes DynamoDB items rather than CDFs. + +How the telemetry differs +------------------------- + +**[DOC]** I-ALiRT data is packed **identically** to the regular science data but +transmitted differently: a trickle of small packets that must be reassembled. +Each packet has the CCSDS header plus: + +.. list-table:: + :header-rows: 1 + :widths: 30 10 60 + + * - Mnemonic + - Bits + - Description + * - ``SHCOARSE`` + - 32 + - Spacecraft time in seconds. + * - ``ACQUISITION_TIME`` + - 32 + - Spacecraft time at the **end** of the acquisition cycle. All packets in a + cycle share this. + * - ``STATUS`` + - 8 + - Data quality / status. + * - ``COUNTER`` + - 8 + - Which block of data this packet carries. 0-231 are valid; 231-239 are + fill (0xFF). + +then **15 data bytes** (Lo) or **5 data bytes** (Hi), a spare byte and a 16-bit +checksum. + +**[DOC]** ``Plan ID``, ``Plan Step`` and ``View ID`` are all assumed to be 0, so +from the Views tab all I-ALiRT packets are **Lossy A + Lossless** compressed. + +**[CODE]** ``COD_LO_COUNTER = 232``, ``COD_HI_COUNTER = 199``, +``COD_LO_RANGE = range(0, 15)``, ``COD_HI_RANGE = range(0, 5)``. +``find_groups`` (from ``ialirt/utils/grouping.py``) assembles complete counter +runs; ``concatenate_bytes`` stacks the ``cod__data_NN`` fields into a +single bytearray. + +CoDICE-Lo I-ALiRT +----------------- + +**[DOC]** Section 10.4.1. On board, data from all spin sectors and the five +sunward positions is summed into a single value, organised **by species** (not +by ESA step). Once a complete 0-231 set arrives it is processed **identically to +the SW_SPECIES_COUNTS product** with a single spin sector, a single azimuth and +128 energies. The only difference is the species list. Nominal species: + +.. math:: + + \mathrm{He^{++}},\ \mathrm{C^{+5}},\ \mathrm{C^{+6}},\ \mathrm{O^{+6}},\ + \mathrm{O^{+7}},\ \mathrm{O^{+8}},\ \mathrm{Mg},\ \mathrm{Fe(lowQ)},\ + \mathrm{Fe(hiQ)} + +**[CODE]** ``LO_IALIRT_VARIABLE_NAMES``, with mass-per-charge +``LO_IALIRT_M_OVER_Q``: + +.. list-table:: + :header-rows: 1 + :widths: 20 14 20 14 20 12 + + * - Species + - m/q + - Species + - m/q + - Species + - m/q + * - ``heplusplus`` + - 2.0 + - ``oplus6`` + - 2.7 + - ``mg`` + - 3.5 + * - ``cplus5`` + - 2.4 + - ``oplus7`` + - 2.28 + - ``fe_hiq`` + - 3.85 + * - ``cplus6`` + - 2.0 + - ``oplus8`` + - 2.0 + - ``fe_loq`` + - 7.25 + +Pipeline +^^^^^^^^ + +**[CODE]** ``process_codice(dataset, l1a_lut_path, l2_lut_path, "codice_lo", +l2_geometric_factor_path)``: + +1. ``find_groups`` -> ``concatenate_bytes`` -> one bytearray per cycle. +2. ``process_ialirt_data_streams`` splits the bit string by + ``IAL_BIT_STRUCTURE`` (a 24-field dict mirroring the ordinary science packet + header, including the post-2026-01-29 RGFO/NSO fields), then takes + ``BYTE_COUNT`` bytes as the science payload. A packet with ``SHCOARSE == 0`` + is discarded. +3. ``create_xarray_dataset`` builds a **fake** dataset with an integer-index + epoch and ``pkt_apid`` set to 1152 (Lo) or 1168 (Hi), so that the ordinary + L1A functions can consume it. +4. ``process_by_table_id(cod_lo_dataset, l1a_lut_path, l1a_lo_species)`` - the + **same** L1A function used for the science product, taking the + ``COD_LO_IAL`` branch. +5. ``convert_to_rates(l1a_lo, "lo-ialirt")`` - the **same** L1B function. + ``n_sectors`` = 12, except 11 at ESA step 127. +6. ``Logical_file_id`` is synthesised from the mid-measurement epoch because + ``compute_geometric_factors`` parses the date out of it. +7. ``calculate_ratios`` -> the public products. + +Pseudo-densities and ratios +^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Section 14.2. First compute the intensity exactly as at L2: + +.. math:: + + J_j(l) = \frac{R_j(l)}{G_m \cdot \varepsilon_{jl1} \cdot (E/q)_l}, + \qquad + U_j(l) = \frac{\sqrt{C_j(l)}} + {G_m \cdot \varepsilon_{jl1} \cdot (E/q)_l \cdot (t_{acquire} \cdot 10^{-3} \cdot 12)} + +Then form **pseudo**-densities. They are "pseudo" because several constant +factors that a real density needs are omitted - they cancel in the ratios: + +.. math:: + + d\_psN_j(l) = J_j(l) \cdot \sqrt{(E/q)_l} \cdot \sqrt{(m/q)_j}, + \qquad + psN_j = \sum_{l=0}^{127} d\_psN_j(l) + +.. math:: + + psU_j = \sqrt{\sum_{l=0}^{127} \left(d\_psU_j(l)\right)^2} + +Then the public products: + +.. math:: + + \frac{C}{O} &= \frac{psN_{C^{+5}} + psN_{C^{+6}}} + {psN_{O^{+6}} + psN_{O^{+7}} + psN_{O^{+8}}} \\ + \frac{Mg}{O} &= \frac{psN_{Mg}} + {psN_{O^{+6}} + psN_{O^{+7}} + psN_{O^{+8}}} \\ + \frac{Fe}{O} &= \frac{psN_{Fe,low} + psN_{Fe,high}} + {psN_{O^{+6}} + psN_{O^{+7}} + psN_{O^{+8}}} \\ + \frac{C^{+6}}{C^{+5}} &= \frac{psN_{C^{+6}}}{psN_{C^{+5}}}, \qquad + \frac{O^{+7}}{O^{+6}} = \frac{psN_{O^{+7}}}{psN_{O^{+6}}}, \qquad + \frac{Fe_{low}}{Fe_{high}} = \frac{psN_{Fe,low}}{psN_{Fe,high}} + +**[DOC]** Ratio uncertainties propagate as + +.. math:: + + \Delta R = \frac{R_{top}}{R_{bot}} + \sqrt{\left(\frac{\Delta R_{top}}{R_{top}}\right)^2 + + \left(\frac{\Delta R_{bot}}{R_{bot}}\right)^2} + +with numerator and denominator uncertainties summed in quadrature first. + +**[CODE]** ``calculate_ratios`` computes the six ratios, reusing +``get_geometric_factor_lut``, ``compute_geometric_factors``, +``get_efficiency_lut`` and ``process_lo_species_intensity`` from +``codice_l2.py`` with ``SOLAR_WIND_POSITIONS`` (position 1 only). Results are +rounded to six decimal places and wrapped in ``Decimal`` so DynamoDB can store +them. **Zero denominators return ``None`` rather than raising or producing +infinity** - the source notes this matches the instrument team's test data. + +.. warning:: + + **[CODE] The ratio uncertainties are not computed.** ``calculate_ratios`` + returns only the six ratios; there is no ``psU`` accumulation and no + quadrature propagation. Section 14.2.2's uncertainty algorithm is unbuilt. + +CoDICE-Hi I-ALiRT +----------------- + +**[DOC]** Sections 10.4.2 and 14.1. On board, H counts are binned into **4 spin +sector bins** (summing every 6 of the 24 sectors), **4 azimuthal look +directions** (each a group of 3 SSDs) and **15 sqrt(2)-spaced energy-per-nucleon +bins**, accumulated over 4 spins at 1 min resolution. + +SSD groups: + +.. list-table:: + :header-rows: 1 + :widths: 16 30 30 24 + + * - Group :math:`g` + - SSD IDs + - Elevation angle + - Ref. spin angle :math:`\theta_{g,0}` + * - 0 + - 0, 1, 3 + - 132.8 deg + - 196.85 deg + * - 1 + - 4, 5, 7 + - 65.7 deg + - 174.55 deg + * - 2 + - 8, 9, 11 + - 47.1 deg + - 253.16 deg + * - 3 + - 12, 13, 15 + - 114.3 deg + - 275.44 deg + +**[CODE]** ``HI_IALIRT_ELEVATION_ANGLE = [132.8, 65.7, 47.1, 114.3]`` matches. +``HI_IALIRT_REF_SPIN_ANGLE = [196.85, 174.55, 253.16, 275.44]`` **matches Rev 3 +Chg 1**. The January 2026 draft printed values 90 deg higher (286.85, 264.55, +343.16, 5.44). ``HI_IALIRT_SPIN_ANGLE`` in +``ialirt/utils/constants.py`` is built by adding 0, 90, 180 and 270 deg (mod +360) to each reference, matching **[DOC]** :math:`\theta_{g,n} = (\theta_{g,0} ++ 90^\circ n) \bmod 360^\circ`. + +Rates and intensities +^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** + +.. math:: + + R(i, n, g) = \frac{C(i, n, g)}{6 \cdot 4 \cdot t_{acquire} \cdot 10^{-3}} + \qquad [\mathrm{counts/s}] + +.. math:: + + I(i, n, g) = \frac{R(i, n, g)}{G_g \cdot \epsilon_{ig} \cdot \Delta E_i} + \qquad [\#/(\mathrm{cm}^2\,\mathrm{sr}\,\mathrm{s}\,\mathrm{MeV/nuc})] + +with :math:`G_g = 3 G_k = 0.039` cm2 sr (the sum over the three SSDs in the +group) and :math:`\epsilon_{ig}` the average of the H efficiencies for those +three SSDs. + +**[CODE]** ``l1a_ialirt_hi`` produces the counts; ``convert_to_rates(l1a_hi, +"hi-ialirt")`` uses ``L1B_DATA_PRODUCT_CONFIGURATIONS["hi-ialirt"]`` = +``{num_spin_sectors: 6, num_spins: 4}`` times ``HI_ACQUISITION_TIME``. +``convert_to_intensities`` reads ``group_0`` ... ``group_3`` columns from the +efficiency CSV plus its ``GF`` row, and divides by ``g_g * eps_ig * +energy_passbands``. Output shape ``(4 spins, 15 energies, 4 spin sectors, +4 groups)``. + +``IALIRT_HI_NUMBER_OF_SSD_PER_GROUP = 3.0`` exists in ``constants.py`` but the +grouping factor is expected to be baked into the ``GF`` row of the CSV. + +Output +------ + +**[DOC]** Table 4 of Rev 3 Chg 1 lists both CoDICE I-ALiRT products (Hi H +intensities at 1 min; Lo C/O, Mg/O, Fe/O, C6+/C5+, O7+/O6+, Fe_loq/Fe_hiq) with +file prefix ``imap_ialirt_l1_realtime`` - the mission-wide I-ALiRT product - +rather than the ``imap_codice_l2-hi-ialirt_`` / ``imap_codice_l2-lo-ialirt_`` +prefixes of the draft. That matches this repository: CoDICE writes no I-ALiRT +CDF of its own. Its fields go into the shared ``imap_ialirt_l1_realtime`` +dataset built by ``ialirt/utils/create_xarray.py``. The synthesised +``Logical_file_id`` in step 6 above uses the same name. + +**[CODE]** ``process_codice`` returns two lists of dicts, ready for DynamoDB: + +.. code-block:: text + + codice_lo_data[i] = { + ... instrument header items ..., + "instrument": "codice_lo", + "codice_lo_epoch": , + "codice_lo_c_over_o_abundance": Decimal | None, + "codice_lo_mg_over_o_abundance": Decimal | None, + "codice_lo_fe_over_o_abundance": Decimal | None, + "codice_lo_c_plus_6_over_c_plus_5": Decimal | None, + "codice_lo_o_plus_7_over_o_plus_6": Decimal | None, + "codice_lo_fe_low_over_fe_high": Decimal | None, + } + + codice_hi_data[i] = { + ... instrument header items ..., + "instrument": "codice_hi", + "codice_hi_epoch": [, ...], + "codice_hi_h": , + } + +Field names, dtypes and the Hi energy bin centres/deltas are declared in +``imap_processing/ialirt/utils/constants.py`` +(``codice_hi_energy_center``, ``codice_hi_energy_minus``, +``codice_hi_energy_plus``); ``ialirt/utils/create_xarray.py`` treats +``codice_lo`` as a one-epoch instrument and ``codice_hi`` as multi-epoch. + +.. note:: + + ``process_codice``'s own docstring still says "This function is incomplete + and will need to be updated". As of this survey the Lo ratio path and the Hi + intensity path are both implemented end to end; the outstanding gap is + uncertainty propagation. diff --git a/docs/source/algorithm-code-documentation/codice/implementation-status.rst b/docs/source/algorithm-code-documentation/codice/implementation-status.rst new file mode 100644 index 0000000000..09896fddcf --- /dev/null +++ b/docs/source/algorithm-code-documentation/codice/implementation-status.rst @@ -0,0 +1,373 @@ +.. _codice-implementation-status: + +Implementation Status and Known Gaps +==================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This page is the honest accounting of where the code stands against the +algorithm document. **Read it before proposing or estimating work.** + +Accurate as of the most recent survey of ``imap_processing/codice`` and +``imap_processing/ialirt/l0/process_codice.py``. The code survey was done +against the January 2026 draft (Rev 3 Chg 0). This page has since been +re-checked against **Rev 3 Chg 1** (CMAD section 4.3.2) without re-surveying +the code, apart from spot checks where the document changed. If you change +something material, update this page in the same commit. + +Summary +------- + +.. list-table:: + :header-rows: 1 + :widths: 14 22 64 + + * - Level + - State + - Notes + * - L1A + - **Mostly complete** + - Eleven of fourteen science products implemented, plus housekeeping. The + SCI-LUT unpacking machinery, all seven compression modes, segmented + direct events and the full P3 NSO masking rules all work. + * - L1B + - **Complete for what L1A produces** + - Every implemented L1A product has a working rate conversion. The three + missing L1A products would crash if fed in. + * - L2 + - **Roughly half** + - Five of eight declared products implemented: ``lo-sw-species``, + ``lo-direct-events``, ``hi-omni``, ``hi-sectored``, + ``hi-direct-events``. The three Lo angular/NSW products are not built. + The pass-through branch for counters/priority products is broken. + * - L3 + - **Out of scope** + - Belongs to a different repository. See :ref:`codice-l3-scope`. + * - I-ALiRT + - **Works end to end** + - Lo abundance/charge-state ratios and Hi H intensities both produced. + Uncertainty propagation is missing. + +There are **no** ``NotImplementedError`` raises anywhere in the CoDICE code, +apart from the generic unknown-data-level branch in ``Codice.do_processing``. + +Not implemented at all +---------------------- + + +Hi Appendix B: true omni-directional intensity +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +The document is explicit that ``hi-omni`` as implemented is a **pseudo** +omni-directional intensity, and derives the solid-angle-corrected version from +the *sectored* intensities in Appendix B: + +.. math:: + + I_{\Omega}(i) = \frac{\sum_k \Omega_{12,k} \sum_n I(i, n, k)}{\Omega}, + \qquad \Omega = 1.956\ \mathrm{sr} + +using per-pixel spin-integrated solid angles :math:`\Omega_k` (0.412282 to +1.17554 sr) and per-sector values :math:`\Omega_{12,k}` (0.05994 to 0.12354 sr). +None of these constants are in ``constants.py`` and the calculation does not +exist. Whether the SDC should produce it is a question for the CoDICE team. + +Ratio uncertainties in I-ALiRT +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Section 14.2.2 defines pseudo-density uncertainties ``psU`` and quadrature +propagation into each ratio. ``calculate_ratios`` computes only the six ratios. + +Frame conversion to spacecraft coordinates +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +:math:`\theta_{SC} = (\theta_{inst} + 46^\circ) \bmod 360^\circ` (Rev 3 Chg 1; ++316 deg against the draft's old instrument-frame tables) is not applied +anywhere. All L2 angles are instrument-frame. This is consistent with the +document, which only needs the SC frame for L3 pitch angles, but it means an L2 +consumer must do the rotation themselves. + +Bugs +---- + +Ordered by how likely they are to affect released data. + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Location + - Issue + * - ``codice_l2.py``, ``process_codice_l2`` + - **``UnboundLocalError`` for six descriptors.** The branch + + .. code-block:: python + + if dataset_name in ["imap_codice_l2_hi-counters-singles", ...]: + pass + + matches, does nothing, and then falls through to + ``if var in l2_dataset.data_vars`` - but ``l2_dataset`` was never + assigned. Any attempt to run L2 for ``hi-counters-singles``, + ``hi-counters-aggregated``, ``lo-counters-singles``, + ``lo-counters-aggregated``, ``lo-sw-priority`` or ``lo-nsw-priority`` + crashes. Those six also have **no ``Logical_source``** entry, so even a + fixed pass-through could not be written. The ``TODO`` above the branch + says "May not even need L2 files for these products" - decide, then + either delete the branch or finish it. + * - ``codice_l2.py``, ``process_codice_l2`` + - Same ``UnboundLocalError`` for ``lo-nsw-species``, ``lo-sw-angular`` and + ``lo-nsw-angular``: they match no branch at all. These *do* have + ``Logical_source`` entries, so an operator can legitimately request them + and get an unhelpful crash. + * - ``codice_l1b.py``, ``convert_to_rates`` + - Same class of problem one level up. ``lo-sw-angular``, + ``lo-nsw-angular`` and ``lo-nsw-species`` reach the ``lo-`` branch that + computes ``energy_per_charge`` but match none of the three + denominator branches, leaving ``denominator`` unbound. + * - ``codice_l2.py``, ``process_hi_omni`` + - **Possible factor of 12.** The docstring says the denominator includes + ``number_of_ssd``; the code does not multiply by it. The document's + formula divides by :math:`\sum_k G_k = 12 \times 0.013 = 0.156`. Whether + this is correct depends entirely on what the ``GF`` row of + ``imap_codice_l2-hi-omni-efficiency_*.csv`` contains. **Verify before + trusting absolute omni intensities.** + * - ``codice_l1a_lo_species.py``, ``codice_l1a_lo_priority.py``, + ``codice_l1a_lo_counters_aggregated.py`` + - **NSO boundary half-spin over-masked before 2026-01-29.** For P0-P2 the + document says the instrument enters NSO on the half-spin *after* + ``NSO_Half_Spin``, so only ``half_spin > NSO_half_spin`` is NaN + (species: section 10.3.3, unchanged since the draft; priority: section + 10.3.5, new in Rev 3 Chg 1). These three modules use ``>=`` for + pre-FSW data. Species and aggregated do so for *all* dates, which is + correct only from 2026-01-29 for species. In the priority module this + contradicts its own comment. ``codice_l1a_lo_counters_singles.py`` uses + ``>``. Effect: one extra half-spin of valid ESA steps is NaN'd in every + cycle where NSO triggered, in P0-P2 data. Confirm the intended rule for + the counters products with the team; the document does not give one + explicitly. + * - ``codice_l2.py``, ``process_lo_direct_events`` + - **``apd_id`` and ``position`` used inconsistently within one function.** + Elevation is looked up from ``apd_id``; the 13-24 spin de-spin shift uses + ``position``. Section 12.2.1 still says elevation is "converted from + position". However, Rev 3 Chg 1 section 9.2 says delay-line position "is + not an accurate identifier of particle direction" and "should only be + used to support APD ID". It also describes the de-spin in terms of + **APDs** 2-12 / 14-24, and L3b now bins by APD ID. **This makes the + elevation lookup the likely-correct half and the ``position``-based + spin shift the likely-wrong half** - the reverse of what this page said + against the draft. Either way the two should agree; confirm with the + team before changing released L2. + * - ``utils.py``, ``get_codice_epoch_time`` + - Sub-seconds are divided by ``65536`` (2^16) for every product, but non-PHA + science packets declare a **20-bit** ``Acq_Start_Subseconds`` field + (PHA packets declare 16). If the field really is 20-bit, epochs are off + by up to 15/16 of a second. The CoDICE team specified ``/ 65536``, so + this may be intentional - but it should be written down. + * - ``utils.py``, ``process_by_table_id`` + - ``view_id``, ``plan_id`` and ``plan_step`` are read from **record 0 only** + and applied to the entire stream. Only ``table_id`` is grouped. A plan or + view change mid-day silently unpacks the rest of the day with the wrong + collapse table. + * - ``codice_l1a_hi_omni.py`` + - Unexplained ``* 2`` in the sub-epoch spacing, flagged in the source as + ``# TODO: why multiply by 2?``. It appears to undo the ``// 2`` in + ``get_codice_epoch_time`` so that sub-epochs are spaced a full + accumulation window apart, which is probably right - but it is + load-bearing arithmetic with no justification. + +Deviations from the algorithm document +-------------------------------------- + +These are design decisions, not bugs, but they will surprise anyone reading the +document first. +Hi energy-delta variable names +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Rev 3 Chg 1 calls the Hi energy-bin deltas ``energy__delta_plus`` / +``energy__delta_minus``. The code uses ``energy__plus`` / +``energy__minus`` at L1A/L1B and reads them by those names at L2. The +two are equivalent; only the names differ. + +Epoch is the window centre, not the start +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Section 7 says "an Epoch variable is written ... which is the start time of the +acquisition ... a DELTA_EPOCH_PLUS is written". The code writes the **centre** +of the window with symmetric ``epoch_delta_minus`` / ``epoch_delta_plus`` in +integer nanoseconds. This is the IMAP project convention and applies to every +instrument. + +Lo TOF conversion is quadratic +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Section 13.2.5 gives :math:`\tau_{ns} = 0.6217\,\tau_{ch} - 7.4437`. The code +uses a quadratic :math:`a\tau^2 + b\tau + c` with coefficients read from the +``l2-lo-onboard-mpq-cal`` ancillary file. The quadratic is the newer form; the +linear fit in the document is stale. + +Negative TOF is filled at L2 +^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +The document says nothing about this. The code sets ``tof_ns < 0`` to NaN in +``process_lo_direct_events``, with the comment that it mirrors "Menlo's L3a +handling" - i.e. the downstream L3 repository already discards them, and doing +it at L2 keeps the two consistent. + +Data caveats are not flagged +^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Section 9.2 lists known problems with the Summer 2026 L2 data (see +:ref:`codice-data-caveats`), e.g. Hi LG energies unusable and Lo SW/NSW +binning by position. These are instrument/on-board issues, and the document +does not ask the ground pipeline to do anything about them. The pipeline sets +no quality flag for any of them. If the team later asks for flags (for +example on Hi LG/MG events), that would be new work here. + +RGFO boundary fill applies to all dates +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Section 12.2.3 makes the "set ``half_spin == RGFO_half_spin`` to fill" rule +specific to P3 (2026-01-29 onwards). ``process_lo_species_intensity`` applies it +unconditionally. For P1/P2 data, where RGFO always triggered on half spin 0, +this NaNs the first half spin's worth of ESA steps regardless. + +Hi geometric factor is read from the efficiency CSV +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +The document quotes :math:`G_k = 0.013` cm2 sr as a fixed value from a SIMION +model. The code never hard-codes it; it reads a ``GF`` row from the efficiency +CSV. That is more flexible and more fragile - it means the geometric factor and +efficiency are versioned together and cannot be varied independently. + +Aggregated counters variable lists are fixed +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Section 10.3.1 lists 30 selectable Lo rate types and notes "the CDF file which +contains this information must have all the variables defined, but only the +returned values will be filled". The code defines only the **nominal six** +(``LO_COUNTERS_AGGREGATED_VARIABLE_NAMES``). Which of those six are present is +read from the SCI-LUT, so turning one off is handled - but selecting a +*different* rate type in flight would require a code change. The same applies to +the seven Hi aggregated counters. + +Direct events skip L1B +^^^^^^^^^^^^^^^^^^^^^^ + +There is no ``imap_codice_l1b_*-direct-events`` product. L2 direct-event +processing loads the **L1A** CDF. This matches the document. + +FSW-version handling +-------------------- + +Two XTCE files exist for the 2026-01-29 FSW change, selected by parsing the date +from the L0 filename. Within a product, the branch is on ``packet_version``. +Both mechanisms are **whole-file**, and the known limitation is called out three +times in the source: + +* ``codice_l1a.py:55`` - ``TODO get the exact time the FSW changed on january 29 + and relabel the xml file``. The switch is at midnight UTC, not the real + changeover time. +* ``codice_l1a_lo_counters_singles.py:150`` and ``codice_l1a_lo_priority.py:186`` + - ``TODO handle boundary days where the FSW changed halfway through the + dataset. E.g. Some packet_version = 1 and some = 2``. The code takes + ``packet_versions[0]``. +* ``codice_l2.py:390`` - ``TODO: Fix this calculation on days when the sci Lut + changes. There may be different packet versions in the same dataset.`` + +**2026-01-29 itself is therefore expected to be wrong** in some products, and so +is any future day on which the SCI-LUT changes mid-day. **2026-04-03** (start of +P4, a Lo SCI_LUT update; see :ref:`codice-timeline`) is one documented example. If you are chasing an anomaly, check the date first. + +Complete TODO inventory +----------------------- + +.. list-table:: + :header-rows: 1 + :widths: 42 58 + + * - Location + - Text + * - ``codice_l1a.py:55`` + - Get the exact FSW change time on 2026-01-29 and relabel the XML file. + * - ``codice_l1a_de.py:455`` + - ``is this possible?`` - guard around an incomplete-event-packet case. + * - ``codice_l1a_hi_omni.py:112`` + - ``why multiply by 2?`` in the sub-epoch spacing. + * - ``codice_l1a_lo_counters_singles.py:150`` + - Handle boundary days with mixed ``packet_version``. + * - ``codice_l1a_lo_priority.py:186`` + - Same. + * - ``codice_l1b.py:113`` + - ``undo this when I get new validation file from Joey`` - + ``acquisition_time_per_esa_step`` is temporarily kept in L1B output. + * - ``codice_l2.py:390`` + - Geometric factors break on days when the SCI-LUT changes. + * - ``codice_l2.py:494`` + - Pickup-ion geometric factor uses only position 0; the team wants this + standardised. + * - ``codice_l2.py:728`` + - Hi-omni L2 attribute workaround, "may go away once Joey and I fix L1B + CDF". + * - ``codice_l2.py:735`` + - L1B needs ``epoch_delta_plus`` / ``epoch_delta_minus`` attributes and an + ``epoch`` dimension. + * - ``codice_l2.py:1465`` + - Update the list of datasets that need geometric factors. + * - ``codice_l2.py:1515`` + - "May not even need L2 files for these products" (the broken pass-through). + * - ``constants.py:806`` + - Read the angular/priority/species variable name lists from the SCI-LUT + instead of hard-coding them. + +Test coverage +------------- + +``imap_processing/tests/codice/``: + +.. list-table:: + :header-rows: 1 + :widths: 34 66 + + * - File + - Coverage + * - ``test_codice_l1a.py`` + - Housekeeping, Lo counters/priority/species, Hi counters/omni/sectored/ + priority, Lo and Hi direct events, incomplete segments. Compares against + validation CDFs. + * - ``test_codice_l1a_lut.py`` + - SCI-LUT JSON parsing: view table lookup, collapse pattern shape + derivation. + * - ``test_codice_l1b.py`` + - Rate conversion for Lo and Hi, uncertainty propagation. + * - ``test_codice_l2.py`` + - Geometric factor and efficiency LUT reading, MPQ/TOF/energy conversions, + Lo species intensity, Lo and Hi direct events. + * - ``test_codice_hi_l2.py`` + - Hi omni and sectored intensities, spin-angle construction. + * - ``test_codice_spin_angles.py`` + - Exact-value regression on every corrected reference angle. **Runs without + external data** - the primary guard against reverting to pre-Rev 3 Chg 1 + PDF tables. + * - ``test_decompress.py`` + - All seven compression modes. + * - ``test_process_by_table_id.py`` + - Table-ID grouping. + +Validation inputs live under ``imap_processing/tests/codice/data/`` +(``l0_data/``, ``l1a_input/``, ``l1b_validation/``, ``l1a_lut/``, ``l2_lut/``) +and are pinned in ``conftest.py`` by ``VALIDATION_FILE_DATE = "20250814"`` and +``VALIDATION_FILE_VERSION = "v015"``. ``conftest.codice_lut_path`` is a +``side_effect`` callable that stands in for +``ProcessingInputCollection.get_file_paths`` - **it raises ``ValueError`` on an +unknown descriptor**, which is the fastest way to discover which ancillary files +a new code path needs. + +Untested paths worth knowing about: + +* The 2026-01-29 packet definition is exercised only by a single + ``fsw-changes`` fixture (``imap_codice_l0_raw_20260130_v001.pkts``). +* ``compute_geometric_factors(angular_product=True)`` is unreachable from + production code and therefore untested against real data. +* The I-ALiRT CoDICE path is tested in ``imap_processing/tests/ialirt/``, not + here. diff --git a/docs/source/algorithm-code-documentation/codice/index.rst b/docs/source/algorithm-code-documentation/codice/index.rst new file mode 100644 index 0000000000..81b8b9578e --- /dev/null +++ b/docs/source/algorithm-code-documentation/codice/index.rst @@ -0,0 +1,227 @@ +:orphan: + +.. _codice-index: + +CoDICE +====== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +.. currentmodule:: imap_processing.codice + +This is the CoDICE (Compact Dual Ion Composition Experiment) instrument module, +which contains the code for processing data from the CoDICE instrument. + +Purpose of these pages +---------------------- + +These pages are a **condensed, self-contained working reference** for the CoDICE +processing algorithms, written so that a developer (human or AI agent) can get +productive without reading the full algorithm document. + +They are a summary of the source document below plus what the code in +``imap_processing/codice`` actually does. Where the two disagree, that is called +out explicitly in :ref:`codice-implementation-status`. + +.. _codice-source-documents: + +Source documents +---------------- + +**None of these are redistributed in this repository.** This is an open-source +repository and the mission documents are not ours to publish. Request them from +the CoDICE instrument team at SwRI or the SDC document store. + +.. list-table:: + :header-rows: 1 + :widths: 22 78 + + * - Short name + - Reference + * - **Algorithm document** + - This revision is publicly released as **section 4.3.2 of the IMAP + Calibration and Measurement Algorithms Document (CMAD)**, + * - **SCI_LUT spreadsheet** + - ``26850.03-SCI-LUT-01.xls`` (and successors). The plan / ESA-sweep / + stepping / views / collapse tables. **This is not optional** - CoDICE + science packets cannot be unpacked without it. The SDC consumes a JSON + rendering of it as an ancillary file with descriptor ``l1a-sci-lut``. + * - **Packet ICD** + - Field-level packet definitions. Superseded in practice by the two XTCE + files in ``imap_processing/codice/packet_definitions/``, which are what + the code parses. + +.. tip:: + + If you hold a copy of the algorithm document, put it in ``docs/reference/``. + That directory is gitignored, so it will never be committed, and the section + index in :ref:`codice-reference-tables` is written against that location. + Even though the CMAD is public, do not commit it: it is ~128 MB and far over + the repository's file-size limit. + +.. important:: + + Two conventions used throughout these pages: + + * **[DOC]** marks a statement taken from the algorithm document. It describes + the intended behavior, which may not be what the code does yet. + * **[CODE]** marks a statement verified against ``imap_processing/codice``. + + When those conflict, the code is what runs and the document is what the + instrument team expects. Both matter; do not silently "fix" one to match the + other without asking. + +.. note:: + + **This repository stops at L2.** CoDICE has a substantial L3 program - + partial densities, abundance and charge-state ratios, 3-D VDFs, pitch-angle + distributions and the combined Hi+Lo L3c products - all described in section + 13 of the algorithm document. **None of that belongs here.** It is produced + by a separate repository closer to the science team. Section 13 is summarised + on :ref:`codice-l3-scope` only so that you can recognise an L3 request when + one arrives and redirect it. + + The one exception is I-ALiRT: the real-time stream computes abundance and + charge-state ratios that look like L3 quantities, but they are produced here + because the whole I-ALiRT chain lives in this repository. See + :ref:`codice-ialirt`. + +Which page to read +------------------ + +Read only what you need. Each page is designed to be loaded on its own. + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Page + - Read it when you need to know... + * - :ref:`codice-overview` + - What CoDICE physically is, the two sensors, the coordinate frames and + angle conventions, the ESA stepping scheme, RGFO/NSO modes, the + commissioning timeline and the instrument team's data caveats. **Start + here if you are new.** + * - :ref:`codice-data-products` + - The full product inventory, exact ``Logical_source`` strings, APID + mapping, what feeds what, and how the CLI is wired. **The "what goes into + what" map.** + * - :ref:`codice-l1a` + - The SCI-LUT unpacking machinery (plan / view / collapse tables), + decompression, de-spinning, NSO masking and direct-event bit unpacking. + **The hardest part of CoDICE.** + * - :ref:`codice-l1b` + - Counts to rates: acquisition times, spin-sector counts, and the + ``energy_table`` derivation. + * - :ref:`codice-l2` + - Rates to intensities: geometric factors, RGFO Full/Reduced selection, + efficiencies, and the direct-event unit conversions. + * - :ref:`codice-ialirt` + - The real-time space-weather stream and the pseudo-density ratios it + produces. + * - :ref:`codice-ancillary` + - Every lookup table and calibration file: who delivers it, what is in it, + and which level consumes it. + * - :ref:`codice-implementation-status` + - What is implemented, what is missing, where the code deviates from the + document, and known/suspected bugs. **Read before proposing or + estimating work.** + * - :ref:`codice-l3-scope` + - What section 13 asks for, and why none of it is built here. + * - :ref:`codice-reference-tables` + - Where the big tables live (SCI-LUT, XTCE, CDF attribute YAML, PDF page + ranges). Deliberately *not* reproduced inline. + +.. toctree:: + :maxdepth: 1 + + overview + data-products + l1a + l1b + l2 + ialirt + ancillary + implementation-status + l3-scope + reference-tables + +Ten-second orientation +---------------------- + +* CoDICE is **two ion sensors sharing one time-of-flight / energy (TOF-E) + subsystem**, with two separate apertures. + + * **CoDICE-Lo** measures ~0.5-80 keV/q ions. An electrostatic analyzer (ESA) + selects energy-per-charge; a -15 kV post-acceleration carbon foil produces + start electrons; 24 avalanche photodiodes (APDs) around 360 degrees of + azimuth measure residual energy. (E/q, TOF, E) gives mass, charge state and + m/q. + * **CoDICE-Hi** measures ~0.03-5 MeV/nuc ions through 12 collimators onto 12 + solid-state detectors (SSDs). (E, TOF) gives mass. + +* The fundamental Lo cadence is a **16-spin (32 half-spin, ~4 minute) cycle** + over which the ESA steps through **128 energy-per-charge steps**. Different + half-spins sample different numbers of ESA steps (1 to 6). Hi products + accumulate over 4 or 16 spins. + +* Almost every science product is **"collapsed" on board** - angles and spins + summed together per a configurable table - and then optionally compressed + (table-based lossy and/or LZMA lossless). Ground unpacking is impossible + without the **SCI_LUT** tables, which are keyed by ``(table_id, plan_id, + plan_step, view_id)`` carried in every science packet. + +* Two on-board protective modes change the meaning of the data and both must be + handled on the ground: **RGFO** (Reduced Geometric Factor Operation, changes + which geometric factor applies) and **NSO** (No-Scan Operation, stops the + energy sweep - affected data must be set to fill). + +* Processing chain:: + + CCSDS packets + -> L1A decompressed, un-collapsed, de-spun raw counts + metadata + -> L1B counts / acquisition time = rates (plus energy_per_charge) + -> L2 rates / (G * efficiency * energy passband) = intensities; + direct events converted to physical units + -> L3 (elsewhere) densities, ratios, VDFs, pitch angles, Hi+Lo combos + +* A separate **I-ALiRT** path takes a trickle-fed subset of the Lo SW species + counts and the Hi sectored H counts all the way to abundance ratios and + intensities for space-weather forecasting. + +Where the code lives +-------------------- + +.. code-block:: text + + imap_processing/codice/ + constants.py APIDs, species lists, lossy tables, angle tables + utils.py SCI-LUT reading, collapse patterns, acq times, epochs + decompress.py lossy A/B, LZMA, 24-bit pack + codice_l1a.py APID dispatch for L1A (+ the L1B housekeeping product) + codice_l1a_de.py direct events (Lo + Hi), segmented packet reassembly + codice_l1a_lo_species.py Lo SW species counts (also used by I-ALiRT Lo) + codice_l1a_lo_priority.py Lo SW/NSW priority counts + codice_l1a_lo_counters_aggregated.py + codice_l1a_lo_counters_singles.py + codice_l1a_hi_omni.py Hi omni-directional species counts + codice_l1a_hi_sectored.py Hi sectored species counts + codice_l1a_hi_priority.py Hi priority counts + codice_l1a_hi_counters_aggregated.py + codice_l1a_hi_counters_singles.py + codice_l1a_ialirt_hi.py Hi I-ALiRT counts (called only by the I-ALiRT pipeline) + codice_l1b.py counts -> rates for every descriptor + codice_l2.py intensities, geometric factors, DE unit conversions + data/esa_sweep_values.csv historical ESA sweep values (not read by the pipeline) + data/lo_stepping_values.csv historical Lo stepping values (not read by the pipeline) + packet_definitions/ + imap_codice_packet-definition_20250101_v001.xml pre-2026-01-29 FSW + imap_codice_packet-definition_20260129_v001.xml post-2026-01-29 FSW + P_COD_NHK.xml housekeeping only + + imap_processing/ialirt/l0/process_codice.py the entire I-ALiRT CoDICE algorithm + imap_processing/ialirt/utils/constants.py I-ALiRT CoDICE field/energy definitions + + imap_processing/cdf/config/imap_codice_*.yaml CDF global + variable attributes + imap_processing/tests/codice/ tests + validation data + imap_processing/cli.py (class Codice) dependency wiring per level diff --git a/docs/source/algorithm-code-documentation/codice/l1a.rst b/docs/source/algorithm-code-documentation/codice/l1a.rst new file mode 100644 index 0000000000..90cdfc9ef5 --- /dev/null +++ b/docs/source/algorithm-code-documentation/codice/l1a.rst @@ -0,0 +1,573 @@ +.. _codice-l1a: + +Level 1A - Unpacking +==================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**[DOC]** Sections 7, 8 and 10. **[CODE]** ``codice_l1a.py``, ``utils.py``, +``decompress.py`` and the eleven ``codice_l1a_*`` product modules. + +L1A is where essentially all of CoDICE's complexity lives. Nothing here is a +science calculation; it is entirely about reversing a very flexible on-board +packing scheme. + +.. important:: + + **You cannot unpack a CoDICE science packet without the SCI-LUT.** Every + science packet carries four identifiers - ``Table_ID``, ``Plan_ID``, + ``Plan_Step``, ``View_ID`` - which must be used *in tandem* to look up how + the data was collapsed and compressed on board. The lookup tables originate + as an Excel spreadsheet (``26850.03-SCI-LUT-01.xls``) and reach the SDC as a + JSON ancillary file with descriptor ``l1a-sci-lut``. + +Packet structure +---------------- + +**[DOC]** All CoDICE packets share the standard CCSDS primary header plus +``SHCOARSE`` (32-bit spacecraft seconds). + +Non-PHA science packets then carry the fields below. The latest version of the CoDICE algorithm document +dropped the "Max Length in Bits" column from these tables. The widths shown here come from a +January 2026 draft, and are included for reference only. The XTCE files are the authority. + +.. list-table:: + :header-rows: 1 + :widths: 30 10 60 + + * - Mnemonic + - Bits + - Description + * - ``Packet_Version`` + - 16 + - Incremented when the packet format changes. **The ground branches on + this** (``<= 1`` = pre-2026-01-29 FSW). + * - ``Spin_Period`` + - 16 + - Spin period in **spin ticks of 320 us**, active when the 16-spin cycle + started. + * - ``Acq_Start_Seconds`` + - 32 + - Whole seconds at the start of the 16-spin cycle. + * - ``Acq_Start_Subseconds`` + - 20 + - Sub-seconds (16 for PHA packets). + * - ``ST_Bias_Gain_Mode_Flag`` + - 2 + - Suprathermal high/low gain bias curve. Not used for Hi. + * - ``SW_Bias_Gain_Mode_Flag`` + - 2 + - Solar-wind high/low gain bias curve. Not used for Hi. + * - ``Table_ID`` + - 32 + - Which SCI-LUT version applies. + * - ``Plan_ID`` + - 16 + - Plan table in use. + * - ``Plan_Step`` + - 4 + - Plan step active during acquisition. + * - ``View_ID`` + - 4 + - How the data was collapsed and/or compressed. + * - ``RGFO_Half_Spin`` + - 6 + - Half spin at which RGFO activated. Not used for Hi. + * - ``NSO_Half_Spin`` + - 6 + - Half spin at which NSO activated. Not used for Hi. + * - ``Data_Quality_Flag`` + - 1 + - Errors during acquisition/processing (EDAC, timing violations, ...). + **[CODE]** decommutated as ``suspect``, written as ``data_quality``. + * - ``Compression_Flag`` + - 3 + - See the compression table below. + * - ``Byte_Count`` + - 23 + - Length of the data array; the **compressed** length if compressed. + * - ``RGFO_esa_step``, ``RGFO_spin_sector``, ``NSO_esa_step``, + ``NSO_spin_sector`` + - - + - **Added by the 2026-01-29 FSW update.** Pin the trigger to an exact bin. + **[DOC]** "the ESA step / spin sector at which the RGFO (NSO) limit is + exceeded. RGFO (NSO) mode is activated for all ESA steps (spin sectors) + after this value." The spin-sector fields count over a **full spin + (0-23)**, while COUNTS-product ``spin_sector`` is half-spin relative + (0-11), so compare using ``% 12``. The same fields, plus + ``RGFO_Half_Spin`` / ``NSO_Half_Spin``, are also in PHA packets since + 2026-01-29. + +**[DOC]** PHA (direct-event) packets differ: 16-bit ``Acq_Start_Subseconds``, a +4-bit ``Priority``, a 32-bit ``Num_Events``, a 1-bit ``Compressed`` flag and a +31-bit ``Byte_Count``. Event data is Rice-compressed if the flag is set. +**[CODE]** ``DE_METADATA_FIELDS`` in ``constants.py`` lists the exact field +widths the code unpacks, and applies **LZMA** (``CoDICECompression.LOSSLESS``) +when ``compressed`` is set. + +**[DOC]** When the data array is too large for one CCSDS packet, CoDICE uses the +**CCSDS grouping flags** to spread it across several packets. Fields are padded +to a 16-bit boundary; pad bits are counted in the CCSDS length field but **not** +in ``Byte_Count``. + +Compression +----------- + +**[DOC]** Section 7: + +.. list-table:: + :header-rows: 1 + :widths: 12 28 60 + + * - Flag + - Name + - Algorithm + * - 0 + - No compression + - - + * - 1 + - Lossy A + - Table-based 24 -> 8 bit compression per counter value; decompress by + looking up the centre of each compression region. + * - 2 + - Lossy B + - As above with a different table. + * - 3 + - Lossless + - LZMA over the whole ``Data`` field as a single unit. + * - 4 + - Lossy A + Lossless + - Undo LZMA first, then the lossy table. + * - 5 + - Lossy B + Lossless + - As above. + +**[CODE]** ``CoDICECompression`` in ``utils.py`` adds a sixth value the document +does not list: + +.. list-table:: + :header-rows: 1 + :widths: 12 30 58 + + * - Value + - Name + - Implementation + * - 6 + - ``PACK_24_BIT`` + - ``_apply_pack_24_bit`` - reads 3-byte big-endian integers into 32-bit + values. Used where the on-board counters are 24-bit packed rather than + compressed. + +``LOSSY_A_TABLE`` and ``LOSSY_B_TABLE`` are 256-entry dictionaries in +``constants.py``, transcribed from Greg Dunn's ``sohis_cdh_utils.v``. The values +are expected to change; the format is not. + +``decompress(compressed_bytes, algorithm)`` dispatches on the enum and applies +LZMA before the lossy table for the combined modes. + +.. note:: + + **[DOC]** The 2026-01-29 FSW load made **Hi and Lo priority counts + lossless-only** (no lossy stage). Because the algorithm is read from the + SCI-LUT view table rather than hard-coded, no code change was needed. + +The unpacking algorithm +----------------------- + +**[DOC]** Section 7, verbatim in structure: + +1. **Plan lookup.** ``PlanTables[Plan_ID][Plan_Step]`` gives ``Iterations``, + ``ESA Sweep`` (0 or 1; nominally 0, descending in energy), ``Lo Stepping`` + (0-4), ``Hi Products`` and ``Lo Products`` table indices. +2. **Energies.** The ESA Sweep table has 128 entries of voltage; multiply by + ``k_factor`` (5.76) and divide by 1000 for keV/e. +3. **Acquisition time.** Index the Lo Stepping table by ESA step; the "Acq Time + (including Sector Margin)" column gives milliseconds. +4. **View lookup.** ``View_ID`` selects a row of the ``Views`` table giving the + expected APID, compression scheme and collapse-table index. +5. **Decompress** in the correct order (lossless first, then lossy). +6. **Extract:** + + * **Lo (APIDs 0x480-0x48F):** 128 frames (one per ESA step); each frame + started as a [12 spin angle x 24 position] matrix; use the collapse table + to index into ``Collapse_Lo``. + * **Hi (APIDs 0x490-0x49F):** 192 frames (one per counter); each frame is + [n spins x 24 spin angles x 16 azimuths]; use ``Collapse_Hi``, and use the + ``3D Collapse`` value to know how many spins went into the matrix. + +7. **Write.** The collapse table ID determines which data product is being + extracted and how it is written. **No transformations at L1A except writing + Electrons as Flashed minus Unflashed.** + +Worked example (section 8) +^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** ``Table_ID = 1``, ``Plan_ID = 0``, ``Plan_Step = 0``, ``View_ID = 3``: + +* Table 1 -> use ``26850.03-SCI-LUT-01.xls``. +* Plan 0 occupies rows 7-10; plan step 0 is row 7: Iterations 1, ESA Sweep 0, + Lo Stepping 0, Hi/Lo products 0. +* ESA Sweep 0 -> 128 decreasing voltages. +* Sweep step 74 with Lo Stepping 0 -> row 28 of the Lo Stepping tab -> 118.75 ms + acquisition time. +* Views tab, Lo, row 10 for ``View_ID`` 3 -> lossy **and** lossless compressed, + collapse table 3. +* Collapse_Lo table 3 is "SW Angular Counts" -> write 128 E/q x 12 spin sectors + x 5 positions to ``imap_codice_l1a_lo-angular`` variable + ``SunwardAngularCounts``. + +How the code implements it +-------------------------- + +**[CODE]** ``codice_l1a.process_l1a`` does three things: + +1. Picks the XTCE file by date (see :ref:`codice-modes`). +2. Calls ``packet_file_to_datasets(science_file, xtce_file)`` -> ``dict[apid, + Dataset]``. +3. Dispatches each APID to a product module, wrapping non-PHA products in + ``utils.process_by_table_id``. + +``process_by_table_id`` +^^^^^^^^^^^^^^^^^^^^^^^ + +The shared wrapper for every non-PHA, non-housekeeping product. It: + +* reads ``view_id``, ``pkt_apid``, ``plan_id`` and ``plan_step`` from the + **first** record (they are assumed uniform across a stream), +* splits the dataset by unique ``table_id``, +* calls the per-product function once per group with signature + ``(group_ds, lut_file, table_id, view_id, apid, plan_id, plan_step)``, +* concatenates on ``epoch`` with ``data_vars="minimal", coords="minimal", + compat="equals"`` and sorts by epoch. + +The ``minimal`` settings keep 1-D support variables such as ``voltage_table`` +and ``k_factor`` from being broadcast along ``epoch``. + +.. warning:: + + ``view_id``, ``plan_id`` and ``plan_step`` are taken from index 0 only. If + any of them change mid-file the rest of the day will be unpacked with the + wrong view. Only ``table_id`` is handled per-group. + +SCI-LUT access helpers +^^^^^^^^^^^^^^^^^^^^^^ + +All in ``utils.py``: + +.. list-table:: + :header-rows: 1 + :widths: 36 64 + + * - Function + - What it does + * - ``read_sci_lut(path, table_id)`` + - Loads the JSON and returns the sub-dict for one ``table_id``. Raises if + absent. + * - ``get_view_tab_info(json, view_id, apid)`` + - Indexes ``view_tab`` by the literal key ``"(, 0x)"``. + * - ``get_view_tab_obj(...)`` + - Returns ``(sci_lut_data, ViewTabInfo)``. ``ViewTabInfo`` carries ``apid``, + ``view_id``, ``sensor`` (0 = Lo, 1 = Hi), ``collapse_table``, + ``compression`` and ``three_d_collapsed``. + * - ``get_collapse_pattern_shape(json, sensor_id, collapse_table_id)`` + - Reads ``collapse_[]["matrix"]`` and derives the **reduced** + shape. Returns ``(1,)`` when every non-zero entry is identical (fully + collapsed), otherwise ``(unique_spin_sectors, unique_inst_azs)``. + * - ``get_counters_aggregated_pattern(...)`` + - Reads ``collapse_[]["variables"]``, drops rows containing a + zero (counter turned off in flight), sorts by first value, and returns + ``{counter_name: n_spin_sectors}``. + * - ``index_to_position(...)`` + - Indices of unique non-zero rows in the collapse matrix - i.e. which + physical positions survived collapsing. + * - ``calculate_acq_time_per_step(lo_stepping_tab)`` + - Appendix C timing, returns a 128-element array in **seconds**. + * - ``get_energy_info(energy_table)`` + - Geometric bin centres ``sqrt(min*max)`` and the plus/minus deltas for Hi + energy-per-nucleon bins. + * - ``get_codice_epoch_time(...)`` + - Epoch centre and delta - see below. + +Epoch construction +^^^^^^^^^^^^^^^^^^ + +**[DOC]** "an Epoch variable is written in CDF_TT2000 format which is the start +time of the acquisition ... For variables with an accumulation period, a +DELTA_EPOCH_PLUS is written with the units of seconds." + +**[CODE]** ``get_codice_epoch_time`` writes the **centre** of the accumulation +window with symmetric ``epoch_delta_minus``/``epoch_delta_plus`` in **integer +nanoseconds**: + +.. code-block:: python + + spin_period_ns = spin_period.astype(np.int64) * 320_000 + delta_times = (num_spins * spin_period_ns) // 2 + center_times_seconds = (acq_start_seconds + + acq_start_subseconds / 65536 + + delta_times / 1e9) + +``num_spins`` is **16** for every Lo product and for Hi PHA; for other Hi +products it is the SCI-LUT ``3d_collapse`` value. + +This is a deliberate deviation - centre-plus-delta is the IMAP project +convention - but note that the subsecond divisor is ``65536`` (2^16) even though +non-PHA packets declare a **20-bit** subseconds field. The CoDICE team specified +``subseconds / 65536``; see :ref:`codice-implementation-status`. + +Per-product notes +----------------- + +Lo species counts (``codice_l1a_lo_species.py``) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Section 10.3.3. Data is collapsed on board over spin sectors and +positions to **two values per energy**: azimuths 1, 2, 3, 23, 24 sum to a +sunward product; 4-22 sum to a non-sunward product. Because the spin axis is +sun-facing these assignments are constant, so the two are separated on board +and telemetered as different products. + +The section 9.2 caveats say the flight instrument +**actually bins by delay-line position**, which lowers the counts and +intensities. This is an on-board issue; the ground cannot undo it. See +:ref:`codice-data-caveats`. + +**[CODE]** Handles ``COD_LO_SW_SPECIES_COUNTS`` and ``COD_LO_IAL``. Reshapes to +``(num_packets, num_species, 128, *collapsed_shape)`` where ``collapsed_shape`` +is normally ``(1,)``. Output dims ``(epoch, esa_step, spin_sector)`` with +``spin_sector`` of length 1. + +Species selection is defensive: the module takes the union of the SCI-LUT's +``desired_species_names`` and the hard-coded list, then for each wanted species +either takes its index in ``actual_species_names`` or logs a warning and fills +NaN. This absorbs the Fe highQ/lowQ label swap. + +Uncertainty is ``sqrt(counts)``, matching **[DOC]** +:math:`\sigma_j(l) = \sqrt{C_j(l)}`. + +NSO masking, **[DOC]** section 10.3.3, branches by date: + +* **Launch - 2026-01-29 (P0-P2):** the instrument enters NSO on the half-spin + *after* ``NSO_Half_Spin``, so set to NaN where ``half_spin > NSO_half_spin``. +* **2026-01-29 onward (P3+):** NSO now starts mid-half-spin at a specific (ESA + step, spin sector). The summed species counts for that half-spin are not + representative, so set to NaN where ``half_spin >= NSO_half_spin``. + +**[CODE]** One rule for all dates: + +.. code-block:: python + + nso_mask = (half_spin_per_esa_step >= nso_half_spin[:, None]) | \ + (half_spin_per_esa_step == HALF_SPIN_FILLVAL) + +i.e. ``half_spin >= NSO_half_spin`` -> NaN, plus anything at a padded (never +sampled) ESA step. ``half_spin_per_esa_step`` is set to ``HALF_SPIN_FILLVAL`` +(63) and ``acquisition_time_per_esa_step`` to NaN in the same places. + +.. warning:: + + **Discrepancy.** For pre-2026-01-29 data the code also NaNs the + ``half_spin == NSO_half_spin`` half-spin, which the document says was still + valid, so one extra half-spin of ESA steps is discarded per NSO cycle. This + rule is unchanged between the January 2026 draft and Rev 3 Chg 1. See + :ref:`codice-implementation-status`. + +Lo priority counts (``codice_l1a_lo_priority.py``) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Section 10.3.5. Summed over all positions; final arrays are +(128 ESA steps x 12 spin sectors), one variable per priority. + +**[DOC]** CMAD adds an explicit NSO rule for this product: + +* **Launch - 2026-01-29:** ``half_spin > NSO_half_spin`` -> NaN. +* **2026-01-29 onward:** the exact-bin rule, identical to the angular + products (section 10.3.4): + + 1. ``half_spin > nso_half_spin`` -> NaN + 2. ``half_spin == nso_half_spin``: + + a. ``spin_sector > nso_spin_sector`` -> NaN + b. ``spin_sector == nso_spin_sector`` and ``esa_step > nso_esa_step`` -> + NaN + + with ``nso_spin_sector`` taken **mod 12**, because it counts over the full + spin (0-23) while ``spin_sector`` is half-spin relative (0-11). + +**[CODE]** When ``packet_version > 1`` the code implements the exact-bin rule +as documented, including the ``% 12``. + +For ``packet_version <= 1`` it uses ``half_spin >= nso_half_spin``. **That +disagrees with the new document rule** (``>``) and with the code's own comment +directly above it ("set all data to NaN where half_spin > nso_half_spin"). +Compare ``codice_l1a_lo_counters_singles.py``, whose pre-FSW branch uses +``>``. See :ref:`codice-implementation-status`. + +Lo instrument counters +^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Section 10.3.1. Two packets. ``AGGREGATED`` carries a selectable +subset of 30 rate types; ``SINGLES`` carries the 24 APD singles. The nominal +aggregated selection is 6 rates - TCR (type 0), DCR (1), Start-A (13), Start-B +(14), Stop (15) and Total Position Count (16 in the document's numbering) - +giving 128 ESA steps x 6 rates x 6 spin sectors. Adjacent 15 deg spin sectors +are paired into 30 deg bins, hence 6 rather than 12. + +**[CODE]** ``LO_COUNTERS_AGGREGATED_VARIABLE_NAMES`` matches that nominal six. +Which counters are present is read from the collapse matrix's ``variables`` dict +via ``get_counters_aggregated_pattern``, so a re-configuration in flight does not +require a code change - but the six variable names are fixed, so a *different* +selection would not be written. + +The singles module applies the same exact-bin NSO rule as priority counts for +``packet_version > 1``, with the extra step of ``nso_spin_sector % 12 // 2`` +to map to the 6 paired spin-sector bins. Its pre-FSW branch uses +``half_spin > nso_half_spin``, which matches the document's priority/angular +rule, **unlike** the priority module. The aggregated module uses +``half_spin >= nso_half_spin`` for all dates, like the species module. Section +10.3.1 gives no counters-specific rule. Section 11.2 says only that +"esa-steps that are in half-spins after NSO_Half_Spin" are NaN, which reads as +``>``. + +Hi omni-directional counts (``codice_l1a_hi_omni.py``) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Section 10.2.3. The product is summed over all spin sectors and SSD +IDs, provided as a function of energy-per-nucleon in **sqrt(2)-spaced** bins. +Nominally summed over every **4 spins** at **1 min** cadence, so there are 4 +values per species in each 16-spin packet. Species: H, He3, He4, C, O, +Ne/Mg/Si, Fe, UH and unknown ("junk"). **The number of energy bins varies by +species.** The CDF must carry each species' energy table with centres and +plus/minus deltas. + +**[DOC]** CMAD names the deltas ``energy__delta_plus`` and +``energy__delta_minus`` (sections 10.2.3 and 10.2.4). **[CODE]** The +variables are ``energy__plus`` and ``energy__minus``, and +``codice_l2.py`` reads them by those names. The meaning is the same; renaming +would be a CDF-schema change touching L1A, L1B and L2. + +**[CODE]** ``n_spins = int(16 / three_d_collapsed)``; each packet's epoch is +expanded into ``n_spins`` epochs: + +.. code-block:: python + + epoch_times = (np.repeat(epoch_center, n_spins) + + np.tile(np.arange(n_spins), num_packets) + * np.repeat(deltas, n_spins) / 1e9 * 2) # TODO: why multiply by 2? + +The ``* 2`` is unexplained and flagged in the source. It is arithmetically the +undoing of the ``// 2`` in ``get_codice_epoch_time`` (which halves the full +window to make a symmetric delta), so the sub-epoch spacing is a full window +rather than a half - which is probably correct, but nobody has written that +down. See :ref:`codice-implementation-status`. + +Hi sectored counts (``codice_l1a_hi_sectored.py``) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Section 10.2.4. Accumulated into **12 spin-sector bins (30 deg)**, +summed over **16 spins (4 min)**, **x2-spaced** energy-per-nucleon bins. +Dimensions :math:`C_j(i, n, k)` = (energy bins, 12 spin sectors, 12 SSD IDs). +Species H, He3He4, CNO, Fe. + +Hi instrument counters +^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Section 10.2.1. Both packets are packed identically but decoded +differently. **Aggregated**: all spin sectors and SSDs summed into a single +value for DCR, STO, SPO, MST and ASIC 1/2 invalid flag events. **Singles**: TCR, +SSDO and STSSD summed over spin sectors but stored **per SSD** (array of 12). +Both summed over 16 spins. + +Rate-type map (**[DOC]**): 0 = TCR (per SSD), 1 = DCR, 2 = STO, 3 = SPO, +4 = SSDO (per SSD), 6 = STSSD (per SSD), 7 = MST, 12 = Low TOF Cutoff, +15 = ASIC 1/2 flag- and channel-invalid counts. Types 5, 8-11, 13-14 reserved. + +Direct events (``codice_l1a_de.py``) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Sections 10.2.2 (Hi) and 10.3.2 (Lo). Bit layouts: + +.. list-table:: + :header-rows: 1 + :widths: 26 12 26 12 24 + + * - Hi field + - Bits + - Lo field + - Bits + - Note + * - SSD Energy + - 11 + - APD Gain + - 1 + - + * - TOF + - 10 + - APD ID + - 5 + - + * - SSD ID + - 4 + - Position + - 5 + - Lo direction comes from ``position``, not ``apd_id`` + * - Energy Range (gain) + - 2 + - APD Energy + - 9 + - + * - Multi-Flag + - 1 + - TOF + - 10 + - + * - PHA Type + - 2 + - Multi-Flag + - 1 + - + * - Spin Sector + - 5 + - PHA Type + - 2 + - Hi spin sector 0-23 + * - Spin Number + - 4 + - Spin Sector + - 5 + - + * - Priority + - 3 + - ESA Step + - 7 + - + * - Spare + - 22 + - Priority + - 3 + - + * - + - + - Spare + - 16 + - + +**[DOC]** Hi gain (``Energy Range``) mapping: 0 = no energy, 1 = low gain, +2 = mid gain, 3 = high gain. **[CODE]** ``GAIN_ID_TO_STR = {1: "LG", 2: "MG", +3: "HG"}``. Per the section 9.2 caveats, LG is currently unusable and only the +upper part of MG is usable; see :ref:`codice-data-caveats`. + +**[DOC]** Three PHA event types: **TCR** (start + stop + SSD), **DCR** (start + +stop, no SSD) and **SSD** (SSD only). Each type populates a different subset of +the fields at different resolutions - a DCR event has no energy, no E range and +no SSD ID. + +**[CODE]** ``DE_DATA_PRODUCT_CONFIGURATIONS`` gives, per APID, ``num_priorities`` +(Hi 6, Lo 8) and the bit structure with ``dtype`` and ``fillval`` per field. +``MAX_DE_EVENTS_PER_PACKET = 10000``. Segmented packets are reassembled using +``SegmentedPacketOrder`` (3 = unsegmented, 1 = first, 0 = continuation, +2 = last), then decompressed with LZMA when the ``compressed`` flag is set, then +bit-unpacked into per-priority arrays with dims ``(epoch, priority, event_num)``. + +**[DOC]** Each priority is written to its own CDF variable with the number of +events and the ``Suspect`` flag stored per priority. diff --git a/docs/source/algorithm-code-documentation/codice/l1b.rst b/docs/source/algorithm-code-documentation/codice/l1b.rst new file mode 100644 index 0000000000..80979effd8 --- /dev/null +++ b/docs/source/algorithm-code-documentation/codice/l1b.rst @@ -0,0 +1,235 @@ +.. _codice-l1b: + +Level 1B - Counts to Rates +========================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**[DOC]** Section 11. **[CODE]** ``codice_l1b.py`` - 219 lines, one function +that does everything. + +L1B is conceptually the simplest level in CoDICE: divide counts by an +accumulation time. All of the difficulty is in knowing **what that accumulation +time is for each product**, which depends on how many spin sectors and spins the +on-board collapse summed together. + +The general form +---------------- + +**[DOC]** + +.. math:: + + R_j = \frac{C_j}{\tau_{FSW}}, \qquad + \tau_{FSW}(n_{sectors}, n_{spins}) = t_{acquire} \cdot n_{sectors} \cdot n_{spins} + +where :math:`t_{acquire}` is the time spent acquiring data in a single 15 deg +spin sector (for Hi) or in a single (ESA step, sector) pair (for Lo). See +:ref:`codice-esa-stepping` for the Appendix C derivation. + +Uncertainties are converted the same way: :math:`\sigma_R = \sigma_C / +\tau_{FSW}`. + +**[CODE]** ``convert_to_rates(dataset, descriptor, cdf_attrs)`` picks the +variable list by name reflection:: + + variables_to_convert = getattr( + constants, f"{descriptor.upper().replace('-', '_')}_VARIABLE_NAMES" + ) + +so ``lo-sw-species`` -> ``LO_SW_SPECIES_VARIABLE_NAMES``. **A descriptor with no +matching constant raises ``AttributeError``** - which is how the missing +products fail. + +Uncertainties are skipped entirely when ``"counters" in descriptor`` +(``calculate_unc = False``), matching the document's treatment of the +engineering counters as diagnostics. + +CoDICE-Hi +--------- + +**[DOC]** Section 11.1. :math:`t_{acquire} = 0.59916` s, configurable via LUT. + +.. list-table:: + :header-rows: 1 + :widths: 30 16 16 38 + + * - Product + - :math:`n_{sectors}` + - :math:`n_{spins}` + - Document reference + * - Instrument rates (both) + - 24 + - 16 + - 11.1.1 + * - Omni-directional species rates + - 24 + - 4 + - 11.1.2 + * - Sectored species rates + - 2 + - 16 + - 11.1.3 + * - Priority rates + - 24 + - 16 + - 11.1.4 + * - I-ALiRT + - 6 + - 4 + - 14.1 + +**[CODE]** ``L1B_DATA_PRODUCT_CONFIGURATIONS`` in ``constants.py`` holds exactly +these values, and the Hi branch is: + +.. code-block:: python + + denominator = ( + constants.L1B_DATA_PRODUCT_CONFIGURATIONS[descriptor]["num_spin_sectors"] + * constants.L1B_DATA_PRODUCT_CONFIGURATIONS[descriptor]["num_spins"] + * constants.HI_ACQUISITION_TIME + ) + +so the denominator is a **scalar** for all Hi products. + +CoDICE-Lo +--------- + +**[DOC]** Section 11.2. Lo is different because :math:`t_{acquire}` **varies +per ESA step** - it depends on how many ESA steps were sampled in the half-spin +that step belonged to. The document writes + +.. math:: + + t_{accum} = t_{acquire} \times 10^{-3} \times n_{sectors} + +with :math:`t_{acquire}` in milliseconds. + +**[CODE]** ``acquisition_time_per_esa_step`` is written at L1A **already in +seconds** as an ``(epoch, esa_step)`` array (``calculate_acq_time_per_step`` +divides by 1e3 before returning), so L1B multiplies by the sector count only. +The denominator is therefore a 2-D array that broadcasts against the species +data. + +.. list-table:: + :header-rows: 1 + :widths: 34 22 44 + + * - Product + - :math:`n_{sectors}` + - Document reference + * - Instrument rates (aggregated, singles) + - 2 + - 11.2.1 - two 15 deg sectors summed into a 30 deg bin + * - Priority rates (SW and NSW) + - 1 + - 11.2.4 + * - Species rates + - **12, except 11 at ESA step 127** + - 11.2.2 + * - Angular rates + - 1 + - 11.2.3 - **not implemented** + * - I-ALiRT + - 12 / 11 + - 14.2.1 + +The flyback step +^^^^^^^^^^^^^^^^ + +**[DOC]** "Due to the ESA stepping scheme, the last spin sector at ESA step 127 +is used as a flyback step and so counts are not accumulated." Hence +:math:`n_{sectors} = 12` for steps 0-126 and **11 for step 127**. + +**[CODE]** + +.. code-block:: python + + n_sector = xr.full_like(dataset.acquisition_time_per_esa_step, 12.0, + dtype=np.float64) + n_sector[:, -1] = 11.0 + denominator = dataset.acquisition_time_per_esa_step * n_sector + +Applied to ``lo-sw-species`` and ``lo-ialirt`` only. + +.. note:: + + Under the post-2025-12-18 stepping scheme the sweep ends at **step 103**, not + 127, so index ``-1`` is a padded, NaN-masked step and the 11-sector + correction lands on a step that carries no data. That is harmless (NaN either + way) but it means the flyback correction is effectively inactive for P2/P3 + data. Confirm with the CoDICE team whether step 103 now needs the correction. + +Energy per charge +----------------- + +**[DOC]** For every Lo product: + +.. math:: + + \mathrm{energy\_table}(l) = \mathrm{voltage\_table}(l) \times k \times 10^{-3} + \quad [\mathrm{keV/e}] + +**[CODE]** Computed once for any descriptor starting with ``lo-``: + +.. code-block:: python + + energy_per_charge = (dataset["voltage_table"].values + * dataset["k_factor"].values * 1e-3) + +and written as both ``energy_per_charge`` and a string +``energy_per_charge_label`` (``f"{value:.3f}"``). The document calls the variable +``energy_table``; the code calls it ``energy_per_charge``. + +``cdf_attrs`` is optional here specifically so that the I-ALiRT pipeline can +call ``convert_to_rates`` for its intermediate values without a CDF attribute +manager. + +Variables dropped at L1B +------------------------ + +**[CODE]** Metadata that only mattered for unpacking is removed: + +* **Lo counters and priority**: ``k_factor``, ``nso_half_spin``, + ``sw_bias_gain_mode``, ``st_bias_gain_mode``, ``spin_period``, + ``voltage_table``, ``nso_esa_step``, ``nso_spin_sector``. +* **Lo species and I-ALiRT**: the same minus ``nso_esa_step`` and + ``nso_spin_sector`` (these two are **kept**, because L2 needs them). +* **All products**: ``spin_period`` if still present. + +``acquisition_time_per_esa_step`` is deliberately **not** dropped, with the +comment ``# TODO: undo this when I get new validation file from Joey``. It is +dropped later, at L2. + +Note that ``rgfo_half_spin``, ``rgfo_esa_step``, ``rgfo_spin_sector``, +``half_spin_per_esa_step`` and ``packet_version`` survive L1B for the Lo species +product - L2 needs every one of them to pick the right geometric factor. + +Housekeeping +------------ + +**[DOC]** Section 11.3: "all voltages and currents are converted to physical +units". + +**[CODE]** This is done by re-running ``packet_file_to_datasets`` with +``use_derived_value=True`` **inside ``process_l1a``**, so the XTCE polynomial +calibrators do the conversion. ``process_codice_l1b`` is never called for +``hskp`` - and could not be, since there is no ``HSKP_VARIABLE_NAMES``. + +Entry point +----------- + +.. code-block:: python + + def process_codice_l1b(file_path: Path) -> xr.Dataset: + l1a_dataset = load_cdf(file_path) + dataset_name = l1a_dataset.attrs["Logical_source"].replace("_l1a_", "_l1b_") + descriptor = dataset_name.removeprefix("imap_codice_l1b_") + ... + l1b_dataset = l1a_dataset.copy(deep=True) + l1b_dataset.attrs = cdf_attrs.get_global_attributes(dataset_name) + return convert_to_rates(l1b_dataset, descriptor, cdf_attrs) + +The descriptor is derived from the **input file's** ``Logical_source``, not from +the CLI ``--descriptor`` argument. L1B is a pure L1A-CDF-in, L1B-CDF-out +transformation with no ancillary inputs. diff --git a/docs/source/algorithm-code-documentation/codice/l2.rst b/docs/source/algorithm-code-documentation/codice/l2.rst new file mode 100644 index 0000000000..25e14bf2fa --- /dev/null +++ b/docs/source/algorithm-code-documentation/codice/l2.rst @@ -0,0 +1,510 @@ +.. _codice-l2: + +Level 2 - Intensities and Physical Units +======================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**[DOC]** Section 12. **[CODE]** ``codice_l2.py`` - ~1570 lines. + +L2 is the lowest level usable for science analysis. It does two different kinds +of work: + +1. **Counts-rate products** -> differential intensity, by dividing by geometric + factor, efficiency and an energy width. +2. **Direct events** -> physical units, by looking up calibration tables. + +Dispatch +-------- + +**[CODE]** ``process_codice_l2(descriptor, dependencies)``: + +.. code-block:: python + + file_path = dependencies.get_file_paths(descriptor=descriptor)[0] + descriptor = ScienceFilePath(file_path).descriptor + dataset_name = f"imap_codice_l2_{descriptor}" + + if dataset_name in ["imap_codice_l2_lo-sw-species"]: + ... # geometric factors + efficiencies + + if dataset_name in [ ... six counters/priority descriptors ... ]: + pass # "no changes needed" + elif dataset_name == "imap_codice_l2_hi-direct-events": + l2_dataset = process_hi_direct_events(dependencies) + elif dataset_name == "imap_codice_l2_hi-sectored": + l2_dataset = process_hi_sectored(dependencies) + elif dataset_name == "imap_codice_l2_hi-omni": + l2_dataset = process_hi_omni(dependencies) + elif dataset_name == "imap_codice_l2_lo-direct-events": + l2_dataset = process_lo_direct_events(dependencies) + + for var in ["acquisition_time_per_esa_step", "rgfo_half_spin", + "half_spin_per_esa_step", "rgfo_esa_step", + "rgfo_spin_sector", "packet_version"]: + if var in l2_dataset.data_vars: + l2_dataset = l2_dataset.drop_vars(var) + +Note the structure: the first ``if`` is separate; the second ``if/elif`` chain +does not have an ``else``. Any descriptor that matches neither leaves +``l2_dataset`` unbound before the drop loop. See +:ref:`codice-implementation-status`. + +The RGFO geometric-factor selection +----------------------------------- + +This is the trickiest logic at L2 and the one that changes with the mission +timeline. **[CODE]** ``compute_geometric_factors(dataset, +geometric_factor_lookup, angular_product=False)``. + +**[DOC]** Rules by period (sections 12.2.2 and 12.2.3): + +**Launch - 2025-11-24 (P0)** + If the half spin for ESA step :math:`l` is **<=** ``RGFO_Half_Spin``, use + **Full** :math:`G_m`; if **>**, use **Reduced**. + +**2025-11-24 - 2026-01-29 (P1, P2)** + The threshold was lowered so RGFO triggers immediately on half spin 0, but the + ESA voltage ratio stayed 1. **Ignore ``RGFO_half_spin`` and always use Full** + :math:`G_m`. + +**2026-01-29 - current (P3+), angular products** + RGFO is now pinned to an exact (ESA step, spin sector). Here ``spin_angle`` is + the de-spun L1B bin (0-23) and ``rgfo_spin_sector`` counts over the full spin + (0-23), so Rev 3 Chg 1 compares everything **mod 12**: + + 1. ``half_spin > rgfo_half_spin`` -> **Reduced** + 2. ``half_spin < rgfo_half_spin`` -> **Full** + 3. ``half_spin == rgfo_half_spin``: + + a. spin-angle bins **less than** ``rgfo_spin_sector mod 12``, and those + plus 12 -> **Full**. E.g. ``rgfo_spin_sector mod 12 = 3`` gives + ``[0, 1, 2, 12, 13, 14]``. + b. spin-angle bins **greater than** it, and those plus 12 -> **Reduced**. + E.g. ``[4..11, 16..23]``. + c. spin-angle bins with ``spin_angle mod 12 == rgfo_spin_sector mod 12``: + ``esa_step <= rgfo_esa_step`` -> **Full**, else **Reduced** + + Worked example from the document: ``rgfo_half_spin = 10``, + ``rgfo_spin_sector = 3``, ``rgfo_esa_step = 19``, with steps [18, 19, 20] + sampled during half-spin 10. Then spin angles **[3, 15]** use Full for steps + [18, 19] and Reduced for step [20]. + + .. note:: + + A draft of the algorithm document printed spin angles **[10, 22]** in this example, + which contradicted its own rule (10 and 22 = 10 + 12 look like the + half-spin number used in place of the spin sector). Rev 3 Chg 1 corrects it to [3, 15] and adds the explicit + ``mod 12``. Two small slips remain in Rev 3 Chg 1: rule (b) omits the + ``mod 12``, and the text says ``spin_sector`` "ranges from 0-24" where + 0-23 is meant. The code (below) applies ``% 12`` throughout, which is the + consistent reading. + +**2026-01-29 - current (P3+), species products** + Because species data is summed over all spin sectors on board, the half-spin + where RGFO triggered **cannot be de-convolved**. Rev 3 Chg 1 spells out the + full rule: + + 1. ``half_spin == rgfo_half_spin`` -> **fill** + 2. ``half_spin > rgfo_half_spin`` -> **Reduced** + 3. ``half_spin < rgfo_half_spin`` -> **Full** + +**[CODE]** The implementation: + +.. code-block:: python + + processing_date = datetime.datetime.strptime( + dataset.attrs["Logical_file_id"].split("_")[4], "%Y%m%d") + date_switch = datetime.datetime(2025, 11, 24) + fsw_switch_date = datetime.datetime(2026, 1, 29) + valid_half_spin = half_spin_per_esa_step != HALF_SPIN_FILLVAL + + if angular_product and dataset.packet_version.data[0] > 1: + # exact-bin rule, as in the document + elif (processing_date < date_switch) | (processing_date >= fsw_switch_date): + modes = valid_half_spin & (half_spin_per_esa_step > rgfo_half_spin) + else: + modes = np.zeros_like(half_spin_per_esa_step, dtype=bool) # always Full + +``modes`` is a boolean array where ``True`` means Reduced. The exact-bin branch +tests ``(spin_sector % 12) > rgfo_spin_sector`` and ``(spin_sector % 12) == +rgfo_spin_sector & esa_step > rgfo_esa_step``, with ``rgfo_spin_sector`` +already reduced ``% 12``. Note: + +* The **date is parsed out of ``Logical_file_id``**, not from the CLI start + date. A missing attribute raises ``ValueError``. +* ``rgfo_spin_sector`` is taken **modulo 12** to bring the packet's 0-23 value + into the half-spin sector range. +* The exact-bin branch is only reachable with ``angular_product=True``, and + **nothing currently passes that**, because the angular products are not + implemented. In practice all Lo L2 processing today uses the half-spin-only + branch. +* The species-product fill of the boundary half-spin **is** implemented, in + ``process_lo_species_intensity``, but unconditionally for all dates rather + than only for P3. + +The lookup tables have shape ``(esa_step, position)`` for each of ``"full"`` and +``"reduced"``. Output shape is ``(epoch, esa_step, inst_az)``, or +``(epoch, esa_step, spin_sector, inst_az)`` when spin-sector information is +present. + +Lo species intensities +---------------------- + +**[DOC]** Section 12.2.3. Three different assumptions apply to three groups. + +**Sunward solar-wind species** - a narrow beam that always lands on azimuthal +position 1, so use the geometric factor for **one** sector: + +.. math:: + + J_j(l) = \frac{R_j(l)}{G_m \cdot \varepsilon_{jl1} \cdot (E/q)_l} + +Species: H+, He++, O5+, O6+, O7+, O8+, C4+, C5+, C6+, Ne, Mg, Si, Fe(low Q), +Fe(high Q). + +**Sunward pickup ions** - isotropic over the sunward range, which is **5** +azimuthal bins: + +.. math:: + + J_j(l) = \frac{R_j(l)}{5 \, G_m \cdot \varepsilon_{jl} \cdot (E/q)_l} + +with :math:`\varepsilon` averaged over the 5 positions. Species: He+ and CNO+. + +**Non-sunward species** - isotropic over the non-sunward range, **19** positions: + +.. math:: + + J_j(l) = \frac{R_j(l)}{19 \, G_m \cdot \varepsilon_{jl} \cdot (E/q)_l} + +with both :math:`G_m` and :math:`\varepsilon` averaged over the 19 positions. +Species: H+, He++, O5-8, C4-6, Ne+Mg+Si, Fe, He+, CNO+. + +Units throughout: particles / (cm2 s sr keV/q). Uncertainties convert +identically. + +**[CODE]** ``calculate_intensity(dataset, species_list, geometric_factors, +efficiency, positions, average_across_positions)`` implements all three via one +expression: + +.. code-block:: python + + if species_list == LO_SW_PICKUP_ION_SPECIES_VARIABLE_NAMES: + geometric_factors = geometric_factors.isel(inst_az=[0]) # position 1 only + else: + geometric_factors = geometric_factors.isel(inst_az=positions) + + if average_across_positions: + geometric_factors = geometric_factors.mean(dim="inst_az") + scalar = len(positions) + else: + scalar = 1 + + denominator = scalar * geometric_factors * species_eff * dataset["energy_per_charge"] + +with position constants + +.. code-block:: python + + SW_POSITIONS = [0, 1, 2, 22, 23] # 0-based -> positions 1,2,3,23,24 + NSW_POSITIONS = list(range(3, 22)) # positions 4-22 + SOLAR_WIND_POSITIONS = [0] # position 1 + PUI_POSITIONS = SW_POSITIONS + +so: + +* solar wind: ``positions=[0]``, ``scalar=1`` -> :math:`G_m` at position 1. +* pickup ions: ``positions=SW_POSITIONS``, ``scalar=5``, :math:`G_m` forced to + position 1 -> :math:`5 G_{m,1}`, efficiency averaged over the 5. Matches the + document, with a ``TODO`` noting the CoDICE team eventually want this + standardised. +* non-sunward: would be ``positions=NSW_POSITIONS``, ``scalar=19`` - **never + called**, because there is no NSW L2 branch. + +``process_lo_species_intensity`` then applies the L2 variable attributes +(``lo-sw-species-attrs`` / ``lo-pui-species-attrs``, with ``{species}`` and +``{species_display}`` substituted by ``apply_replacements_to_attrs``), NaNs the +``half_spin == rgfo_half_spin`` boundary, and restores ``nso_esa_step`` / +``nso_spin_sector`` to ``uint8`` with their fill values. + +Lo angular intensities +---------------------- + +**[DOC]** Section 12.2.2. **[CODE] Not implemented.** + +For reference, the algorithm is: + +.. math:: + + J_j(l, n, k) = \frac{R_j(l, n, k)}{G_m \cdot \varepsilon_{jlk} \cdot (E/q)_l} + +where :math:`k` is position 1-24, :math:`n` is spin-angle index 0-23 and +:math:`l` is ESA step 0-127. :math:`G_m` is delivered as a +(128 ESA steps x 24 APD IDs) array per mode. + +Then the array is transformed from position to elevation angle using the +:ref:`codice-frames` mapping, with an extra step: **positions 1 and 13 see only +half the spin angles**, so their intensities are replicated into the other half +depending on the pixel orientation: + +.. math:: + + \text{SW position 1:} \quad + & J_j(l_A, 12{:}23, 0) = J_j(l_A, 0{:}11, 0), \quad + J_j(l_B, 0{:}11, 0) = J_j(l_B, 12{:}23, 0) \\ + \text{NSW position 13:} \quad + & J_j(l_A, 0{:}11, 9) = J_j(l_A, 12{:}23, 9), \quad + J_j(l_B, 12{:}23, 9) = J_j(l_B, 0{:}11, 9) + +Final shapes: SW (128 E/q x 24 spin angles x **3** elevation angles), NSW +(128 E/q x 24 spin angles x **10** elevation angles). + +**[DOC]** Spin Angle: the product is de-spun at L1A with spin +angle index 0 corresponding to the direction of APDs 2-12, so + +.. math:: + + \theta_{Lo,n} = n \cdot 15^\circ + 7.5^\circ + 270^\circ + +where the 270 deg is the spin angle of APDs 2-12 at spin sector 0. The draft +gave :math:`n \cdot 15^\circ + 7.5^\circ`. + +Hi omni-directional intensities +------------------------------- + +**[DOC]** Section 12.1.3. Assumes isotropy over the full sky: + +.. math:: + + I_j(i) = \frac{R_j(i)}{\left(\sum_k G_k\right) \cdot \epsilon_{ji} + \cdot \Delta E_{ji}} + \quad [\#/(\mathrm{cm}^2\,\mathrm{sr}\,\mathrm{s}\,\mathrm{MeV/nuc})] + +with :math:`G_k = 0.013` cm2 sr per SSD (from a SIMION instrument model), 12 +SSDs, and the energy passband + +.. math:: + + \Delta E_{ji} = E_{i,\mathrm{minus}} + E_{i,\mathrm{plus}} + +**[DOC]** This is explicitly a **pseudo**-omni-directional intensity: each SSD +sees a different solid angle over a spin, and because the species rates are +summed over all angles on board that cannot be corrected from this product. +Appendix B derives the true solid-angle-corrected version from the *sectored* +intensities - **[CODE]** not implemented. (Section 12.1.3 still refers to it +as "Appendix ##", an unresolved cross-reference.) + +**[CODE]** ``process_hi_omni(dependencies)``: + +.. code-block:: python + + geometric_factor = efficiencies_df[efficiencies_df["species"] == "GF"].values[0][-1] + for species in HI_OMNI_VARIABLE_NAMES: + species_efficiencies = species_data["average_efficiency"].values[np.newaxis, :] + energy_passbands = (l1b_dataset[f"energy_{species}_plus"] + + l1b_dataset[f"energy_{species}_minus"]).values[np.newaxis, :] + omni = l1b_dataset[species] / (geometric_factor + * species_efficiencies * energy_passbands) + +Each species has a **different number of energy bins**, which is why this is a +Python loop rather than a vectorised operation. The function also builds the +``energy_`` and ``energy__label`` coordinates (labels formatted +as ``"H int @0.123 MeV/nuc"`` by ``_format_hi_energy_labels``). + +.. warning:: + + The docstring says the denominator is ``geometric_factor * number_of_ssd * + efficiency * energy_passband``, but the code has **no ``number_of_ssd`` + term**. Either the ``GF`` row of ``imap_codice_l2-hi-omni-efficiency_*.csv`` + already contains the summed :math:`\sum_k G_k = 12 \times 0.013 = 0.156`, or + there is a factor-of-12 error. Verify against the delivered CSV before + trusting absolute omni intensities. + +Hi sectored intensities +----------------------- + +**[DOC]** Section 12.1.2: + +.. math:: + + I_j(i, n, k) = \frac{R_j(i, n, k)}{G_k \cdot \epsilon_{jik} \cdot \Delta E_{ji}} + +with :math:`G_k = 0.013` cm2 sr for a single SSD and :math:`\epsilon_{jik}` +depending on species, energy bin, position **and the particle priority scheme**. + +**[CODE]** ``process_hi_sectored(dependencies)`` builds a brand-new +``xr.Dataset`` (rather than mutating the L1B one) with coordinates +``spin_sector``, ``energy_``, ``elevation_angle`` (= +``HI_L2_ELEVATION_ANGLE``) and epoch deltas, then per species: + +.. code-block:: python + + sectored_intensities = l1b_dataset[species] / ( + geometric_factor_da * species_efficiencies * energy_passbands + ) + +Here the geometric factor comes from the ``GF`` **row** of the sectored +efficiency CSV and is per-``inst_az`` (a 12-element vector), which is the correct +per-SSD form. Output dims: ``(epoch, energy_, spin_sector, +elevation_angle)``. + +Spin angle: + +**[DOC]** :math:`\theta_{k,n} = (\theta_{k,0} + 30^\circ n) \bmod 360^\circ` +for :math:`n = 0 \dots 11`. + +**[CODE]** + +.. code-block:: python + + elevation_offsets = np.arange(n_elev)[np.newaxis, :] * 30.0 + base_angles = L2_HI_SECTORED_ANGLE.reshape(n_spin, 1) + spin_angle = ((base_angles + elevation_offsets) % 360.0).T + +which is transposed so that ``spin_angle`` increments by 30 deg **across the +elevation-angle axis** for a fixed spin sector - the CoDICE team confirmed that +as the intended layout. + +.. note:: + + The CMAD labels the columns of this table by **SSD index** (0-11, + the array position), with the SSD ID in a second row, as the direct-event + table already did. The draft used to create this table labelled the sectored table by SSD ID only. The code arrays are indexed the + same way: ``L2_HI_SECTORED_ANGLE`` by index, ``SSD_ID_TO_SPIN_ANGLE`` by ID + with NaN gaps. + +Hi direct events +---------------- + +**[DOC]** Section 12.1.1. Conversions applied *in addition* to the L1A +variables: + +.. list-table:: + :header-rows: 1 + :widths: 26 74 + + * - Field + - Conversion + * - SSD Energy + - To **MeV** using calibration data per SSD and the recorded gain setting. + * - TOF + - To **ns**. + * - Elevation Angle + - From SSD ID, degrees relative to the spin axis. + * - Spin Angle + - From spin sector, :math:`\theta_{n,k} = (\theta_{0,k} + 15^\circ n) + \bmod 360^\circ` for :math:`n = 0 \dots 23`. :math:`\theta_{0,k}` = + 187.50, 146.61, 131.19, 127.50, ... (Rev 3 Chg 1), identical to + ``SSD_ID_TO_SPIN_ANGLE``. + * - Energy per Nuc + - From the TOF channel via a calibration lookup, in **MeV/nuc**. + * - Gain, Multi-Flag, PHA Type, Spin Number + - No change. + +**[CODE]** ``process_hi_direct_events(dependencies)``: + +.. code-block:: python + + elevation[valid_ssd] = SSD_ID_TO_ELEVATION[ssd_id[valid_ssd]] + + cols = ssd_id * 3 + (gain - 1) # energy table columns: ssd0-LG, ssd0-MG, ssd0-HG, ... + ssd_energy_converted[valid_mask] = energy_table[ssd_energy[valid_mask], cols[valid_mask]] + + theta_angles[valid_ssd] = SSD_ID_TO_SPIN_ANGLE[ssd_id[valid_ssd]] + spin_angle = (theta_angles + 15.0 * spin_sector) % 360.0 + + tof_ns[valid_tof_mask] = tof_table[tof[valid_tof_mask], 0] # column 0: ns + energy_nuc[valid_tof_mask] = tof_table[tof[valid_tof_mask], 1] # column 1: MeV/n + +Invalid SSD IDs (2, 6, 10, 14), out-of-range gains and out-of-range TOF indices +all produce NaN. ``rgfo_*``/``nso_*`` and bias-gain-mode variables are dropped. + +Lo direct events +---------------- + +**[DOC]** Section 12.2.1: + +.. list-table:: + :header-rows: 1 + :widths: 26 74 + + * - Field + - Conversion + * - APD Energy + - To **keV** using calibration per APD, based on APD ID, APD Gain and the + ST/SW Bias Gain Mode flag. + * - TOF + - To **ns** using calibration data. + * - Elevation Angle + - From **position**, degrees relative to the spin axis. + * - Spin Angle + - The L1A spin sector (0-23, 15 deg bins, relative to a full spin) is + **"de-spun" at L2 to account for APDs 2-12 and 14-24 viewing opposite + directions**, then :math:`\theta_{Lo,n} = n \cdot 15^\circ + + 277.5^\circ` for :math:`k` = 1-24. + * - ESA Step + - To **keV/e** using the voltage table and analyzer constant. + * - APD Gain, APD ID, Position, Multi-Flag, PHA Type + - No change. + +**[CODE]** ``process_lo_direct_events(dependencies)``: + +.. code-block:: python + + # elevation + pos_to_els = (LO_POSITION_TO_ELEVATION_ANGLE["sw"] + | LO_POSITION_TO_ELEVATION_ANGLE["nsw"]) + elevation_angle = [pos_to_els.get(pos, np.nan) + for pos in l2_dataset["apd_id"].values.flat] + + # spin angle: shift positions 13-24 by half a spin, then convert + l2_dataset["spin_sector"] = xr.where( + (l2_dataset["position"] >= 13) & (l2_dataset["position"] <= 24), + (l2_dataset["spin_sector"] + 12) % 24, + l2_dataset["spin_sector"]) + l2_dataset["spin_angle"] = (l2_dataset["spin_sector"] * 15.0 + 277.5) % 360.0 + + # APD energy: two-stage lookup + col_inds = apd_ids * 2 + gains # columns APD-1-LG, APD-1-HG, ... + energy_bins_inds = energy_table[apd_energy, col_inds] + energy_kev = energy_bins[energy_bins_inds] + + # TOF and E/q from the mpq-cal LUT + tof_ns = tof_bits**2 * ns_channel_sq + tof_bits * ns_channel + tof_offset + esa_kev = esa_v * k_factor / 1000 + +Points to note: + +* **Which field selects the half-spin shift?** The latest version of the CMAD describes the + opposite-facing groups as **APDs** 2-12 and 14-24. The code shifts on + ``position`` 13-24. Apart from the choice of field, the only difference is + whether 13 is shifted, and 13's spin angle is meaningless anyway (section + 4.2). The choice of field matters, though: ``position`` and ``apd_id`` can + disagree for the same event. + Given the section 9.2 caveat that position "is not an accurate identifier of + particle direction", ``apd_id`` looks like the intended field. Confirm with + the team. +* **Elevation is derived from ``apd_id``, but the spin shift uses + ``position``.** The document text still says elevation is "converted from + position". The caveat above favours ``apd_id``, so the elevation lookup may + be the right one and the spin shift the odd one out. Both fields exist + independently in the L1A direct-event record - see + :ref:`codice-implementation-status`. +* **Negative TOF values are set to NaN**, mirroring the L3a handling at Menlo, + because negative TOF is unphysical. +* TOF conversion is a **quadratic** read from ``l2-lo-onboard-mpq-cal``, not the + linear :math:`\tau_{ns} = 0.6217\,\tau_{ch} - 7.4437` printed in the document's + section 13.2.5. The quadratic is the newer form. + +Uncertainties +------------- + +**[DOC]** "The uncertainty in intensity is computed from the rates uncertainties +using the same conversions above" - i.e. the same denominator is applied to +``unc_``. Ultimately these all trace back to +:math:`\sigma = \sqrt{\mathrm{counts}}` at L1A. + +**[CODE]** Implemented for Lo species (with a warning + NaN fill when the +uncertainty variable is missing), Hi omni and Hi sectored. Not applicable to +direct events. diff --git a/docs/source/algorithm-code-documentation/codice/l3-scope.rst b/docs/source/algorithm-code-documentation/codice/l3-scope.rst new file mode 100644 index 0000000000..10ebf4faec --- /dev/null +++ b/docs/source/algorithm-code-documentation/codice/l3-scope.rst @@ -0,0 +1,185 @@ +.. _codice-l3-scope: + +Level 3 - Out of Scope Here +=========================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**This repository stops at L2.** CoDICE L3 products are produced by a separate +repository run closer to the science team. Nothing in section 13 of the +algorithm document should be implemented in ``imap_processing``. + +This page exists so that you can **recognise an L3 request and redirect it**, +and so that you can tell which L2 variables exist purely to feed L3. + +What L3 consists of +------------------- + +**[DOC]** Section 13 and the L3 product table in section 6. + +.. list-table:: + :header-rows: 1 + :widths: 26 20 54 + + * - Product + - File prefix + - Description + * - Lo direct events + - ``imap_codice_l3a_lo-direct-events_`` + - L2 direct events plus per-event **mass** and **mass-per-charge**, and a + normalisation factor derived from the priority counts. + * - Lo SW 3-D VDFs + - ``imap_codice_l3a_lo--3d-distribution_`` + (was ``lo-sw-3d-vdf``) + - Intensity vs (azimuthal sector, spin sector, energy) in the instrument + frame, built from direct events. Binned by **APD ID**, + output 128 E/q x 24 spin angles x 13 elevations. + * - Lo SW partial densities + - ``imap_codice_l3a_lo-partial-densities_`` + - Per-species partial densities from the L2 SW species intensities. + * - Lo SW elemental abundance ratios + - ``imap_codice_l3a_lo-sw-ratios_`` + - C/O, Mg/O, Fe/O at 12 minute cadence. + * - Lo SW charge state ratios + - (shares the partial-densities prefix) + - O7+/O6+, C6+/C4+, C6+/C5+, Fe_low/Fe_high at 12 minute cadence. + * - Lo SW charge state distributions + - ``imap_codice_l3a_lo-sw-charge-state-distributions_`` + - Relative abundances of O charge states 5-8 and C charge states 4-6. + * - Hi direct events + - ``imap_codice_l3_hi-direct-events_`` + - L2 direct events plus energy-per-nucleon and estimated mass. + * - Hi pitch angle distributions + - ``imap_codice_l3_hi-pitch-angle_`` + - Intensity vs (energy, pitch angle) for H, 4He, O, Fe at 4 min cadence. + * - Combined energy-time spectrograms + - (L3c) + - H, He, O, Fe intensities spanning the full Lo + Hi energy range. + * - Combined pitch angle distributions + - (L3c) + - He and O pitch-angle distributions spanning Lo + Hi. + +Why some of it looks like it belongs here +----------------------------------------- + +Three things blur the line. Be alert to them: + +**1. I-ALiRT computes ratios that look like L3a.** + The real-time stream produces C/O, Mg/O, Fe/O, C6+/C5+, O7+/O6+ and + Fe_low/Fe_high - the same quantities as the L3a ratio products. They are + produced **here** because the entire I-ALiRT chain lives in this repository + and must run in under five minutes. They use *pseudo*-densities (constant + factors omitted, since they cancel in a ratio), whereas L3a uses real partial + densities with the unit conversion factor + :math:`C = 2.283 \times 10^{-8}`. **The two are not interchangeable.** See + :ref:`codice-ialirt`. + +**2. The L2 direct-event products carry fields L3 needs.** + ``energy_per_charge``, ``spin_angle``, ``elevation_angle``, + ``energy_per_nuc``, converted ``tof`` and converted energies exist at L2 + precisely so the L3 repository does not have to re-read calibration tables. + Do not remove them because "nothing here uses them". + +**3. Negative TOF filling.** + ``process_lo_direct_events`` NaNs negative TOF values with a comment that it + "mirrors Menlo's L3a handling". That is a deliberate coordination with the L3 + repository, not a stray heuristic. + +Quantities L3 derives (for orientation only) +-------------------------------------------- + +Do not implement these. They are summarised so you can recognise them. + +Lo mass and mass-per-charge (13.2.5) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +.. math:: + + MQ = \frac{2 \left(E/q + PAC - \alpha\right)}{K} + \left(\frac{\tau}{d}\right)^{2} + +with :math:`PAC` = 15 kV (post-acceleration), :math:`\alpha` an approximation +for carbon-foil energy loss (default 0), :math:`\tau` the TOF in ns, :math:`d` = +10.64 cm the path length, and :math:`K` the conversion constant assembled from + +.. math:: + + K = \frac{2 \times 1.602 \times 10^{-16}\,\mathrm{J/keV}} + {1.673 \times 10^{-27}\,\mathrm{kg/AMU}} + \left(\frac{1}{10.64\,\mathrm{cm}}\right)^{2} + \left(\frac{100\,\mathrm{cm}}{\mathrm{m}}\right)^{2} + \left(\frac{1\,\mathrm{s}}{10^{9}\,\mathrm{ns}}\right)^{2} + +Mass comes from an empirical fit in log space: + +.. math:: + + X &= \ln E, \quad Y = \ln \tau \\ + Z &= A_0 + A_1 X + A_2 Y + A_3 XY + A_4 X^2 + A_5 Y^3 \\ + \mathrm{Mass} &= e^{Z} \quad [\mathrm{AMU}] + +with preliminary coefficients :math:`A = [-0.85674, -0.363739, -1.36612, +0.30347, 0.0290419, 0.0557362]`. + +Lo partial densities (13.2.1) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +.. math:: + + dN_j(l) = C \cdot \Delta\vartheta \cdot \Delta\varphi \cdot \frac{\Delta E}{E} + \cdot J_j(l) \cdot (E/q)_l \cdot \sqrt{(m/q)_j}, + \qquad N_j = \sum_{l=0}^{127} dN_j(l) + +with :math:`C = 2.283 \times 10^{-8}` converting to cm-3. The m/q table for the +14 SW species is in section 13.2.1. + +Pitch angles (13.1.2, 13.3.2) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +This is the one place the **instrument -> spacecraft frame rotation** is needed: + +.. math:: + + \theta_{SC} = (\theta_{inst} + 46^\circ) \bmod 360^\circ, \qquad + \phi_{SC} = \phi_{inst} + +.. math:: + + \hat{n}_{SC} = (\cos\theta_{SC}\sin\phi_{SC},\; + \sin\theta_{SC}\sin\phi_{SC},\; + \cos\phi_{SC}) + +.. math:: + + \alpha = \cos^{-1}\!\left(\frac{-\hat{n}_{SC} \cdot \vec{B}_{SC}} + {|\vec{B}_{SC}|}\right) \cdot \frac{180}{\pi} + +binned into 30 deg pitch-angle bins. Note the minus sign: the look direction is +where particles come *from*, so the direction of travel is its negative. The +magnetic field is averaged over the accumulation period, which makes L3 pitch +angles **dependent on MAG data** - a cross-instrument dependency the SDC would +have to broker if it were ever built here. + +L3c combined products (13.3) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +The Hi + Lo combination converts Lo energy-per-charge to energy-per-nucleon +using an assumed charge state and mass, then re-bins Lo onto a **common grid of +20 sqrt(2)-spaced energy-per-nucleon bins** spanning +:math:`E_{min} = 10^{-4}` to :math:`E_{max} \approx 0.1` MeV/n (lower edges +:math:`E_{min} 2^{i/2}`, geometric centres). It then restricts the Hi range to avoid overlap +(H > 0.08, He > 0.04, O > 0.03, Fe > 0.02 MeV/n), time-averages Hi to the 4 min +Lo cadence, and concatenates. + +If you are asked to build any of this +------------------------------------- + +Say no, and point at this page. Then check: + +1. Does the requested quantity actually need an **L2 change** to enable it? For + example, if L3 needs a variable that L2 drops, that *is* work for this + repository. +2. Is the request really about **I-ALiRT**? The ratio products exist in both + places under similar names. +3. Is it a **validation** request - "compare our L2 to their L3"? That is fine + and does not require implementing L3. diff --git a/docs/source/algorithm-code-documentation/codice/overview.rst b/docs/source/algorithm-code-documentation/codice/overview.rst new file mode 100644 index 0000000000..1140c20479 --- /dev/null +++ b/docs/source/algorithm-code-documentation/codice/overview.rst @@ -0,0 +1,780 @@ +.. _codice-overview: + +Instrument Overview +=================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +Everything on this page is background you need before any algorithm page makes +sense. If you only read one thing, read +:ref:`codice-esa-stepping` and :ref:`codice-modes` - those two are where almost +all CoDICE-specific processing complexity comes from. + +What CoDICE measures +-------------------- + +**[DOC]** CoDICE combines an ElectroStatic Analyzer (ESA) with a common +Time-Of-Flight versus energy (TOF-E) subsystem to simultaneously measure: + +1. the 3-D velocity distribution functions and ionic charge state and mass + composition of ~0.5-80 keV/q ions (**CoDICE-Lo**), and +2. the mass composition and arrival direction of ~0.03-5 MeV/nuc ions + (**CoDICE-Hi**). + +The science goals are to determine Local InterStellar Medium composition and +flow properties, and to understand the origin of suprathermal tails and particle +acceleration in the heliosphere. + +The final science products cover three populations: **solar wind heavy ions**, +**pickup ions** (interstellar and inner source), and **suprathermal particles**. + +CoDICE-Lo +^^^^^^^^^ + +**[DOC]** Ions between ~0.5 and 80 keV/q enter a collimated aperture and are +selected by energy-per-charge in the ESA, which focuses them onto a carbon foil +at the entrance to the TOF-E subsystem. + +* The ESA steps through **128 steps over 16 spacecraft spins (~4 minutes)**. +* The carbon-foil subassembly is biased at **-15 kV** (post-acceleration, PAC), + accelerating ions before they strike the ~1 ug/cm2 foil. +* Secondary electrons from the foil go to the outer annulus of the **Start MCP**; + the neutralized ion traverses the flight path and strikes one of **24 APDs**, + whose secondary electrons go to the centre of the **Stop MCP**. +* TOF between Start and Stop gives velocity; the APD gives residual energy E. +* Combined (E/q, TOF, E) determines mass M, charge state q and M/q. +* Arrival direction in azimuth comes from a **delay-line anode** on the Start + MCP ("position"), not from the APD ID - by design (section 4.1). + +.. warning:: + + **[DOC]** The post-launch data caveats (section 9.2) reverse that in + practice: the delay-line position "is not an accurate identifier of particle + direction" and "should only be used to support APD ID". The L3b 3-D VDFs now + bin by APD ID. When the document says "position", check whether the team + now means APD ID. See :ref:`codice-data-caveats`. + +**[DOC]** One Lo azimuth sector always points sunward, so **one sector measures +the solar wind continuously** while the others sweep the sky as the spacecraft +spins. The 24 sectors cover 0 deg and 180 deg and +/-15, 30, 45, 60, 75, 90, +105, 120, 135, 150, 165 deg from the sunward direction. + +CoDICE-Hi +^^^^^^^^^ + +**[DOC]** Ions between ~0.03 and 5 MeV/nuc enter through **12 separate 12 deg x +7 deg FOV collimators** with 150 nm Al-polyimide foils (UV attenuation to +< 0.1%, stops < 10 keV protons). Start/Stop MCP signals come from secondary +electrons produced at a ~1 ug/cm2 carbon foil, and the ion strikes one of **12 +SSDs** (700 um thick, 15 x 15 mm active area) distributed over 360 deg azimuth. +(E, TOF) yields velocity and mass. **CoDICE-Hi covers 1.8 pi sr per spin.** + +The Hi FOV is a cone centred **30 deg above the CoDICE-Lo FOV plane**, so ++/-~20 deg from the sun and anti-sun directions are not covered. + +.. important:: + + **[DOC]** There are 12 SSDs but **16 SSD ID values (0-15)**. The original + design had four dual-pixel SSDs for electrons; that was not flown, but the + flight software still emits 16 IDs. The **valid SSD IDs are 0, 1, 3, 4, 5, 7, + 8, 9, 11, 12, 13, 15**; the remaining four are not valid. + + **[CODE]** ``SSD_ID_TO_ELEVATION`` and ``SSD_ID_TO_SPIN_ANGLE`` in + ``constants.py`` are 16-element arrays indexed by SSD ID with ``np.nan`` at + indices 2, 6, 10 and 14. Any code that indexes by SSD ID must tolerate NaN. + +Vocabulary +---------- + +You will see these subscripts and terms everywhere. The document uses single +letters; the code uses names. + +.. list-table:: + :header-rows: 1 + :widths: 14 12 74 + + * - Doc + - Code + - Meaning + * - :math:`k` + - ``inst_az`` / ``position`` / ``ssd_id`` + - Azimuthal look direction in the instrument frame. Lo: **position 1-24** + (delay-line anode) in most of the document, but **APD ID 1-24** in the + newer text (L1A species split, L3b VDFs) - see + :ref:`codice-data-caveats`. Hi: **SSD ID**. + * - :math:`n` + - ``spin_sector`` / ``spin_angle`` + - Spin phase bin. Lo reports **12 half-spin sectors (0-11)** which de-spin + to **24 spin angles (0-23, 15 deg each)**. Hi reports **24 sectors + (15 deg)** for direct events and **12 sectors (30 deg)** for sectored + counts. + * - :math:`l` + - ``esa_step`` / ``energy_step`` + - ESA energy-per-charge step, **0-127** (Lo only). Step 0 is the highest + energy; the sweep descends. + * - :math:`i` + - ``energy_`` + - Energy-per-nucleon bin index (Hi only). The number of bins **varies by + species**. + * - :math:`j` + - species variable name + - Ion species. + * - :math:`m` + - ``full`` / ``reduced`` + - Geometric-factor mode. See :ref:`codice-modes`. + * - "half spin" + - ``half_spin_per_esa_step`` + - Index 0-31 within the 16-spin cycle. The Lo ESA stepping table is + expressed per half-spin. + * - "collapsing" + - collapse table + - On-board summing of adjacent angle/spin bins to reduce telemetry. All + collapsed regions must be rectangular and contiguous. + * - "view" + - ``view_id`` + - Selects, per APID, the collapse table and compression scheme in use. + * - PHA + - direct events + - Pulse-Height-Analysis event, i.e. a single particle's full record. + +.. _codice-frames: + +Coordinate frames and angle conventions +--------------------------------------- + +**[DOC] CoDICE instrument frame** + +Z points towards the Sun. +X points from the instrument towards the + spacecraft. +Y completes the right-hand rule. The **azimuth angle** + :math:`\varphi` is measured clockwise from +Z when viewing the Y-Z plane from + the +X direction (i.e. looking towards the instrument from the spacecraft). + This is in the plane of the APD FOVs. + +**[DOC] IMAP spacecraft (SC) frame (de-spun)** + +Z is the average spin vector, +X is the North Ecliptic Pole, +Y completes the + right-hand rule. + +**[DOC] Spin phase and the spin-angle reference** + Spacecraft spin phase 0 deg is the moment the **+Y**\ :sub:`SC` axis crosses + the ecliptic plane. The SC-frame spin angle is measured from +X\ :sub:`SC` + (north ecliptic pole) to the detector's central look direction. Because the + instrument frame rotates with the spacecraft, the **instrument-frame spin + angle is measured from a fixed reference: the direction of the +X**\ + :sub:`Co` **axis at spin phase 0**. At spin phase 0, +X\ :sub:`Co` is offset + **~46.0 deg** from +X\ :sub:`SC`. + +**[DOC] Frame conversion** + +.. math:: + + \theta_{SC} = (\theta_{inst} + 46^\circ) \bmod 360^\circ + +The elevation angle is identical in both frames. This applies to both Hi SSDs +and Lo APDs. + +.. note:: + + **[CODE]** The +46 deg rotation is **not applied anywhere in this + repository.** All CoDICE L2 angle variables are in the **instrument frame**. + The conversion appears in the document only in the L3 pitch-angle sections + (13.1.2, 13.3.2), which are out of scope here. Section 4.2 now also says + spin angles "SHOULD" be computed from the look-direction unit vectors with + SPICE and the instrument kernel. The pipeline does not do that; it uses the + tabulated constants below. + +Lo azimuth (position) to angle +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Position 1 is the sunward channel; positions increment clockwise in +15 deg steps: + +.. math:: + + \varphi_k = (k - 1) \times 15^\circ , \quad k = 1 \dots 24 + +Hi SSD azimuth +^^^^^^^^^^^^^^ + +**[DOC]** + +.. list-table:: + :header-rows: 1 + :widths: 20 20 20 20 + + * - SSD ID + - :math:`\varphi_k` + - SSD ID + - :math:`\varphi_k` + * - 0 + - 180 deg + - 8 + - 0 deg + * - 1 + - 210 deg + - 9 + - 30 deg + * - 3 + - 240 deg + - 11 + - 60 deg + * - 4 + - 270 deg + - 12 + - 90 deg + * - 5 + - 300 deg + - 13 + - 120 deg + * - 7 + - 330 deg + - 15 + - 150 deg + +Lo position to elevation angle +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Multiple positions sample the same physical elevation angle at +different spin phases. Products that retain angular information collapse the +24 positions to 13 elevation angles relative to the spin axis: + +**[CODE]** ``LO_POSITION_TO_ELEVATION_ANGLE`` in ``constants.py``, split into +``"sw"`` and ``"nsw"`` sub-dictionaries. + +.. list-table:: + :header-rows: 1 + :widths: 24 14 18 22 22 + + * - Positions + - Product + - Elevation + - SW array index + - NSW array index + * - 1 + - SW + - 0 deg + - 0 + - - + * - 2, 24 + - SW + - 15 deg + - 1 + - - + * - 3, 23 + - SW + - 30 deg + - 2 + - - + * - 4, 22 + - NSW + - 45 deg + - - + - 0 + * - 5, 21 + - NSW + - 60 deg + - - + - 1 + * - 6, 20 + - NSW + - 75 deg + - - + - 2 + * - 7, 19 + - NSW + - 90 deg + - - + - 3 + * - 8, 18 + - NSW + - 105 deg + - - + - 4 + * - 9, 17 + - NSW + - 120 deg + - - + - 5 + * - 10, 16 + - NSW + - 135 deg + - - + - 6 + * - 11, 15 + - NSW + - 150 deg + - - + - 7 + * - 12, 14 + - NSW + - 165 deg + - - + - 8 + * - 13 + - NSW + - 180 deg + - - + - 9 + +**[DOC]** Positions 1 and 13 only observe half of the 24 spin angles for a given +ESA step, because they sit on the spin axis. Angular products must **replicate** +their counts into the unobserved half, choosing 0-11 or 12-23 depending on the +pixel orientation (A or B) of that half-spin. + +Hi SSD to elevation angle +^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** = **[CODE]** ``SSD_ID_TO_ELEVATION`` (indexed by SSD ID) and +``HI_L2_ELEVATION_ANGLE`` (indexed by array position 0-11). + +.. list-table:: + :header-rows: 1 + :widths: 16 16 16 16 16 20 + + * - Index + - SSD ID + - Elevation + - Index + - SSD ID + - Elevation + * - 0 + - 0 + - 150.0 deg + - 6 + - 8 + - 30.0 deg + * - 1 + - 1 + - 138.6 deg + - 7 + - 9 + - 41.4 deg + * - 2 + - 3 + - 115.7 deg + - 8 + - 11 + - 64.3 deg + * - 3 + - 4 + - 90.0 deg + - 9 + - 12 + - 90.0 deg + * - 4 + - 5 + - 64.3 deg + - 10 + - 13 + - 115.7 deg + * - 5 + - 7 + - 41.4 deg + - 11 + - 15 + - 138.6 deg + +Spin angles in the instrument frame +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +All spin angles below are measured from the +X\ :sub:`Co` reference defined in +:ref:`codice-frames` and are bin centres. + +**[DOC] Lo** (section 4.2). The FOV of APDs 1 and 13 is perpendicular to the X-Y +plane, so their spin angle is meaningless. For de-spun data (spin-angle index +:math:`n` = 0-23 spanning 360 deg): + +.. math:: + + \theta_{Lo,n} = n \cdot 15^\circ + 277.5^\circ \quad (\bmod 360^\circ) + +Sections 12.2.2 and 13.2.6 write the same constant as +:math:`7.5^\circ + 270^\circ`: the 270 deg is the spin angle of APDs 2-12 at +spin sector 0. + +**[DOC] Hi** (section 4.2). The SSD FOVs are inclined 30 deg to the Lo FOV +plane, so each SSD sees a different spin angle for the same sector. Using the +look-direction unit-vector components :math:`n_{x,k}, n_{y,k}` for SSD +:math:`k`: + +.. math:: + + \theta_{Hi,0,k} = 180^\circ + \tan^{-1}\!\left(\frac{n_{y,k}}{n_{x,k}}\right) + + 15^\circ + +This is the centre of the first 30 deg (sectored) bin; add 30 deg per sector. +The direct-event table (section 12.1.1, 15 deg bins) is the same value minus +7.5 deg, incremented by 15 deg per sector. The draft used two expressions in +:math:`\varphi_k`, split by SSD group. Rev 3 Chg 1 replaces them with this +single unit-vector form. + + +.. _codice-esa-stepping: + +The Lo ESA stepping scheme +-------------------------- + +This is the single most important CoDICE-Lo concept. + +**[DOC]** The Lo FOV covers the full sky in **one half-spin**, so data is +accumulated on board at half-spin cadence for **32 half-spins = 16 spins ~ 4 +minutes**. Different ESA steps are sampled during each half-spin, from 1 step +per half-spin at the top of the sweep to 6 at the bottom. + +The **pixel orientation** (A for even half-spins, B for odd) describes the APD +FOV orientation relative to the start of a full spin: the spin angles measured +by a given APD are offset by 180 deg between A and B. Because accumulation is at +half-spin cadence, **the instrument always reports spin sector 0-11 regardless +of orientation** - this is what makes de-spinning necessary. + +Terminology, from this point on: + +* **spin sector** = the half-spin-relative bin the instrument reports, 0-11. +* **spin angle** = the physical spin angle, 0-23 indices covering 0-360 deg. + +Two stepping schemes have been used: + +**[DOC] Launch - 2025-12-18** - all 128 steps are sampled, 1/1/1/1 for +half-spins 0-3, then 2 per half-spin for 4-7, 3 for 8-11, 4 for 12-15, 5 for +16-23, and 6 for 24-31 (steps 80-127). + +**[DOC] 2025-12-18 - current** - identical through half-spin 23 (steps 0-79), +then only **3 steps per half-spin** for half-spins 24-31, ending at **step 103**. +Steps 104-127 are never sampled. + +.. note:: + + **[CODE]** The pipeline does not hard-code either table. The per-ESA-step + half-spin number is read from the SCI-LUT + (``lo_stepping_tab["row_number"]["data"]``) and padded to 128 with + ``HALF_SPIN_FILLVAL = 63`` when the table is shorter. Data at a fill-valued + half spin is set to NaN. This is why the second scheme "just works" - the + trailing 24 steps come back padded and get masked. + +De-spinning +^^^^^^^^^^^ + +**[DOC]** After unpacking, Lo count arrays are (128 ESA steps x 5 or 19 +positions x 12 spin sectors). They must be reformatted to (128 ESA steps x 24 +spin angles x 5 or 19 positions) using: + +.. list-table:: + :header-rows: 1 + :widths: 16 14 18 18 18 16 + + * - Half-spin parity + - Orientation + - Positions + - L0 spin sector + - L1A spin angle + - Position index + * - Even + - A + - 1-12 + - 0-11 + - 0-11 + - SW 0-2 / NSW 0-9 + * - Even + - A + - 13-24 + - 0-11 + - 12-23 + - SW 3-4 / NSW 10-18 + * - Odd + - B + - 1-12 + - 0-11 + - 12-23 + - SW 0-2 / NSW 0-9 + * - Odd + - B + - 13-24 + - 0-11 + - 0-11 + - SW 3-4 / NSW 10-18 + +**[CODE]** ``LO_DESPIN_SPIN_SECTORS = 24``. This full de-spin mapping is only +required for the **angular** products, which are not implemented (see +:ref:`codice-implementation-status`). The species and priority products are +summed over position on board and keep the instrument's 0-11 (or fully +collapsed) spin-sector dimension. + +Acquisition timing +------------------ + +.. note:: + + In the CMAD (pages 623-709), Appendix C contains **only the first table** + (values common to Hi and Lo). The Hi :math:`t_{acquire}` derivation and the + Lo equations below are transcribed from the January 2026 draft, where they + continued onto the next page. Sections 11.1 and 11.2 of Rev 3 Chg 1 still + quote :math:`t_{acquire} = 0.59916` s for Hi and still point to "Appendix + B" (a stale cross-reference; the timing appendix is C). The section 7 + pseudocode for Lo timing is unchanged and still omits the + :math:`t_{minHvSettle}` floor. + +**[DOC]** Appendix C. Values common to both sensors, all in microseconds unless +noted: + +.. math:: + + t_{sectorTime} = \mathrm{int}\!\left(\frac{P_{spinPeriod} \times 320} + {\mathrm{NumSectors}}\right) + +* ``NumSectors`` = 24 (configurable, but should not change). +* ``P_spinPeriod`` is in **spin ticks of 320 us**, initially 45687 ticks + (14.62 s, deliberately shorter than the real spin), range 45687-48031. + Sector times range 0.609-0.640 s; the baseline 15 s spin gives 0.625 s. +* ``t_sectorMargin`` nominal 5000 us - FSW dead time at the end of each sector. +* ``t_minHvSettle`` nominal 5000 us, ``t_maxHvSettle`` nominal 100000 us. + +**CoDICE-Hi:** + +.. math:: + + t_{acquire} = (t_{sectorTime} - t_{sectorMargin} - t_{minHvSettle}) + \times 10^{-6} = 0.59916\ \mathrm{s} + +**[CODE]** ``HI_ACQUISITION_TIME = 0.59916`` in ``constants.py``, hard-coded. +Hi does not actually need HV settling, but the same collection algorithm is used +for both sensors. + +**CoDICE-Lo:** the number of ESA steps per sector varies, so: + +.. math:: + + t_{acquire} = \left(\frac{t_{sectorTime} - t_{sectorMargin}} + {\mathrm{NumAcqSteps}} - t_{hvSettlePerStep}\right) \times 10^{-3}\ \mathrm{[ms]} + +where + +.. math:: + + t_{nonAcquire} &= \mathrm{int}\!\left(\frac{t_{sectorTime} + \times (100 - P_{dwellFraction})}{100}\right) \\ + t_{totalHvSettle} &= t_{nonAcquire} - t_{sectorMargin} \\ + t'_{hvSettlePerStep} &= \mathrm{int}\!\left(\frac{t_{totalHvSettle}} + {\mathrm{NumAcqSteps}}\right) \\ + t_{hvSettlePerStep} &= \max\!\left(t_{minHvSettle},\ + \min(t'_{hvSettlePerStep},\ t_{maxHvSettle})\right) + +``P_dwellFraction`` is nominally **95%** (to meet the L3 requirement of +collecting 95% of the time). ``NumAcqSteps`` varies 1-6 and is looked up from +the SCI-LUT Lo Stepping table. + +**[CODE]** ``utils.calculate_acq_time_per_step`` implements exactly this, reading +``lo_stepping_tab["tunable_values"]`` (``spin_time_ms``, ``num_sectors_ms``, +``sector_margin_ms``, ``dwell_fraction_percentage``, ``min_hv_settle_ms``, +``max_hv_settle_ms``) and ``lo_stepping_tab["num_steps"]["data"]``. It returns a +128-element array **in seconds**, padded with NaN, and is written to L1A as +``acquisition_time_per_esa_step``. + +Energy per charge +----------------- + +**[DOC]** ESA sweep table entries are voltages. Energy-per-charge is: + +.. math:: + + E/q\ [\mathrm{keV/e}] = V \times k \times 10^{-3}, \qquad k = 5.76 + +**[CODE]** ``K_FACTOR = 5.76``. The voltage table for the active +``(plan_id, plan_step)`` is read from ``esa_sweep_tab`` in the SCI-LUT and +written to L1A as ``voltage_table``; ``energy_per_charge`` is derived at L1B. + +.. _codice-modes: + +RGFO and NSO operating modes +---------------------------- + +**[DOC]** Both are on-board count-rate protections on CoDICE-Lo. Neither applies +to CoDICE-Hi. + +RGFO - Reduced Geometric Factor Operation +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +The ratio of voltages on the upper and lower ESA plates is reduced from 1, +cutting the number of ions that pass the ESA. **The geometric factor changes**, +so the ground must know which :math:`G_m` (Full or Reduced) applied to every +bin. RGFO persists until NSO triggers or the 32 half-spin cycle ends. In the +launch configuration the FSW triggers RGFO on the **total counts on the +START-A and START-B MCP** over a half-spin (~7.5 s), with a limit tunable in +on-board LUTs. Section 4.1 gives the nominal reduction as the entrance ESA +running at 70% of the main ESA voltage (configurable). + +NSO - No-Scan Operation +^^^^^^^^^^^^^^^^^^^^^^^ + +The ESA voltage is pinned to the highest value (step 0) and stays there until +the end of the cycle. **No further energy stepping happens**, so the summed +counts for the affected bins are not representative and **must be set to fill**. +NSO can only trigger while already in RGFO. + +Telemetry +^^^^^^^^^ + +Every COUNTS packet carries ``RGFO_Half_Spin`` and ``NSO_Half_Spin`` (6 bits +each, 0-31). After the 2026-01-29 FSW load, both COUNTS **and PHA** packets also +carry ``RGFO_spin_sector``, ``RGFO_esa_step``, ``NSO_spin_sector`` and +``NSO_esa_step``, which pin the trigger to an exact bin rather than a half-spin. + +**[CODE]** The FSW-version split is handled by having **two XTCE files**; +``codice_l1a.process_l1a`` picks one based on the date parsed out of the L0 +filename: + +.. code-block:: python + + if start_date >= datetime.datetime(2026, 1, 29): + xtce_file = path / "imap_codice_packet-definition_20260129_v001.xml" + else: + xtce_file = path / "imap_codice_packet-definition_20250101_v001.xml" + +Within a product, the branch is taken on ``packet_version`` (``<= 1`` = old +behaviour, ``> 1`` = exact-bin behaviour). When the new fields are absent they +are still created in L1A, filled with NaN, for SPDF consistency. + +.. _codice-timeline: + +Commissioning timeline (section 9.1) +------------------------------------ + +**[DOC]** Instrument behaviour has changed four times, giving five periods +(P0-P4). **Any algorithm that touches RGFO, NSO or the ESA sweep must branch on +the date.** This table is the authority for those branches. The document marks +the last period "Current" as of its publication and says to check with the +instrument team for later changes. + +.. list-table:: + :header-rows: 1 + :widths: 8 22 70 + + * - Period + - Dates + - Behaviour + * - **P0** + - Launch - 2025-11-24 + - RGFO triggers on total counts accumulated over a half spin; if the limit + is passed the instrument enters RGFO on the **next half spin**. NSO + triggers the same way while in RGFO. + * - **P1** + - 2025-11-24 - 2025-12-18 + - RGFO limit reduced so that **RGFO always triggers on half spin 0**, but + the ESA voltage ratio remains 1, so the instrument is **not actually + reduced-G**. ``RGFO_half_spin`` must be **ignored** and Full :math:`G_m` + used. NSO still triggers on half-spin boundaries. + * - **P2** + - 2025-12-18 - 2026-01-29 + - ESA sweep table updated: only **3 ESA steps per half-spin from step 80** + (576 V, 3.3 keV/e). RGFO/NSO as P1. + * - **P3** + - 2026-01-29 - 2026-04-03 + - FSW v1.5. RGFO/NSO trigger on **count rate within a single (ESA step, + spin sector)** pair, switching on the following pair. New spin-sector and + e-step fields added to all COUNTS and PHA packets. Hi/Lo priority counts + are **lossless-only** (no lossy). **Fe highQ/lowQ labels swapped** + (affects Lo I-ALiRT view 0, Lo SW species view 5, Lo SW angular view 7 - + which also changed from species 13 to species 14). Two species added to + Lo angular counts (SW He+ view 7 APID 0x486, NSW He+ view 8 APID 0x487). + PHA allocation increased: Lo 5760 -> 11520 and Hi 5000 -> 10000 events + per cycle. Hi priority scheme updated to P5 = Heavies, P4 = Helium, + P3 = Protons. + * - **P4** + - 2026-04-03 - current + - **Lo SCI_LUT update**: fixed an indexing issue that assigned species + rates to the wrong rate box. **Lo on-board species classification LUT + update**: refined mass boundaries for key species; binned **H+ is now + TCRs only** (it was DCRs during P0-P3). **Hi on-board species + classification LUT update**: fixed a bug that sent low-TOF events to low + priority. Only LUT updates are listed; no FSW update. + +.. note:: + + The document lists **only LUT changes for P4, no FSW update**, so a new XTCE + file or ``packet_version`` bump is not expected. (That is an inference; the + document does not say so.) On the ground P4 arrives as a new + ``l1a-sci-lut`` table. It does change what the Lo species (and H+) counts + *mean* before and after 2026-04-03. The code has no date branch for P4, and + none is needed for unpacking. Anyone comparing Lo species across that date + should know about it. 2026-04-03 is also a mid-mission SCI-LUT change, the + case the ``codice_l2.py`` TODO about SCI-LUT change days warns about (see + :ref:`codice-implementation-status`). + +.. warning:: + + **[CODE]** The Fe highQ/lowQ label swap is handled indirectly: L1A reads the + species ordering from the SCI-LUT (``desired_species_names`` vs + ``actual_species_names``) and warns + fills with NaN when a wanted species is + absent. The comment in ``codice_l1a_lo_species.py`` calls this out as + handling "the bug in which the spacecraft was sending data down 'off by one' + and getting mislabeled". If you see unexplained NaN species columns, check + the SCI-LUT version first. + +.. _codice-data-caveats: + +Data caveats (section 9.2) +-------------------------- + +**[DOC]** New in Rev 3 Chg 1. "Due to issues identified post-launch", these +three Lo products are **not currently produced at any level (L1a-L2)**: +``lo-nsw-species``, ``lo-sw-angular``, ``lo-nsw-angular``. That matches the +repository, where none of the three is built. See +:ref:`codice-implementation-status`. + +The caveats below apply to all CoDICE L2 data in the **first official IMAP data +release (Summer 2026)**. The document says most of them will be fixed or +mitigated in later releases. They describe the *data*, not the ground code, so +nothing here is a pipeline bug. Keep them in mind when validating output or +answering "why does this look wrong". + +.. list-table:: + :header-rows: 1 + :widths: 20 40 40 + + * - L2 product + - Caveat + - Effect on data + * - ``lo-sw-species`` + - Counts are binned into SW/NSW products using **delay-line position + instead of APD ID** (on board). + - Lower counts than expected, so lower intensities. + * - ``lo-sw-species`` + - Species classification boxes need fine tuning. + - Charge-state species are not well separated / identified. + * - ``lo-direct-events`` + - Delay-line position is **not an accurate identifier of particle + direction**; position should only be used to support APD ID. + - (as stated) + * - ``lo-direct-events`` + - Automatic TOF correction based on position. + - Larger TOF uncertainty. + * - ``hi-omni``, ``hi-sectored`` + - Detection threshold depends on SSD ID. + - Increased noise at lower energies. + * - ``hi-omni``, ``hi-sectored`` + - ASIC baseline fluctuates with count rate, producing pseudo-TCRs for + low-energy data. + - Increased noise / false counts at lower energies. + * - ``hi-omni``, ``hi-sectored`` + - On-board mass computation needs correction. + - Mass tracks trend downward at low energy-per-nuc. + * - ``hi-omni``, ``hi-sectored``, ``hi-direct-events`` + - MG and LG energy channels not well calibrated. + - Upper part of MG usable; lower part of MG and **all of LG not currently + usable**. + +.. note:: + + Two of these bear on ground-code choices already documented here: + + * The Lo **position vs APD ID** caveat is relevant to + ``process_lo_direct_events``, which derives elevation from ``apd_id`` but + de-spins spin sectors using ``position``. See + :ref:`codice-implementation-status`. + * The Hi **LG/MG** caveat means ``GAIN_ID_TO_STR`` values 1 (LG) and 2 (MG) + label events whose energies are, for now, partly or wholly unreliable. The + pipeline still converts them; it does not flag them. + +Operational modes +----------------- + +**[DOC]** CoDICE has four operational modes: **SAFE**, **LVENG** (low-voltage +engineering), **HVENG** (high-voltage engineering) and **Science**. Everything +in the algorithm document and in this pipeline applies to Science mode only. diff --git a/docs/source/algorithm-code-documentation/codice/reference-tables.rst b/docs/source/algorithm-code-documentation/codice/reference-tables.rst new file mode 100644 index 0000000000..622bc51077 --- /dev/null +++ b/docs/source/algorithm-code-documentation/codice/reference-tables.rst @@ -0,0 +1,263 @@ +.. _codice-reference-tables: + +Reference Tables - Where to Look Them Up +======================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +The algorithm document's large tables are **deliberately not reproduced** here: +they go stale, and in almost every case a machine-readable version already +exists that the code actually reads. + +The document itself is **not in this repository** - see +:ref:`codice-source-documents`. CoDICE is unusual in that the single most +important table set - the SCI-LUT - is neither in the document nor in this +repository: it is an operational ancillary file that changes in flight. + +Rule of thumb +------------- + +.. list-table:: + :header-rows: 1 + :widths: 44 56 + + * - If you need... + - Go to + * - A packet field's name, bit offset, width or type + - ``imap_processing/codice/packet_definitions/imap_codice_packet-definition_20250101_v001.xml`` + (pre-2026-01-29) or ``..._20260129_v001.xml`` (post) + * - A housekeeping field or its engineering-unit calibrator + - ``imap_processing/codice/packet_definitions/P_COD_NHK.xml`` + * - The plan / view / collapse / ESA sweep / Lo stepping tables + - The ``l1a-sci-lut`` JSON ancillary file. A test copy is at + ``imap_processing/tests/codice/data/l1a_lut/imap_codice_l1a-sci-lut_20251007_v005.json`` + (and ``..._20260129_v002.json``). Structure documented in + :ref:`codice-ancillary`. + * - The lossy A/B decompression tables + - ``LOSSY_A_TABLE`` / ``LOSSY_B_TABLE`` in + ``imap_processing/codice/constants.py`` (256 entries each) + * - APIDs + - ``CODICEAPID`` in ``imap_processing/codice/constants.py`` + * - Species lists, position groupings, angle tables, acquisition time + - ``imap_processing/codice/constants.py`` - see the inventory on + :ref:`codice-data-products` + * - Direct-event bit layouts + - ``DE_DATA_PRODUCT_CONFIGURATIONS`` and ``DE_METADATA_FIELDS`` in + ``constants.py`` + * - The I-ALiRT packet layout + - ``IAL_BIT_STRUCTURE`` in ``constants.py``, plus + ``imap_processing/ialirt/utils/constants.py`` + * - A CDF variable's units, fill value, valid range or description + - ``imap_processing/cdf/config/imap_codice_l1a_variable_attrs.yaml``, + ``imap_codice_l1b_variable_attrs.yaml`` and the six + ``imap_codice_l2-*_variable_attrs.yaml`` files + * - The complete list of CoDICE products + - ``imap_processing/cdf/config/imap_codice_global_cdf_attrs.yaml`` + * - Geometric factors or efficiencies + - The ``l2-lo-gfactor`` / ``l2-*-efficiency`` ancillary CSVs; test copies + in ``imap_processing/tests/codice/data/l2_lut/`` + * - The spin-angle reference values (now equal to Rev 3 Chg 1; 90 deg below + older drafts) + - See ``imap_processing/tests/codice/test_codice_spin_angles.py`` + * - The commissioning timeline / RGFO / NSO rules + - :ref:`codice-timeline` and :ref:`codice-modes` - transcribed + * - The instrument team's known data caveats + - :ref:`codice-data-caveats` - transcribed + * - The acquisition-time equations (Appendix C) + - :ref:`codice-esa-stepping` - transcribed + * - Anything else + - the algorithm document, using the section index below + +Machine-readable tables in the repository +------------------------------------------ + +Packet definitions +^^^^^^^^^^^^^^^^^^ + +Two XTCE files, selected by date in ``codice_l1a.process_l1a``: + +.. code-block:: text + + imap_codice_packet-definition_20250101_v001.xml ~335 KB, pre-2026-01-29 FSW + imap_codice_packet-definition_20260129_v001.xml ~372 KB, post-2026-01-29 FSW + P_COD_NHK.xml ~37 KB, housekeeping only + +The 2026-01-29 file adds ``RGFO_SPIN_SECTOR``, ``RGFO_ENERGY_STEP``, +``NSO_SPIN_SECTOR`` and ``NSO_ENERGY_STEP`` to the COUNTS and PHA packets. + +.. code-block:: bash + + # every parameter defined for one APID + grep -o 'name="COD_LO_SW_SPECIES_COUNTS\.[^"]*"' \ + imap_processing/codice/packet_definitions/imap_codice_packet-definition_20260129_v001.xml + + # diff the two FSW versions + diff <(grep -o 'name="[^"]*"' .../20250101_v001.xml | sort -u) \ + <(grep -o 'name="[^"]*"' .../20260129_v001.xml | sort -u) + +CDF metadata +^^^^^^^^^^^^ + +``imap_processing/cdf/config/``: + +* ``imap_codice_global_cdf_attrs.yaml`` - **the definitive product list.** Every + ``Logical_source`` CoDICE can emit has an entry here; if a string is not in + this file, ``get_global_attributes`` raises. 41 entries: 18 L1A, 15 L1B, + 8 L2. +* ``imap_codice_l1a_variable_attrs.yaml`` - L1A counts and metadata variables, + including the ``lo-species-attrs`` / ``lo-species-unc-attrs`` templates with + ``{species}`` and ``{direction}`` placeholders. +* ``imap_codice_l1b_variable_attrs.yaml`` - rates, ``energy_per_charge``. +* ``imap_codice_l2-lo-species_variable_attrs.yaml`` - the ``lo-sw-species-attrs``, + ``lo-pui-species-attrs`` and matching ``-unc-attrs`` templates. +* ``imap_codice_l2-lo-angular_variable_attrs.yaml`` - **written, but the product + is not.** +* ``imap_codice_l2-lo-direct-events_variable_attrs.yaml`` +* ``imap_codice_l2-hi-omni_variable_attrs.yaml`` +* ``imap_codice_l2-hi-sectored_variable_attrs.yaml`` +* ``imap_codice_l2-hi-direct-events_variable_attrs.yaml`` + +.. code-block:: bash + + grep 'Logical_source:' imap_processing/cdf/config/imap_codice_global_cdf_attrs.yaml + +Placeholder substitution is done by ``utils.apply_replacements_to_attrs``, which +uses plain ``str.replace`` rather than ``str.format`` so that braces elsewhere in +a template do not raise. + +Validation data +^^^^^^^^^^^^^^^ + +``imap_processing/tests/codice/data/``: + +.. code-block:: text + + l0_data/ imap_codice_l0_raw_20241110_v001.pkts + l1a_input/ per-product .pkts slices, dated 20250814 + + imap_codice_l0_raw_20260130_v001.pkts (FSW change) + l1a_lut/ SCI-LUT JSON, v005 (20251007) and v002 (20260129) + l1b_validation/ per-product L1B CDFs, 20250814 v015 + l2_lut/ the seven L2 ancillary CSVs + +Pinned in ``conftest.py`` as ``VALIDATION_FILE_DATE = "20250814"`` and +``VALIDATION_FILE_VERSION = "v015"``. The ``codice_lut_path`` fixture maps +``(descriptor, data_type)`` to a path and **raises on anything unknown** - the +quickest way to enumerate what a code path needs. + +Algorithm document section index +-------------------------------- + +Section numbers are those of **Rev 3 Chg 1**, as embedded in the CMAD +(``IMAP_CMAD_20260722.pdf`` section 4.3.2, PDF pages 623-709). To find a +printed body page in the CMAD, add 628 to it (printed page 1 = CMAD page 629). +The front matter is on CMAD pages 625-628, and Appendices A, B and C start on +CMAD pages 704, 705 and 709. Printed page numbers moved by one or two pages +relative to the January 2026 draft, and sections 4-5 were re-numbered as shown. + +.. list-table:: + :header-rows: 1 + :widths: 12 52 36 + + * - Section + - Title + - Covered on + * - 1-3 + - Introduction, product overview, heritage instruments + - :ref:`codice-overview` + * - 4.1 + - Physical description + - :ref:`codice-overview` + * - 4.2 + - Angular mappings in sensor coordinates (look-direction unit vectors, + azimuth tables, spin-phase / +X\ :sub:`Co` reference, spin-angle tables, + 46 deg SC frame offset) + - :ref:`codice-frames` + * - 5.1 + - Algorithm description: the four packet IDs and the SCI-LUT (was 4.3 + "Algorithm Input" in the draft) + - :ref:`codice-l1a`, :ref:`codice-ancillary` + * - 5.2 + - Processing pipeline (was 5.1) + - :ref:`codice-data-products` + * - 5.3 + - Data validation: pre-launch synthetic-data validation, independent + re-implementation by the SDC, post-launch re-validation on flight data. + - - + * - 6 + - Data products overview (Tables 1-4) + - :ref:`codice-data-products` + * - 7 + - Common processing algorithms: CCSDS header, packet structures, + compression, the unpacking algorithm, acquisition timing + - :ref:`codice-l1a` + * - 8 + - Worked unpacking example + - :ref:`codice-l1a` + * - 9.1 + - Instrument operation timeline and changes (P0-P4) + - :ref:`codice-timeline` + * - 9.2 + - Data caveats for the Summer 2026 release; products not being produced + - :ref:`codice-data-caveats` + * - 10.1 + - L1A housekeeping + - :ref:`codice-l1b` + * - 10.2 + - L1A CoDICE-Hi (counters, direct events, omni, sectored, priority) + - :ref:`codice-l1a` + * - 10.3 + - L1A CoDICE-Lo (counters, direct events, species, angular, priority), + ESA stepping tables, de-spinning + - :ref:`codice-l1a`, :ref:`codice-esa-stepping` + * - 10.4 + - L1A I-ALiRT packet formats + - :ref:`codice-ialirt` + * - 11 + - L1B rates for both sensors + - :ref:`codice-l1b` + * - 12.1 + - L2 Hi: direct events, sectored intensities, omni intensities + - :ref:`codice-l2` + * - 12.2 + - L2 Lo: direct events, angular intensities, species intensities, RGFO + rules + - :ref:`codice-l2` + * - 13 + - L3 (all) + - :ref:`codice-l3-scope` - **out of scope** + * - 14 + - I-ALiRT rates, intensities, pseudo-densities, ratios + - :ref:`codice-ialirt` + * - 15 + - Operation modes + - :ref:`codice-overview` + * - App. A + - Acronyms + - - + * - App. B + - Hi true omni-directional intensity (solid-angle correction) + - :ref:`codice-implementation-status` - **not implemented** + * - App. C + - Spin-sector acquisition times. **Truncated in the CMAD** after the + common-values table; the Hi/Lo equations are only in an older draft of the algorithm document. + - :ref:`codice-esa-stepping` - transcribed + +Things you genuinely need the document for +------------------------------------------- + +Everything else on this page has a machine-readable equivalent. These do not: + +* **The look-direction unit vectors** (section 4.2) for all 12 SSDs and 24 APDs + in the instrument frame. Not in the code - the document notes they can be + obtained from SPICE with the instrument kernel instead. +* **The full Hi spin-angle table** for all 12 SSDs x 12 spin sectors (section + 4.2). The code stores only the sector-0 reference column and increments by + 30 deg. +* **The 30 Lo aggregated rate types and 15 Hi rate types** (sections 10.3.1 and + 10.2.1). Only the nominal subsets are named in ``constants.py``. +* **The PHA event-type field tables** (section 10.2.2) showing which fields are + populated for TCR vs DCR vs SSD events at what resolution. +* **Appendix B** in full - the solid-angle derivation and the + :math:`\Omega_k` / :math:`\Omega_{12,k}` table. +* **Section 13** in full, if you need to understand what the L3 repository + expects from our L2 products. diff --git a/docs/source/algorithm-code-documentation/glows.rst b/docs/source/algorithm-code-documentation/glows.rst index b0d4d22023..84a03f291b 100644 --- a/docs/source/algorithm-code-documentation/glows.rst +++ b/docs/source/algorithm-code-documentation/glows.rst @@ -27,4 +27,4 @@ data from the GLOWS instrument. l0.decom_glows l0.glows_l0_data l1a.glows_l1a_data - utils.constants + utils.constants \ No newline at end of file diff --git a/docs/source/algorithm-code-documentation/glows/ancillary.rst b/docs/source/algorithm-code-documentation/glows/ancillary.rst new file mode 100644 index 0000000000..1cee9312ca --- /dev/null +++ b/docs/source/algorithm-code-documentation/glows/ancillary.rst @@ -0,0 +1,335 @@ +.. _glows-ancillary: + +Ancillary and Settings Files +============================ + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +GLOWS is unusually dependent on instrument-team-supplied files. Almost every +decision the pipeline makes about *which data to trust* comes out of one of +these, and none of the science-relevant numbers are hard-coded in the repository. + +**[DOC §3.12]** defines nine files. Seven are used at or below L2; two are +L3-only and are listed here only so you recognise them. + +The canonical list +------------------ + +Names follow ``imap_glows__YYYYMMDD_vXXX.``. + +.. list-table:: + :header-rows: 1 + :widths: 5 34 10 51 + + * - # + - File / descriptor + - Level + - Purpose + * - 1 + - ``l1b-map-of-excluded-regions`` (``.dat``) + - L1B + - Points in the sky (ecliptic J2000) that densely cover regions to be + excluded - principally the neighbourhood of the galactic plane. Drives + ``is_inside_excluded_region``. + * - 2 + - ``l1b-conversion-table-for-anc-data`` (``.json``) + - L1B + - ``min``/``max``/``n_bits`` (and unused ``p01``-``p04``) for decoding + filter temperature, HV voltage, spin period, spin phase and pulse length + from integers to physical units. + * - 3 + - ``l1b-map-of-uv-sources`` (``.dat``) + - L1B + - Catalogue of bright UV point sources with a **per-source masking + radius**. Drives ``is_close_to_uv_source``. + * - 4 + - ``l1b-exclusions-by-instr-team`` (``.dat``) + - L1B + - ``unique_block_identifier`` + a 0/1 mask string per block, for anything + the GLOWS team wants excluded by hand. Drives + ``is_excluded_by_instr_team``. + * - 5 + - ``l1b-suspected-transients`` (``.dat``) + - L1B + - Same format; for bins likely to carry transient signal (comets, etc.). + Drives ``is_suspected_transient``. + * - 6 + - ``l2-calibration`` (``.dat``) + - L2 + - ``start_time_UTC cps_per_R``. Each value applies from its start time + until the next; the last row is open ended. + * - 7 + - ``l3a-map-of-extra-helio-bckgrd`` (``.dat``) + - L3A + - All-sky HEALPix map of the time-independent extra-heliospheric + background. **Not used in this repository.** + * - 8 + - ``l3a-time-dep-bckgrd`` (``.dat``) + - L3A + - Per-pointing background corrections on a 3600-bin grid, to be linearly + interpolated onto the L3A grid. **Not used in this repository.** + * - 9 + - ``pipeline-settings`` (``.json``) + - L1B, L2 + - Thresholds, the two active-flag masks, day/night offsets, the number of + L3A bins. The control panel for the whole pipeline. + +Document §4.15 additionally lists L3A-to-L3E inputs - +``imap_glows_bad_days_list``, ``imap_glows_WawHelioIonMP``, +``imap_glows_uv-anisotropy-*``, ``imap_glows_p-density-*``, +``imap_glows_sw-speed-*``, ``imap_glows_lya-*``, ``imap_glows_phion-*``, +``imap_glows_e-density-*``. None of these are relevant here. + +How they are loaded +------------------- + +**[CODE]** ``GlowsAncillaryCombiner`` in +``imap_processing/ancillary/ancillary_dataset_combiner.py``, a subclass of the +generic ``AncillaryCombiner``. The CLI constructs one per descriptor with a +3-day end-date buffer, and passes ``.combined_dataset`` into the processing +function. The combiner handles time-ranged, versioned ancillary files: multiple +deliveries are merged into one dataset with an ``epoch`` dimension representing +validity, and the consumer selects with ``.sel(epoch=day, method="nearest")``. + +``convert_file_to_dataset`` dispatches on the **filename substring**: + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Substring + - Resulting variables + * - ``excluded-regions`` + - ``ecliptic_longitude_deg``, ``ecliptic_latitude_deg`` on dim ``region``. + Handles the empty-file case explicitly. + * - ``uv-sources`` + - ``object_name``, ``ecliptic_longitude_deg``, ``ecliptic_latitude_deg``, + ``angular_radius_for_masking`` on dim ``source``. + * - ``suspected-transients`` + - ``l1b_unique_block_identifier``, ``histogram_mask_array`` on dim + ``time_block``. + * - ``exclusions-by-instr-team`` + - Identical to the above. + * - ``l2-calibration`` + - ``start_time_utc``, ``cps_per_r`` on dim ``time_block``. + * - ``*.json`` + - Generic ``convert_json_to_dataset``, which **flattens** nested JSON into + dotted-ish variable names. + * - anything else + - ``ValueError: Unknown GLOWS ancillary file type``. + +.. warning:: + + Dispatch is by substring on the file name. Rename an ancillary file and the + pipeline will raise ``ValueError`` rather than mis-parse - but a name that + happens to contain two of these substrings would match whichever branch comes + first. + +The conversion table is the one exception: it is **not** run through the +combiner. The CLI opens it with ``json.load`` and passes the raw ``dict`` to +``AncillaryParameters``. + +Bundled example copies +---------------------- + +**[CODE]** ``imap_processing/glows/ancillary/`` ships example copies of the +instrument-team files. These are for development and tests; in production the +SDC supplies them through ``ProcessingInputCollection``. + +.. list-table:: + :header-rows: 1 + :widths: 56 44 + + * - File + - Notes + * - ``imap_glows_pipeline-settings_20250923_v002.json`` + - Version "0.1", created 2023-05-27. **No** ``sunrise_offset``, + ``sunset_offset`` or ``spin_offset_correction``. + * - ``imap_glows_pipeline_settings_v001.json`` + - Older, underscore-separated name. Would **not** match the SDC descriptor + convention. + * - ``l1b_conversion_table_v001.json`` + - Also non-conforming name (underscores, no ``imap_glows_`` prefix). + * - ``imap_glows_map-of-uv-sources_20250923_v002.dat`` + - Version 0.2. Header records + ``star_background_max: 2 cts/s, min_masking_angle: 0.5 deg``. + * - ``imap_glows_map-of-excluded-regions_20250923_v002.dat`` + - Version 0.2, generated by the team's + ``generate_map_of_excluded_regions.py``. + * - ``imap_glows_exclusions-by-instr-team_20250923_v002.dat`` + - Header only - no exclusions defined yet. + * - ``imap_glows_suspected-transients_20250923_v002.dat`` + - Header only. + +File formats +------------ + +``map-of-uv-sources`` +^^^^^^^^^^^^^^^^^^^^^ + +Whitespace-separated, ``#``-comment header, four columns: + +.. code-block:: text + + # columns: object_name, ecliptic_longitude_deg, ecliptic_latitude_deg, angular_radius_for_masking + **HDO221A 220.5517651786544 -45.67397641472249 1.6606970217308294 + HD100600 167.4557237863621 12.89523400723998 1.6077588256414592 + HD10144 345.2999797928628 -59.3822555493876 3.949782859665976 + HD10205 38.90283578666903 27.9330576386139 0.5 + +**[DOC §12.7.6]** The radius depends on the source's brightness: up to ~4° for +the brightest, with a floor of 0.2-0.6° (comfortably larger than the 3σ +nutation). The catalogue is expected to hold "probably hundreds" of sources and +to be updated during the mission. Read with ``np.loadtxt(dtype=str)``, so the +name column must not contain spaces. + +``map-of-excluded-regions`` +^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Two columns, longitude and latitude in ecliptic J2000 degrees. **No radius +column** - the region is defined by dense point coverage. + +.. code-block:: text + + # columns: ecliptic_longitude_deg, ecliptic_latitude_deg + 8.437499999999998579e+01 2.074237995448714500e+01 + 8.718749999999998579e+01 2.074237995448714500e+01 + +``exclusions-by-instr-team`` / ``suspected-transients`` +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +.. code-block:: text + + # Value 1 in mask_array means that a given bin is to be excluded + # columns: l1b_unique_block_identifier, histogram_mask_array + 2026-04-15T03:24:36 0000000011110000....0000 + +Split on the **first space**: everything before is the block identifier, +everything after is the mask string of ``n_bin`` characters. Blocks not listed +default to all zeros. + +``l2-calibration`` +^^^^^^^^^^^^^^^^^^ + +.. code-block:: text + + # columns: start_time_UTC, cps_per_R + 2025-09-01T00:00:00 3.37 + +Same first-space split; the second field is parsed as ``float``. + +``pipeline-settings`` +^^^^^^^^^^^^^^^^^^^^^ + +The bundled file, annotated with what the code does with each key: + +.. list-table:: + :header-rows: 1 + :widths: 46 14 40 + + * - Key + - Bundled value + - Used? + * - ``filter_based_on_daily_statistical_error.n_sigma_threshold_lower`` / + ``_upper`` + - 3.0 / 3.0 + - **No.** ``is_beyond_daily_statistical_error`` is a stub. + * - ``filter_based_on_comparison_of_spin_periods.relative_difference_threshold`` + - 7.0e-5 + - **No.** ``is_spin_period_difference_beyond_threshold`` is a stub. + * - ``filter_based_on_temperature_std_dev.std_dev_threshold__celsius_deg`` + - 2.03 + - Yes - flag 12. + * - ``filter_based_on_hv_voltage_std_dev.std_dev_threshold__volt`` + - 50.0 + - Yes - flag 13. + * - ``filter_based_on_spin_period_std_dev.std_dev_threshold__sec`` + - 0.033333 + - Yes - flag 14. + * - ``filter_based_on_pulse_length_std_dev.std_dev_threshold__usec`` + - 1.0 + - Yes - flag 15. + * - ``filter_based_on_maps.angular_radius_for_excl_regions__deg`` + - 2.0 + - **No.** L1B hard-codes 0.05°. + * - ``active_bad_time_flags`` (17 booleans) + - see below + - Yes - the L2 good-time mask. + * - ``active_bad_angle_flags`` (4 booleans) + - all ``true`` + - Parsed into ``PipelineSettings`` but **never applied**. + * - ``number_of_good_histograms_at_night`` + - 3 + - **No.** + * - ``l3a_nominal_number_of_bins`` + - 90 + - **No** - L3A is not produced here. + * - ``sunrise_offset`` / ``sunset_offset`` + - *absent* + - Default 0.0 when absent. Used by ``apply_is_night_offsets``. + * - ``spin_offset_correction`` + - *absent* + - Default 0.0 when absent. Added to the position-angle offset at both L1B + and L2. + +``active_bad_time_flags`` in the bundled file has ``is_night: false`` and +``is_spin_period_difference_beyond_threshold: false``, all others ``true``. The +validation file used by the tests differs: ``is_overexposed``, +``is_hv_test_in_progress`` and ``is_beyond_daily_statistical_error`` are +``false``, ``is_night`` is ``true``, and both offsets plus +``spin_offset_correction: 1.047`` are present. + +.. important:: + + **The pipeline settings file materially changes which data survives to L2.** + When investigating "why did this pointing produce no output", check which + settings file version was served **before** looking at the code. + +``PipelineSettings`` parsing +---------------------------- + +**[CODE]** ``PipelineSettings`` in ``glows/l1b/glows_l1b_data.py`` handles two +shapes, because ``convert_json_to_dataset`` may or may not have flattened the +nested objects: + +* ``active_bad_angle_flags`` as a single array variable, **or** four separate + ``active_bad_angle_flags_`` variables, **or** absent → default + ``[True] * 4``. +* ``active_bad_time_flags`` likewise, keyed on ``BAD_TIME_FLAG_NAMES``, → default + ``[True] * 17``. +* ``sunrise_offset``, ``sunset_offset``, ``spin_offset_correction`` via + ``pipeline_dataset.get(name, 0.0)``. +* ``processing_thresholds``: every variable whose name contains ``"threshold"`` + or ``"limit"``, looked up later by ``get_threshold(suffix)`` using + ``str.endswith``. + +.. warning:: + + Every default is permissive: a settings file that fails to parse, or is + missing a section, silently produces "all flags active" rather than an error. + That is safe for the flag masks (more data rejected) but it means a + mis-delivered file will not announce itself. The one exception is a missing + *threshold*, which crashes with ``TypeError`` when compared against + ``None``. + +What the SDC provides versus what GLOWS provides +------------------------------------------------ + +**[DOC §3.2]** splits the inputs three ways: + +1. **GLOWS instrument telemetry** - the L0 packets. +2. **Ancillary data from the SDC/POC/MOC** - spin period at high cadence, + position-angle offset for GLOWS, spin-axis orientation, spacecraft state + vectors. **[CODE]** All of these arrive through **SPICE kernels and the spin + table**, not through GLOWS-specific ancillary files. See + :ref:`glows-l1b`. +3. **Input provided by the GLOWS team** - the nine files above. + +The document also notes an IMAP-level dependency that is *not* satisfied by +anything here: a **thruster operation flag**, so GLOWS can know whether the +thrusters fired during a given second. Document §12.1.2 item 4 lists it as TBD. + +Finally, §3.14 item 4 notes that data from **SWE** (electron fluxes) and **HIT** +(high-energy cosmic rays) would be useful for identifying periods of elevated +particle background - but explicitly states that other-instrument data are **not +anticipated in the L0-to-L2 pipeline**. They belong at L3. Do not add them here. diff --git a/docs/source/algorithm-code-documentation/glows/data-products.rst b/docs/source/algorithm-code-documentation/glows/data-products.rst new file mode 100644 index 0000000000..ef617fc825 --- /dev/null +++ b/docs/source/algorithm-code-documentation/glows/data-products.rst @@ -0,0 +1,411 @@ +.. _glows-data-products: + +Data Products and Pipeline +========================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This page is the map of **what exists, what feeds what, and what it is called**. +Use it to find the right module and the right ``Logical_source`` before diving +into an algorithm page. + +Level definitions +----------------- + +.. list-table:: + :header-rows: 1 + :widths: 8 24 68 + + * - Level + - Units / organisation + - Meaning for GLOWS + * - L0 + - Raw bits, per pointing + - Binary CCSDS packet stream, + ``imap_glows_l0_raw_{date}-repoint{n}_v{vvv}.pkts``. **Not produced by + this repository** - it is the input. + * - L1A + - Integer-encoded, per pointing + - Unpacked telemetry blocks. Histogram bins and all ancillary values are + still in onboard integer encoding. Direct events are decompressed from + the timestamp/offset stream into explicit ``(sec, subsec, pulse_length, + multi_event)`` tuples. + * - L1B + - Physical units, per pointing + - Ancillary values decoded (°C, V, s, µs). Times become floats. Bad-time + flags decoded and extended with ground-computed ones. Bad-angle flag + arrays computed against the sky masks. Spacecraft position, velocity, + spin axis and ground spin period added from SPICE. **Histogram counts + are untouched** - no calibration happens at L1B. + * - L2 + - Rayleighs, per pointing + - One **daily lightcurve**: good-time L1B histograms co-added, divided by + exposure, converted to photon flux in Rayleighs by the instrument team's + cps-per-Rayleigh factor. Bins are re-indexed from IMAP spin angle to + GLOWS position angle. Bad-angle flags are OR-ed across contributing + blocks. Sky coordinates per bin added. + * - L3A-L3E + - -- + - **Not produced by this repository.** See below. + +.. note:: + + **Direct events stop at L1B.** The document is explicit (§3.8): "Automated + processing of direct events in the SDC ends at Level-1B." There is no L2 DE + product and none is planned in the current revision. + +What happens after L2 (elsewhere) +--------------------------------- + +You do not implement any of this here, but you should know what your L2 output +feeds so you can reason about requirements. + +.. list-table:: + :header-rows: 1 + :widths: 12 88 + + * - Level + - Product + * - L3A + - Daily **low-resolution** lightcurve of Lyman-α photon flux. Nominally 90 + bins of 4°, rebinned from L2's 3600 bins, with star- and + region-contaminated bins **removed** (not just masked), plus estimates of + the time-independent extra-heliospheric background and a time-dependent + background. Uses SWE electron spectra for background correction. + * - L3B + - Carrington-period-averaged **ionization rate profiles** (charge exchange + + photoionization) versus heliolatitude, via the WawHelioIon-MP model. + Needs F10.7 and SWAPI/OMNI2. + * - L3C + - Latitudinal profiles of **solar wind speed and density**, Carrington + averaged. + * - L3D + - Time **history of solar parameters** (WawHelioIon time series), using the + composite Lyman-α index and OMNI2. + * - L3E + - Daily **ENA survival probabilities** for IMAP-Lo, Hi and Ultra. These are + consumed by the other instruments' own L3 pipelines. + +L3 also needs a **bad-day list** from the GLOWS team (days ruined by flares, +CMEs, etc.), and an **averaged spin-axis product** from the SDC shared with the +ENA instruments so that everyone uses identical pointing directions. + +Source packets +-------------- + +**[CODE]** ``GlowsParams`` enum in ``glows/l0/decom_glows.py``. Fields are +defined in ``glows/packet_definitions/``. + +.. list-table:: + :header-rows: 1 + :widths: 12 16 20 52 + + * - APID + - Hex + - Enum + - Contents + * - 1480 + - ``0x5c8`` + - ``GlowsParams.HIST_APID`` + - One block histogram plus its ancillary data. 24 fields after the CCSDS + header. Defined in ``P_GLX_TMSCHIST.xml``. + * - 1481 + - ``0x5c9`` + - ``GlowsParams.DE_APID`` + - One second of direct events (possibly split across several packets). + 4 fields after the header. Defined in ``P_GLX_TMSCDE.xml``. + +Both are loaded from the single master document +``packet_definitions/GLX_COMBINED.xml``, which is what ``decom_packets`` passes +to ``packet_generator``. Any other APID in the file (housekeeping, +telecommands, memory dumps, boot reports) is **silently ignored** - GLOWS +housekeeping is out of scope for this repository. + +Histogram packet fields (APID 1480) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC §3.4.1]** The uppercase name is the telemetry-definition mnemonic; the +lowercase name is what the GLOWS Python bundle and these docs call it. Big +endian throughout. + +.. list-table:: + :header-rows: 1 + :widths: 16 30 8 46 + + * - Mnemonic + - Name + - Bits + - Meaning + * - ``STARTID`` + - ``first_spin_id_in_block`` + - 32 + - Ordinal ID of the first IMAP spin in the block. + * - ``ENDID`` + - ``diff_spin_id_in_block`` + - 16 + - **Difference** from the first ID, not the last ID. The mnemonic is + misleading and the document says so. + * - ``FLAGS`` + - ``histogram_validity_flags`` + - 16 + - The 10 onboard bad-time flags; upper 6 bits reserved, zero. + * - ``SWVER`` + - ``software_version`` + - 24 + - Flight software version used to generate the histogram. + * - ``SEC`` / ``SUBSEC`` + - ``imap_start_time_second`` / ``_subsecond`` + - 32 / 24 + - IMAP-clock block start. Subseconds are interpolated onboard from the + GLOWS clock; limit 2 000 000. + * - ``OFFSETSEC`` / ``OFFSETSUBSEC`` + - ``imap_diff_second`` / ``_subsecond`` + - 16 / 24 + - IMAP-clock **end-time offset** (duration), not an absolute end time. + * - ``GLXSEC`` / ``GLXSUBSEC`` + - ``glows_start_time_second`` / ``_subsecond`` + - 32 / 24 + - Same, from the GLOWS internal SCIENCE timer. + * - ``GLXOFFSEC`` / ``GLXOFFSUBSEC`` + - ``glows_diff_second`` / ``_subsecond`` + - 16 / 24 + - Same, GLOWS clock. + * - ``SPINS`` + - ``number_of_spins_per_block`` + - 8 + - ``n_block``. Taken from FSW configuration; included so the ground can + cross-check it against ``ENDID``. + * - ``NBINS`` + - ``number_of_bins_per_histogram`` + - 16 + - ``n_bin``. + * - ``TEMPAVG`` / ``TEMPVAR`` + - ``filter_temperature_average`` / ``_variance`` + - 8 / 16 + - Encoded per Eq. 37 / Eq. 43. + * - ``HVAVG`` / ``HVVAR`` + - ``hv_voltage_average`` / ``_variance`` + - 16 / 32 + - CEM high voltage. + * - ``SPAVG`` / ``SPVAR`` + - ``spin_period_average`` / ``_variance`` + - 16 / 32 + - The spin period **used onboard for histogramming**. + * - ``ELAVG`` / ``ELVAR`` + - ``pulse_length_average`` / ``_variance`` + - 8 / 16 + - Event impulse length. + * - ``EVENTS`` + - ``number_of_events`` + - 32 + - Total events in the histogram. + * - ``HISTTAB`` + - ``histogram`` + - ``NBINS`` × 8 + - The counts, in **IMAP spin angle** order. + +Direct-event packet fields (APID 1481) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +.. list-table:: + :header-rows: 1 + :widths: 16 34 12 38 + + * - Mnemonic + - Name + - Bits + - Meaning + * - ``SEC`` + - ``imap_start_time_second`` + - 32 + - IMAP-clock whole second this packet's data belongs to. + * - ``LEN`` + - ``number_of_de_packets`` + - 16 + - How many CCSDS packets carry this one second of data. + * - ``SEQ`` + - ``seq_num_of_de_packet`` + - 16 + - Index of this packet within that sequence. + * - ``DATA`` + - ``de_data`` + - variable + - Payload. Structure depends on ``LEN``/``SEQ`` - see :ref:`glows-l1a`. + +``LEN == 0 and SEQ == 0`` means the packet carries **no useful data** (e.g. a +requested second was not found in instrument memory). + +Products produced by this repository +------------------------------------ + +**[CODE]** Exact ``Logical_source`` strings from +``imap_processing/cdf/config/imap_glows_global_cdf_attrs.yaml``. Anything not in +this table does not exist. + +.. list-table:: + :header-rows: 1 + :widths: 28 14 58 + + * - ``Logical_source`` + - Descriptor + - Contents + * - ``imap_glows_l1a_hist`` + - ``hist`` + - Per-block histograms, integer-encoded ancillary, raw onboard flag word. + * - ``imap_glows_l1a_de`` + - ``de`` + - Per-second direct-event arrays plus the ``data_every_second`` + housekeeping structure. + * - ``imap_glows_l1b_hist`` + - ``hist`` + - Per-block histograms in physical units, 17 bad-time flags, 4 × ``n_bin`` + bad-angle flags, SPICE-derived spacecraft state. + * - ``imap_glows_l1b_de`` + - ``de`` + - Per-second direct-event times and pulse lengths, decoded housekeeping, + 11 DE flags. + * - ``imap_glows_l2_hist`` + - ``hist`` + - The daily lightcurve. One epoch per file. + +.. note:: + + ``docs/source/filename-convention/naming-conventions.rst`` lists ``hist``, + ``de``, ``lightcurve``, ``ionization-rate`` and ``survival-probabilities`` as + GLOWS descriptors. Only ``hist`` and ``de`` are produced here; the L2 product + uses ``hist``, **not** ``lightcurve``, despite being a lightcurve. + +Data flow +--------- + +.. code-block:: text + + ┌─────────────────────────── instrument team supplies ────────────────────────┐ + │ l1b-conversion-table-for-anc-data l1b-map-of-uv-sources │ + │ l1b-map-of-excluded-regions l1b-exclusions-by-instr-team │ + │ l1b-suspected-transients pipeline-settings l2-calibration │ + └────────────────────────────────────────────────────────────────────────────┘ + │ │ │ + L0 .pkts │ │ │ + │ │ │ │ + ▼ │ │ │ + ┌──────────────────┐ │ │ + │ glows_l1a() │ decom_packets -> APID split │ │ + └────────┬─────────┘ │ │ + ├──────────────► imap_glows_l1a_hist ─────┼──┐ │ + └──────────────► imap_glows_l1a_de ──┐ │ │ │ + │ │ │ │ + conversion table only ──────┘ │ │ │ + │ │ │ │ + ▼ ▼ ▼ │ + ┌──────────────────┐ ┌────────────────────────┐ │ + │ glows_l1b_de() │ │ glows_l1b() │ │ + └────────┬─────────┘ └───────────┬────────────┘ │ + ▼ ▼ │ + imap_glows_l1b_de imap_glows_l1b_hist │ + (pipeline ends) │ │ + ▼ ▼ + ┌────────────────────────────┐ + │ glows_l2() │ + │ + pipeline-settings │ + │ + l2-calibration │ + └────────────┬───────────────┘ + ▼ + imap_glows_l2_hist + │ + ▼ + L3A..L3E (separate repository) + + SPICE (spin table, CK, SPK, frame kernels) feeds glows_l1b() and glows_l2(). + +CLI wiring +---------- + +**[CODE]** ``class Glows(ProcessInstrument)`` in ``imap_processing/cli.py``. +Supported ``data_level`` values are ``l1a``, ``l1b`` and ``l2``; anything else +raises ``NotImplementedError``. + +L1A +^^^ + +.. code-block:: python + + science_files = dependencies.get_file_paths(source="glows", data_type="l0") + # exactly one file required, else ValueError + datasets = glows_l1a(science_files[0]) + +Returns a list of **one or two** datasets - histogram and/or direct event, +whichever the L0 file contained. A single call produces both products. + +L1B +^^^ + +Requires exactly one L1A CDF. The conversion table is loaded for **both** +branches: + +.. code-block:: python + + conversion_table_file = dependencies.get_processing_inputs( + descriptor="l1b-conversion-table-for-anc-data")[0] + with open(conversion_table_file.imap_file_paths[0].construct_path()) as f: + conversion_table_dict = json.load(f) + + current_day = np.datetime64(...) # from self.start_date + day_buffer = current_day + np.timedelta64(3, "D") + +Then the branch is chosen on ``"hist" in self.descriptor``: + +* **Histogram branch** additionally pulls five ancillary inputs, each wrapped in + a ``GlowsAncillaryCombiner`` with the 3-day buffer: + ``l1b-map-of-excluded-regions``, ``l1b-map-of-uv-sources``, + ``l1b-suspected-transients``, ``l1b-exclusions-by-instr-team``, + ``pipeline-settings``. All five ``.combined_dataset`` values plus the + conversion table go into ``glows_l1b(...)``. +* **Direct-event branch** is just + ``glows_l1b_de(input_dataset, conversion_table_dict)``. + +.. warning:: + + The branch condition is a substring test on the **descriptor**, so a + descriptor containing "hist" anywhere selects the histogram path. There is no + validation that the input CDF's ``Logical_source`` matches the descriptor. + +L2 +^^ + +Requires exactly one L1B CDF plus ``pipeline-settings`` and ``l2-calibration``, +again through ``GlowsAncillaryCombiner`` with the 3-day buffer: + +.. code-block:: python + + datasets = glows_l2( + input_dataset, + pipeline_settings_combiner.combined_dataset, + calibration_combiner.combined_dataset, + ) + +``glows_l2`` returns an **empty list** - and therefore no file is written - when +there are no good-time L1B blocks, or when flux, uncertainties and exposure are +all zero. That is intentional and not an error. + +.. note:: + + The 3-day ``day_buffer`` exists because instrument-team ancillary files are + time-ranged and open-ended: the combiner needs an end date to close the last + validity interval. It is not a data selection window. + +CDF attribute configuration +--------------------------- + +**[CODE]** ``imap_processing/cdf/config/``: + +* ``imap_glows_global_cdf_attrs.yaml`` - **the definitive product list**. All + five ``Logical_source`` values above. +* ``imap_glows_l1a_variable_attrs.yaml`` +* ``imap_glows_l1b_variable_attrs.yaml`` +* ``imap_glows_l2_variable_attrs.yaml`` + +Each level's module calls ``ImapCdfAttributes.add_instrument_global_attrs("glows")`` +and ``.add_instrument_variable_attrs("glows", "")``. If a variable name is +missing from the YAML, ``get_variable_attributes`` raises - which is the usual +first failure when adding a new output variable. diff --git a/docs/source/algorithm-code-documentation/glows/implementation-status.rst b/docs/source/algorithm-code-documentation/glows/implementation-status.rst new file mode 100644 index 0000000000..29d96cd284 --- /dev/null +++ b/docs/source/algorithm-code-documentation/glows/implementation-status.rst @@ -0,0 +1,458 @@ +.. _glows-implementation-status: + +Implementation Status and Known Gaps +==================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This page is the honest accounting of where the code stands against the +algorithm document. **Read it before proposing or estimating work.** + +Accurate as of the most recent survey of ``imap_processing/glows``, +``imap_processing/cli.py`` (class ``Glows``), +``imap_processing/ancillary/ancillary_dataset_combiner.py`` and +``imap_processing/quality_flags.py``. If you change something material, update +this page in the same commit. + +Summary +------- + +.. list-table:: + :header-rows: 1 + :widths: 12 24 64 + + * - Level + - State + - Notes + * - L0 / L1A + - **Mature** + - Both APIDs decommutated; all three direct-event compression markers + implemented; multi-packet reassembly with gap tracking; dedup and + invalid-time filters. Validated against the GLOWS team's JSON. + * - L1B histograms + - **Substantially complete, several stubs** + - Decoding, SPICE geometry and sky masking all work. Two of the seventeen + bad-time flags are hard-coded to "good", and the excluded-region radius + ignores the configured value. + * - L1B direct events + - **Complete for what it is** + - Times, housekeeping decode and the 11 DE flags are all there. Pulse + lengths are not converted to µs and ``unique_identifier`` is missing. + This is the end of the DE pipeline by design. + * - L2 + - **Complete for the nominal path** + - Co-adding, exposure, calibration, position-angle conversion and sky + coordinates all work. Per-bin exclusion, the active-bad-angle mask, and + the rejection counters are not implemented. + * - L3A-L3E + - **Out of scope** + - Belongs to a separate repository closer to the science team. + +There are **no** ``NotImplementedError`` raises anywhere in the GLOWS code apart +from the generic unknown-data-level branch in ``Glows.do_processing``. + +Not implemented at all +---------------------- + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Feature + - Detail + * - **Ground-generated histograms from direct events** + - **[DOC §8.3, §3.5.1, §12.5.2]** GLOWS can rebuild block histograms on the + ground from downlinked DEs, using "Algorithm 2" (bin incrementing once + per block, because the full GLOWS-vs-IMAP clock history is not + downlinked). The resulting structures would be tagged + ``is_generated_on_ground`` and fed into the normal pipeline. **None of + this exists.** ``is_generated_on_ground`` is hard-coded ``False`` at L1A. + The document calls this a contingency capability, not routine operations. + * - **``is_beyond_daily_statistical_error`` (flag 11)** + - **[DOC §12.7.1, Eqs. 45-46]** Reject blocks whose total counts fall + outside ``C_block ± n_reject·√C_block``, with ``n_reject`` between 3 and 4 + and ``C_block`` determined **daily**. This is the primary + particle-background rejection mechanism and arguably the most important + single filter GLOWS has. **[CODE]** ``np.uint8(1)`` with the comment + *"Placeholder until daily histogram is available in glows_l1b.py"* and + *"TODO: this equation needs to be clarified"*. The thresholds + (``n_sigma_threshold_lower``/``_upper``) are sitting unused in the + settings file. + * - **``is_spin_period_difference_beyond_threshold`` (flag 16)** + - **[DOC Table 3.10 item 30.17]** Compare the onboard and ground spin + periods and flag when they disagree. Both values are computed at L1B and + the threshold (``relative_difference_threshold``) is in the settings file. + **[CODE]** The slot is filled with ``np.uint8(1)`` under the name + ``is_beyond_background_error`` - a *different* condition, which the + document marks TBC. The comparison is never made. + * - **``bad_time_flag_occurrences``** + - **[DOC §3.9.1 item 1, Table 3.13 item 24]** Count how many blocks were + rejected for each of the 17 flags, so an analyst can see why a day is + thin. **[CODE]** ``np.zeros((1, FLAG_LENGTH))`` with ``# TODO fill this + in``. The information is trivially available at the point where + ``return_good_times`` runs. + * - **Per-bin exclusion at L2** + - **[DOC §3.9.1 item 5]** ``HistogramL2.filter_bad_bins`` exists but returns + its input unchanged, with two TODOs. There are also + ``# TODO: bad angle filter`` / ``# TODO: filter bad bins out`` in + ``__init__``. Today all bins from all good blocks contribute to the sum + regardless of their bad-angle flags. That matches the document's *"the + signal values are kept at Level-2 as they are"* for the **flux**, but the + document also wants the active-flag mask applied to the **flags**. + * - **``active_bad_angle_flags`` mask** + - **[DOC §3.9.1, bad-angle masking]** *"inactive flags have zeroes set in + Level-2 even if they are not zeroed in Level-1B"*. **[CODE]** + ``PipelineSettings.active_bad_angle_flags`` is parsed and never read. + * - **HV-test pre-filter before ``is_night`` transition detection** + - **[DOC §3.9.1 item 2]** Blocks with ``is_hv_test_in_progress`` raised must + be excluded before locating ``is_night`` transitions, because the monthly + gain test's time-tagged command loads produce **fake transitions**. + **[CODE]** ``apply_is_night_offsets`` looks at the raw ``is_night`` column + only. This will misbehave once a month unless ``is_hv_test_in_progress`` + is separately active in the good-time mask (which it is in the bundled + settings, but is *not* in the test settings file). + * - **``angular_radius_for_excl_regions__deg``** + - **[DOC §12.7.6]** *"will be probably set to the half of the nominal radius + of the GLOWS FOV"*, delivered in pipeline settings. **[CODE]** L1B + hard-codes ``np.deg2rad(0.1 / 2)`` = 0.05°, i.e. half a **bin width**, + roughly 30× smaller than the bundled 2.0° setting and 10× smaller than the + validation file's 0.5°. Excluded regions are therefore masked far more + narrowly than the instrument team intends. **This is probably the most + consequential single deviation on this page.** + * - **``unique_identifier`` for direct events** + - **[DOC Table 3.12 item 2]** Commented out in ``DirectEventL1B`` with a + note that strings belong in attributes, not the data section. + * - **Direct-event pulse length in µs** + - **[DOC Table 3.12 item 13]** ``direct_event_pulse_lengths`` is the raw + encoded value copied straight from ``de[2]``, not converted with the + ``pulse_length`` entry of the conversion table. The document marks the + unit "µs TBC". + * - **``multi_event`` beyond L1A** + - Carried in the L1A ``direct_events`` array, dropped at L1B with + ``# TODO: where does the multi-event flag go?``. Harmless today - KPLabs + say the flight software never sets it. + * - **Full ancillary time series product** + - **[DOC §12.6.1]** muses that a companion product holding the full + spacecraft time series (rather than block averages and spreads) would be + useful. Not implemented, and not formally required. + * - **Point-source / temperature / HV-sensitivity / flight-calibrated + products** + - **[DOC §15]** Five future data products are sketched in the concluding + remarks. None are specified in enough detail to implement and none exist. + * - **Thruster operation flag** + - **[DOC §12.1.2 item 4]** GLOWS wants to know whether thrusters fired in a + given second. Marked TBD in the document; no corresponding input exists. + +Suspected bugs +-------------- + +These need confirmation with the GLOWS team or a test before being changed. +Ordered by how likely they are to affect released data. + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Location + - Issue + * - ``glows_l2_data.py``, ``HistogramL2.__init__`` + - ``circstd(np.radians(spin_axis_data[:, 0]), low=0, high=360)`` - the data + are converted to **radians** but the range is given in **degrees**. The + adjacent ``circmean`` call correctly uses ``high=2*np.pi``. The result is + then passed through ``np.degrees``. This will produce a wrong + ``spin_axis_orientation_std_dev`` longitude. The equivalent code in + ``HistogramL1B.update_spice_parameters`` uses ``high=2*np.pi`` + consistently, so L1B is fine and only L2 is affected. + * - ``glows_l1b_data.py`` vs. ``glows_l2_data.py``, position angle + - ``position_angle_offset_average`` is computed **two different ways**. + L1B: ``360 - get_spin_angle(get_instrument_spin_phase(imap_start_time, + instrument=IMAP_GLOWS), degrees=True) + spin_offset_correction``, i.e. a + per-block SPICE spin-phase lookup, with **no** modulo. L2: + ``(360 - get_instrument_mounting_az_el(IMAP_GLOWS)[0] + + spin_offset_correction) % 360``, i.e. a static mounting azimuth. L2 uses + its own value for the ψ→ψ\ :sub:`PA` conversion and ignores the L1B + variable it is writing out. The document (§10.6) says the quantity is + ``360° - ψ_GLOWS``, constant, which matches the L2 form. Both are written + into products; they need not agree. + * - ``glows_l1b.py``, ``create_l1b_hist_output`` + - The ``flags`` variable is emitted on a dimension called ``flag_dim`` (from + ``output_dimension_mapping``) while the dataset declares a coordinate + ``bad_time_flags``. Both are length 17. Likewise ``glows_l2.py`` writes + ``spin_axis_orientation_*`` on ``latitudinal``, which is never declared as + a coordinate. Confirm against a written CDF; ISTP compliance may be + affected. + * - ``glows_l2_data.py``, ``HistogramL2.__init__`` + - ``self.identifier = int(repointing.replace("repoint", ""))`` where + ``repointing = l1b_dataset.attrs.get("Repointing")``. If the attribute is + absent this raises ``AttributeError: 'NoneType'``. Since GLOWS is + organised strictly per pointing this should never happen, but the failure + mode is opaque. + * - ``glows_l1a_data.py``, ``_build_uncompressed_event`` + - ``seconds = values[0]`` takes all 32 bits where the document specifies a + 2-bit marker plus 30 bits of seconds. It happens to be correct because + the marker bits are zero in both paths that reach this function, but it is + correct by accident rather than by masking. + * - ``glows_l1a.py``, ``generate_de_dataset`` + - The ``within_the_second`` axis is **zero padded** with no fill value and + no per-epoch valid count, so a padded slot is indistinguishable from a + genuine event at GLOWS time ``(0, 0)`` with zero pulse length. L1B then + converts every slot, padding included, into + ``direct_event_glows_times``. + * - ``glows_l1b_data.py``, ``get_threshold`` + - Returns ``None`` when no key matches; the caller then evaluates + ``value <= None`` and raises ``TypeError``. A settings file missing one + threshold crashes L1B rather than failing loudly with a useful message. + * - ``glows_l2_data.py``, ``return_good_times`` + - Uses ``print()`` rather than the module logger when the active-flag mask + length does not match, and then continues with a mismatched boolean index. + Repository convention is ``logger``; ``print`` is also used in + ``Glows.do_processing``. + * - ``glows_l0_data.py``, ``within_same_sequence`` + - Compares only ``SEC`` and ``LEN``, with ``# TODO: What other fields need + to match?``. Two genuinely different second-groups with the same second + and packet count would merge silently. + * - ``glows_l1a_data.py``, ``HistogramL1A.__post_init__`` + - The ``ENDID`` vs. ``SPINS`` cross-check the document asks for (§3.4.1 + item 13) is present only as commented-out code, disabled because the + emulator did not populate the fields correctly. Worth re-enabling once + flight data is available. + +Deviations from the algorithm document +-------------------------------------- + +These are design decisions, not bugs, but they will surprise anyone reading the +document first. + +Flag polarity is inverted +^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC Table 3.10]** says ``false`` = normal, ``true`` = problem. **[CODE]** +``HistogramL1B.compute_flags`` returns ``1 = good, 0 = bad``, and +``return_good_times`` selects rows where all active flags are ``1``. Every +comparison against the GLOWS team's JSON validation output has to invert. This +is consistent within the codebase and there is no reason to change it, but it +must be documented at every boundary. + +The DE flags at L1B are **not** inverted - they are copied straight through. + +ψ→ψ\ :sub:`PA` conversion happens at L2, not L1B +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC §12.6.2 item 6]** says the L1B histogram is re-arranged to the position +angle. **[DOC §3.9.1 item 9 and §3.14 item 2]** say the conversion happens at +L2. The document states that §3 supersedes §12 where they conflict, and the code +follows §3. L1B carries ``imap_spin_angle_bin_cntr`` in raw ψ. + +Calibration is a division, not a multiplication +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC Eq. 53]** ``I_m = S_m × α``. **[CODE]** ``photon_flux = (counts / +exposure) / calibration_factor``. Since ``α`` is in **cps per Rayleigh**, the +code is dimensionally correct and the equation as printed is not. Do not +"correct" the code. + +Conversion table values differ from Table 11.1 +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +The document's Table 11.1 gives ``hv_voltage`` as 16 bits over 0-56012.82 V (a +consequence of promoting a 12-bit ADC value to a 16-bit field). The delivered +conversion table uses **12 bits over 0-3500 V**. The code uses whatever the +delivered file says, which is right. Table 11.1 also flags the filter +temperature relation as *probably nonlinear*, needing a lookup table; the +``p01``-``p04`` slots in the JSON appear to be reserved for that and are +currently all zero and unused. + +Excluded-region masking radius +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Covered above under "not implemented", but repeated here because it is a silent +science-affecting difference rather than a missing feature: the code masks +within **half a bin width (0.05°)** of an excluded-region point, not within the +configured ``angular_radius_for_excl_regions__deg``. + +Masking algorithm implementation +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC §3.7.1]** describes the GLOWS team's implementation using +``astropy.coordinates.search_around_sky``. **[CODE]** uses SPICE frame +transforms and dot products of unit vectors instead. This is a legitimate +reimplementation - no astropy dependency, and it reuses the repository's +standard geometry helpers - but the two will not be bit-identical. + +Single-time geometry approximations +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +* L1B bad-angle masking transforms look vectors at the **block start time** + only, for a block spanning ~2 minutes. +* L2 ecliptic coordinates are computed at the **midpoint block** of the day and + applied to the whole day. + +Both are defensible (the spin axis is nominally fixed within a pointing) but +neither is what a strict reading of the document implies. + +L2 start/end times +^^^^^^^^^^^^^^^^^^ + +**[DOC Table 3.13 items 3-4]** wants the UTC start and end of the observational +day. **[CODE]** uses the first and last *good block* epoch, each of which is a +block **midpoint**. The reported day is therefore shorter than the real one by +up to a block at each end, plus whatever was culled at the edges. + +Exposure is uniform across bins +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Correct today, since no bins are dropped. It becomes wrong the moment per-bin +exclusion is implemented. Whoever implements ``filter_bad_bins`` must also +convert ``exposure_times`` from a scalar broadcast to a genuine per-bin +accumulation (document Eq. 49 is already written per bin). + +Housekeeping is out of scope +^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC §3, item 3]** lists HK Full and HK Brief as GLOWS telemetry categories, +and §12.5.3 points at a separate engineering document for their structure. No +GLOWS housekeeping APID is decommutated here, and none is planned. Everything +science processing needs (filter temperature, HV, spin period, pulse length) is +carried inside the science packets. + +Cross-cutting notes for anyone touching GLOWS +--------------------------------------------- + +Field order is load-bearing +^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +``HistogramL1B``, ``DirectEventL1B`` and ``HistogramL2`` are unpacked positionally +by ``xr.apply_ufunc`` / ``dataclasses.asdict``. Inserting a field in the middle +of a dataclass silently shifts every following variable. Add at the end, and +update ``output_dimension_mapping`` in ``glows_l1b.py`` and the special-case +lists in ``create_l2_dataset``. + +Fill values +^^^^^^^^^^^ + +``GlowsConstants.HISTOGRAM_FILLVAL = 65535`` propagates through L1A and L1B +untouched and is zeroed at L2 before summing. ``DailyLightcurve`` then chops all +bin arrays to ``number_of_bins`` and ``glows_l2.create_l2_dataset`` re-expands +them using each variable's CDF ``FILLVAL``. A new lightcurve variable without a +``FILLVAL`` in ``imap_glows_l2_variable_attrs.yaml`` will raise during padding. + +Empty output is normal +^^^^^^^^^^^^^^^^^^^^^^ + +``glows_l2`` returning ``[]`` (no good-time blocks, or all-zero flux/exposure) +is an expected outcome, not a failure. Batch jobs must tolerate a pointing that +produces no L2 file. + +Configuration dominates behaviour +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Whether ``is_night`` blocks survive, whether HV-test blocks survive, how wide the +sky masks are, and what the calibration factor is are **all** decided by +ancillary files, not by code. When triaging a data anomaly, identify the exact +``pipeline-settings`` and ``l2-calibration`` versions that were used first. + +Full TODO inventory +------------------- + +**[CODE]** Every ``TODO``/placeholder marker currently in +``imap_processing/glows``: + +.. list-table:: + :header-rows: 1 + :widths: 42 8 50 + + * - File + - Line + - Comment + * - ``l0/glows_l0_data.py`` + - 205 + - What other fields need to match? (``within_same_sequence``) + * - ``l1a/glows_l1a.py`` + - 163 + - Block header per second, or global attribute? + * - ``l1a/glows_l1a_data.py`` + - 248 + - Sanity check should exist in final code (``ENDID`` vs ``SPINS``) + * - ``l1b/glows_l1b.py`` + - 387 + - The four spacecraft location/velocity values should each get their own + dimension/attributes + * - ``l1b/glows_l1b_data.py`` + - 460 + - ``number_of_de_packets`` missing from algorithm document + * - ``l1b/glows_l1b_data.py`` + - 522 + - Is ``number_of_de_packets`` required in L1B? + * - ``l1b/glows_l1b_data.py`` + - 547 + - First two values of DE are sec/subsec + * - ``l1b/glows_l1b_data.py`` + - 551 + - Where does the multi-event flag go? + * - ``l1b/glows_l1b_data.py`` + - 608-609 + - Double check ``unique_identifier`` time base; strings must go in + attributes + * - ``l1b/glows_l1b_data.py`` + - 761 + - ``flags_set_onboard`` should be renamed in L1B + * - ``l1b/glows_l1b_data.py`` + - 800 + - Determine human-readable flag output; bad-angle algorithm using SPICE; + move ancillary file to AWS + * - ``l1b/glows_l1b_data.py`` + - 847-848 + - Ancillary should be an AWS file; pass ``AncillaryParameters`` in rather + than reading here (note: the CLI *does* now pass it in) + * - ``l1b/glows_l1b_data.py`` + - 1035-1036 + - ``is_beyond_daily_statistical_error`` placeholder; equation needs + clarification + * - ``l1b/glows_l1b_data.py`` + - 1053-1054 + - ``is_beyond_background_error`` listed as TBC in the document; placeholder + * - ``l2/glows_l2.py`` + - 163 + - Create CDF attributes (epoch) + * - ``l2/glows_l2_data.py`` + - 398-399 + - Bad angle filter; filter bad bins out + * - ``l2/glows_l2_data.py`` + - 406 + - Fill in ``bad_time_flag_occurrences`` + * - ``l2/glows_l2_data.py`` + - 528-529 + - ``filter_bad_bins`` needs the exclusions ancillary file and a working + ``unique_block_identifier`` + +The CDF attribute YAMLs additionally carry ``TODO: Remove unneeded attributes +once SAMMI is fixed`` (all three levels), ``TODO: I am not sure what the +FIELDNAM should be`` (L1B) and ``TODO: Update validmin and validmax`` (L2). + +Test coverage notes +------------------- + +* Validation against the GLOWS team's own JSON exists for L1A histograms + (``glows_l1a_hist_validation.json``), L1B histograms + (``imap_glows_l1b_hist_full_output.json``) and L1B direct events + (``imap_glows_l1b_de_output.json``). **There is no equivalent L2 validation + file in the repository** - L2 tests exercise the arithmetic against + synthetic fixtures instead. +* Tests requiring SPICE kernels are marked ``@pytest.mark.external_kernel``; + tests requiring the larger in-flight packet file are marked + ``@pytest.mark.external_test_data``. Both are excluded by the repository's + default ``-m "not external_kernel and not external_test_data"`` selection, so + a green local run does **not** exercise the SPICE geometry paths. +* ``test_glows_l1b.py`` around line 467 carries the comment *"This needs to be + added eventually, but is skipped for now."* +* The bundled packet file ``glows_test_packet_20110921_v01.pkts`` contains 505 + histogram and 1088 direct-event packets. The date in the name is not + meaningful. +* **[DOC §3.13.1]** describes Validation Set 1 + (``data_products_cbk_implementation_2025-07-25_validation_set_1.zip``): five + pointings, of which 1-3 are real EQM data taken under a deuterium lamp (real + flight software v3.0.1, but **flat histograms** because the UV beam was not + modulated with spin phase, plus monthly calibration tests visible as count-rate + drops) and 4-5 are purely synthetic packets with realistic helioglow + modulation from the WawHelioGlow model plus stars (**histograms only, no + DEs**). Only a subset of this is checked into the repository. diff --git a/docs/source/algorithm-code-documentation/glows/index.rst b/docs/source/algorithm-code-documentation/glows/index.rst new file mode 100644 index 0000000000..c7346032b2 --- /dev/null +++ b/docs/source/algorithm-code-documentation/glows/index.rst @@ -0,0 +1,219 @@ +:orphan: + +.. _glows-index: + +GLOWS +===== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +.. currentmodule:: imap_processing.glows + +This is the GLOWS (GLObal solar Wind Structure) instrument module, which +contains the code for processing data from the GLOWS instrument. + +Purpose of these pages +---------------------- + +These pages are a **condensed, self-contained working reference** for the GLOWS +processing algorithms, written so that a developer (human or AI agent) can get +productive without reading the full ~120-page data-products document. + +They are a summary of the source document below plus what the code in +``imap_processing/glows`` actually does. Where the two disagree, that is called +out explicitly in :ref:`glows-implementation-status`. + +.. _glows-source-documents: + +Source documents +---------------- + +**None of these are redistributed in this repository.** This is an open-source +repository and the mission documents are not ours to publish. Request them from +the GLOWS instrument team at CBK PAN (Warsaw) or from the SDC document store. + +.. list-table:: + :header-rows: 1 + :widths: 22 78 + + * - Short name + - Reference + * - **Algorithm document** + - M. Bzowski, M. Strumik, I. Kowalska-Leszczyńska, M. A. Kubiak, + R. Wawrzaszek, K. Ber, P. Orleański, *GLOWS data products*, revision + 4.4.7, 25 July 2025. CBK PAN, Bartycka 18a, Warsaw. 123 pages. The + primary source for these pages. Referred to below simply as "the + document". + * - **Telemetry definition** + - ``TLM_GLX_YYYY_MM_DD.xlsx``, sheets ``P_GLX_TMSCHIST`` (histograms) and + ``P_GLX_TMSCDE`` (direct events). Superseded in practice by the XTCE in + ``imap_processing/glows/packet_definitions/``, which is what the code + parses. + * - **Python-script bundle** + - The GLOWS team's own reference implementation of the L0-to-L3A pipeline, + delivered to the SDC together with validation data sets. It emits JSON + for every level. **When the document is ambiguous, this bundle is the + tie-breaker** (document §3.11). The JSON outputs in + ``imap_processing/tests/glows/validation_data/`` come from it. + * - **Supporting reports** + - *GLOWS Signal Evolution Report* (Kowalska-Leszczyńska et al., 2021), + *PSF Report* / *PSF Definition Report* (Strumik, 2020), *Baffle design + report* (Kaźmierczak et al., 2021), *GLOWS Entrance System Writeup* + (Bzowski et al., 2021), MICD (Kowalski, 2020). Referenced by the + document for instrument physics; not needed for any code here. + +.. tip:: + + If you hold a copy of the algorithm document, put it in ``docs/reference/``. + That directory is gitignored, so it will never be committed, and the section + index in :ref:`glows-reference-tables` is written against that location. + +.. important:: + + Two conventions used throughout these pages: + + * **[DOC]** marks a statement taken from the algorithm document. It describes + the intended behavior, which may not be what the code does yet. + * **[CODE]** marks a statement verified against ``imap_processing/glows``. + + When those conflict, the code is what runs and the document is what the + instrument team expects. Both matter; do not silently "fix" one to match the + other without asking. + +.. note:: + + **This repository stops at L2.** GLOWS has a rich L3A-L3E chain (low-res + lightcurves, ionization rates, solar-wind latitude profiles, and ENA survival + probabilities for Lo/Hi/Ultra), but none of it is produced here. It is the + responsibility of a separate repository closer to the science team. Sections + 4 and 13 of the algorithm document describe L3; they are summarised only + briefly in :ref:`glows-data-products` so that you know what your L2 output is + feeding. + +.. warning:: + + GLOWS has a **flag polarity trap**. The document says a bad-time flag value + of ``true`` means "there is a problem". The code writes the opposite: + ``1 = good, 0 = bad``. See :ref:`glows-l1b` before you touch anything + flag-related. + +Which page to read +------------------ + +Read only what you need. Each page is designed to be loaded on its own. + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Page + - Read it when you need to know... + * - :ref:`glows-overview` + - What GLOWS physically is, how a measurement happens, and the vocabulary + (helioglow, spin block, histogram bin, observational day, spin angle vs. + position angle, day/night mode). **Start here if you are new.** + * - :ref:`glows-data-products` + - The full product inventory, exact ``Logical_source`` strings, APIDs, what + feeds what, and how the CLI is wired. **The "what goes into what" map.** + * - :ref:`glows-l1a` + - Packet decommutation, the direct-event compression markers, packet + merging, and the L1A CDF variables. + * - :ref:`glows-l1b` + - Integer-to-physical decoding, the 17 bad-time flags, the 4 bad-angle + flags, sky masking, and everything SPICE contributes. + * - :ref:`glows-l2` + - The daily lightcurve: good-time selection, day/night offsets, co-adding, + exposure, the Rayleigh calibration, and the position-angle conversion. + * - :ref:`glows-ancillary` + - Every instrument-team-supplied file: format, descriptor, who reads it, + and which of them are currently ignored. + * - :ref:`glows-implementation-status` + - What is implemented, what is stubbed, where the code deviates from the + document, and what is not written at all. **Read before proposing work.** + * - :ref:`glows-reference-tables` + - Where the big tables live (XTCE, ancillary ``.dat``, CDF attribute YAML, + validation JSON) and the document's section index. Deliberately *not* + reproduced inline. + +.. toctree:: + :maxdepth: 1 + + overview + data-products + l1a + l1b + l2 + ancillary + implementation-status + reference-tables + +Ten-second orientation +---------------------- + +* GLOWS is a **single-pixel, non-imaging Lyman-α photometer**. It has no + imaging optics and no energy or mass analysis. It counts photons. Everything + else is bookkeeping. +* The spacecraft spins at ~4 RPM (15 s period). GLOWS' boresight is fixed at an + angle to the spin axis, so one spin sweeps a **small circle of 75° angular + radius** on the sky. The daily product is the **modulation of the helioglow + brightness around that circle** - a "lightcurve". +* Photon arrival times ("direct events") are histogrammed **onboard** in the + spin-angle domain: **3600 bins of 0.1°**, accumulated over a **block of 8 + spins (~2 minutes)**. One CCSDS packet = one block histogram. That is the + primary science telemetry and all of it is downlinked. +* Direct events are **supporting data**, not science. Only ~13 blocks per day + are downlinked. This repository takes them to L1B and stops. +* Data are organised **per pointing** ("observational day"), i.e. the interval + between IMAP repointing maneuvers, nominally 24 h. +* Processing chain:: + + CCSDS packets (APID 1480 hist, 1481 DE) + -> L1A unpacked telemetry; still integer-encoded; DEs decompressed + -> L1B physical units, ancillary from SPICE, bad-time + bad-angle flags + -> L2 daily lightcurve of photon flux in Rayleighs + -> L3A..L3E (NOT in this repository) + +* The whole scientific point of L1B/L2 is **culling**: deciding which blocks + (bad *times*) and which bins (bad *angles*) are trustworthy. The arithmetic is + trivial; the flag bookkeeping is where the complexity and the bugs live. + +Where the code lives +-------------------- + +.. code-block:: text + + imap_processing/glows/ + __init__.py BAD_TIME_FLAG_NAMES (17 names), FLAG_LENGTH + utils/constants.py TimeTuple, DirectEvent, GlowsConstants + packet_definitions/ + GLX_COMBINED.xml master XTCE, loaded by decom_packets + P_GLX_TMSCHIST.xml APID 1480 histogram packet + P_GLX_TMSCDE.xml APID 1481 direct-event packet + l0/ + decom_glows.py GlowsParams APID enum; decom_packets() + glows_l0_data.py HistogramL0, DirectEventL0 dataclasses + l1a/ + glows_l1a.py orchestration + both L1A xr.Datasets + glows_l1a_data.py StatusData, HistogramL1A, DirectEventL1A + l1b/ + glows_l1b.py orchestration, apply_ufunc wiring, CDF assembly + glows_l1b_data.py the big one: PipelineSettings, + AncillaryExclusions, AncillaryParameters, + DirectEventL1B, HistogramL1B + l2/ + glows_l2.py orchestration + L2 xr.Dataset assembly + glows_l2_data.py DailyLightcurve, HistogramL2 + ancillary/ bundled example instrument-team files + + imap_processing/ancillary/ancillary_dataset_combiner.py GlowsAncillaryCombiner + imap_processing/quality_flags.py GLOWSL1bFlags (bad-angle) + imap_processing/cdf/config/imap_glows_*.yaml CDF attributes + imap_processing/tests/glows/ tests + validation JSON + imap_processing/cli.py (class Glows) dependency wiring per level + +.. note:: + + ``imap_processing/glows/l1a/``, ``l1b/``, ``l2/`` and ``ancillary/`` have **no** + ``__init__.py``. Only ``l0/`` and ``utils/`` do. Imports work because the + package is installed and Python treats these as namespace packages, but be + aware of it if you are debugging an import or packaging problem. diff --git a/docs/source/algorithm-code-documentation/glows/l1a.rst b/docs/source/algorithm-code-documentation/glows/l1a.rst new file mode 100644 index 0000000000..5b48e08817 --- /dev/null +++ b/docs/source/algorithm-code-documentation/glows/l1a.rst @@ -0,0 +1,454 @@ +.. _glows-l1a: + +Level 1A - Unpacked Telemetry +============================= + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**Goal:** turn the binary CCSDS stream into human-readable data structures. +Nothing is calibrated, nothing is masked, no SPICE is involved. The only real +algorithm here is **direct-event decompression**. + +Entry point: ``glows_l1a(packet_filepath) -> list[xr.Dataset]`` in +``glows/l1a/glows_l1a.py``. One L0 file (one observational day) in, one or two +datasets out. + +.. code-block:: python + + hist_l0, de_l0 = decom_packets(packet_filepath) + + if hist_l0: + l1a_hists = [HistogramL1A(hist) for hist in hist_l0] + output_datasets.append(generate_histogram_dataset(l1a_hists, glows_attrs)) + + if de_l0: + l1a_de = process_de_l0(de_l0) + output_datasets.append(generate_de_dataset(l1a_de, glows_attrs)) + +Decommutation +------------- + +**[CODE]** ``glows/l0/decom_glows.py``. + +``decom_packets`` iterates ``packet_generator(packet_file_path, xtce_document)`` +where the XTCE is ``glows/packet_definitions/GLX_COMBINED.xml``, and dispatches +on ``PKT_APID``: + +* **1480** - ``separate_ccsds_header_userdata`` splits the packet, and a + ``HistogramL0`` is built from ``(__version__, filename, CcsdsData(header), + *userdata.values())``. +* **1481** - the first 7 fields are treated as the header and the remainder are + taken as ``item.raw_value``. This raw-value path matters: ``DE_DATA`` must + stay as uninterpreted bytes. + +.. note:: + + ``glows.__version__`` (``"v001"``) is passed in as ``ground_sw_version``. It + is the GLOWS module's own version string, not the package version. + +``DirectEventL0`` defines ``within_same_sequence(other)`` (currently checks +``SEC`` and ``LEN``) and ``__lt__`` on ``SEQ`` so packet groups can be sorted. + +Histograms: L0 → L1A +-------------------- + +**[CODE]** ``HistogramL1A`` in ``glows/l1a/glows_l1a_data.py``. It is a +mechanical field copy plus four ``TimeTuple`` constructions: + +.. code-block:: python + + self.imap_start_time = TimeTuple(l0.SEC, l0.SUBSEC) + self.imap_time_offset = TimeTuple(l0.OFFSETSEC, l0.OFFSETSUBSEC) + self.glows_start_time = TimeTuple(l0.GLXSEC, l0.GLXSUBSEC) + self.glows_time_offset= TimeTuple(l0.GLXOFFSEC, l0.GLXOFFSUBSEC) + + self.last_spin_id = l0.STARTID + l0.ENDID # ENDID is a difference + self.first_spin_id = l0.STARTID + + self.flags = {"flags_set_onboard": l0.FLAGS, "is_generated_on_ground": False} + +Two quirks worth knowing: + +* **Odd bin counts lose a byte.** CCSDS packets must have an even number of + bytes, so when ``NBINS`` is odd the payload carries a pad byte. The code + drops the last element: ``if self.number_of_bins_per_histogram % 2 == 1: + self.histogram = self.histogram[:-1]``. +* The ``ENDID`` vs. ``SPINS`` consistency check the document asks for exists + only as a **commented-out block** (``glows_l1a_data.py`` line ~248) because + the emulator did not set the values correctly. A length mismatch between + ``NBINS`` and the decoded histogram logs a warning but does not raise. + +``is_generated_on_ground`` is hard-coded ``False``: GLOWS can in principle build +histograms on the ground from direct events (document §8.3), but that path does +not exist here. + +Direct events: L0 → L1A +----------------------- + +This is the interesting part. Three problems have to be solved in order: +**reassembly**, **decompression** and **padding into a rectangular array**. + +Reassembly across packets +^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC §3.4.2]** One second of direct events may not fit in one CCSDS packet. +When it is split, the split happens **at the byte layer, not at the direct-event +layer** - a single event can straddle two packets. So you must concatenate +first, parse second. + +**[CODE]** ``process_de_l0`` in ``glows_l1a.py``: + +.. code-block:: python + + sorted_l0 = sorted(de_l0, key=lambda x: x.SEC) + for sec, de in groupby(sorted_l0, lambda x: x.SEC): + ... + +For each second: + +* One packet with ``LEN == 1`` → construct ``DirectEventL1A`` and parse + immediately. +* One packet with ``LEN != 1`` → packets are missing off the end; + ``finish_incomplete_packet()`` records the missing sequence numbers and + populates ``status_data`` only, leaving ``direct_events`` unset. +* Several packets → sort by ``SEQ``. **If ``SEQ != 0`` is missing, the whole + second is skipped with a warning**, because the ``data_every_second`` + structure lives only in the first packet. Otherwise each subsequent packet is + appended via ``merge_de_packets``, which validates ordering and + ``within_same_sequence``, and accumulates gaps into ``missing_seq``. + +Finally records with no ``direct_events`` are filtered out entirely. Missing +sequence numbers are joined into the dataset global attribute +``missing_packets_sequence``. + +.. code-block:: python + + l1a_output = [de for de in l1a_output if de.direct_events] + +The ``data_every_second`` structure +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC Table 3.2]** / **[CODE]** ``StatusData``, built from the **first 40 +bytes** (320 bits) of the reassembled payload. This is the onboard housekeeping +that the flight software's histogramming algorithm itself uses, so downlinking +it is what makes ground-side histogram regeneration possible in principle. + +.. list-table:: + :header-rows: 1 + :widths: 40 10 50 + + * - Field + - Bits + - Notes + * - ``imap_sclk_last_pps`` + - 32 + - IMAP seconds at the last PPS. + * - ``glows_sclk_last_pps`` + - 32 + - GLOWS seconds at the last PPS. + * - ``glows_ssclk_last_pps`` + - 32 + - GLOWS subseconds at the last PPS. + * - ``imap_sclk_next_pps`` + - 32 + - IMAP seconds at the next PPS. + * - ``catbed_heater_active`` + - 8 + - Flag. Thruster catbed heaters on → repointing imminent. + * - ``spin_period_valid`` + - 8 + - Flag. + * - ``spin_phase_at_next_pps_valid`` + - 8 + - Flag. + * - ``spin_period_source`` + - 8 + - Flag. + * - ``spin_period`` + - 16 + - Integer-encoded; decoded at L1B. + * - ``spin_phase_at_next_pps`` + - 16 + - Integer-encoded; decoded at L1B. + * - ``number_of_completed_spins`` + - 32 + - Provided to GLOWS by IMAP. + * - ``filter_temperature`` + - 16 + - Integer-encoded. + * - ``hv_voltage`` + - 16 + - Integer-encoded. + * - ``glows_time_on_pps_valid`` + - 8 + - Flag. + * - ``time_status_valid`` + - 8 + - Flag. + * - ``housekeeping_valid`` + - 8 + - Flag. + * - ``is_pps_autogenerated`` + - 8 + - Flag. + * - ``hv_test_in_progress`` + - 8 + - Flag. + * - ``pulse_test_in_progress`` + - 8 + - Flag. + * - ``memory_error_detected`` + - 8 + - Flag. + * - ``zero_padding`` + - 8 + - Pad to an even byte count. Present in the document's table; not carried + as a field in ``StatusData``. + +Direct-event decompression +^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC §3.4.2 / §3.5.2, Tables 3.3 and 3.4]**. Events are stored as time +*differences* from the previous event, with the first event of each second +carried as a full timestamp. The **top two bits of the first byte** of every +record are a marker: + +.. list-table:: + :header-rows: 1 + :widths: 12 20 68 + + * - Marker + - Record size + - Layout + * - ``0x0`` + - 8 bytes + - **Full timestamp.** 2-bit marker, 30-bit GLOWS seconds, 8-bit + ``impulse_length`` (50 ns units), 1-bit ``multi_event``, 2 unused bits + (not guaranteed zero), 21-bit GLOWS subseconds. + * - ``0x2`` + - 2 bytes + - **14-bit offset** (``16 - 2``) in GLOWS clock ticks relative to the + previous event, plus 8-bit ``impulse_length``. + * - ``0x3`` + - 3 bytes + - **22-bit offset** (``24 - 2``), plus 8-bit ``impulse_length``. + +An event is written as a full timestamp whenever the offset from the previous +event will not fit in 22 bits. + +``multi_event`` is set by the FPGA. Per KPLabs it is **not currently used** by +the flight software and is always ``False``. + +**[CODE]** ``DirectEventL1A._generate_direct_events`` implements the loop: + +.. code-block:: python + + current_event = self._build_uncompressed_event(direct_events[:8]) + processed_events = [current_event] + + i = 8 + while i < len(direct_events) - 1: + first_byte = int(direct_events[i]); i += 1 + oldest_diff = first_byte & 0x3F # low 6 bits + marker = first_byte >> 6 + + if marker == 0x0: # 7 more bytes; oldest_diff becomes the top byte + part = bytearray([oldest_diff]); part.extend(direct_events[i:i+7]); i += 7 + current_event = self._build_uncompressed_event(part) + elif marker == 0x2: # 2 more bytes + current_event = self._build_compressed_event( + direct_events[i:i+2], oldest_diff, current_event.timestamp); i += 2 + elif marker == 0x3: # 3 more bytes + current_event = self._build_compressed_event( + direct_events[i:i+3], oldest_diff, current_event.timestamp); i += 3 + else: + raise IndexError(...) + processed_events.append(current_event) + +Offset reconstruction, ``_build_compressed_event``: + +.. code-block:: python + + # 2-byte: diff = oldest_diff << 8 | raw[0]; length = raw[1] + # 3-byte: diff = oldest_diff << 16 | int(raw[0:2]); length = raw[2] + subseconds = previous_time.subseconds + diff + seconds = previous_time.seconds + return DirectEvent(TimeTuple(seconds, subseconds), length, False) + +Carry from subseconds into seconds is handled by ``TimeTuple.__post_init__``, +which folds anything ``>= 2 000 000`` into whole seconds. **This is the only +place the carry happens** - do not add manual carry logic. + +Full timestamp reconstruction, ``_build_uncompressed_event``: + +.. code-block:: python + + values = struct.unpack(">II", raw) # 8 bytes + seconds = values[0] + subseconds = values[1] & 0x1FFFFF # low 21 bits + impulse_length = (values[1] >> 24) & 0xFF # top byte + multi_event = bool((values[1] >> 23) & 0b1) + +.. warning:: + + ``seconds = values[0]`` uses all 32 bits, but the document says the field is + a 2-bit marker followed by **30** bits of seconds. For the first event of a + second the marker is ``0x0`` so the top two bits are zero and the two agree; + for a mid-stream full timestamp the marker bits have already been consumed + into ``oldest_diff`` and re-prepended, so they are also zero. It works, but it + works by construction rather than by masking. + +Robustness +^^^^^^^^^^ + +* Payloads shorter than 8 bytes after the status block log a warning and yield + an empty event list. +* Any unexpected marker or a truncated record raises ``IndexError``, which + ``process_de_l0`` catches per packet, logs, and continues - **DE errors never + stop processing.** + +L1A output datasets +------------------- + +``imap_glows_l1a_hist`` +^^^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** ``generate_histogram_dataset``. Three filters run before anything is +written: + +1. Histograms with ``number_of_bins_per_histogram == 0`` are dropped. +2. Histograms with ``imap_start_time.seconds == 0`` are dropped as invalid + timing (warning logged). +3. **Deduplication** on the 4-tuple ``(imap_start_time.seconds, + .subseconds, imap_time_offset.seconds, .subseconds)``, keeping the first + occurrence (warning logged). + +Coordinates: ``epoch``, ``bins`` (0-3599), ``bins_label``. + +.. code-block:: python + + epoch_time = met_to_ttj2000ns( + hist.imap_start_time.to_seconds() + hist.imap_time_offset.to_seconds() / 2 + ) + +i.e. **epoch is the block midpoint**, not the block start. + +The histogram array is always allocated at ``GlowsConstants.STANDARD_BIN_COUNT`` +(3600) and pre-filled with ``GlowsConstants.HISTOGRAM_FILLVAL`` (65535, i.e. +``uint16`` max); shorter histograms occupy the leading bins and the rest stay +fill. Every downstream stage must respect that fill value. + +.. list-table:: + :header-rows: 1 + :widths: 40 14 46 + + * - Variable + - dtype + - Notes + * - ``histogram`` + - uint16 + - ``(epoch, bins)``. Counts. 65535 = unused bin. + * - ``seq_count_in_pkts_file`` + - uint16 + - CCSDS ``SRC_SEQ_CTR``. + * - ``first_spin_id`` / ``last_spin_id`` + - uint32 + - ``STARTID`` and ``STARTID + ENDID``. + * - ``flags_set_onboard`` + - uint16 + - The raw 16-bit onboard flag word, undecoded. + * - ``is_generated_on_ground`` + - uint8 + - Always 0. + * - ``number_of_spins_per_block`` + - uint8 + - ``n_block``. + * - ``number_of_bins_per_histogram`` + - uint16 + - ``n_bin``. + * - ``number_of_events`` + - uint32 + - Total counts. + * - ``filter_temperature_average`` / ``_variance`` + - uint32 + - **Still encoded.** + * - ``hv_voltage_average`` / ``_variance`` + - uint32 + - Still encoded. + * - ``spin_period_average`` / ``_variance`` + - uint32 + - Still encoded. + * - ``pulse_length_average`` / ``_variance`` + - uint32 + - Still encoded. + * - ``imap_start_time`` / ``imap_time_offset`` + - float64 + - Seconds with subseconds as decimals. + * - ``glows_start_time`` / ``glows_time_offset`` + - float64 + - Seconds with subseconds as decimals. + +Global attribute: ``flight_software_version``, taken from the first histogram. + +.. note:: + + The document (Table 3.5) keeps L1A times as **integer** second/subsecond + pairs and defers the float conversion to L1B. **[CODE]** The L1A CDF already + writes floats, because ``TimeTuple.to_seconds()`` is applied on the way out. + The ``TimeTuple`` objects themselves are integer-valued inside + ``HistogramL1A``. + +``imap_glows_l1a_de`` +^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** ``generate_de_dataset``. One epoch per **second** of direct events +(``met_to_ttj2000ns(de.l0.MET)``). + +Coordinates: ``epoch``, ``within_the_second``, ``direct_event_components``. + +.. list-table:: + :header-rows: 1 + :widths: 40 14 46 + + * - Variable + - dtype + - Notes + * - ``direct_events`` + - float64 + - ``(epoch, within_the_second, direct_event_components)``. The last axis is + ``[seconds, subseconds, impulse_length, multi_event]``. + * - ``seq_count_in_pkts_file`` + - uint16 + - CCSDS sequence counter. + * - ``number_of_de_packets`` + - uint32 + - ``LEN``. + * - 20 ``StatusData`` fields + - uint32/uint8/float64 + - See the table above. + +The ``within_the_second`` dimension is sized to the **longest** second in the +file; the array is grown with ``np.pad`` as longer seconds are encountered and +shorter seconds are **zero padded**. There is no fill value and no explicit +count of valid events per epoch, so downstream code cannot cheaply distinguish a +padded slot from a genuine event at GLOWS time zero. In practice +``TimeTuple(0, 0)`` never occurs in flight data, but it is a latent trap. + +Global attribute: ``missing_packets_sequence``, a comma-joined list of the +``missing_seq`` lists. + +Testing +------- + +**[CODE]** ``imap_processing/tests/glows/``: + +* ``test_glows_decom.py`` - packet counts (505 histogram + 1088 DE packets in + the bundled test file), header fields, byte-array handling. +* ``test_glows_l1a_data.py`` - all three compression markers individually, + sequential events, packet merging with and without gaps, ``StatusData`` + parsing, and comparison against ``glows_l1a_hist_validation.json``. +* ``test_glows_l1a_cdf.py`` - dataset generation, the empty/zero-time filters, + and deduplication. + +Tests using the larger in-flight packet file are marked +``@pytest.mark.external_test_data`` and are skipped by the default selection. diff --git a/docs/source/algorithm-code-documentation/glows/l1b.rst b/docs/source/algorithm-code-documentation/glows/l1b.rst new file mode 100644 index 0000000000..d76f6804ca --- /dev/null +++ b/docs/source/algorithm-code-documentation/glows/l1b.rst @@ -0,0 +1,625 @@ +.. _glows-l1b: + +Level 1B - Physical Units, Flags and Geometry +============================================= + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**Goal:** decode onboard integer encodings into physical units, attach +everything the spacecraft knows (position, velocity, spin axis, spin period), +and compute the two flag systems that decide what L2 is allowed to use. + +Nothing is calibrated here. **The histogram counts pass through untouched.** + +Entry points in ``glows/l1b/glows_l1b.py``: + +* ``glows_l1b(input_dataset, excluded_regions, uv_sources, suspected_transients, + exclusions_by_instr_team, pipeline_settings_dataset, conversion_table_dict)`` + → one histogram ``xr.Dataset`` +* ``glows_l1b_de(input_dataset, conversion_table_dict)`` → one direct-event + ``xr.Dataset`` + +Both are driven by ``xr.apply_ufunc(..., vectorize=True)``, which constructs one +``HistogramL1B`` / ``DirectEventL1B`` dataclass **per epoch** and unpacks its +fields back into aligned ``DataArray``\ s. The ordering of the dataclass fields +is therefore load-bearing: + +.. warning:: + + ``HistogramL1B``'s docstring says it outright: *"IMPORTANT: The order of the + fields inherited from L1A must match the order of the fields in the DataSet + created in decom_glows.py."* Reordering a field, or inserting one in the + middle, silently mis-assigns every subsequent variable. Add new fields at the + end and check ``output_dimension_mapping`` in ``glows_l1b.py``. + +Integer decoding +---------------- + +**[DOC §11.4]** Every averaged ancillary quantity is encoded onboard into +``n_T`` bits over a fixed range: + +.. math:: + + \tau = \left\lfloor \frac{T - T_{min}}{T_{max} - T_{min}}(2^{n_T} - 1) + \right\rfloor = A\,T + B + +.. math:: + + A = \frac{2^{n_T} - 1}{T_{max} - T_{min}}, \qquad B = -T_{min} A + +Decoding (Eq. 39/40): + +.. math:: + + T_d = (\tau - B) / A + +Spreads are downlinked as the **variance**, encoded with the same ``A`` but over +``2 n_T`` bits, because ``⟨τ²⟩ - ⟨τ⟩² = A²(⟨T²⟩ - ⟨T⟩²)`` (Eq. 43). So (Eq. 44): + +.. math:: + + \Delta T_d^2 = \Delta\tau^2 / A^2, \qquad \sigma = \sqrt{\Delta T_d^2} + +.. important:: + + **At L0/L1A the spread measure is a variance. At L1B and above it is a + standard deviation.** The variable names change accordingly + (``*_variance`` → ``*_std_dev``). + +**[CODE]** ``AncillaryParameters`` in ``glows_l1b_data.py`` implements exactly +this: + +.. code-block:: python + + def decode(self, param_key, encoded_value): + params = getattr(self, param_key) + param_a = (2 ** params["n_bits"] - 1) / (params["max"] - params["min"]) + param_b = -params["min"] * param_a + return np.double((encoded_value - param_b) / param_a) + + def decode_std_dev(self, param_key, encoded_value): + params = getattr(self, param_key) + param_a = (2 ** params["n_bits"] - 1) / (params["max"] - params["min"]) + return np.double(np.sqrt(encoded_value / (param_a ** 2))) + +The parameters come from the ``l1b-conversion-table-for-anc-data`` JSON. The +bundled copy (``glows/ancillary/l1b_conversion_table_v001.json``): + +.. list-table:: + :header-rows: 1 + :widths: 26 12 12 10 40 + + * - Parameter + - min + - max + - n_bits + - Physical unit + * - ``filter_temperature`` + - -30.0 + - 80.0 + - 8 + - °C + * - ``hv_voltage`` + - 0.0 + - 3500.0 + - 12 + - V + * - ``spin_period`` + - 0.0 + - 20.9712 + - 16 + - s + * - ``spin_phase`` + - 0.0 + - 360.0 + - 16 + - deg + * - ``pulse_length`` + - 0.0 + - 255.0 + - 8 + - µs + +``AncillaryParameters.__init__`` validates the key sets and raises ``KeyError`` +if the file does not conform. ``filter_temperature``, ``hv_voltage`` and +``pulse_length`` may additionally carry ``p01``-``p04`` polynomial coefficients; +they are all ``0.0`` in the bundled file and **the code never uses them**. + +.. note:: + + **[DOC Table 11.1 note 1]** The filter temperature relation is expected to be + *nonlinear* in reality (12-bit ADC values squeezed into 8 bits), and may + eventually need a lookup table. The ``p01``-``p04`` slots are presumably where + that would live. Also note the document quotes ``hv_voltage`` max as + ``56012.82 V`` for a 16-bit field, while the delivered table uses 3500 V with + 12 bits - **use the delivered table**, which is what the code does. + +Histogram L1B +------------- + +Processing steps +^^^^^^^^^^^^^^^^ + +**[DOC §3.7.1]** lists eight steps. **[CODE]** ``HistogramL1B.__post_init__``: + +.. list-table:: + :header-rows: 1 + :widths: 6 46 48 + + * - # + - Document step + - Code + * - 1 + - Times from int sec/subsec to float + - Already floats out of L1A; passed straight through. + * - 2 + - Decode onboard flag word to a readable structure + - ``deserialize_flags`` → 10 booleans, then inverted (see below). + * - 3 + - Copy ``is_generated_on_ground`` + - Copied and inverted into flag slot 11. + * - 4 + - Bad-time masking - ground-computed flags + - ``compute_flags`` adds 7 more flags. + * - 5 + - Bad-angle masking - per-bin flag array + - ``_compute_histogram_flag_array`` → shape ``(4, n_bin)``. + * - 6 + - Decode ancillary values and spreads + - ``AncillaryParameters.decode`` / ``decode_std_dev``. + * - 7 + - Add SDC-supplied ancillary (spin period, spin axis, ephemeris, + position-angle offset) + - ``update_spice_parameters``. + * - 8 + - Generate the IMAP spin-angle grid for bin centres + - ``imap_spin_angle_bin_cntr``. + +Bin centres +^^^^^^^^^^^ + +**[DOC §3.2]** ``ψ_i = (360°/n_bin)(i - 1/2) = 0.1°(i - 0.5)``, so the **left +edge** of bin 1 is at ψ = 0, not its centre. + +**[CODE]** + +.. code-block:: python + + n_bins = len(self.histogram) + phi = (np.arange(n_bins, dtype=np.float64) + 0.5) / n_bins + self.imap_spin_angle_bin_cntr = phi * 360.0 + +Same thing, zero-indexed. Note that ``n_bins`` comes from the **length of the +histogram array**, which after L1A is always 3600 including fill bins - +``number_of_bins_per_histogram`` is carried separately. + +Unique block identifier +^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC Table 3.8 item 2]** ``YYYY-MM-DDThh:mm:ss`` from the IMAP UTC time. This +string is the join key for two of the ancillary mask files. + +**[CODE]** + +.. code-block:: python + + datetime64_time = met_to_datetime64(self.imap_start_time) + self.unique_block_identifier = np.datetime_as_string(datetime64_time, "s") + +The 17 bad-time flags +^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** ``BAD_TIME_FLAG_NAMES`` in ``glows/__init__.py`` fixes both the names +and the array order. ``FLAG_LENGTH = 17``. + +.. list-table:: + :header-rows: 1 + :widths: 5 44 12 39 + + * - Idx + - Name + - Source + - How it is set **[CODE]** + * - 0 + - ``is_pps_missing`` + - onboard + - Bit 0 of ``flags_set_onboard``. + * - 1 + - ``is_time_status_missing`` + - onboard + - Bit 1. + * - 2 + - ``is_phase_missing`` + - onboard + - Bit 2. + * - 3 + - ``is_spin_period_missing`` + - onboard + - Bit 3. + * - 4 + - ``is_overexposed`` + - onboard + - Bit 4. At least one bin overexposed. + * - 5 + - ``is_direct_event_non_monotonic`` + - onboard + - Bit 5. + * - 6 + - ``is_night`` + - onboard + - Bit 6. **Index referenced by + ``GlowsConstants.IS_NIGHT_FLAG_IDX = 6``.** + * - 7 + - ``is_hv_test_in_progress`` + - onboard + - Bit 7. Monthly gain test. + * - 8 + - ``is_test_pulse_in_progress`` + - onboard + - Bit 8. + * - 9 + - ``is_memory_error_detected`` + - onboard + - Bit 9. + * - 10 + - ``is_generated_on_ground`` + - ground + - ``1 - is_generated_on_ground`` from L1A. Always 1 today. + * - 11 + - ``is_beyond_daily_statistical_error`` + - ground + - **Hard-coded to 1 (good).** Placeholder. + * - 12 + - ``is_temperature_std_dev_beyond_threshold`` + - ground + - ``filter_temperature_std_dev <= std_dev_threshold__celsius_deg``. + * - 13 + - ``is_hv_voltage_std_dev_beyond_threshold`` + - ground + - ``hv_voltage_std_dev <= std_dev_threshold__volt``. + * - 14 + - ``is_spin_period_std_dev_beyond_threshold`` + - ground + - ``spin_period_std_dev <= std_dev_threshold__sec``. + * - 15 + - ``is_pulse_length_std_dev_beyond_threshold`` + - ground + - ``pulse_length_std_dev <= std_dev_threshold__usec``. + * - 16 + - ``is_spin_period_difference_beyond_threshold`` + - ground + - **Hard-coded to 1 (good).** The code comment calls this slot + ``is_beyond_background_error``, which is a *different* condition from the + document's flag 30.17. + +.. danger:: + + **Polarity is inverted relative to the document.** + + **[DOC Table 3.10]**: *"GLOWS uses a convention for bad-time flags, where + false value corresponds to normal conditions and true value indicates a + problem."* + + **[CODE]** ``compute_flags`` returns ``1 = good, 0 = bad``: + + .. code-block:: python + + onboard_flags = (1 - self.deserialize_flags(int(self.flags_set_onboard))).astype(np.uint8) + ... + is_temp_ok = np.uint8(self.filter_temperature_std_dev <= temp_threshold) + + and L2 selects blocks where **all active flags equal 1**. When comparing + against the GLOWS team's JSON validation output, or reading the document, + remember to invert. This is the most common source of confusion in GLOWS + code review. + +Threshold lookup +^^^^^^^^^^^^^^^^ + +``PipelineSettings.processing_thresholds`` collects every pipeline-settings +variable whose name contains ``"threshold"`` or ``"limit"``, and +``get_threshold(suffix)`` returns the first whose key **ends with** the given +suffix. The suffixes used are ``std_dev_threshold__celsius_deg``, +``std_dev_threshold__volt``, ``std_dev_threshold__sec`` and +``std_dev_threshold__usec``. This suffix matching exists because +``convert_json_to_dataset`` flattens nested JSON into names like +``filter_based_on_temperature_std_dev_std_dev_threshold__celsius_deg``. + +.. warning:: + + ``get_threshold`` returns ``None`` when nothing matches, and the comparison + ``value <= None`` then raises ``TypeError``. A pipeline-settings file missing + a threshold key will crash L1B rather than degrade gracefully. + +The 4 bad-angle flags +^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** ``GLOWSL1bFlags`` in ``imap_processing/quality_flags.py``: + +.. code-block:: python + + IS_CLOSE_TO_UV_SOURCE = 2**0 + IS_INSIDE_EXCLUDED_REGION = 2**1 + IS_EXCLUDED_BY_INSTR_TEAM = 2**2 + IS_SUSPECTED_TRANSIENT = 2**3 + +``_compute_histogram_flag_array`` returns a ``(4, n_bin)`` ``uint8`` array. Row +``k`` holds the bit value ``2**k`` where set, ``0`` elsewhere - so each row is +effectively a scaled boolean, and the four rows can be OR-ed into a single +bitmask (which is what L2 does). + +Sky masking geometry +"""""""""""""""""""" + +**[DOC §3.7.1, §12.7.6]** For each bin, compute where on the sky the boresight +was pointing, and compare to the UV-source catalogue and the excluded-region +point set. The GLOWS team's own implementation uses +``astropy.coordinates.search_around_sky`` to compare 3600 bin positions against +thousands of mask points quickly. + +**[CODE]** ``flag_uv_and_excluded`` does it with dot products instead: + +.. code-block:: python + + # 1. Bin look directions in the despun frame + azimuth = (imap_spin_angle_bin_cntr - position_angle_offset_average + 360.0) % 360.0 + elevation = get_instrument_mounting_az_el(SpiceFrame.IMAP_GLOWS)[1] + look_vecs_dps = spherical_to_cartesian( + np.stack([np.ones_like(azimuth), azimuth, + np.full_like(azimuth, elevation)], axis=-1)) + + # 2. Rotate to the ecliptic frame at the block start time + look_vecs_ecl = frame_transform( + data_start_time_et, look_vecs_dps, + SpiceFrame.IMAP_DPS, SpiceFrame.ECLIPJ2000, + allow_spice_noframeconnect=True) + + # 3. cos(separation) = dot product of unit vectors + uv_cos_sep = look_vecs_ecl @ uv_vecs.T # (nbin, n_src) + close_to_uv_source = np.any(uv_cos_sep >= np.cos(uv_radius)[None, :], axis=1) + + region_cos_sep = look_vecs_ecl @ region_vecs.T # (nbin, n_region) + half_bin_rad = np.deg2rad(0.1 / 2) + inside_excluded_region = np.any(region_cos_sep >= np.cos(half_bin_rad), axis=1) + +Points worth noting: + +* Each UV source carries **its own masking radius**, from the catalogue's fourth + column. Bright sources get up to ~4°; the floor is 0.2-0.6° (larger than the + 3σ nutation). +* Excluded regions have **no per-point radius**. The code uses a hard-coded + half-bin-width of **0.05°**, on the reasoning that the region point set + densely covers the area. **[DOC §12.7.6]** expects + ``angular_radius_for_excl_regions__deg`` from pipeline settings, "probably set + to half the nominal radius of the GLOWS FOV". The bundled settings file + supplies ``2.0``; the validation settings file supplies ``0.5``. **The code + ignores both.** See :ref:`glows-implementation-status`. +* ``allow_spice_noframeconnect=True`` is deliberate: the DPS CK intentionally + excludes the ~2 minute repointing transition, and a histogram block can land + in that gap. +* The transform uses the **block start time only**, not a time range - a block + spans ~2 minutes so this is a small approximation. + +Instrument-team masks +""""""""""""""""""""" + +``flag_from_mask_dataset`` looks up ``unique_block_identifier`` in the +``l1b-exclusions-by-instr-team`` / ``l1b-suspected-transients`` dataset and +parses the matching ``"0"``/``"1"`` character string into a boolean array. **No +match → all zeros**, which is the document's specified default. + +What SPICE contributes +^^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** ``HistogramL1B.update_spice_parameters``. The time range used is +``np.arange(start_et, end_et)`` - i.e. **one sample per second** across the +block. + +.. list-table:: + :header-rows: 1 + :widths: 32 68 + + * - Output + - How + * - ``spin_period_ground_average``, ``spin_period_ground_std_dev`` + - ``get_spin_data()``, filtered to + ``data_start_met <= spin_start_met <= data_end_met``, then + ``np.average`` / ``np.std`` of ``spin_period_sec``. + * - ``position_angle_offset_average`` + - ``360 - get_spin_angle(get_instrument_spin_phase(imap_start_time, + instrument=SpiceFrame.IMAP_GLOWS), degrees=True)`` plus + ``spin_offset_correction`` from pipeline settings. + * - ``position_angle_offset_std_dev`` + - Hard-coded ``0.0`` per document §10.6. + * - ``spin_axis_orientation_average`` / ``_std_dev`` + - Transform ``[0, 0, 1]`` from ``IMAP_SPACECRAFT`` to ``ECLIPJ2000`` at + every second, convert to spherical, then **circular** mean/std + (``scipy.stats.circmean`` / ``circstd``) for both longitude and latitude. + Output in degrees, ``[lon, lat]``. + * - ``spacecraft_location_average`` / ``_std_dev`` + - ``geometry.imap_state(et=time_range, ref_frame=ECLIPJ2000, + observer=SpiceBody.SUN)`` columns 0-2. km. + * - ``spacecraft_velocity_average`` / ``_std_dev`` + - Same call, columns 3-5. km/s. + +**[DOC §12.6.1]** asks for exactly these ten quantities. The document also notes +that L1B carries only averages and spreads, and muses that a companion product +with the **full time series** might be useful. That does not exist. + +.. note:: + + **Two spin periods coexist from L1B onward.** ``spin_period_average`` is what + the instrument used onboard for histogramming (from the star tracker, via the + 1 PPS messages). ``spin_period_ground_average`` is recomputed on the ground + from the SDC spin table, which is more accurate because it can assume + constancy over long timescales. **[DOC Table 3.13 footnote]** L2 is expected + to report the **onboard** value provided the two agree to better than about + ``(0.2/360)·⟨P⟩`` - and flag 16 exists to detect when they do not. + +Direct event L1B +---------------- + +**[DOC §3.7.2]** Four steps: convert times to floats, group the onboard flags +into one substructure, decode the encoded ancillary values, and split the L1A +direct-event array into times and pulse lengths. + +**[CODE]** ``DirectEventL1B``: + +.. code-block:: python + + self.direct_event_glows_times, self.direct_event_pulse_lengths = \ + self.process_direct_events(direct_events) # de[0],de[1] -> secs; de[2] + + self.glows_time_last_pps = TimeTuple(int(self.glows_time_last_pps), + glows_ssclk_last_pps).to_seconds() + + self.filter_temperature = anc.decode("filter_temperature", ...) + self.hv_voltage = anc.decode("hv_voltage", ...) + self.spin_period = anc.decode("spin_period", ...) + self.spin_phase_at_next_pps = anc.decode("spin_phase", ...) + + self.de_flags = np.array([catbed_heater_active, spin_period_valid, + spin_phase_at_next_pps_valid, spin_period_source, + glows_time_on_pps_valid, time_status_valid, + housekeeping_valid, is_pps_autogenerated, + hv_test_in_progress, pulse_test_in_progress, + memory_error_detected]) + +The 11 DE flags match **[DOC Table 3.12 item 11]** in name and order. They are +copied through **without polarity inversion**, unlike the histogram flags. + +Two things the document asks for that are **not** produced: + +* ``unique_identifier`` (Table 3.12 item 2) - the code has it commented out with + a note that strings cannot live in the data section and should be an + attribute. +* ``direct_event_pulse_lengths`` in **µs** - the code copies the raw encoded + value (``de[2]``) with no conversion. The document says µs (and marks it TBC). + +The ``multi_event`` element of the L1A direct-event tuple is dropped; a ``TODO`` +asks where it should go. + +L1B output datasets +------------------- + +``imap_glows_l1b_hist`` +^^^^^^^^^^^^^^^^^^^^^^^ + +Before processing, histograms with ``imap_start_time == 0.0`` are dropped with a +warning (a second line of defence after the L1A filter). + +Coordinates created by ``create_l1b_hist_output``: ``epoch``, ``bins``, +``bins_label``, ``bad_angle_flags`` (0-3), ``bad_time_flags`` (0-16), +``ecliptic`` (0-2), ``latitudinal`` (0-1). + +Variables, in the order of the ``HistogramL1B`` dataclass fields: + +.. list-table:: + :header-rows: 1 + :widths: 40 24 36 + + * - Variable + - Dims + - Notes + * - ``histogram`` + - ``(epoch, bins)`` + - Raw counts, unchanged from L1A, fill 65535. + * - ``seq_count_in_pkts_file``, ``first_spin_id``, ``last_spin_id`` + - ``(epoch,)`` + - Passthrough. + * - ``flags_set_onboard`` + - ``(epoch,)`` + - The raw 16-bit word, still carried. A ``TODO`` says it should be + renamed at L1B. + * - ``is_generated_on_ground`` + - ``(epoch,)`` + - Passthrough. + * - ``number_of_spins_per_block``, ``number_of_bins_per_histogram``, + ``number_of_events`` + - ``(epoch,)`` + - Passthrough. + * - ``filter_temperature_average`` / ``_std_dev`` + - ``(epoch,)`` + - °C. + * - ``hv_voltage_average`` / ``_std_dev`` + - ``(epoch,)`` + - V. + * - ``spin_period_average`` / ``_std_dev`` + - ``(epoch,)`` + - s, onboard value. + * - ``pulse_length_average`` / ``_std_dev`` + - ``(epoch,)`` + - µs. + * - ``imap_start_time``, ``imap_time_offset``, ``glows_start_time``, + ``glows_time_offset`` + - ``(epoch,)`` + - Floats. + * - ``unique_block_identifier`` + - ``(epoch,)`` + - ISO-8601 string. + * - ``imap_spin_angle_bin_cntr`` + - ``(epoch, bins)`` + - ψ in degrees. **Not** ψ\ :sub:`PA`. + * - ``histogram_flag_array`` + - ``(epoch, bad_angle_flags, bins)`` + - The ``(4, n_bin)`` bad-angle array. + * - ``spin_period_ground_average`` / ``_std_dev`` + - ``(epoch,)`` + - s, from the SDC spin table. + * - ``position_angle_offset_average`` / ``_std_dev`` + - ``(epoch,)`` + - deg; the std dev is always 0. + * - ``spin_axis_orientation_average`` / ``_std_dev`` + - ``(epoch, latitudinal)`` + - ``[lon, lat]`` in degrees. + * - ``spacecraft_location_average`` / ``_std_dev`` + - ``(epoch, ecliptic)`` + - ``[X, Y, Z]`` km, ecliptic, Sun-centred. + * - ``spacecraft_velocity_average`` / ``_std_dev`` + - ``(epoch, ecliptic)`` + - ``[Vx, Vy, Vz]`` km/s. + * - ``flags`` + - ``(epoch, flag_dim)`` + - The 17 bad-time flags, ``1 = good``. + +Global attributes: ``flight_software_version`` (from L1A) and +``pkts_file_name`` (the ``.pkts`` entries of the input ``Parents`` attribute). + +.. warning:: + + ``flags`` is emitted on a dimension named ``flag_dim`` (from + ``output_dimension_mapping``) while the dataset declares a coordinate named + ``bad_time_flags``. Both have length 17. This looks like an oversight - the + flags variable ends up on an unlabelled dimension. Same pattern appears at L2 + with ``latitudinal``. Worth confirming against a written CDF before relying + on either name. + +``imap_glows_l1b_de`` +^^^^^^^^^^^^^^^^^^^^^ + +Coordinates: ``epoch``, ``within_the_second``, ``within_the_second_label``, +``flags`` (0-10, i.e. **11** DE flags - not the 17 histogram flags). + +Variables follow the ``DirectEventL1B`` field order: +``seq_count_in_pkts_file``, ``number_of_de_packets``, ``imap_time_last_pps``, +``glows_time_last_pps``, ``imap_time_next_pps``, ``spin_period``, +``spin_phase_at_next_pps``, ``number_of_completed_spins``, +``filter_temperature``, ``hv_voltage``, ``de_flags``, +``direct_event_glows_times``, ``direct_event_pulse_lengths``. + +Global attribute: ``missing_packets_sequence``, propagated from L1A. + +Testing +------- + +* ``test_glows_l1b_data.py`` - ``AncillaryParameters`` validation, + ``deserialize_flags`` parametrised over flag words, ``PipelineSettings`` + construction from the flattened JSON form, ``get_threshold``, spin-axis + circular statistics near the 0/360 wrap, and number-for-number comparison + against ``imap_glows_l1b_hist_full_output.json`` and + ``imap_glows_l1b_de_output.json`` from the GLOWS team's bundle. +* ``test_glows_l1b.py`` - the ``apply_ufunc`` dimension mapping, the + zero-start-time filter, the case where the histogram length differs from + ``NBINS``, and the look-vector azimuth formula. The SPICE-dependent test is + marked ``@pytest.mark.external_kernel`` and uses the + ``use_fake_spin_data_for_time`` fixture. diff --git a/docs/source/algorithm-code-documentation/glows/l2.rst b/docs/source/algorithm-code-documentation/glows/l2.rst new file mode 100644 index 0000000000..4623ba936d --- /dev/null +++ b/docs/source/algorithm-code-documentation/glows/l2.rst @@ -0,0 +1,564 @@ +.. _glows-l2: + +Level 2 - The Daily Lightcurve +============================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**Goal:** the lowest level of research-grade GLOWS science data. One +observational day's good-time L1B histograms, co-added into a single +high-resolution lightcurve of **photon flux in Rayleighs** along the GLOWS +scanning circle. + +**[DOC §12.7]** L2 is conceptually the GLOWS equivalent of a SOHO/SWAN +measurement, with the caveat that it is a modulation along one scanning circle +rather than an all-sky map. It contains **all** natural signal components - +helioglow **plus** stars **plus** background. Separating them is L3's job. + +.. important:: + + L2 keeps the full 3600-bin resolution and **masks rather than removes** + problematic bins. Values in flagged bins are left as they are: not zeroed, + not NaN-ed. The mask travels alongside. Rebinning to 90 bins and actually + discarding contaminated bins happens at L3A, elsewhere. + +Entry point: ``glows_l2(input_dataset, pipeline_settings_dataset, +calibration_dataset) -> list[xr.Dataset]`` in ``glows/l2/glows_l2.py``. The work +is done by ``HistogramL2`` and ``DailyLightcurve`` in ``glows_l2_data.py``. + +Direct events do not have an L2 product. + +Processing steps +---------------- + +**[DOC §3.9.1]** lists nine steps. Mapping to code: + +.. list-table:: + :header-rows: 1 + :widths: 6 46 48 + + * - # + - Document step + - Code + * - 1 + - Identify good times via the ``active_bad_time_flags`` mask; keep counters + of why blocks were rejected + - ``return_good_times``. **Rejection counters are a stub.** + * - 2 + - Restrict to ``(t₀ⁿ + Δ₀, t₀ⁿ⁺¹ - Δ₁)`` using ``is_night`` transitions + shifted by ``sunrise_offset`` / ``sunset_offset`` + - ``apply_is_night_offsets``. + * - 3 + - Add good-time histograms into a daily raw histogram + - ``DailyLightcurve.calculate_histogram_sums``. + * - 4 + - Add exposure times + - ``exposure_per_epoch`` summation. + * - 5 + - Set daily bad-angle flags using the active-flag mask + - Bitwise OR across blocks. **The active-flag mask is not applied.** + * - 6 + - Daily count rates from counts and exposure + - Inside the flux calculation. + * - 7 + - Apply the calibration factor to get Rayleighs + - ``get_calibration_factor`` then divide. + * - 8 + - Daily averages and std devs of all ancillary parameters + - ``mean``/``std`` over epoch in ``HistogramL2.__init__``. + * - 9 + - Angular grids: spin angle from north, ecliptic lon/lat + - ``compute_position_angle`` + ``compute_ecliptic_coords_of_bin_centers``. + +Good-time selection +------------------- + +**[CODE]** In ``HistogramL2.__init__``, in this order: + +.. code-block:: python + + active_flags = np.array(pipeline_settings.active_bad_time_flags, dtype=float) + + flags = self.apply_is_night_offsets( + l1b_dataset["flags"].data, + is_night_idx=GlowsConstants.IS_NIGHT_FLAG_IDX, # 6 + sunrise_offset=int(pipeline_settings.sunrise_offset), + sunset_offset=int(pipeline_settings.sunset_offset)) + + good_data = l1b_dataset.isel(epoch=self.return_good_times(flags_da, active_flags)) + + # then drop any histogram where *every* bin carries IS_EXCLUDED_BY_INSTR_TEAM + excl_row = good_data["histogram_flag_array"].data[:, 2, :] + not_all_excl = ~np.all(excl_row == GLOWSL1bFlags.IS_EXCLUDED_BY_INSTR_TEAM.value, axis=1) + good_data = good_data.isel(epoch=np.where(not_all_excl)[0]) + +``return_good_times`` is simply: + +.. code-block:: python + + good_times = np.where(np.all(flags[:, active_flags == 1] == 1, axis=1))[0] + +i.e. **a block is good when every activated flag equals 1**, which given the +inverted polarity described in :ref:`glows-l1b` means "no activated problem was +detected". If the mask length does not match the flag array width, the code +*prints* a message and carries on - it does not raise. + +The "all bins excluded by the instrument team" rule is not in the document; it +comes from the GLOWS team's own reference implementation, where marking every +bin of a block is how they say "throw this whole block away". + +Day/night windowing +------------------- + +**[DOC §3.9.1 item 2]** L2 must exclude histograms taken close to the repointing +maneuver. The onboard ``is_night`` flag already marks the interval, but the +science team wants to be able to shift the two transitions independently: + +* ``is_night`` 0→1 (sunset) is shifted by ``sunset_offset`` histograms; +* ``is_night`` 1→0 (sunrise) is shifted by ``sunrise_offset`` histograms; +* positive ``sunrise_offset`` **extends** night, positive ``sunset_offset`` + **shortens** it; zero means "trust the onboard transition". + +**[CODE]** ``HistogramL2.apply_is_night_offsets`` is a static method operating +directly on the ``(n_epoch, 17)`` flag array, in the code's own inverted +polarity where the ``is_night`` column is ``0 = night (bad)`` and +``1 = day (good)``. It returns the input array unchanged when both offsets are +zero, and otherwise walks the transitions found with ``np.diff``: + +.. code-block:: python + + diff = np.diff(is_night_col.astype(int)) + sunset_index = np.where(diff == -1)[0] # good -> bad, i.e. entering night + sunrise_index = np.where(diff == 1)[0] # bad -> good, i.e. leaving night + + if sunrise_offset > 0: zero out sunrise_offset epochs after each sunrise + if sunrise_offset < 0: set to 1 |sunrise_offset| epochs before each sunrise + if sunset_offset > 0: set to 1 sunset_offset epochs after each sunset + if sunset_offset < 0: zero out |sunset_offset| epochs before each sunset + +Both offsets are counted in **histograms (blocks)**, not hours - despite +``PipelineSettings``' docstring saying "hours". The value is cast with ``int()``. + +.. warning:: + + **[DOC §3.9.1]** requires that blocks with ``is_hv_test_in_progress`` raised + be excluded **before** looking for ``is_night`` transitions, because the + time-tagged command loads used for the monthly gain tests can produce + spurious ``is_night`` transitions. **[CODE]** No such pre-filter exists. + +.. note:: + + The bundled ``imap_glows_pipeline-settings_20250923_v002.json`` contains + **neither** ``sunrise_offset`` **nor** ``sunset_offset``, so both default to + ``0.0`` and ``apply_is_night_offsets`` returns immediately. That file also + sets ``"is_night": false`` in ``active_bad_time_flags``, i.e. night blocks are + *not* excluded at all. The test settings file + (``imap_glows_pipeline-settings_20251112_v001.json``) sets both offsets to 0 + and ``"is_night": true``. **Which behaviour you get is entirely a function of + which ancillary file the SDC serves.** + +Co-adding +--------- + +**[DOC Eq. 47]** + +.. math:: + + H_m = \sum_{k \in K} H^{block}_{k,m} + +where ``K`` is the set of good-time blocks and ``m`` indexes bins. + +**[CODE]** ``DailyLightcurve.calculate_histogram_sums`` first replaces +``HISTOGRAM_FILLVAL`` (65535) with 0, then sums along the epoch axis in +``int64``. The result is truncated to ``number_of_bins_per_histogram`` taken +from the **first** good epoch. + +.. note:: + + All bin-dimensioned arrays inside ``DailyLightcurve`` are **chopped** to + ``number_of_bins``. ``glows_l2.create_l2_dataset`` re-expands them back to + 3600 with each variable's own CDF ``FILLVAL`` via + ``_pad_daily_lightcurve_bins`` before writing. If you add a lightcurve + variable, it must have a ``FILLVAL`` in the L2 YAML or that padding raises. + +Exposure time +------------- + +**[DOC Eqs. 48-49]** Assuming a constant angular rate over a block, + +.. math:: + + \Delta^{block}_{k,m} = \frac{n_{block}\,\langle P_{GLOWS,k}\rangle}{n_{bin}}, + \qquad + \Delta_m = \sum_{k \in K} \Delta^{block}_{k,m} + +**[CODE]** + +.. code-block:: python + + exposure_per_epoch = (l1b_data["spin_period_average"].data + * l1b_data["number_of_spins_per_block"].data + / self.number_of_bins) + self.exposure_times = np.full(self.number_of_bins, np.sum(exposure_per_epoch)) + +The **onboard** spin period is used, consistent with the document's stated +preference (Table 3.13 footnote a). + +.. note:: + + Exposure is therefore **identical in every bin**. That is correct as long as + every good block contributes to every bin, which is true today because + bad-angle bins are masked rather than dropped. It will stop being true the + moment per-bin exclusion is implemented (document §3.9.1 step 5 and the + ``filter_bad_bins`` stub), and the exposure array will then have to be + accumulated per bin. + +Count rate, flux and uncertainty +-------------------------------- + +**[DOC Eqs. 50, 53-55]** + +.. math:: + + S_m = \frac{H_m}{\Delta_m}\ \ [\mathrm{cts/s}], + \qquad + I_m = S_m \times \alpha\ \ [\mathrm{R}], + \qquad + \sigma^S_m = \frac{\sqrt{H_m}}{\Delta_m}, + \qquad + \sigma^I_m = \sigma^S_m \times \alpha + +The calibration factor (document §12.7.5) is derived from ground calibration at +the PTB synchrotron: + +.. math:: + + \alpha_0 = L_{1R} \times QE \times A \times \Omega = 3.37\ \mathrm{cps/R} + +with ``L_1R = 10⁶/(4π)`` photons s⁻¹ cm⁻² sr⁻¹, ``QE = 0.0041 cts/photon``, +``Ω = 0.00416 sr``, ``A = 5.067 cm²``. For comparison, TWINS/LAD had +``α ≈ 2 cps/R``. The full form is expected to be + +.. math:: + + \alpha(HV, THRS, COMP, t) = \alpha_0 \times f(HV, THRS, COMP, t) + +with ``f = 1`` at the reference settings ``HV₀ = 1600 V``, ``THRS₀ = 1.74 V``, +``COMP₀ = 3.2 V`` and the PTB calibration epoch. The time dependence (detector +aging) is to be tracked by monitoring stars in the FOV. The SDC never evaluates +``f``; it just reads whatever ``cps_per_R`` the instrument team delivers. + +**[CODE]** + +.. code-block:: python + + raw_uncertainties = np.sqrt(self.raw_histograms) + if self.number_of_bins > 0 and self.exposure_times[0] > 0 and calibration_factor: + self.photon_flux = (self.raw_histograms / self.exposure_times) / calibration_factor + self.flux_uncertainties = (raw_uncertainties / self.exposure_times) / calibration_factor + +.. important:: + + The code **divides** by the calibration factor; the document's Eq. 53 shows a + multiplication. The code is dimensionally right - ``α`` is in **cps per + Rayleigh**, so ``[cts/s] / [cps/R] = [R]``. Treat Eq. 53 as a typo, not as a + specification. If you ever "fix" the code to match the equation you will + produce fluxes wrong by a factor of ``α² ≈ 11``. + +Uncertainties are pure Poisson, scaled the same way as the counts. The document +(§12.7.7) notes that the correlation matrix is treated as diagonal at this stage +(bin-to-bin correlations are an L3 problem), and that the systematic uncertainty +``σ_α`` on the calibration factor is deliberately **not** folded in here. + +If any of the guards fail - no bins, zero exposure, or no calibration factor - +flux and uncertainties stay at zero, and ``glows_l2`` then returns an empty list +so no file is written. + +Selecting the calibration factor +-------------------------------- + +**[CODE]** ``HistogramL2.get_calibration_factor``. The assumption is stated +explicitly in the docstring: **the calibration is constant over an observational +day**. + +.. code-block:: python + + mid_idx = len(epoch_values) // 2 + mid_epoch_utc = et_to_datetime64(ttj2000ns_to_et(epoch_values[mid_idx].item())) + + cal_at_epoch = calibration_dataset.sel(epoch=mid_epoch_utc, method="pad") + start_times = np.array(cal_at_epoch["start_time_utc"].values, dtype="datetime64[ns]") + nearest_idx = np.searchsorted(start_times, mid_epoch_utc, side="right") - 1 + return float(cal_at_epoch["cps_per_r"].isel(cps_per_r_dim_0=nearest_idx)) + +Two levels of selection, because the ancillary combiner produces a dataset with +its own ``epoch`` (file validity) *and* a ``start_time_utc`` variable (the +calibration table's own step times). The ``l2-calibration`` ``.dat`` file is two +columns, ``start_time_utc cps_per_R``, where each value applies from its start +time until the next one, and the last row is open ended. + +Spin angle → position angle +--------------------------- + +**[DOC §10.6, Eqs. 29-30]** and :ref:`glows-overview`. At L2 the lightcurve is +re-expressed on ψ\ :sub:`PA`, measured from the **northernmost point** of the +scanning circle. + +**[CODE]** ``HistogramL2.compute_position_angle``: + +.. code-block:: python + + glows_mounting_azimuth, _ = get_instrument_mounting_az_el(SpiceFrame.IMAP_GLOWS) + return (360.0 - glows_mounting_azimuth + spin_offset_correction) % 360.0 + +``spin_offset_correction`` is a constant from pipeline settings that corrects a +systematic bias observed in star positions (the validation settings file uses +``1.047``). It is the code's implementation of ``δψ_G,eff``, which the document +sets to zero. + +Then in ``DailyLightcurve``: + +.. code-block:: python + + self.spin_angle = (spin_angle_bin_cntr - position_angle + 360.0) % 360.0 + + roll = -np.argmin(self.spin_angle) + for arr in (spin_angle, raw_histograms, photon_flux, exposure_times, + flux_uncertainties, histogram_flag_array): + arr = np.roll(arr, roll) + +so that **bin 0 of the L2 product is the northernmost point** and the arrays +increase monotonically in ψ\ :sub:`PA`. + +.. note:: + + ``imap_spin_angle_bin_cntr`` is taken from **epoch 0 only** + (``l1b_data["imap_spin_angle_bin_cntr"].data[0]``). That is fine because the + grid is identical for every block - it is a pure function of ``n_bin``. + +Sky coordinates per bin +----------------------- + +**[CODE]** ``DailyLightcurve.compute_ecliptic_coords_of_bin_centers``. Uses the +**midpoint** block of the day: + +.. code-block:: python + + mid_idx = len(l1b_data["imap_start_time"]) // 2 + pointing_midpoint_time_et = sct_to_et(met_to_sclkticks(l1b_data["imap_start_time"][mid_idx].data)) + + azimuth = spin_angle_bin_centers # already psi_PA + elevation = get_instrument_mounting_az_el(SpiceFrame.IMAP_GLOWS)[1] + az_el = np.stack((azimuth, np.full_like(azimuth, elevation)), axis=-1) + + ecliptic_coords = frame_transform_az_el( + data_time_et, az_el, SpiceFrame.IMAP_DPS, SpiceFrame.ECLIPJ2000) + return ecliptic_coords[:, 0], ecliptic_coords[:, 1] + +Giving ``ecliptic_lon`` and ``ecliptic_lat`` for each of the 3600 bin centres. +This is a single-time snapshot for the whole day - acceptable because the spin +axis is nominally fixed within a pointing, but it does mean the coordinates are +representative rather than exact for blocks far from the midpoint. + +Bad-angle flag propagation +-------------------------- + +**[DOC §12.3.4]** *"If a given flag in a bin is True in any histogram of Level-1b, +it is also equal to True at Level-2."* + +**[CODE]** + +.. code-block:: python + + flags = l1b_data["histogram_flag_array"].data # (n_epoch, 4, n_bins) + flags_2d = flags.reshape(-1, self.number_of_bins) # (n_epoch * 4, n_bins) + self.histogram_flag_array = np.bitwise_or.reduce(flags_2d, axis=0).astype(np.uint8) + +The four separate L1B rows collapse into **one bitmask per bin**, where the bits +are the ``GLOWSL1bFlags`` values. This is a deliberate shape change from L1B's +``(4, n_bin)`` to L2's ``(n_bin,)``. + +.. warning:: + + **[DOC §3.9.1 step 5]** requires the ``active_bad_angle_flags`` mask from + pipeline settings to be applied here: *"inactive flags have zeroes set in + Level-2 even if they are not zeroed in Level-1B"*. **[CODE]** The mask is + parsed into ``PipelineSettings.active_bad_angle_flags`` and then never used. + All four flags always propagate. + +Daily averages +-------------- + +**[DOC §3.9.1 step 8]** *"The daily average is computed as the mean of block +averages. The daily standard deviation is computed as the spread of block +averages (to account for possible drifts of ancillary parameters)."* Note that +this is deliberately **not** an error-propagated combination of the per-block +standard deviations. + +**[CODE]** For ``filter_temperature``, ``hv_voltage``, ``spin_period``, +``pulse_length`` and ``spin_period_ground``: + +.. code-block:: python + + self.X_average = good_data["X_average"].mean(dim="epoch", keepdims=True).data + self.X_std_dev = good_data["X_average"].std(dim="epoch", keepdims=True).data + +For spacecraft location and velocity, ``mean``/``std`` over epoch on the +3-vector. For the spin axis, longitude uses ``scipy.stats.circmean`` / +``circstd`` (it sits near the 0/360 wrap) and latitude uses plain +``mean``/``std``: + +.. code-block:: python + + lon_avg = circmean(np.radians(spin_axis_data[:, 0]), low=0, high=2 * np.pi) + lon_std = circstd(np.radians(spin_axis_data[:, 0]), low=0, high=360) + lat_avg = float(np.mean(spin_axis_data[:, 1])) + lat_std = float(np.std(spin_axis_data[:, 1])) + +.. warning:: + + ``circstd`` is called with ``high=360`` on data that has already been + converted to **radians**, while ``circmean`` on the next line correctly uses + ``high=2π``. This looks like a bug; see + :ref:`glows-implementation-status`. + +Product identity and times +-------------------------- + +.. code-block:: python + + repointing = l1b_dataset.attrs.get("Repointing") + self.identifier = int(repointing.replace("repoint", "")) + + self.total_l1b_inputs = len(l1b_dataset["epoch"]) + self.number_of_good_l1b_inputs = len(good_data["epoch"]) + + self.start_time = good_data["epoch"].data[0] # first good block midpoint + self.end_time = good_data["epoch"].data[-1] # last good block midpoint + +The document (Table 3.13 items 3-4) asks for the **UTC start and end time of the +observational day**. The code uses the first and last *good block epoch*, each of +which is a block midpoint, so both ends are short by up to a block and by however +much was culled at the edges. Both are written out as UTC strings via +``met_to_utc(ttj2000ns_to_met(value))``. + +If there are no good blocks at all, the times fall back to the first L1B block's +``imap_start_time`` and ``imap_start_time + imap_time_offset`` - but that path +ends in an empty return anyway. + +``bad_time_flag_occurrences`` (document Table 3.13 item 24: how many blocks were +rejected for each flag) is **all zeros**: + +.. code-block:: python + + # TODO fill this in + self.bad_time_flag_occurrences = np.zeros((1, FLAG_LENGTH), dtype=np.uint16) + +L2 output dataset +----------------- + +``imap_glows_l2_hist``. **One epoch per file**, at the midpoint: + +.. code-block:: python + + time_data = np.array([(histogram_l2.start_time + histogram_l2.end_time) / 2], + dtype=np.float64) + +Coordinates: ``epoch``, ``bins`` (0-3599), ``flags`` (0-16), ``ecliptic`` (0-2). +Data variables ``bins_label`` and ``flags_label`` carry human-readable names; +``flags_label`` is literally ``BAD_TIME_FLAG_NAMES``. + +.. list-table:: + :header-rows: 1 + :widths: 38 22 40 + + * - Variable + - Dims + - Notes + * - ``spin_angle`` + - ``(epoch, bins)`` + - ψ\ :sub:`PA` degrees, starting at the northernmost point. + * - ``photon_flux`` + - ``(epoch, bins)`` + - Rayleigh. + * - ``flux_uncertainties`` + - ``(epoch, bins)`` + - Rayleigh, Poisson only. + * - ``raw_histograms`` + - ``(epoch, bins)`` + - Daily summed counts. + * - ``exposure_times`` + - ``(epoch, bins)`` + - Seconds; currently identical in every bin. + * - ``histogram_flag_array`` + - ``(epoch, bins)`` + - OR-ed ``GLOWSL1bFlags`` bitmask. + * - ``ecliptic_lon`` / ``ecliptic_lat`` + - ``(epoch, bins)`` + - Degrees, ECLIPJ2000. + * - ``number_of_bins`` + - ``(epoch,)`` + - The valid bin count before padding. + * - ``identifier`` + - ``(epoch,)`` + - Pointing number. + * - ``start_time`` / ``end_time`` + - ``(epoch,)`` + - UTC strings. + * - ``number_of_good_l1b_inputs`` / ``total_l1b_inputs`` + - ``(epoch,)`` + - Document Table 3.14 items 1.5/1.6. + * - ``filter_temperature_average`` / ``_std_dev`` + - ``(epoch,)`` + - °C. + * - ``hv_voltage_average`` / ``_std_dev`` + - ``(epoch,)`` + - V. + * - ``spin_period_average`` / ``_std_dev`` + - ``(epoch,)`` + - s, onboard. + * - ``spin_period_ground_average`` / ``_std_dev`` + - ``(epoch,)`` + - s, ground. + * - ``pulse_length_average`` / ``_std_dev`` + - ``(epoch,)`` + - µs. + * - ``position_angle_offset_average`` / ``_std_dev`` + - ``(epoch,)`` + - deg; std dev always 0. + * - ``spin_axis_orientation_average`` / ``_std_dev`` + - ``(epoch, latitudinal)`` + - ``[lon, lat]`` deg. + * - ``spacecraft_location_average`` / ``_std_dev`` + - ``(epoch, ecliptic)`` + - km. + * - ``spacecraft_velocity_average`` / ``_std_dev`` + - ``(epoch, ecliptic)`` + - km/s. + * - ``bad_time_flag_occurrences`` + - ``(epoch, flags)`` + - **Always zero.** + +Global attributes ``flight_software_version`` and ``pkts_file_name`` are +propagated from the L1B input, the former through +``_normalize_global_attr_to_string`` because CDF ``CDF_CHAR`` attributes cannot +take arrays. + +.. note:: + + ``spin_axis_orientation_*`` is written on a ``latitudinal`` dimension that is + never declared as a coordinate in ``create_l2_dataset`` (only ``epoch``, + ``bins``, ``flags`` and ``ecliptic`` are). Same wart as at L1B. + +Testing +------- + +* ``test_glows_l2_data.py`` - calibration factor selection, flux and uncertainty + arithmetic, zero-exposure guards, bin-count handling, the bitwise-OR flag + propagation (including the zero-epoch case), good-time filtering, + ``apply_is_night_offsets`` parametrised over offset combinations, the + ψ→ψ\ :sub:`PA` formula, and that the rolled array starts at the minimum. The + ecliptic-coordinate test is marked ``@pytest.mark.external_kernel``. +* ``test_glows_l2.py`` - end-to-end dataset generation, CDF metadata and fill + values, the global-attribute normaliser, and the (currently no-op) bin + exclusion path. diff --git a/docs/source/algorithm-code-documentation/glows/overview.rst b/docs/source/algorithm-code-documentation/glows/overview.rst new file mode 100644 index 0000000000..e411f16889 --- /dev/null +++ b/docs/source/algorithm-code-documentation/glows/overview.rst @@ -0,0 +1,386 @@ +.. _glows-overview: + +Instrument and Measurement Overview +=================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +Everything on this page is **[DOC]** unless marked otherwise. It exists so that +the algorithm pages can use GLOWS vocabulary without stopping to define it. + +What GLOWS measures +------------------- + +GLOWS (GLObal solar Wind Structure) observes the **helioglow**: the heliospheric +backscatter glow of interstellar neutral hydrogen (ISN H) in the solar Lyman-α +line at **121.567 nm**. + +The physics chain is short: + +1. ISN H atoms flow through the inner heliosphere, collisionless, within a few + au of the Sun. +2. Intense solar Lyman-α resonantly excites them; they immediately re-emit in + random directions. Those re-emitted photons are the helioglow. +3. The density and velocity distribution of ISN H is sculpted by solar + gravity, Lyman-α radiation pressure, and **ionization losses** - charge + exchange with solar wind protons and alphas, photoionization below ~91.2 nm, + and (within 1-2 au) electron-impact ionization. +4. The solar wind has a **latitudinal structure that evolves over the solar + cycle** (slow/dense ~400 km/s, ~5 cm⁻³ near the equator at low activity; + fast/rarefied ~750 km/s, ~2.5 cm⁻³ at the poles). Different charge-exchange + rates at different heliolatitudes carve a 3D structure into the ISN H + density. +5. That structure shows up in the sky distribution of the helioglow. Observing + the helioglow over the mission therefore **infers the latitude structure of + the solar wind and how it evolves.** + +Secondary objectives: the ISN H distribution itself, and the solar radiation +pressure acting on ISN H. + +Helioglow intensity at ~1 au is of order **180-900 Rayleigh**, varying across +the sky and with observer position. + +The instrument +-------------- + +Conceptually descended from the TWINS/LaD photometer (Nass et al. 2006; +McComas et al. 2009). Built by CBK PAN. + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Element + - Notes + * - Collimator with baffle + - Defines the field of view and suppresses stray light. + * - Spectral filter + - Narrow band around Lyman-α. Its **temperature** matters for sensitivity + and is telemetered per block. + * - Channeltron (CEM) detector + - Single-pixel electron multiplier. Effective area ``A = 5.067 cm²``. + Quantum efficiency ``QE ≈ 0.0041 cts/photon`` at 121.6 nm. + * - Electronics block + - Discriminates pulses, timestamps events, histograms them, packs + telemetry. + +Field of view: **~3.31° FWHM** (PSF report), nominal full diameter ~6.8-8° +depending on how you define the edge (transmission is 0.01 at 8°). + +.. important:: + + **GLOWS cannot distinguish photons from particles.** Any event above + threshold is a count. Separating the helioglow from background is entirely a + ground-processing problem, and it is why the flag machinery in L1B/L2 is as + elaborate as it is. + +Sources contributing to the observed counts (document §5), in rough order of +how much they annoy you: + +1. Heliospheric backscatter glow in Lyman-α - **the science signal.** +2. Extraheliospheric sources: stars, the Milky Way, quasars, planets, comets. + Bright stars are *also useful*: they are the in-flight photometric standards + used to track detector aging. +3. Particle background - **bursty**, non-directional, detectable statistically + as an anomalous total count rate in a block. +4. Solar Lyman-α scattered off interplanetary dust. +5. Reflections of solar flares/active regions off interplanetary hydrogen. +6. ENA glow (may itself become a science topic). +7. Stray light from strong EUV sources. +8. Detector dark counts. + +Expected count rates: **~200-1000 cps** from the helioglow, up to ~1000 cps for +the brightest star, so the detector must not saturate below **~2000 cps**. The +document's nominal working number is ``s_mean = 600 cps``. + +How an observation is built +--------------------------- + +.. code-block:: text + + photon -> CEM pulse -> "direct event" (timestamp + pulse length) + -> binned by spin angle into a 3600-bin histogram + -> accumulated over 8 spins = one "block" = one CCSDS packet + -> ~720 blocks = one "observational day" = one pointing + +Scanning geometry +^^^^^^^^^^^^^^^^^ + +The IMAP spin axis points near the Sun, offset by **4°** towards lower ecliptic +longitudes. GLOWS' boresight is mounted at a fixed angle to the spin axis, so +one spin traces a **small circle of angular radius 75°** on the sky. + +**[CODE]** ``GlowsConstants.SCAN_CIRCLE_ANGULAR_RADIUS = 75.0`` in +``glows/utils/constants.py``. Note that the value is currently *unused* by the +pipeline - bin sky positions are obtained from SPICE frame transforms instead +(see :ref:`glows-l1b`). + +After one observational day the spin axis is re-pointed by ~1° to maintain the +4° Sun offset, and the observed strip of sky shifts accordingly. A given star +stays inside the FOV for **~7-8 consecutive days**, crossing at a different +distance from the boresight each day, which is what makes the extrapolation to +zero elongation (and hence absolute stellar brightness, and hence absolute +calibration) possible. + +The key numbers +--------------- + +From the document's Table 0.1. Superscript **c** = configurable in flight, +**g** = configurable on the ground. + +.. list-table:: + :header-rows: 1 + :widths: 30 16 54 + + * - Quantity + - Value + - Notes + * - IMAP day length ``T_IMAP_day`` + - 1.0 day + - Bounded 0.5-3.0 days. Time between spin-axis changes. + * - IMAP spin period ``P_IMAP`` + - 15 s + - Bounded 14.63-15.38 s (4 ± 0.1 RPM). + * - Spins per block ``n_block`` + - 8 :sup:`c` + - 1-256. Set by particle-background detection capability (§10.2). + ~120 s per block. + * - Bins per histogram ``n_bin`` + - 3600 :sup:`c` + - 225-3600. Set by star-calibration resolution needs (§10.3). + * - Bin width ``b`` + - 0.1° + - ``360°/n_bin``. Range 0.1°-1.6°. + * - Bits per bin ``d`` + - 8 :sup:`g` + - Fixed at 8 in practice; max ~66.67 counts/bin expected. + * - Time per bin ``t_bin`` + - 4167 µs + - ``P_IMAP/n_bin``; 4065-4272 µs. + * - Typical counts per block ``C_block`` + - 72 000 ± 268 + - 0.37 % 1-σ Poisson scatter - the basis of background detection. + * - Blocks per day ``N_block`` + - 720 + - 702-738 nominal; 351-2215 for extreme day lengths. + * - GLOWS clock ``f_counter`` + - 2 MHz + - 0.5 µs resolution. Subsecond limit = 2 000 000. + * - DE blocks downlinked/day ``T_dirEv`` + - 13 + - 48 in an older revision of Table 0.1; §3.3.2 and §9.3 say 13. + * - Direct events per day ``C_day`` + - 5.184 × 10⁷ + - Only a tiny fraction is downlinked. + * - Low-res L3A bins ``n_bin_lores`` + - 90 + - 4° bins. **L3A is not produced in this repository.** + * - GLOWS boresight azimuth ``ψ_GLOWS`` + - 217° + - ``90 + 127`` in the IMAP frame, measured from the X axis. Exact value + TBD from as-built measurement; **[CODE]** the pipeline reads it from + SPICE instead of hard-coding it. + +Vocabulary you must not mix up +------------------------------ + +Block vs. histogram vs. lightcurve +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +* **Block** - ``n_block`` consecutive spins (~2 min). The atomic unit of + culling. Everything at L1A/L1B is per-block. +* **Histogram** - the ``n_bin``-element array of **counts** accumulated over one + block. L1A and L1B carry histograms. +* **Lightcurve** - the ``n_bin``-element array of **photon flux in Rayleighs** + accumulated over one observational day. That is what L2 is. The document is + deliberate about this distinction: histograms have counts, lightcurves have + physical units. + +Spin angle ψ vs. position angle ψ\ :sub:`PA` +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +This is the single most confusing thing in GLOWS. There are two angular +coordinates around the scanning circle: + +* **ψ (IMAP spin angle)** - defined in the GI ICD: zero is where the + spacecraft Y-axis crosses the plane parallel to the ecliptic, moving + north-to-south, with +Z rotation. This is what the observatory broadcasts and + what the onboard histogramming uses. **Histograms at L0, L1A and L1B are + organised by ψ.** +* **ψ**\ :sub:`PA` **(GLOWS position angle)** - same rotation sense, but + measured **from the northernmost point of the GLOWS scanning circle**. This + is what the science team wants. **The conversion happens at L2.** + +.. math:: + + \psi_{PA} = \left[\psi - \psi_{G,\mathrm{eff}}\right] \bmod 360^\circ + \qquad\text{(Eq. 29)} + +.. math:: + + \psi_{G,\mathrm{eff}} = 360^\circ - \psi_{GLOWS} + \delta\psi_{G,\mathrm{eff}} + \qquad\text{(Eq. 30)} + +with ``ψ_GLOWS = 217°`` the boresight azimuth in the spacecraft frame. The +instrument team **decided to set** ``δψ_G,eff = 0``: precession and nutation +contribute <0.25° (3σ) and average out over a block, let alone a day. The +document therefore says explicitly that +``position_angle_offset_average = 360° - ψ_GLOWS`` and +``position_angle_offset_std_dev = 0``. + +.. note:: + + **[CODE]** ``position_angle_offset_std_dev`` is hard-coded to ``0.0`` at both + L1B and L2, exactly as the document requires. But + ``position_angle_offset_average`` is computed **two different ways** in two + different places - see :ref:`glows-implementation-status`. + +.. warning:: + + Document §12.6.2 item 6 says the **L1B** histogram is re-arranged from ψ to + ψ\ :sub:`PA`. Document §3.9.1 item 9 and §3.14 item 2 say the conversion + happens at **L2**. §12 is the older text and the document states that §3 + supersedes it. **[CODE]** The code converts at L2, i.e. it follows §3. + +Bad times vs. bad angles +^^^^^^^^^^^^^^^^^^^^^^^^ + +Burn this into memory; it structures the whole flag system. + +* **Bad time** - the *entire block* is unusable. One flag set per block. 17 of + them. A bad-time block is dropped completely before co-adding at L2. +* **Bad angle** - *individual bins* within an otherwise good block are + unusable, because the instrument was looking at a star, the galactic plane, a + comet, or something the instrument team flagged by hand. 4 flags, each a + ``n_bin``-long array. Bad-angle bins are **masked, not removed**: the values + stay in the L2 product and the mask travels with them. Removal happens at L3A. + +Observational day, pointing, day mode, night mode +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +* **Observational day** = **pointing** = the interval between two IMAP + repointing maneuvers. Data products from L1A up are organised on this basis, + and the file name carries a ``-repointNNNNN`` token. GLOWS follows the + official POC/MOC repointing table. +* GLOWS keeps observing through daily repointings; the HV bias is normally + **not** ramped down. It does get turned off for ΔV maneuvers (station + keeping, TCMs), which produces genuine gaps of two hours or more. +* Around the repointing, GLOWS walks through a state sequence (document Fig. + 8.1): + + .. code-block:: text + + Day Mode Night Mode Day Mode + ---------|--------------------------------------------------------|-------- + Evening | Sunset | Night | Sunrise (30 blocks) | + ^ ^ + IsNight raised RepointingPending cleared + (20 blocks ≈ 40 min after (34 min after maneuver start) + RepointingPending set, + which is 1 h before maneuver) + + Data collection and histogramming continue throughout. The ``is_night`` flag + simply marks the interval where spacecraft activity may perturb the spin. + The extra 30-block Sunrise wait exists to let spin-axis instabilities from + the **IMAP-Lo pivot platform** motion damp out. + +* L2 excludes histograms near the repointing. The exclusion window is + ``(t₀ⁿ + Δ₀, t₀ⁿ⁺¹ - Δ₁)``, driven by the ``is_night`` transitions and tuned + by ``sunrise_offset`` / ``sunset_offset`` in the pipeline settings file. See + :ref:`glows-l2`. + +Operation modes +--------------- + +.. list-table:: + :header-rows: 1 + :widths: 32 68 + + * - Mode + - Notes + * - Normal science operations + - Day and Night modes as above. The overwhelming majority of the data. + * - Regular in-flight tests + - **Monthly.** HV gain test, comparison voltage test, threshold voltage + test. Data structures are *identical* to normal science, so they arrive + in the same packets and must be flagged out: + ``is_hv_test_in_progress`` / ``is_test_pulse_in_progress``. + * - Initial calibration / HV ramp-up + - Commissioning, and after any safing or reboot. + +.. warning:: + + **[DOC §3.9.1]** During HV tests the time-tagged command loads can cause + **fake ``is_night`` transitions**. When detecting is_night transitions for + the L2 day/night windowing, blocks with ``is_hv_test_in_progress`` raised + must be excluded first. **[CODE]** This is not currently done - see + :ref:`glows-implementation-status`. + +Time systems +------------ + +Three clocks are in play and the pipeline touches all of them. + +.. list-table:: + :header-rows: 1 + :widths: 24 76 + + * - Clock + - Notes + * - **GLOWS internal clock** + - Free-running, 2 MHz, **not synchronised** to the IMAP clock. All direct + event timestamps are in this clock. Subseconds are counted in + 1/2 000 000 s ticks. The relationship to the IMAP clock is known only + through the PPS timestamps carried in telemetry. + * - **IMAP clock (MET/SCLK)** + - Seconds since 2010-01-01T00:00:00 UTC. 3σ accuracy ±50 µs; distributed + to instruments at 1 PPS with ±30 µs accuracy; aligned to UTC within + ±500 ms (3σ) after ground post-processing. + * - **CDF epoch** + - **[CODE]** TT2000 nanoseconds, via + ``imap_processing.spice.time.met_to_ttj2000ns``. Never hand-roll this. + +Representation by level: + +* **L0/L1A** - integer ``seconds`` + integer ``subseconds`` pairs. Modelled by + ``TimeTuple`` in ``glows/utils/constants.py``, which normalises subseconds + above the 2 000 000 limit into whole seconds. +* **L1B** - floats, subseconds as the decimal part. +* **L2 and above** - UTC / J2000. + +Coordinate frames +----------------- + +**[DOC §14]** GLOWS has two instrument frames and you must not confuse them: + +* **Science (SPICE) frame** - ``+Z`` along the GLOWS boresight but *pointing + opposite* to it, so the **boresight vector is (0, 0, -1)**. ``+Y`` lies in the + plane defined by the boresight and the IMAP body ``Z`` axis, pointing + anti-sunward. Right-handed, so ``X`` follows. +* **MICD ("mechanical") frame** - ``Z`` perpendicular to the mounting plane, + ``X`` perpendicular to the EBOX PCBs. Used only for mechanical-interface + discussions. Ignore it for data processing. + +**[CODE]** The frames the pipeline actually names are +``SpiceFrame.IMAP_GLOWS``, ``SpiceFrame.IMAP_DPS`` (despun pointing frame), +``SpiceFrame.IMAP_SPACECRAFT`` and ``SpiceFrame.ECLIPJ2000``. Spacecraft state +vectors are taken relative to ``SpiceBody.SUN``. + +What the data system must guarantee +----------------------------------- + +The document (§6) states four objectives; they are a useful sanity check on any +change you make: + +1. Identify and remove unwanted **particle background**. +2. Identify and remove the contribution from **extraheliospheric sources**. +3. Track the **day-to-day evolution of selected calibration stars** when they + are in the FOV. +4. Bin the pure helioglow signal on the ground into **90 well-defined spin-angle + bins without smearing**. + +Objectives 1 and 2 are what L1B/L2 flagging is for. Objective 3 is not yet a +data product anywhere. Objective 4 is L3A, and therefore not this repository's +problem - but it is why L2 must stay at full 3600-bin resolution rather than +rebinning: masking must happen at high resolution first, so that as few good +counts as possible are thrown away. diff --git a/docs/source/algorithm-code-documentation/glows/reference-tables.rst b/docs/source/algorithm-code-documentation/glows/reference-tables.rst new file mode 100644 index 0000000000..3492c90bd6 --- /dev/null +++ b/docs/source/algorithm-code-documentation/glows/reference-tables.rst @@ -0,0 +1,346 @@ +.. _glows-reference-tables: + +Reference Tables - Where to Look Them Up +======================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +The algorithm document's large tables are **deliberately not reproduced** here: +they go stale, and in almost every case a machine-readable version already +exists in the repository that the code actually reads. + +The document itself is **not in this repository** - see +:ref:`glows-source-documents`. + +Rule of thumb +------------- + +.. list-table:: + :header-rows: 1 + :widths: 42 58 + + * - If you need... + - Go to + * - A packet field's name, bit offset, width or type + - ``imap_processing/glows/packet_definitions/P_GLX_TMSCHIST.xml`` (APID + 1480) or ``P_GLX_TMSCDE.xml`` (APID 1481). ``GLX_COMBINED.xml`` is the + master document ``decom_packets`` loads. + * - The list of bad-time flags, in order + - ``BAD_TIME_FLAG_NAMES`` in ``imap_processing/glows/__init__.py``. This is + authoritative; the L2 ``flags_label`` variable is written from it. + * - The bad-angle flag bit values + - ``GLOWSL1bFlags`` in ``imap_processing/quality_flags.py``. + * - Fill values, the scan-circle radius, the standard bin count, the + subsecond limit + - ``GlowsConstants`` in ``imap_processing/glows/utils/constants.py``. + * - The integer-to-physical conversion ranges + - The ``l1b-conversion-table-for-anc-data`` file. Bundled example: + ``imap_processing/glows/ancillary/l1b_conversion_table_v001.json``. + * - Which flags are active, and every threshold + - The ``pipeline-settings`` file. Bundled example: + ``imap_processing/glows/ancillary/imap_glows_pipeline-settings_20250923_v002.json``; + test copy in ``imap_processing/tests/glows/validation_data/``. + * - The star catalogue or the excluded-region point set + - ``imap_processing/glows/ancillary/imap_glows_map-of-uv-sources_*.dat`` + and ``imap_glows_map-of-excluded-regions_*.dat``. + * - A CDF variable's units, fill value, valid range or description + - ``imap_processing/cdf/config/imap_glows_l{1a,1b,2}_variable_attrs.yaml``. + * - The complete list of GLOWS products + - ``imap_processing/cdf/config/imap_glows_global_cdf_attrs.yaml``. + * - The direct-event compression scheme + - :ref:`glows-l1a` - fully transcribed. + * - The encoding/decoding equations + - :ref:`glows-l1b` - fully transcribed. + * - The L2 co-adding, exposure, flux and uncertainty equations + - :ref:`glows-l2` - fully transcribed. + * - The spin angle ↔ position angle conversion + - :ref:`glows-overview` - fully transcribed. + * - Expected numeric values for a level + - The GLOWS team's JSON in + ``imap_processing/tests/glows/validation_data/``. + * - Anything else + - the algorithm document, using the section index below. + +Machine-readable tables in the repository +------------------------------------------ + +Packet definitions +^^^^^^^^^^^^^^^^^^ + +.. code-block:: bash + + grep -o 'name="[^"]*"' imap_processing/glows/packet_definitions/P_GLX_TMSCHIST.xml | sort -u + grep -o 'name="[^"]*"' imap_processing/glows/packet_definitions/P_GLX_TMSCDE.xml | sort -u + +Both APIDs are reachable from ``GLX_COMBINED.xml``, which is the only file +``decom_packets`` references. The variable-length payloads +(``HISTOGRAM_DATA``, ``DE_DATA``) are declared as dynamically sized byte +sequences (``BYTEHIST`` / ``BYTEDE``). + +CDF metadata +^^^^^^^^^^^^ + +``imap_processing/cdf/config/``: + +* ``imap_glows_global_cdf_attrs.yaml`` - **the definitive product list.** The + five ``Logical_source`` values are ``imap_glows_l1a_hist``, + ``imap_glows_l1a_de``, ``imap_glows_l1b_hist``, ``imap_glows_l1b_de`` and + ``imap_glows_l2_hist``. If a string is not in this file, + ``get_global_attributes`` will raise. +* ``imap_glows_l1a_variable_attrs.yaml`` +* ``imap_glows_l1b_variable_attrs.yaml`` +* ``imap_glows_l2_variable_attrs.yaml`` + +Validation data +^^^^^^^^^^^^^^^ + +``imap_processing/tests/glows/validation_data/``: + +.. list-table:: + :header-rows: 1 + :widths: 50 50 + + * - File + - Contents + * - ``glows_test_packet_20110921_v01.pkts`` + - 505 histogram + 1088 direct-event CCSDS packets. + * - ``glows_l1a_hist_validation.json`` + - Expected L1A histogram values from the GLOWS team's bundle. + * - ``imap_glows_l1b_hist_full_output.json`` + - Expected L1B histogram values. + * - ``imap_glows_l1b_de_output.json`` + - Expected L1B direct-event values. + * - ``imap_glows_pipeline-settings_20251112_v001.json`` + - Pipeline settings used by the tests. Differs materially from the bundled + production example - see :ref:`glows-ancillary`. + +**[DOC §3.11]** The team's Python bundle emits JSON for every level from L1A to +L3A, and the SDC's CDF output is compared to it "number vs number". The document +is explicit that *"looking into the source code in the scripts (and/or asking +the GLOWS team) is presumably the best way of understanding the intentions of +the instrument team, when something in this document needs clarification."* +Treat the bundle as the tie-breaker, not the prose. + +Algorithm document section index +--------------------------------- + +Revision 4.4.7, 25 July 2025, 123 pages. Page numbers are the document's own. + +.. list-table:: + :header-rows: 1 + :widths: 10 12 78 + + * - Section + - Pages + - Contents + * - Table 0.1 + - 2 + - **The parameter table.** Every nominal value and its bounds. Start here + for any "what is the nominal X" question. + * - 1 + - 5-6 + - Introduction: what the helioglow is and why it maps the solar wind. + * - 2 + - 6-9 + - Onboard processing, science-team view. Table 2.1 operation modes, + Table 2.2 data types. + * - 3 + - 9-32 + - **The primary specification for L0-L2.** Supersedes §12 where they + conflict. + * - 3.2 + - 10-11 + - Inputs: telemetry, SDC-provided ancillary, GLOWS-team-provided files. + * - 3.3.3 + - 13-14 + - "Per pointing" organisation; repointing and ΔV handling. + * - 3.4 + - 14-19 + - **L0 packet layouts.** Table 3.1 CCSDS header, §3.4.1 histogram fields, + §3.4.2 DE fields with Tables 3.2 (data_every_second), 3.3 (full + timestamp) and 3.4 (time offset). + * - 3.5 + - 19-20 + - L0→L1A processing, including the DE parsing loop. + * - 3.6 + - 20-21 + - **L1A data products.** Tables 3.5 (histogram), 3.6 (DE), 3.7 (single DE). + * - 3.7 + - 21-23 + - L1A→L1B processing; bad-angle masking algorithm. + * - 3.8 + - 23 + - L1B data products. Tables 3.8 (histogram), 3.9 (header), + **3.10 (17 bad-time flags)**, **3.11 (4 bad-angle flags)**, 3.12 (DE). + * - 3.9 + - 23-27 + - **L1B→L2 processing.** The nine steps, day/night offsets, bad-angle + masking for L2. + * - 3.10 + - 28-29 + - **L2 data products.** Tables 3.13 (product), 3.14 (header), + 3.15 (daily_lightcurve). + * - 3.11 + - 30 + - The Python-script bundle. + * - 3.12 + - 30-31 + - **The nine ancillary files.** Names and contents. + * - 3.13 + - 31 + - Validation data sets; §3.13.1 describes Validation Set 1 in detail. + * - 3.14 + - 32 + - Comments: quick-look nature of L3, ψ vs ψ\ :sub:`PA`, HK needs, the + SWE/HIT background question. + * - 4 + - 32-49 + - **L3 pipeline.** Not this repository. §4.1 lists internal (SWE, SWAPI, + averaged spin axis) and external (F10.7, composite Lyman-α, OMNI2) + dependencies; §4.15 lists the L3 ancillary files. + * - 5 + - 49 + - **Sources of the observed signal** - the eight contributions. + * - 6 + - 50 + - **What the data system must guarantee** - the four objectives. + * - 7 + - 50-51 + - Characteristics of the observed signal; count-rate expectations; star + visibility windows; Eqs. 1-2 for day length and spin period; onboard time + keeping and attitude accuracy. + * - 8 + - 51-54 + - **Operation modes.** §8.1 and Figure 8.1 are the definitive description + of the Evening/Sunset/Night/Sunrise sequence. §8.2 in-flight tests. + §8.3 ground histogram generation from DEs. + * - 9 + - 54-61 + - Direct events in depth: collection, onboard file format, downlink + selection (min/max blocks, ±k\ :sub:`DE` neighbours, bisection ordering), + data volume. + * - 10 + - 61-74 + - **Histograms in depth.** §10.2 block length from background-detection + capability (Eqs. 13-20). §10.3 bin width from star calibration + (Eqs. 21-23). §10.4 the two onboard histogramming algorithms and the χ² + testing. §10.5 bit rate. **§10.6 the spin-angle offset, Eqs. 29-30 - + essential reading.** + * - 11 + - 74-77 + - Ancillary onboard data. §11.2 average/variance definitions (Eqs. 31-33). + §11.3 event signals. **§11.4 encoding/decoding, Eqs. 35-44.** + **Table 11.1 conversion coefficients** (superseded in practice by the + delivered JSON). + * - 12 + - 77-91 + - The older, more discursive L0-L2 treatment. §12.3 the flag taxonomy. + §12.7.1 initial culling and **Eqs. 45-46, the count-based background + rejection**. §12.7.3 co-adding, Eqs. 47-49. §12.7.4 count rate, Eq. 50. + **§12.7.5 the calibration factor, Eqs. 51-53.** §12.7.6 bad-angle masking + rationale. §12.7.7 uncertainties, Eqs. 54-55. + * - 13 + - 91-102 + - L3 in depth, including the survival-probability programs. Not this + repository. + * - 14 + - 102-104 + - **Coordinate frames.** §14.1.1 the science (SPICE) instrument frame, + §14.1.2 the MICD frame. + * - 15 + - 104-105 + - Concluding remarks: five future data products the team intends to add. + * - Change log + - 107-123 + - Revision history. Useful for working out when a definition changed. + +Equations worth knowing by number +---------------------------------- + +.. list-table:: + :header-rows: 1 + :widths: 14 86 + + * - Eq. + - What it is + * - 29, 30 + - ψ\ :sub:`PA` = mod[ψ - ψ\ :sub:`G,eff`, 360°] and ψ\ :sub:`G,eff` = 360° - + ψ\ :sub:`GLOWS` + δψ\ :sub:`G,eff`. Implemented at L2. + * - 35-39 + - Onboard integer encoding and ground decoding of a scalar. + * - 40 + - Decoding an **averaged** quantity - same form as Eq. 39. + * - 43, 44 + - Decoding an encoded **variance**, then taking the square root. + * - 45, 46 + - Upper and lower count-based rejection thresholds for particle background. + **Not implemented.** + * - 47 + - Daily histogram = sum of good-time block histograms. + * - 48, 49 + - Per-block and daily exposure time. + * - 50 + - Daily count rate S\ :sub:`m` = H\ :sub:`m` / Δ\ :sub:`m`. + * - 51, 52 + - The calibration factor α\ :sub:`0` = 3.37 cps/R and its parameterised + form α(HV, THRS, COMP, t). + * - 53 + - Intensity in Rayleighs. **Printed as a multiplication; the code divides, + which is dimensionally correct.** + * - 54, 55 + - Poisson uncertainty on the count rate and on the intensity. + +Glossary +-------- + +.. list-table:: + :header-rows: 1 + :widths: 26 74 + + * - Term + - Meaning + * - **Block** + - ``n_block`` = 8 consecutive spins, ~120 s. The unit of bad-time culling. + One histogram CCSDS packet. + * - **Bin** + - One of ``n_bin`` = 3600 spin-angle slots, 0.1° wide. The unit of + bad-angle culling. + * - **Observational day / pointing** + - Interval between IMAP repointing maneuvers. The unit of L2 accumulation. + * - **DE** + - Direct event: a single photon detection - GLOWS-clock timestamp plus + impulse length. + * - **HB** + - Histogram block - the document's abbreviation for the histogram data + structure. + * - **CEM** + - Channeltron electron multiplier - the detector. + * - **Helioglow** + - The heliospheric backscatter glow of ISN H in Lyman-α. The science + signal. + * - **ISN H** + - Interstellar neutral hydrogen. + * - **Rayleigh (R)** + - Surface-brightness unit. 1 R corresponds to a radiance of + ``10⁶/(4π)`` photons s⁻¹ cm⁻² sr⁻¹. + * - **cps/R** + - Counts per second per Rayleigh - the calibration factor's unit. + ``α₀ = 3.37``. + * - **ψ (spin angle)** + - IMAP spin phase as defined in the GI ICD. Used at L0/L1A/L1B. + * - **ψ**\ :sub:`PA` **(position angle)** + - Angle from the northernmost point of the GLOWS scanning circle. Used at + L2 and above. + * - **Bad time** + - Whole block unusable. 17 flags. Blocks are dropped. + * - **Bad angle** + - Individual bins unusable. 4 flags. Bins are masked, not dropped. + * - **PPS** + - Pulse per second - the spacecraft timing signal that lets GLOWS relate + its free-running clock to the IMAP clock. + * - **BT / PT** + - The document's "Bad Times" and "Prohibited Times" data structures + (§12.7.1). Conceptual only; the implementation uses the flag array. + * - **WawHelioGlow / WawHelioIon** + - The GLOWS team's forward models, used to generate synthetic validation + data and, at L3, to invert lightcurves into ionization rates. diff --git a/docs/source/algorithm-code-documentation/hit.rst b/docs/source/algorithm-code-documentation/hit.rst index 3b9a9cf77e..51d1a16222 100644 --- a/docs/source/algorithm-code-documentation/hit.rst +++ b/docs/source/algorithm-code-documentation/hit.rst @@ -15,4 +15,4 @@ Level 1A Processing code. :template: autosummary.rst :recursive: - l1a.hit_l1a + l1a.hit_l1a \ No newline at end of file diff --git a/docs/source/algorithm-code-documentation/hit/ancillary.rst b/docs/source/algorithm-code-documentation/hit/ancillary.rst new file mode 100644 index 0000000000..d69b97eb45 --- /dev/null +++ b/docs/source/algorithm-code-documentation/hit/ancillary.rst @@ -0,0 +1,211 @@ +.. _hit-ancillary: + +Ancillary Data and External Dependencies +======================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +HIT is unusually light on ancillary data. There is exactly **one** family of +ancillary files, and it is only used at L2. + +The factor CSVs +--------------- + +**[CODE]** Twelve files: three product families x four dynamic threshold +states. + +.. code-block:: text + + imap_hit_standard-dt0-factors__v.csv + imap_hit_standard-dt1-factors__v.csv + imap_hit_standard-dt2-factors__v.csv + imap_hit_standard-dt3-factors__v.csv + imap_hit_summed-dt0-factors_... (and dt1, dt2, dt3) + imap_hit_sectored-dt0-factors_... (and dt1, dt2, dt3) + +Copies used by the test suite live in +``imap_processing/tests/hit/test_data/ancillary/`` at version +``20250219_v002``. + +The CLI selects them by the descriptor substring ``-dt``; ``load_ancillary_data`` +then picks the right one per state by the substring ``dt-factors``. + +Format +^^^^^^ + +**Standard** and **summed** share a 7-column layout: + +.. code-block:: text + + Species,Lower Energy (MeV),Upper Energy (MeV),Delta E (MeV),Geometry Factor (cm2 sr),Efficiency,b + H,1.8,2.2,0.4,3.4146,1,0 + H,2.2,2.7,0.5,3.44142,1,0 + ... + +**Sectored** inserts a ``Sector`` column after the species: + +.. code-block:: text + + Species, Sector, Lower Energy (MeV), Upper Energy (MeV), Delta E (MeV), Geometry Factor (cm2 sr), Efficiency, b + H ,0,1.8,4,2.2,0.41686,1,0 + H ,1,1.8,4,2.2,0.49253,1,0 + ... + +Row counts, which are a useful sanity check: + +.. list-table:: + :header-rows: 1 + :widths: 24 16 60 + + * - Family + - Data rows + - Structure + * - ``standard`` + - **204** + - one per (species, energy bin); matches + ``STANDARD_PARTICLE_ENERGY_RANGE_MAPPING`` + * - ``summed`` + - **67** + - one per (species, energy bin); matches + ``SUMMED_PARTICLE_ENERGY_RANGE_MAPPING`` + * - ``sectored`` + - **80** + - 10 species/energy combinations x 8 declination sectors + +Quirks to be aware of +^^^^^^^^^^^^^^^^^^^^^ + +* **The files are UTF-8 with a BOM.** ``pandas.read_csv`` handles it, but a + naive reader will see ``Species`` as the first column name. +* **Whitespace is inconsistent.** The sectored file has spaces after commas in + the header and trailing spaces in the species values. The code compensates: + ``load_ancillary_data`` does ``df.columns.str.lower().str.strip()`` and + lowercases the species, and ``get_species_ancillary_data`` additionally + strips every string cell. +* **The ``Sector`` column is never read by name.** ``get_species_ancillary_data`` + groups by ``lower energy (mev)`` and relies on the **row order within the + group** being sector 0-7. If a future delivery reorders the rows, the + declination assignment will silently scramble. See + :ref:`hit-gap-sector-order`. +* **``Efficiency`` is 1 everywhere and ``b`` is 0 everywhere**, exactly as the + document says they should be until measured in flight. + +Where the numbers come from +^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Algorithm document Tables 32-37: + +.. list-table:: + :header-rows: 1 + :widths: 14 20 22 44 + + * - Table + - Family + - State + - PDF pages + * - 32 + - standard + - DT0 + - 107-115 + * - 33 + - standard + - DT1 + - 115-122 + * - 34 + - standard + - DT2 + - 122-130 + * - 35 + - standard + - DT3 + - 130-138 + * - 36 + - sectored + - all (8 declination sectors) + - 138-142 + * - 37 + - summed + - all + - 142-145 + +Page numbers are **PDF page numbers** in +``HIT_Algorithm_Document_v1p11p00_06_02_2026.pdf``; the printed page number in +the footer is one lower. These tables are deliberately **not** reproduced +here - the CSVs are the machine-readable form the code actually reads. See +:ref:`hit-reference-tables`. + +Housekeeping conversions +------------------------ + +Not a separate file: the raw-to-engineering-unit conversions for housekeeping +(algorithm document Tables 29 and 30, originally from HIT-ELEC-HDBK-0008) live +inside the XTCE at +``imap_processing/hit/packet_definitions/hit_packet_definitions.xml``: + +* **156 ``PolynomialCalibrator`` elements** for the linear voltage + conversions, including the 64 ``LEAK_I_NN`` channels which all share + ``0.00488758553274682``. +* **Eight ``ContextCalibrator`` chains** for the thermistors (Table 30 is + 191 rows, -40 to +150 degrees C) - + ``TEMP0``-``TEMP3`` with 22 context ranges each (FEE board), + ``ANALOG_TEMP``, ``HVPS_TEMP``, ``IDPU_TEMP``, ``LVPS_TEMP`` with 20 each + (analog board). Together these express the piecewise Table 30 lookup. + +``hit_l1b`` gets engineering units purely by re-parsing the L0 packet with +``use_derived_value=True``. **To change a housekeeping conversion, edit the +XTCE, not Python.** + +HIT SPICE Usage +--------------- + +**[CODE]** HIT uses SPICE for **time conversion only**: + +* ``hit_l1a`` imports ``et_to_datetime64``, ``met_to_datetime64`` and + ``ttj2000ns_to_et`` for the processing-day filter. +* ``ialirt/l0/process_hit.py`` imports ``met_to_ttj2000ns``. +* The CLI declares SPICE time kernels as one of the two L1A dependencies. + +**No geometry is used.** ``imap_processing/spice/geometry.py`` does define +``SpiceFrame.IMAP_HIT = -43500``, a boresight of ``[0, 1, 0]``, a mounting +normal of ``[0, 1, 0]`` and a spacecraft-to-instrument spin phase offset of +``119.6452 / 360`` (nominally 30 + 90 = 120 degrees, from the frame kernel +``imap_130.tf``) - but nothing in HIT processing calls any of it. Pointing is +needed at L3, which is a different repository. + +.. note:: + + The one place SPICE geometry *would* be needed inside this repository is the + spin-rate correction to the 15th inclination bin (see + :ref:`hit-gap-spinrate`). That correction is required by the document and is + not implemented. + +Ancillary data needed by L3 but not by this repository +------------------------------------------------------ + +Listed here so that nobody adds them to ``imap_processing`` by mistake. All of +these belong to the L3 repository; see :ref:`hit-l3-scope`. + +.. list-table:: + :header-rows: 1 + :widths: 34 66 + + * - Item + - Algorithm document reference + * - ADC-to-MeV coefficients (6 sets: L1/L2/L3 x low/high gain) + - Table 42, section 9.1.2 + * - ``WINCORR2`` / ``WINCORR3`` Kapton foil correction arrays + - Table 41, section 9.1.1 + * - Cosine correction :math:`K_\theta` tables (150 L1 x L2 segment + combinations, per range) + - Tables 43-45, section 9.1.3 + * - Double-power-law ion-track fit parameters (15 species x 5 parameters x + 3 ranges, x2 for the A0/B0 apertures) + - Tables 47-52, section 9.1.4 + * - Charge (Z) lookup tables per range + - section 9.1 + * - Event classification / range table + - Table 40, section 9.1 + * - MAG L1D despun magnetic field vectors + - section 9.2 + * - Electron response matrices (L4-only and L3 vs L4) + - section 9.3 diff --git a/docs/source/algorithm-code-documentation/hit/data-products.rst b/docs/source/algorithm-code-documentation/hit/data-products.rst new file mode 100644 index 0000000000..0d0e81d106 --- /dev/null +++ b/docs/source/algorithm-code-documentation/hit/data-products.rst @@ -0,0 +1,307 @@ +.. _hit-data-products: + +Data Products and What Feeds What +================================= + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This is the "what goes into what" map. It is the page to read before touching +``cli.py`` or adding a product. + +Product inventory +----------------- + +**[CODE]** Everything this repository produces for HIT. The ``logical_source`` +strings come from ``imap_processing/cdf/config/imap_hit_global_cdf_attrs.yaml`` +and are what ``write_cdf()`` uses to name the output file. + +.. list-table:: + :header-rows: 1 + :widths: 32 8 12 48 + + * - ``logical_source`` + - Level + - Cadence + - Contents + * - ``imap_hit_l1a_hk`` + - L1A + - 1 min + - Housekeeping, **raw** DN values. The 64 ``leak_i_NN`` fields are + collapsed into a single 2-D ``leak_i`` on an ``adc_channels`` (64) + dimension. + * - ``imap_hit_l1a_counts-standard`` + - L1A + - 1 min + - Decompressed **counts** for every fixed-format counter in the science + frame: singles, coincidence, event-processing, priority buffer, and + all six FG/BG rate arrays, plus ``ialirtrates``, ``l4fgrates``, + ``l4bgrates``, the livetime counter and the frame header fields. Each + gets ``_stat_uncert_plus`` / ``_stat_uncert_minus`` companions. + * - ``imap_hit_l1a_counts-sectored`` + - L1A + - 1 min records, in complete 10-min sets + - Sectored **counts** reorganised by species: + ``h_sectored_counts``, ``he4_sectored_counts``, + ``cno_sectored_counts``, ``nemgsi_sectored_counts``, + ``fe_sectored_counts``, each ``(epoch, _energy_mean, + azimuth, zenith)``. Plus per-species energy mean/delta variables, + ``livetime_counter`` on its own ``epoch_livetime`` coordinate, and + ``hdr_dynamic_threshold_state``. + * - ``imap_hit_l1a_direct-events`` + - L1A + - 1 min + - ``pha_raw`` only - the **undecoded** concatenated binary of packets + 6-19. See :ref:`hit-gap-events`. + * - ``imap_hit_l1b_hk`` + - L1B + - 1 min + - Housekeeping, **derived / engineering-unit** values. Same packet, + re-parsed with ``use_derived_value=True``. + * - ``imap_hit_l1b_standard-rates`` + - L1B + - 1 min + - Every L1A standard counter divided by the livetime fraction, plus the + scaled uncertainties and ``dynamic_threshold_state``. + * - ``imap_hit_l1b_summed-rates`` + - L1B + - 1 min + - 17 species (``h``, ``he3``, ``he4``, ``he``, ``c``, ``n``, ``o``, + ``ne``, ``na``, ``mg``, ``al``, ``si``, ``s``, ``ar``, ``ca``, + ``fe``, ``ni``), **67 wide energy bins total**, summed across + penetration ranges then divided by livetime. + * - ``imap_hit_l1b_sectored-rates`` + - L1B + - 1 min records, in complete 10-min sets + - Sectored counts divided by ``15 x`` the **previous** 10 minutes' + summed livetime. Variables lose the ``_sectored_counts`` suffix and + become plain ``h``, ``he4``, ``cno``, ``nemgsi``, ``fe``. + * - ``imap_hit_l2_standard-intensity`` + - L2 + - 1 min + - 17 species, **204 native energy bins total**, in + :math:`\mathrm{cm^{-2}s^{-1}sr^{-1}(MeV/nuc)^{-1}}`. Variables are + renamed ``_standard_intensity``, with ``_stat_uncert_*``, + ``_sys_err_*`` and ``_total_uncert_*``. + * - ``imap_hit_l2_summed-intensity`` + - L2 + - 1 min + - Same, from the 67 summed bins. Variables + ``_summed_intensity``. + * - ``imap_hit_l2_macropixel-intensity`` + - L2 + - **10 min** + - 5 species (``h``, ``he4``, ``cno``, ``nemgsi``, ``fe``), 10 + species/energy combinations, 120 look directions. Variables + ``_macropixel_intensity``, dimensions ``(epoch, + _energy_mean, azimuth, zenith)``. Carries ``epoch_delta`` of + 5 minutes. + +Plus one non-CDF product: + +.. list-table:: + :header-rows: 1 + :widths: 30 10 12 48 + + * - Product + - Level + - Cadence + - Contents + * - HIT I-ALiRT record + - -- + - 1 min + - ``list[dict]`` of 6 electron, 4 proton and 2 helium space-weather + rates destined for the I-ALiRT database, **not** a CDF. See + :ref:`hit-ialirt`. + +Not produced here +----------------- + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Product + - Status + * - L3 PHA products (time, ion energy in MeV/nuc, charge Z per event) + - **Separate repository.** Algorithm document section 9.1. Needs the + decoded event records from L1A, which do not exist yet either - see + :ref:`hit-gap-events`. + * - L3 sectored products (pitch angle, gyrophase, 22.5 x 24 degree + skymaps) + - **Separate repository.** Algorithm document section 9.2. Consumes + ``imap_hit_l2_macropixel-intensity`` + MAG L1D. + * - L3 electron science products + - **Separate repository.** Algorithm document section 9.3. Consumes the + 6 I-ALiRT electron rates plus modelled response matrices. + * - Quicklook plots + - **No HIT quicklook code exists in this repository.** The algorithm + document does not specify any either. + +See :ref:`hit-l3-scope` for what those products need from L1A and L2. + +The processing chain +-------------------- + +**[DOC]** Algorithm document Figure 8, annotated with what is and is not here. + +.. code-block:: text + + CCSDS packets (APID 1251 hk, 1252 science, 1253 I-ALiRT) + | + | decommutation table (the frame byte map) + v + L1A raw particle counts + | imap_hit_l1a_hk + | imap_hit_l1a_counts-standard + | imap_hit_l1a_counts-sectored + | imap_hit_l1a_direct-events <-- raw binary only, not decoded + | + | livetimes + v + L1B particle count rates + | imap_hit_l1b_hk <-- from L0 again, not from L1A + | imap_hit_l1b_standard-rates + | imap_hit_l1b_summed-rates + | imap_hit_l1b_sectored-rates + | + | geometry factors, energy bin widths, efficiencies, + | selected by dynamic threshold state + v + L2 particle intensities + | imap_hit_l2_standard-intensity + | imap_hit_l2_summed-intensity + | imap_hit_l2_macropixel-intensity + | + | MAG L1D B-field, ADC-MeV coefficients, charge lookup tables, + | electron response matrices + v + L3 pitch angles, ion charge, science-quality electrons + *** NOT IN THIS REPOSITORY *** + +Note the two things that are **not** a simple level chain: + +#. **L1B housekeeping does not come from L1A housekeeping.** It re-reads the + same L0 CCSDS file with ``use_derived_value=True``, letting the XTCE + calibrators do the conversion. The L1A and L1B HK products are the same + packet, raw and derived. +#. **L2 standard intensity does not come from an L1B "standard" *product* in + the obvious way.** It takes the L1B standard rates (which are per-Particle- + ID arrays) and *then* does the cross-range summation into species/energy + bins. The equivalent summation for the summed product happens one level + earlier, at L1B. See :ref:`hit-l2-where-summing-happens`. + +How the CLI is wired +-------------------- + +**[CODE]** ``imap_processing/cli.py``, class ``Hit``. + +.. list-table:: + :header-rows: 1 + :widths: 10 18 36 36 + + * - Level + - Descriptor + - Dependencies expected + - Entry point + * - ``l1a`` + - (any) + - Exactly **2**: the L0 ``raw`` CCSDS file and the SPICE time kernels. + - ``hit_l1a(science_file, start_date)`` + * - ``l1b`` + - ``hk`` + - One L0 ``raw`` file. + - ``hit_l1b(path, "hk")`` + * - ``l1b`` + - ``standard-rates``, ``summed-rates``, ``sectored-rates`` + - Exactly one L1A CDF (loaded with ``load_cdf``). + - ``hit_l1b(dataset, descriptor)`` + * - ``l2`` + - (derived from the input) + - Exactly **5**: one L1B science CDF matching ``-rates``, plus **4** + ancillary CSVs matching ``-dt`` (one per dynamic threshold state). + - ``hit_l2(l1b_dataset, ancillary_files)`` + +Two things to know about this wiring: + +* **L1A takes a ``start_date``.** ``hit_l1a`` raises ``ValueError`` without + one. It is used to trim the day-boundary buffer - see + :ref:`hit-l1a-day-boundary`. +* **L2 dispatches on the input's ``Logical_source``**, not on the descriptor. + ``hit_l2`` inspects ``dependency_sci.attrs["Logical_source"]`` for + ``imap_hit_l1b_summed-rates`` / ``standard-rates`` / ``sectored-rates`` and + picks the processing function from that. If none match it silently returns + ``None``. + +Which ancillary files each L2 product needs +------------------------------------------- + +**[CODE]** The CLI hands ``hit_l2`` all four files it was given; the code then +picks by filename substring ``dt-factors``. The product determines which +*family* of four: + +.. list-table:: + :header-rows: 1 + :widths: 36 64 + + * - L2 product + - Ancillary family + * - ``imap_hit_l2_standard-intensity`` + - ``imap_hit_standard-dt{0,1,2,3}-factors__v.csv`` + * - ``imap_hit_l2_summed-intensity`` + - ``imap_hit_summed-dt{0,1,2,3}-factors__v.csv`` + * - ``imap_hit_l2_macropixel-intensity`` + - ``imap_hit_sectored-dt{0,1,2,3}-factors__v.csv`` + +Only the states actually present in the data are read - ``load_ancillary_data`` +takes ``set(dataset["dynamic_threshold_state"].values)``. Passing the wrong +family will raise ``StopIteration`` from the ``next(...)`` generator lookup, not +a friendly error. + +See :ref:`hit-ancillary` for the file format. + +Dimensions and coordinates +-------------------------- + +**[CODE]** Worth having in front of you when reading the code. + +.. list-table:: + :header-rows: 1 + :widths: 28 12 60 + + * - Coordinate + - Size + - Where it comes from + * - ``epoch`` + - n frames + - Mean of the first and last packet epoch in the science frame + (``calculate_epoch_mean``). For the L2 macropixel product it is + recomputed to the centre of the **collection** window, 10 minutes + before transmission. + * - ``sc_tick`` + - n packets + - Per-packet spacecraft time. Used as the dimension for the CCSDS header + fields, which are per-packet not per-frame. + * - ``epoch_livetime`` + - n frames + - A shadow of ``epoch`` attached to ``livetime_counter`` so that + filtering ``epoch`` for complete sectored sets does not also filter + livetime. See :ref:`hit-l1b-sectored`. + * - ``gain`` + - 2 + - 0 = high gain, 1 = low gain. Only used by ``sngrates``. + * - ``zenith`` + - 8 + - Declination bin centres, 11.25 to 168.75 degrees. + * - ``azimuth`` + - 15 + - Inclination bin centres, 12 to 348 degrees. + * - ``_energy_mean`` + - varies + - Per-species energy bin identifier, with + ``_energy_delta_plus`` / ``_delta_minus`` giving the bin + edges. + * - ``_index`` + - varies + - Plain integer index into a raw counter array, e.g. + ``l3fgrates_index`` runs 0-166. **This is the Particle ID** for the + FG arrays. diff --git a/docs/source/algorithm-code-documentation/hit/ialirt.rst b/docs/source/algorithm-code-documentation/hit/ialirt.rst new file mode 100644 index 0000000000..6cf4dbc5f5 --- /dev/null +++ b/docs/source/algorithm-code-documentation/hit/ialirt.rst @@ -0,0 +1,278 @@ +.. _hit-ialirt: + +I-ALiRT: The Real-Time Space Weather Product +============================================ + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**[DOC]** Sections 2.5.1, 2.5.3, 4.4 and 8 of the algorithm document. + +I-ALiRT (IMAP Active Link for Real-Time) is the continuous low-rate broadcast +used for space weather monitoring. HIT's contribution is **12 rates at a +1-minute cadence**: 6 electron, 4 proton and 2 helium. + +This product does **not** produce a CDF. It produces a ``list[dict]`` destined +for the I-ALiRT database, and it lives outside the ``hit`` package, in +``imap_processing/ialirt/l0/process_hit.py``. + +Why HIT can measure electrons at all +------------------------------------ + +**[DOC]** Two of HIT's ten apertures (``A0`` and ``B0``) are **I-ALiRT +apertures**. Behind their L1 detectors sits a 1500 um **L4** detector split +into an inner region (``L4Ai``/``L4Bi``) and an outer annulus +(``L4Ao``/``L4Bo``). + +Energetic ~0.5-1 MeV electrons are identified by requiring L1 **and** the outer +annulus of L4, **in anticoincidence with the central region of L4** - a +particle that also lights up the centre is a proton, not an electron. This is +the one place where HIT's detector stack is optimised for electrons rather than +ions. + +The I-ALiRT apertures are also **never affected by the dynamic thresholds** - +they stay in their nominal configuration even during the largest SEP events, +which is precisely when space weather monitoring matters most. + +The packet: a 60-slot subcommutation +------------------------------------ + +**[DOC]** Section 8.1. APID **1253**, 54 bytes, **one packet per second**. + +Each second carries **6 bytes of rate data** arranged as three 2-byte fields: + +.. code-block:: text + + FAST_RATE_1 2 bytes cycles through 4 values (period 4) + FAST_RATE_2 2 bytes cycles through 4 values (period 4) + SLOW_RATE 2 bytes cycles through 60 values (period 60) + +A **subcom counter** 0-59 says which slot this second carries. **A full I-ALiRT +set is 60 consecutive seconds = 1 minute**, which is why the public products +have a 1-minute cadence even though the packets arrive at 1 Hz. + +The fast rates repeat every 4 slots, so each of them is sampled 15 times per +minute: + +.. list-table:: + :header-rows: 1 + :widths: 16 42 42 + + * - subcom mod 4 + - ``FAST_RATE_1`` + - ``FAST_RATE_2`` + * - 0 + - L1A Trigger + - L1B Trigger + * - 1 + - IAevent Trigger + - IBevent Trigger + * - 2 + - NFORMAT + - Aevent Trigger + * - 3 + - L3A Trigger + - L3B Trigger + +The slow rate is the interesting one - 60 distinct quantities per minute, +covering singles, the 20 dedicated I-ALiRT rates, event counters, the ERATES +livetime block, coincidence rates, and five duplicated H/He science rates. +The full mapping is algorithm document **Table 38** (PDF pages 145-147). + +The 20 dedicated I-ALiRT rates +------------------------------ + +**[DOC]** Section 8.2 and Table 39. Slow-rate slots 16-35 carry +``I-ALiRT Rate 1`` through ``I-ALiRT Rate 20``. These are the **same** 20 rates +that appear in the science frame as ``ialirtrates`` (bytes 1148-1187, Table +23), so the two paths are cross-checkable. + +.. list-table:: + :header-rows: 1 + :widths: 22 24 54 + + * - Rates + - Quantity + - Meaning + * - 1-6 + - ``L4Ai`` dE bins 0-5 + - Six horizontal slices of energy deposit in the A-side inner L4. + Rates 1-2 are **below** the minimum-ionising peak (< ~270 keV); 3-4 + are **within** it (~270 keV - 1 MeV); 5-6 are **above** it (> ~1 MeV). + * - 7-10 + - ``L4Ai`` vs ``L3A`` bins 0-3 + - Four boxes in the (L4, L3) energy-loss plane. In order of increasing + L3 deposit: (1) ~1-10 MeV solar electrons, (2) ~30-200 MeV solar + protons, (3) >~500 MeV galactic protons, (4) everything else + (background). + * - 11-16 + - ``L4Bi`` dE bins 0-5 + - B-side mirror of 1-6. + * - 17-20 + - ``L4Bi`` vs ``L3B`` bins 0-3 + - B-side mirror of 7-10. + +Algorithm document Figure 15 shows the simulated L4 energy-loss distributions +that these boxes were drawn on. + +The 12 public products +---------------------- + +**[DOC]** Section 8.2, and this is the whole algorithm: + +.. list-table:: + :header-rows: 1 + :widths: 36 64 + + * - Product + - Formula + * - HIT low energy e- A side + - ``I-ALiRT Rate 1 + I-ALiRT Rate 2`` + * - HIT medium energy e- A side + - ``I-ALiRT Rate 5 + I-ALiRT Rate 6`` + * - HIT high energy e- A side + - ``I-ALiRT Rate 7`` + * - HIT low energy e- B side + - ``I-ALiRT Rate 11 + I-ALiRT Rate 12`` + * - HIT medium energy e- B side + - ``I-ALiRT Rate 15 + I-ALiRT Rate 16`` + * - HIT high energy e- B side + - ``I-ALiRT Rate 17`` + * - HIT low energy H omni + - ``H (6.0-8.0 MeV/nuc) L23FG`` + * - HIT medium energy H omni + - ``H (12.0-15.0 MeV/nuc) L23FG`` + * - HIT high energy H A side + - ``I-ALiRT Rate 8`` + * - HIT high energy H B side + - ``I-ALiRT Rate 18`` + * - HIT low energy He omni + - ``He-4 (6.0-8.0 MeV/nuc) L23FG`` + * - HIT high energy He omni + - ``He-4 (15.0-70.0 MeV/nuc) L23FG`` + +**[DOC]** These are explicitly labelled *"simplified algorithms (TBD)"*. +Section 8.2 states: *"After launch, the algorithms will be updated to use the +high energy SEP proton box as a background subtraction for the high energy +electrons."* In other words, **rate 8 will eventually be subtracted from rate +7** (and 18 from 17) with some coefficient. That is not defined yet. + +The supplemental (non-public) rates +----------------------------------- + +**[DOC]** Section 4.4 lists what else rides along in the I-ALiRT stream, all at +60-second cadence, so that the algorithms can be improved in flight: + +* 26 singles and trigger rates (13 per side) +* 5 engineering rates (the ERATES block: livetime, NUMTRIG, NUMREJECT, + NUMACCPHA, NUMACCNPHA) +* 3 species-identified proton rates +* 2 species-identified helium rates +* 12 rates (6 per side) of 1-D cuts on inner-L4 energy deposit - these are + I-ALiRT Rates 1-6 and 11-16 +* 8 rates (4 per side) of events triggering inner L4 and the corresponding L3 + but **not** the opposite-side L3 - these are I-ALiRT Rates 7-10 and 17-20 + +The implementation +------------------ + +**[CODE]** ``imap_processing/ialirt/l0/process_hit.py``. + +``HIT_PREFIX_TO_RATE_TYPE`` is a dict of three lists that name each +subcommutation slot. ``FAST_RATE_1`` and ``FAST_RATE_2`` are generated as 15 +repetitions of a 4-element pattern; ``SLOW_RATE`` is written out as 60 explicit +names. + +``process_hit`` then: + +#. Computes MET from ``sc_sclk_sec`` and ``sc_sclk_sub_sec`` with + ``calculate_time(..., 256)`` - **LSB = 1/256 s**, per the 7516-9054 GSW-FSW + ICD. +#. Calls ``find_groups(xarray_data, (0, 59), "hit_subcom", "met")`` to collect + each minute's 60 packets. +#. **Rejects any group whose ``hit_subcom`` values are not exactly + ``np.arange(60)``** - no duplicates, no gaps, in order. Rejected groups are + logged at INFO and skipped entirely; there is no partial-set path. +#. Logs a **warning** for any group containing a zero ``hit_status`` value, but + still emits the record. +#. ``create_l1`` zips the three name lists against the three data arrays into a + flat dict. +#. Emits one dict per group with the 12 products above, each as a + ``Decimal`` formatted to 3 decimal places, plus ``instrument: "hit"`` and + ``hit_epoch`` from ``met_to_ttj2000ns(hit_met)``. + +The 12 output keys are: + +.. code-block:: text + + hit_e_a_side_low_en hit_e_a_side_med_en hit_e_a_side_high_en + hit_e_b_side_low_en hit_e_b_side_med_en hit_e_b_side_high_en + hit_h_omni_low_en hit_h_omni_med_en + hit_h_a_side_high_en hit_h_b_side_high_en + hit_he_omni_low_en hit_he_omni_high_en + +which map one-to-one onto the document's 12 products. + +.. warning:: + + **[CODE]** The slow-rate slot names in + ``HIT_PREFIX_TO_RATE_TYPE["SLOW_RATE"]`` **do not match Table 38 of + document version 1.11.00** in three places. The document's revision history + for 1.11.00 says *"JGM updated Table 38 to reflect what is actually in the + packets"*, so the code is almost certainly working from the previous + revision. + + .. list-table:: + :header-rows: 1 + :widths: 14 40 46 + + * - Slot + - Code name + - Table 38 (v1.11.00) + * - 8, 9, 10 + - ``SLOW_RATE_08``, ``SLOW_RATE_09``, ``SLOW_RATE_10`` + - ``L1B``, ``L2B``, ``L3B`` + * - 38, 39 + - ``NASIDE_IALRT``, ``NBSIDE_IALRT`` + - ``NFORMAT``, ``NASIDE`` + * - 47-54 + - ``L12B``, ``L123B``, ``PENB``, ``SLOW_RATE_51``..``SLOW_RATE_54`` + - ``L12B``, ``2TEL``, ``PENA``, ``PENA?``, ``L12B``, ``2TEL``, + ``PENA``, ``PENA?`` + + **None of the 12 public products read any of the affected slots**, so the + output is currently correct. But the labels are wrong, and anyone adding a + product that uses them will get the wrong quantity. Note also that Table 38 + itself repeats ``L12B``/``2TEL``/``PENA``/``PENA?`` at slots 47-50 and + 51-54, which looks like a copy-paste artefact in the document - confirm with + the HIT team before changing the code. See :ref:`hit-gap-ialirt-slots`. + + The code comment also cites "Table 37 of the HIT Algorithm Document"; in + v1.11.00 this is **Table 38** (Table 37 is now the summed-rate geometry + factors). + +Relationship to the science frame +--------------------------------- + +Worth keeping straight, because the same numbers appear twice: + +.. list-table:: + :header-rows: 1 + :widths: 26 36 38 + + * - Quantity + - In the science frame + - In the I-ALiRT stream + * - The 20 I-ALiRT rates + - ``ialirtrates`` (APID 1252, bytes 1148-1187), decompressed at L1A, + rate-converted at L1B, **not used at L2** + - Slow-rate slots 16-35 (APID 1253), used to build the 12 public + products + * - The five H/He science rates + - ``l3fgrates`` entries (Range 3 foreground) + - Slow-rate slots 55-59, duplicated for real-time use + * - The ERATES block + - bytes 6-23 + - Slow-rate slots 40-44 + +The science-frame copies go through the full livetime and intensity chain; the +I-ALiRT copies do not. They are **raw counts** and are reported as such. diff --git a/docs/source/algorithm-code-documentation/hit/implementation-status.rst b/docs/source/algorithm-code-documentation/hit/implementation-status.rst new file mode 100644 index 0000000000..4e864f6dc4 --- /dev/null +++ b/docs/source/algorithm-code-documentation/hit/implementation-status.rst @@ -0,0 +1,520 @@ +.. _hit-implementation-status: + +Implementation Status and Known Gaps +==================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This page is the honest accounting of where the code stands against the +algorithm document. **Read it before proposing or estimating work.** + +Accurate as of a survey of ``imap_processing/hit`` and +``imap_processing/ialirt/l0/process_hit.py`` against algorithm document +*IMAP/HIT Science Algorithms* version 1.11.00, 2 June 2026 (Draft). If you +change something material, update this page in the same commit. + +Summary +------- + +.. list-table:: + :header-rows: 1 + :widths: 20 20 60 + + * - Level + - State + - Notes + * - L0 / L1A counts + - **Complete and exact** + - The frame byte map matches document Tables 11-26 byte for byte, and the + decompression matches the heritage C exactly. Validated against the + sample CSV exports in ``tests/hit/validation_data/``. + * - L1A direct events + - **Stub** + - The product exists and carries the raw binary. **Nothing decodes it.** + The single largest gap in HIT. + * - L1A sectored counts + - **Complete** + - The mod-10 subcommutation, complete-set detection and 10-minute + livetime shift all work, with the caveat that the shift is by array + position rather than by time. + * - L1A / L1B housekeeping + - **Complete** + - No Python algorithm - the same packet decommutated raw and derived, with + all conversions (including the piecewise thermistors) in the XTCE. + * - L1B science + - **Complete** + - Livetime fraction, rates, summed rates and sectored rates all + implemented and tested. Uncertainty summing is linear, as the document + specifies. + * - L2 + - **Complete for what it claims** + - All three intensity products work. Real gaps: no spin-rate correction, + the ``b`` subtraction is in the wrong place if ``b`` ever becomes + non-zero, and the L4 rates never become intensities. + * - I-ALiRT + - **Working, labels stale** + - Produces all 12 public products correctly. Three blocks of slow-rate + slot names do not match Table 38 as revised in v1.11.00. + * - L3 + - **Out of scope, correctly** + - Separate repository. See :ref:`hit-l3-scope`. + * - Quicklook + - **Not started** + - No HIT quicklook code exists here, and the document does not specify + any. + +Ranked list of things to fix +---------------------------- + +Highest value first, in the judgement of whoever last surveyed this. Each has a +detailed entry below. + +#. :ref:`hit-gap-events` - PHA event records are never decoded. Blocks all of + L3 section 9.1. +#. :ref:`hit-gap-spinrate` - the 15th inclination bin is never corrected for + spin-rate deviation, which the document explicitly assigns to the ground. +#. :ref:`hit-gap-geometric-mean` - arithmetic mean used where the document + specifies the geometric mean, in every product above L1A. +#. :ref:`hit-gap-livetime-shift` - the sectored livetime shift is by array + position, not by time; a dropped frame silently mispairs the data. +#. :ref:`hit-gap-background` - the background term ``b`` is subtracted after + the division rather than from the counts. +#. :ref:`hit-gap-incomplete-frames` - incomplete science frames are silently + discarded with no quality flag. +#. :ref:`hit-gap-ialirt-slots` - three blocks of I-ALiRT slow-rate slot names + disagree with the revised Table 38. +#. :ref:`hit-gap-livetime-range` - no detection of out-of-range livetime, which + the document asks for. +#. :ref:`hit-gap-l4rates` - ``l4fgrates`` / ``l4bgrates`` and ``ialirtrates`` + are carried to L1B and then dropped. +#. :ref:`hit-gap-sector-order` - the sectored ancillary table relies on row + order for the declination assignment. +#. :ref:`hit-gap-systematics` - the chance-coincidence and livetime systematic + corrections are defined but unimplemented (and unquantified by the + instrument team). +#. :ref:`hit-gap-dead-code` - a handful of unused constants and stale + comments. + +Hard failures in the code +------------------------- + +Explicit exceptions, so you know what a bad input looks like: + +.. list-table:: + :header-rows: 1 + :widths: 42 58 + + * - Location + - Condition + * - ``cli.py`` ``Hit.do_processing`` + - ``NotImplementedError`` for any data level other than l1a/l1b/l2. + * - ``cli.py`` ``Hit.do_processing`` + - ``ValueError`` if more than 2 dependencies for l1a, more than one L1A + file for l1b, not exactly 5 dependencies for l2, more than one science + file for l2, or not exactly 4 ancillary files for l2. + * - ``hit_l1a.hit_l1a`` + - ``ValueError`` if ``packet_date`` is missing. + * - ``hit_l1a.subset_sectored_counts`` + - ``ValueError`` if no complete mod-10 set is found. + * - ``hit_l1a.subset_livetime`` + - ``ValueError`` if the epoch array is empty, or if the first complete + set starts fewer than 10 frames into the file. + * - ``hit_l1b.hit_l1b`` + - ``ValueError`` for a descriptor other than ``hk``, ``standard-rates``, + ``summed-rates`` or ``sectored-rates``. + * - ``hit_l2.load_ancillary_data`` + - Bare ``StopIteration`` if no ancillary file matches + ``dt-factors`` for a state present in the data. **Not a friendly + error.** + * - ``decom_hit.assemble_science_frames`` + - ``IndexError`` on ``starting_indices[0]`` if no valid science frame is + found in the file. + +Silent behaviours worth knowing +------------------------------- + +* ``hit_l2.hit_l2`` returns ``None`` if the input's ``Logical_source`` matches + none of the three expected L1B products. +* ``hit_l1b.process_science_data`` returns ``None`` for an unrecognised + descriptor (unreachable - ``hit_l1b`` validates first). +* Incomplete science frames are dropped with a ``print``, not a log or a flag. +* Sectored frames outside a complete 10-frame run are dropped entirely. +* ``add_cdf_attributes`` at L1A warns and continues for a variable missing from + the attribute YAML; at L2 it logs an error and continues. + +Detailed entries +---------------- + +.. _hit-gap-events: + +1. PHA event records are never decoded +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Severity: high. The single biggest missing piece in HIT.** + +**[CODE]** ``imap_hit_l1a_direct-events`` contains exactly one data variable, +``pha_raw``: the concatenated binary of packets 6-19 of each science frame, as +a string. Nothing parses it. + +**[DOC]** The format is fully specified across sections 4.2.5-4.2.8 (PDF pages +25-33): + +* **Event Buffer header** - 2 bytes at frame bytes 1572-1573, the number of + event records. Unused buffer space is filled with ``0x00``; empty packets are + not transmitted. +* **Event Record Header (Table 4)** - 32 bits: Particle ID (bits 0-7), + priority buffer number (8-12), STIM tag (13), HAZ tag (14), time tag / + latency bits (15-18), A/B tag (19), unread-ADCs flag (20), long event flag + (21), culling flag (22), spare (23). +* **ADC fields (Table 5, Figure 9)** - 20 bits each: 12-bit signal (11 bits + plus overflow), 6-bit detector ID, 1 gain bit (0 = high, 1 = low), 1 + End-of-Record bit. The **EOR bit**, not a count, marks the end of the record. +* **Padding (Table 6)** - records with an odd number of ADC fields are padded + with 4 bits to a byte boundary. Minimum record is 2 ADC fields = 72 bits = 9 + bytes. Maximum is 54 ADC hits. +* **Extended Header Block** - 3 bytes appended when bit 21 is set: detector + group flags (Table 7), ``DEINDEX`` (9 bits, 0-399) and ``EPINDEX`` (7 bits, + 0-127), the onboard matrix indices. +* **STIM Information Block** - 3 more bytes when bits 13 **and** 21 are both + set: a seconds-within-minute counter, ``DACLEVEL`` (0-31) and + ``DACCONFIGURATION`` (0-14). +* **Detector address table (Table 8)** - the 6-bit detector ID to name + mapping, 0-63. + +Two design notes from the document worth carrying into any implementation: + +* **A single bit error in an EOR bit can make every subsequent event in the + buffer unreconstructible.** The document acknowledges this and floats + possible mitigations. A decoder should fail the rest of the buffer + gracefully rather than produce garbage. +* **Latency bits do not necessarily correspond to the frame's timestamp.** An + event may sit in a priority buffer for more than a minute before being + telemetered. The 4 latency bits duplicate the low 4 bits of the onboard + minute counter; the document says explicitly that *"it is up to the user of + the data to determine what value of the latency bits precisely corresponds to + times associated with given Science Frames."* + +**Consequence:** all of L3 section 9.1 (incident ion energy, ADC-MeV +conversion, cosine correction, charge calculation) is blocked, and the early- +mission plan to characterise ion contamination in the I-ALiRT electron channels +from PHA data is blocked with it. + +.. _hit-gap-spinrate: + +2. No spin-rate correction to the inclination bins +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Severity: high for anisotropy science. Affects the macropixel product.** + +**[DOC]** Section 4.2.2: + + *"The 15 inclination 24 degree bins (labeled 0-14) are determined from + timing (incrementing every second), with the assumption that the spin rate + is the nominal 4 rpm. Deviation of the spin rate from 4 rpm will change the + angular width of the fifteenth bin (smaller than 24 degrees for spin <= 4 + rpm, and larger for spin >= 4 rpm). This needs to be corrected on the + ground. No onboard correction is planned. A new spin phase zero resets the + inclination bin index."* + +**[CODE]** ``AZIMUTH_ANGLES`` is a fixed array of 15 bin centres at 12, 36, +..., 348 degrees. Nothing reads the actual spin period or spin phase; no HIT +code touches SPICE geometry at all. + +**Consequence:** the 15th inclination bin (``azimuth = 348``) has an incorrect +angular width, and therefore an incorrect effective geometry factor, whenever +the spin rate is not exactly 4 rpm - which is always. The document does not +specify the correction, only that it is required, so implementing this needs +the HIT team to define what "corrected" means (rebinning? a per-bin width +array? an effective geometry factor scaling?). + +.. _hit-gap-geometric-mean: + +3. Arithmetic mean used instead of geometric mean +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Severity: medium. Affects every energy coordinate above L1A.** + +**[DOC]** Section 6.1 specifies the geometric mean +:math:`\sqrt{E_{min} E_{max}}` as the characteristic energy, with +``delta_plus`` and ``delta_minus`` measured from it. The document explains the +reasoning: the true mean energy of particles in a bin depends on the spectrum, +so the label is only characteristic and the **edges** are the real quantity. + +**[CODE]** ``hit_utils.add_energy_variables``: + +.. code-block:: python + + energy_mean = np.round( + np.mean(np.array([energy_min_values, energy_max_values]), axis=0), 3 + ).astype(np.float32) + +**Consequence:** ``_energy_mean``, ``_energy_delta_plus`` and +``_energy_delta_minus`` are wrong in ``imap_hit_l1a_counts-sectored``, +``imap_hit_l1b_summed-rates``, ``imap_hit_l1b_sectored-rates``, and all three +L2 products. The error is largest for the widest bins. For the sectored Fe +4.0-12.0 MeV/nuc bin the arithmetic mean is 8.00 and the geometric mean is +6.93 - a 15% shift in the plotted energy. For narrow standard bins such as +H 3.2-3.6 MeV/nuc the difference is under 0.2%. + +The **bin edges are recoverable** in either convention +(``mean - delta_minus`` and ``mean + delta_plus``), so no information is lost - +but any consumer plotting against ``energy_mean``, or comparing to another +instrument that uses the geometric convention, will be off. + +**Fix:** one line in ``add_energy_variables``. Check the CDF variable +attributes and any validation CSVs at the same time. + +.. _hit-gap-livetime-shift: + +4. The sectored livetime shift is positional, not temporal +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Severity: medium. Silent wrong answer in the presence of data gaps.** + +**[CODE]** ``hit_l1a.subset_livetime`` locates the current block's first and +last epochs in ``epoch_livetime`` and then slices **10 array positions +earlier**: + +.. code-block:: python + + start_trimmed = max(start_idx - 10, 0) + end_trimmed = max(end_idx - 10, 0) + +Similarly ``hit_l1b.sum_livetime_10min`` sums in fixed 10-element windows, and +``hit_l1a.find_complete_mod10_sets`` assumes consecutive array entries are +consecutive minutes. + +**[DOC]** Section 6.2 and Figure 14 define the relationship in **time**: block +*n*'s counts pair with the livetimes accumulated during the preceding 10 +minutes. + +**Consequence:** these coincide only when there are no missing science frames. +A single dropped frame anywhere in or before a sectored block shifts the +pairing by one minute, silently, with no flag. Since incomplete frames are +already discarded without a trace (:ref:`hit-gap-incomplete-frames`), this is +reachable in real data. + +**Fix:** index the livetime by epoch difference rather than array position, and +raise or flag when the expected 10 minutes are not all present. + +.. _hit-gap-background: + +5. The background term is subtracted in the wrong place +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Severity: low today, high the day ``b`` becomes non-zero.** + +**[DOC]** Equations 12, 14 and 16 subtract background **counts** from the +numerator before dividing: + +.. math:: + + j_i \propto \frac{\sum_j q_{ij} - b_{ij}}{\sum_j \sigma_{ij}\lambda_{ij}} + +**[CODE]** ``hit_l2.calculate_intensities`` subtracts ``b`` from the finished +intensity: + +.. code-block:: python + + intensity = (rates / (delta_time * delta_e * geometry_factor * efficiency)) - b + +These agree only if ``b`` is delivered in intensity units rather than counts. +**Every ``b`` in the current ancillary CSVs is 0**, so there is no numerical +difference today. + +There is a second issue in the same place: the uncertainty arrays go through +the identical call, so a non-zero ``b`` would also be subtracted from the +uncertainties, which is meaningless. + +**Fix when needed:** establish the units of ``b`` with the HIT team, then +either move the subtraction into the numerator or document that the CSV column +is an intensity - and give the uncertainty arrays their own path. + +.. _hit-gap-incomplete-frames: + +6. Incomplete science frames vanish silently +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Severity: medium.** + +**[CODE]** ``decom_hit.assemble_science_frames`` keeps only runs of 20 packets +whose grouping flags match ``FLAG_PATTERN`` **and** whose sequence counters are +consecutive. Everything else is dropped. There is an explicit ``TODO``: + + *"The code currently skips all incomplete science frames. Only discard + incomplete science frames in the middle of the CCSDS file or use fill + values?"* + +Notification of dropped packets at the file boundaries is done with ``print``, +not ``logger``, so it does not reach the batch job's logs in a structured way. +There is no quality flag anywhere in the HIT products (contrast SWE, which has +``SweL1bFlags`` in ``imap_processing/quality_flags.py`` - HIT has no entry +there). + +**Consequence:** a user cannot tell a quiet minute from a lost minute, and the +positional livetime shift above turns lost minutes into wrong answers. + +.. _hit-gap-ialirt-slots: + +7. I-ALiRT slow-rate slot names are stale +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Severity: low today (no product reads the affected slots), medium for anyone +extending the product.** + +See the warning in :ref:`hit-ialirt` for the slot-by-slot comparison. Slots +8-10, 38-39 and 47-54 in ``HIT_PREFIX_TO_RATE_TYPE["SLOW_RATE"]`` disagree with +Table 38 as revised in document version 1.11.00. The code comment also cites +"Table 37", which was the table number in an earlier revision. + +Note that Table 38 itself repeats a block of four names at slots 47-50 and +51-54, which looks like an error in the document. **Confirm with the HIT team +before changing anything.** + +.. _hit-gap-livetime-range: + +8. No detection of out-of-range livetime +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Severity: low.** + +**[DOC]** Section 6.2: the piecewise livetime conversion *"will be incorrect +for values of live time below 0.001667% (count 16,000). This situation should +be detectable via other mnemonics, such as NUMTRIG (Table 12) or STIM event +counts (PBUFRATES #29 and #30, Table 16)."* + +**[CODE]** ``livetime_fraction_calculation`` applies the piecewise function +unconditionally. Nothing cross-checks ``num_trig`` or ``pbufrates[29]`` / +``pbufrates[30]``, and there is no flag on the output. + +The document does not say what the check should be, so this needs HIT team +input before it can be implemented. + +.. _hit-gap-l4rates: + +9. The L4 and I-ALiRT rate arrays stop at L1B +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Severity: low. Possibly correct by design.** + +**[CODE]** ``l4fgrates`` (48), ``l4bgrates`` (24) and ``ialirtrates`` (20) are +decommutated at L1A, livetime-corrected at L1B, and then **never referenced +again**. ``STANDARD_PARTICLE_ENERGY_RANGE_MAPPING`` covers only R2, R3 and R4. + +**[DOC]** Section 4.2.1 introduces ``RNG2I``/``RNG3I``/``RNG4I`` - ions that +entered through an I-ALiRT aperture, lost energy in L4, and therefore sit at +higher incident energy - as first-class new-for-HIT ranges. Tables 25 and 26 +give the 48 foreground and 24 background rates with their own Particle IDs. +But **Table 31 (the L2 standard products) does not include them**, and none of +the geometry-factor tables (32-37) has entries for them. + +So the current behaviour matches the document. The open question is whether +the instrument team intends them to become an L2 product later. Worth asking +before anyone "cleans up" the unused arrays. + +.. _hit-gap-sector-order: + +10. Sectored ancillary data depends on CSV row order +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Severity: low, but it is a silent failure mode.** + +**[CODE]** ``hit_l2.get_species_ancillary_data`` groups the ancillary +DataFrame by ``lower energy (mev)`` and collects each column with +``.apply(list)``. For the sectored family, the resulting inner list is assumed +to be sectors 0-7 **in file order**. The ``Sector`` column is present in the +CSV and is never read. + +**Consequence:** a redelivered ancillary file with the rows sorted differently +would scramble the declination assignment with no error. + +**Fix:** sort by, or index on, the ``sector`` column. + +.. _hit-gap-systematics: + +11. Systematic uncertainty corrections are unimplemented +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Severity: none yet - the instrument team has not quantified them.** + +**[DOC]** Section 4.3.2 names two sources: + +* **Chance coincidences.** Mitigated by detector segmentation; *"further + consistency checks will be employed during ground processing"* - unspecified. +* **Livetime errors at high rates.** The counter does not include the + coincidence window opened by each trigger. The proposed correction is + :math:`\mathrm{livetime_{corr}} = \mathrm{livetime} + (\Delta t \times N_{trig})`, + with :math:`\Delta t` *"to be explored using high intensity runs at the + accelerators during HIT calibrations"*. + +**[CODE]** Neither correction exists. ``add_systematic_uncertainties`` writes +zeros and ``add_total_uncertainties`` does the quadrature sum - which is +exactly what the document prescribes for launch. The plumbing is in place; only +the numbers are missing. + +Note that ``num_trig`` (NUMTRIG) **is** available at L1A and L1B, so when +:math:`\Delta t` arrives the correction is a small change in +``livetime_fraction_calculation``. Remember the document's caveat that NUMTRIG +is accumulated **only during even-numbered seconds**, so it represents half the +triggers in the minute; the ``.HAZ`` counters cover the odd seconds. + +.. _hit-gap-dead-code: + +12. Dead constants and stale comments +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Severity: cosmetic.** + +* ``l1b/constants.py`` ``LIVESTIM_PULSES = 270`` - documented as being used to + calculate the fractional livetime; never referenced. +* ``l0/constants.py`` ``MOD_10_PATTERN`` - superseded by + ``np.arange(10)`` inside ``find_complete_mod10_sets``; never referenced. +* ``ialirt/l0/process_hit.py`` cites "Table 37"; the correct reference in + v1.11.00 is Table 38. +* ``hit_l1a.hit_l1a`` docstring says the L0 file has "a 20-minute buffer before + and after the processing day"; the document specifies 5 minutes before and + 15 after. +* ``hit_l1b.process_standard_rates_data`` omits ``l4fgrates_index`` and + ``l4bgrates_index`` from its explicit coordinate list (they get created + implicitly). +* ``hit_utils.initialize_particle_data_arrays`` creates + ``_energy_mean`` as an ``int8`` zero array, which + ``add_energy_variables`` immediately overwrites with ``float32``. + +Test coverage +------------- + +**[CODE]** ``imap_processing/tests/hit/`` is reasonably thorough for what is +implemented: + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - File + - Covers + * - ``test_decom_hit.py`` + - Bit parsing, frame flag matching, sequence checks, decompression, epoch + mean, full ``decom_hit``. + * - ``test_hit_l1a.py`` + - Sectored subcommutation, livetime coordinate handling, complete-set + detection, uncertainties, CDF attributes, processing-day filtering, plus + **validation against** ``hskp_sample_raw.csv`` and ``sci_sample_raw.csv``. + * - ``test_hit_l1b.py`` + - Rates, 10-minute livetime summing, all three science products, + housekeeping, the livetime piecewise fit, plus **validation against** + ``hskp_sample_eu_3_6_2025.csv`` and + ``hit_l1b_standard_sample2_nsrl_v4_3decimals.csv``. + * - ``test_hit_l2.py`` + - Ancillary loading and reshaping, intensity calculation, systematic and + total uncertainties, all three L2 products, the 10-minute regrouping. + * - ``test_hit_utils.py`` + - APID lookup, leak-variable concatenation, housekeeping, energy + variables, cross-range summing. + * - ``tests/ialirt/unit/test_process_hit.py`` + - The I-ALiRT grouping and the 12 products, against + ``hit_ialirt_sample.ccsds`` / ``.csv``. + +Notable coverage gaps: nothing exercises ``pha_raw`` beyond its presence, +nothing tests a science file with missing or corrupt frames, and there is no +end-to-end L1A-to-L2 validation against instrument-team-supplied L2 values. diff --git a/docs/source/algorithm-code-documentation/hit/index.rst b/docs/source/algorithm-code-documentation/hit/index.rst new file mode 100644 index 0000000000..51cb5ed111 --- /dev/null +++ b/docs/source/algorithm-code-documentation/hit/index.rst @@ -0,0 +1,266 @@ +:orphan: + +.. _hit-index: + +HIT +=== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +.. currentmodule:: imap_processing.hit + +This is the HIT (High-energy Ion Telescope) instrument module, which contains +the code for processing data from the HIT instrument. + +Purpose of these pages +---------------------- + +These pages are a **condensed, self-contained working reference** for the HIT +processing algorithms, written so that a developer (human or AI agent) can get +productive without reading the full algorithm document. + +They are a summary of the source document below plus what the code in +``imap_processing/hit`` actually does. Where the two disagree, that is called +out explicitly in :ref:`hit-implementation-status`. + +.. _hit-source-documents: + +Source documents +---------------- + +**None of these are redistributed in this repository.** This is an open-source +repository and the mission documents are not ours to publish. Request them from +the HIT instrument team at NASA Goddard Space Flight Center or the SDC document +store. + +.. list-table:: + :header-rows: 1 + :widths: 22 78 + + * - Short name + - Reference + * - **Algorithm document** + - *IMAP/HIT Science Algorithms*, Version 1.11.00, dated 2 June 2026. + Prepared by J. Grant Mitchell, Eric Christian and Alessandro Bruno, + NASA Goddard Space Flight Center. 169 pages, marked **Draft**. The + primary source for these pages. Referred to below as "the algorithm + document". + * - **HIT_L1A_L1B_Table_JGM_2_6_24** + - Cited by section 6.2 of the algorithm document as the authoritative + "full list of all of the data products expected to be livetime + corrected". Not held by the SDC; the code derives the same list from + the frame format tables instead. + * - **HIT-FSW-DESC-004** + - Flight software description cited by section 4.2.2 for the + ``SECTORRATES`` subcommutation scheme (120 look directions, one + species/energy per minute). + * - **HIT-ELEC-HDBK-0008** + - Electronics handbook; the source of the housekeeping voltage and + thermistor conversions reproduced in algorithm document Tables 29-30. + In this repository those conversions live in the XTCE, not in Python. + * - **STEREO-CIT-CIT-002.F** + - *SEP HIT and Central MISC Processors Flight Software Requirements*. + Heritage document cited for Particle ID semantics. + * - **HIT-SYS-TRD-0004** + - FPGA trade study. Background only; nothing in the code depends on it. + * - **Gehrels 1986** + - DOI `10.1086/164079 `_. The source of + the asymmetric Poisson uncertainties used at L1A. + * - **Heritage instruments** + - STEREO/LET is the direct ancestor (matrices, rate definitions, the + rate compression scheme, the priority buffers). Parker Solar + Probe/EPI-Hi contributed the PHASICs. ACE/SIS, Voyager/CRS, + Voyager/LECP and STEREO/HET are cited for the dE/dx vs residual-E + technique. + +.. tip:: + + If you hold a copy of the algorithm document, put it in ``docs/reference/``. + That directory is gitignored, so it will never be committed, and the section + index in :ref:`hit-reference-tables` is written against that location. + +.. important:: + + Two conventions used throughout these pages: + + * **[DOC]** marks a statement taken from the algorithm document. It + describes the intended behavior, which may not be what the code does yet. + * **[CODE]** marks a statement verified against ``imap_processing/hit`` + (or ``imap_processing/ialirt/l0/process_hit.py`` for the real-time + product). + + When those conflict, the code is what runs and the document is what the + instrument team expects. Both matter; do not silently "fix" one to match + the other without asking. The document is also still marked **Draft**, and + its most recent revision note says Table 38 was changed "to reflect what is + actually in the packets" - so for HIT the document is a moving target. + +.. warning:: + + **This repository stops at L2.** HIT's L3 (pitch angle and gyrophase + distributions from L2 sectored intensities plus MAG, ion charge and energy + from the raw PHA events, and science-quality electron intensities) is + produced by a separate repository run closer to the science team. Section 9 + of the algorithm document - 16 of its 169 pages, and by far the most + algorithmically dense part - describes work that does **not** belong here. + See :ref:`hit-l3-scope` before starting anything that looks like a charge + calculation, a cosine correction, or a pitch angle. + + The one thing that *is* ours and looks like L3 work is **decoding the PHA + event records at L1A**. That is currently unimplemented. See + :ref:`hit-gap-events`. + +Which page to read +------------------ + +Read only what you need. Each page is designed to be loaded on its own. + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Page + - Read it when you need to know... + * - :ref:`hit-overview` + - What HIT physically is, how a measurement happens, and the vocabulary + (aperture, range, matrix, Particle ID, FGRATES/BGRATES, dynamic + threshold, sector, science frame). **Start here if you are new.** + * - :ref:`hit-data-products` + - The full product inventory, exact ``logical_source`` strings, what + feeds what, and how the CLI is wired. **The "what goes into what" + map.** + * - :ref:`hit-l1a` + - Science frame assembly from 20 packets, the 16-to-32 bit rate + decompression, the byte-for-byte frame layout, the sectored-rate + subcommutation, Poisson uncertainties, and the day-boundary buffer + rules. + * - :ref:`hit-l1b` + - The livetime fraction piecewise fit, counts to rates, the summed-rate + definitions, the 10-minute sectored livetime shift, and housekeeping + engineering-unit conversion. + * - :ref:`hit-l2` + - The intensity equation, geometry factors and efficiencies, dynamic + threshold state selection, range summation, and the macropixel + 10-minute regrouping. + * - :ref:`hit-ancillary` + - The twelve ``*-dt-factors`` CSVs, their exact column format, the + XTCE calibrators, and what SPICE is (and is not) used for. + * - :ref:`hit-ialirt` + - The 60-second subcommutated real-time product and the 12 space-weather + rates derived from it. + * - :ref:`hit-l3-scope` + - What L3 is, why it is not in this repository, and exactly what L1A and + L2 have to hand it. **Read before writing any charge or pitch angle + code.** + * - :ref:`hit-implementation-status` + - What is implemented, what is stubbed, where the code deviates from the + document, and what is not written at all. **Read before proposing + work.** + * - :ref:`hit-reference-tables` + - Where the big tables live (PDF page ranges, ancillary CSVs, XTCE). + Deliberately *not* reproduced inline. + +.. toctree:: + :maxdepth: 1 + + overview + data-products + l1a + l1b + l2 + ancillary + ialirt + l3-scope + implementation-status + reference-tables + +Ten-second orientation +---------------------- + +* HIT is a **stack of segmented solid-state detectors (SSDs)** measuring + ~2-40 MeV/nucleon ions from H to Ni. It is a direct descendant of + STEREO/LET, with PHASIC front-end chips from Parker Solar Probe/EPI-Hi. + Together with SWAPI and CoDICE it gives IMAP continuous ion coverage from + 0.1 keV to 40 MeV/nucleon. +* The measurement technique is **dE/dx vs residual E**: the energy a particle + deposits in the detector it passes *through*, plotted against the energy it + deposits in the detector it *stops* in, identifies the species and energy. + Which detectors were hit also gives the arrival direction. +* **Almost all the species identification happens on board.** The flight + software sorts each event through a *matrix* (a 128 x 400 lookup in + dE vs E' space) and increments a counter. What comes down is therefore + already "counts of Fe between 12 and 15 MeV/nuc that stopped in L3" - not + raw physics. Ground processing is overwhelmingly **bookkeeping, livetime + correction, and unit conversion**, not particle identification. +* Three **penetration ranges** define three sets of counters: + ``R2`` (L1L2, stopped in L2), ``R3`` (L2L3, stopped in L3), ``R4`` + (L3AL3B, penetrating). Their counters are ``l2fgrates``/``l2bgrates``, + ``l3fgrates``/``l3bgrates``, ``penfgrates``/``penbgrates``. "FG" = + foreground (identified species and energy), "BG" = background (broad + regions of the matrix that are not on an element track). +* The unit of telemetry is the **HIT Science Frame**: one minute of data, + 5240 bytes, spread over **20 CCSDS packets** (APID 1252, 262 bytes each). + The first 6 packets are the fixed-format counters and rates (bytes + 0-1571); the last 14 are the variable-length **Event Buffer** of raw pulse + heights. +* Every rate in the frame is **compressed from 24 (or 32) bits to 16** by a + biased-exponent / hidden-one scheme inherited from STEREO/LET, and must be + expanded on the ground. +* **Sectored rates** give anisotropy: the 8 science apertures plus the + spacecraft spin divide the sky into **8 declination x 15 inclination = 120 + look directions**. Only one of **10 species/energy combinations** is sent + per minute, so a complete sectored set takes **10 minutes** - and it is + telemetered **10 minutes after it was collected**, so it must be paired + with the *previous* block's livetime. +* Processing chain in this repository: + ``CCSDS packets -> L1A (decompressed counts) -> L1B (livetime-corrected + rates) -> L2 (intensities in cm^-2 s^-1 sr^-1 (MeV/nuc)^-1)``. L3 is not + ours. +* There is also a **1-minute I-ALiRT** product built from a 60-slot + subcommutated 1 Hz packet (APID 1253), living under + ``imap_processing/ialirt/``. + +Where the code lives +-------------------- + +.. code-block:: text + + imap_processing/hit/ + hit_utils.py HitAPID, CDF attr manager, housekeeping + reshaping, energy-bin variables, and the + cross-range summing helpers shared by L1B + summed rates and L2 standard intensity + l0/constants.py COUNTS_DATA_STRUCTURE (the frame byte map), + FLAG_PATTERN, FRAME_SIZE=20, MOD_10_MAPPING, + ZENITH_ANGLES, AZIMUTH_ANGLES, + MANTISSA_BITS=12, EXPONENT_BITS=4 + l0/decom_hit.py science frame assembly, bit parsing, + 16->32 bit rate decompression + l1a/hit_l1a.py L1A products: housekeeping, standard counts, + sectored counts, direct events; Gehrels + uncertainties; processing-day filtering + l1b/constants.py SECTORS=15, fill values, + SUMMED_PARTICLE_ENERGY_RANGE_MAPPING + l1b/hit_l1b.py livetime fraction, counts->rates, summed + rates, sectored rates, housekeeping (derived) + l2/constants.py VALID_SPECIES, VALID_SECTORED_SPECIES, + STANDARD_PARTICLE_ENERGY_RANGE_MAPPING, + SECONDS_PER_MIN/SECONDS_PER_10_MIN, N_AZIMUTH + l2/hit_l2.py intensity calculation, ancillary loading by + dynamic threshold state, systematic and total + uncertainties, macropixel 10-minute regrouping + packet_definitions/ + hit_packet_definitions.xml HIT_HSKP (APID 1251) + HIT_SCIENCE (APID 1252), + including all housekeeping EU calibrators + + imap_processing/ialirt/ + l0/process_hit.py the whole HIT I-ALiRT algorithm + packet_definitions/ialirt_hit.xml HIT I-ALiRT packet fields + + imap_processing/cdf/config/imap_hit_global_cdf_attrs.yaml + imap_processing/cdf/config/imap_hit_l1a_variable_attrs.yaml + imap_processing/cdf/config/imap_hit_l1b_variable_attrs.yaml + imap_processing/cdf/config/imap_hit_l2_variable_attrs.yaml + imap_processing/cli.py (class Hit) dependency wiring per level + imap_processing/tests/hit/ tests, L0 test data, ancillary CSVs, + validation CSVs diff --git a/docs/source/algorithm-code-documentation/hit/l1a.rst b/docs/source/algorithm-code-documentation/hit/l1a.rst new file mode 100644 index 0000000000..146677c5f4 --- /dev/null +++ b/docs/source/algorithm-code-documentation/hit/l1a.rst @@ -0,0 +1,381 @@ +.. _hit-l1a: + +L0 to L1A: Frame Assembly, Decompression and Counts +=================================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**[DOC]** Section 5 of the algorithm document. + +* **Input**: CCSDS packets (APID 1251 housekeeping, 1252 science) +* **Processing requirement**: the decommutation table (the frame byte map) +* **Output**: L1A CDF files + +L1A does no physics. It turns a byte stream into labelled integer arrays, +expands the onboard rate compression, and attaches Poisson uncertainties. + +.. _hit-l1a-day-boundary: + +Day boundaries and the input buffer +----------------------------------- + +**[DOC]** A day's L0 request must include **all science packets for the day +plus a 5-minute buffer at the start (from 23:55 of the previous day) and a +15-minute buffer at the end (to 00:15 of the following day)**. The reason is +the 10-minute sectored block: without the buffer, a block straddling midnight +cannot be completed, and the previous block's livetime would be missing. + +.. note:: + + The document writes the leading buffer as "11:55 of the previous day", + which is a 12-hour-clock slip for 23:55. The 10-minute figure quoted in the + same paragraph is the sectored block length, not the buffer length. + +**[CODE]** ``hit_l1a`` takes ``packet_date`` (``YYYYMMDD``, wired to +``self.start_date`` in the CLI) and raises ``ValueError`` if it is missing. Its +docstring says the L0 file has "a 20-minute buffer before and after the +processing day", which does not match the document's asymmetric 5/15 split. +The code does not depend on the buffer being any particular size - it just +trims - so this is a docstring inaccuracy rather than a bug, but it is worth +knowing which number to believe. + +Trimming is done by ``filter_dataset_to_processing_day``: + +* Convert ``epoch`` (TT2000 ns) to ``datetime64`` via + ``et_to_datetime64(ttj2000ns_to_et(...))`` and keep indices whose date + equals the processing day. +* Optionally (``sc_tick=True``, used for the standard counts product) do the + same on ``sc_tick`` via ``met_to_datetime64``, because the CCSDS header + fields live on a per-packet dimension rather than per-frame. +* For **sectored** data the filter is applied to an array of *mean epochs per + 10-minute set*, repeated 10 times, rather than to the per-frame epochs. That + keeps a block that straddles midnight together, assigned to whichever day + its centre falls in. + +.. note:: + + **[CODE]** There is a standing ``TODO`` in ``process_science`` about this: + a frame whose mean epoch lands in the processing day may still contain + packets from the previous day, and the ``sc_tick`` filter will drop those + header rows. Nobody has decided whether that matters. + +Assembling a science frame +-------------------------- + +**[CODE]** ``decom_hit.assemble_science_frames``. + +A valid frame is **20 consecutive packets** whose CCSDS grouping flags match: + +.. code-block:: text + + FLAG_PATTERN = [1, 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0, 2] + ^ first packet 18 middle ^ last packet + +``get_valid_starting_indices`` slides a 20-wide window over the flag array +(``np.lib.stride_tricks.sliding_window_view``), keeps the matches, and then +additionally requires the ``src_seq_ctr`` values within each match to be +**sequential modulo 16384** (``is_sequential``). + +For each valid start index: + +* Packets 0-5 are concatenated into ``count_rates_raw`` (1572 bytes). +* Packets 6-19 are concatenated into ``pha_raw``. +* ``epoch`` for the frame is the **mean of the first and last packet epoch** + (``calculate_epoch_mean``) - the centre of the collection minute. + +Before any of this, ``update_ccsds_header_dims`` swaps the dataset's dimension +from ``epoch`` to ``sc_tick``, because at this point ``epoch`` is per-packet and +is about to become per-frame. + +.. warning:: + + **[CODE]** **Incomplete frames are silently dropped.** There is no fill-value + path and no quality flag. The code ``print``\ s (does not log) a note when + packets at the start or end of the file belong to a neighbouring day's frame, + and there is a ``TODO`` about handling those when processing multiple files. + ``get_valid_starting_indices`` returning an empty array will also raise + ``IndexError`` on ``starting_indices[0]`` rather than a useful message. + See :ref:`hit-gap-incomplete-frames`. + +Rate compression and decompression +---------------------------------- + +**[DOC]** Section 5.1. Counters are held in 24-bit (some 32-bit) onboard +registers and squeezed into 16 bits by a **modified biased-exponent, +hidden-one** scheme originally suggested by Don Reames for STEREO/LET: +**12 mantissa bits, 4 exponent bits**. Values up to :math:`2^{12}` are stored +exactly; values up to :math:`2^{13}` decompress with no error. + +The document reproduces the heritage C verbatim (``pack_rate``, ``long_rate``, +``dbl_rate``). Only unpacking is needed on the ground. + +**[CODE]** ``decom_hit.decompress_rates_16_to_32``, with ``MANTISSA_BITS = 12`` +and ``EXPONENT_BITS = 4``: + +.. code-block:: python + + power = packed >> 12 # top 4 bits = exponent + if power > 1: + mantissa = packed & 0x0FFF # bottom 12 bits + value = (mantissa | 0x1000) << (power - 1) # restore the hidden one + else: + value = packed # stored exactly + +This matches the heritage ``long_rate()`` exactly, including the ``power > 1`` +(not ``>= 1``) boundary. + +.. important:: + + **The one exception the document calls out is not implemented.** Section + 4.2.4 states that the **Front End Electronics livetime counter** + (frame bytes 6-7) is *"scaled from 24 bits to 16 bits"* rather than + compressed with this algorithm. The code runs every non-header field + through ``decompress_rates_16_to_32``, including ``livetime_counter``. + + In practice this appears to be intentional and correct: the L1B livetime + conversion (section 6.2, and :ref:`hit-l1b-livetime`) is a piecewise linear + fit **defined on the decompressed value**, with explicit breakpoints at + 4101 and 16000 and an explicit warning about rollover. Do not "fix" the + decompression without checking the L1B fit with the HIT team. + +Fields that skip decompression are those whose name contains ``hdr``, +``spare`` or ``pha``. + +The frame byte map +------------------ + +**[CODE]** ``COUNTS_DATA_STRUCTURE`` in ``hit/l0/constants.py`` is an ordered +dict of ``HITPacking(bit_length, section_length, shape)``. +``parse_count_rates`` walks it in order, slicing the binary string, so **the +order of that dict is the wire format**. Do not reorder it. + +It has been checked byte-for-byte against algorithm document Tables 11-26 and +agrees exactly: + +.. list-table:: + :header-rows: 1 + :widths: 22 14 14 22 28 + + * - Field(s) + - Bytes + - Shape + - Doc table + - Contents + * - ``hdr_*`` + - 0-2 + - scalar + - 11 (MISCBITS) + - Unit/version, code-ok, heater duty cycle, leak conv, **dynamic + threshold state**, minute counter. + * - ``spare`` + - 3-5 + - -- + - 11 + - Dropped by ``decom_hit``. + * - ``livetime_counter`` .. ``num_haz_acc_no_pha`` + - 6-23 + - scalar x9 + - 12 (ERATES) + - FEE livetime, then NUMTRIG / NUMREJECT / NUMACCPHA / NUMACCNPHA and + their ``.HAZ`` counterparts. + * - ``sngrates`` + - 24-255 + - (2, 58) + - 13 (SNGRATES) + - 58 ADCs x high/low gain. **Interleaved on the wire** as + (ADC0 high, ADC0 low, ADC1 high, ...); the code de-interleaves with + ``data[::2]`` / ``data[1::2]`` into a ``gain`` dimension. Ordered by + detector address, starting at ``L2A9``. + * - ``nread`` .. ``nbadtags`` + - 256-289 + - scalar x17 + - 14 (EVPRATES) + - Event-processing counters. + * - ``coinrates`` + - 290-341 + - (26,) + - 15 (COINRATES) + - Coincidence rates: ``L12A``, ``L123A``, ``L12B``, ``L123B``, ``2TEL``, + ``PENA``, ``PENA?``, ``PENB``, ``PENB?``, ``ILA``, ``IHA``, ``ILB``, + ``IHB``, ``L142A``, ``L1423A``, ``L142B``, ``L1423B``, ``PEN4A``, + ``PEN4B``, ``ERROR``, then the six singles-by-layer counters ``L1A``, + ``L2A``, ``L3A``, ``L1B``, ``L2B``, ``L3B``. + * - ``pbufrates`` + - 342-405 + - (32,) + - 16 (PBUFRATES) + - The 32 priority buffers. #29/#30 are the "clean"/"poor" livetime STIM + buffers; #31 is the onboard processing error counter. + * - ``l2fgrates`` + - 406-669 + - (132,) + - 17 + - **Range 2 foreground.** Index = Particle ID. + * - ``l2bgrates`` + - 670-693 + - (12,) + - 18 + - Range 2 background. All Particle ID 255. + * - ``l3fgrates`` + - 694-1027 + - (167,) + - 19 + - **Range 3 foreground.** Index = Particle ID. + * - ``l3bgrates`` + - 1028-1051 + - (12,) + - 20 + - Range 3 background. + * - ``penfgrates`` + - 1052-1117 + - (33,) + - 21 + - **Range 4 foreground.** Index = Particle ID. + * - ``penbgrates`` + - 1118-1147 + - (15,) + - 22 + - Range 4 background. + * - ``ialirtrates`` + - 1148-1187 + - (20,) + - 23 + - The 20 I-ALiRT rates, also present in the 1 Hz I-ALiRT packet: 6 + ``L4Ai dE`` bins, 4 ``L4Ai vs L3A`` bins, 6 ``L4Bi dE`` bins, 4 + ``L4Bi vs L3B`` bins. + * - ``sectorates`` + - 1188-1427 + - (8, 15) + - 24 (SECTORRATES) + - 120 look directions for **one** species/energy combination. + * - ``l4fgrates`` + - 1428-1523 + - (48,) + - 25 + - I-ALiRT-aperture ion foreground rates (ranges L1L4L2, L1L4L2L3, + L1L4L2L3L3). + * - ``l4bgrates`` + - 1524-1571 + - (24,) + - 26 + - I-ALiRT-aperture ion background rates. + * - (Event Buffer) + - 1572-5239 + - -- + - 27 + - 2-byte event count header, then variable-length event records. Handled + as ``pha_raw``. + +Total fixed-format section: **12576 bits = 1572 bytes = exactly 6 packets**. + +.. note:: + + Section 4.2.1's "371 Science Rates ... for a total of 742 bytes" checks out + exactly: 132 + 12 + 167 + 12 + 33 + 15 = 371 entries in the six FG/BG + arrays, at 2 bytes each. + + One place where the prose does **not** match its own table: section 4.2.3 + says "space is allocated for 20 coincidence rates", while Table 15 lists + **26**. The code follows the table. + +Sectored-rate subcommutation +---------------------------- + +**[CODE]** ``hit_l1a.subcom_sectorates``. + +Each frame's ``sectorates`` array holds 120 look directions for **one** +species/energy combination, selected by ``hdr_minute_cnt % 10`` via +``MOD_10_MAPPING`` (see :ref:`hit-overview`). The function: + +#. Builds, for each of the 10 mod-10 slots, an array of shape + ``(n_frames, 15, 8)`` filled with ``-9223372036854775808`` (the int64 fill + value). +#. Writes each frame's ``sectorates`` into the slot its minute counter + selects. Every other slot for that frame stays fill. +#. Regroups the 10 slots into 5 species (H gets 3 energy bins, 4He/CNO/NeMgSi + get 2, Fe gets 1) and transposes to ``(epoch, energy_mean, azimuth, + zenith)``. +#. Adds ``_energy_mean``, ``_energy_delta_plus``, + ``_energy_delta_minus`` via ``add_energy_variables``. + +So the resulting arrays are **9/10 fill by construction** at 1-minute +resolution. They only become dense when regrouped into 10-minute records at +L2 (``transform_to_10_minute_chunks``, see :ref:`hit-l2`). + +Selecting complete 10-minute sets +--------------------------------- + +**[CODE]** ``hit_l1a.subset_sectored_counts``, and its helpers. + +#. ``update_livetime_coord`` attaches a **shadow coordinate** + ``epoch_livetime`` to ``livetime_counter`` and swaps its dimension, so that + subsequent filtering of ``epoch`` does not also filter livetime. This is the + trick that lets the 10-minute livetime offset survive the trimming. +#. ``find_complete_mod10_sets`` slides a 10-wide window over + ``hdr_minute_cnt % 10`` and returns the indices where it equals exactly + ``[0,1,2,...,9]``. +#. Start indices **< 10 are discarded**, because the previous 10 minutes' + livetime would not be available. +#. The dataset is subset to those 10-frame runs. +#. A mean epoch per set is computed and repeated 10 times, and the set is + trimmed to the processing day on that array. +#. ``subset_livetime`` then slices ``epoch_livetime`` to the same length but + **shifted 10 indices earlier**. + +Failure modes, all ``ValueError``: + +* No complete mod-10 set found at all. +* Empty epoch values after filtering. +* ``start_idx < 10`` in ``subset_livetime`` (dataset too small to shift). + +.. warning:: + + **[CODE]** This machinery assumes **one frame per minute with no gaps**. + ``subset_livetime`` shifts by **10 array positions**, not by 10 minutes of + wall-clock time. A dropped science frame inside or just before a sectored + block will silently pair the counts with the wrong block's livetime. See + :ref:`hit-gap-livetime-shift`. + +Statistical uncertainties +------------------------- + +**[DOC]** Section 5.5. Asymmetric Poisson at 1 sigma (0.8413), Gehrels 1986 +with S = 1: + +.. math:: + + \lambda_u = n + \sqrt{n+1} + 1 \quad\Rightarrow\quad \delta_u = \sqrt{n+1} + 1 + +.. math:: + + \lambda_l = n - \sqrt{n} \quad\Rightarrow\quad \delta_l = \sqrt{n} + +``DELTA_PLUS`` and ``DELTA_MINUS`` in the CDF carry :math:`\delta_u` and +:math:`\delta_l`. **Uncertainties for fill values must themselves be fill +values.** + +**[CODE]** ``hit_l1a.calculate_uncertainties``. It applies the formulas to +every data variable *except* an explicit ignore list (CCSDS header fields, +``hdr_*``, ``livetime_counter``, and the ``*_energy_delta_*`` variables), and +uses ``np.where(mask, ..., dataset[var])`` so that fill values pass through +unchanged. ``np.maximum(..., 0)`` guards the square root. + +.. note:: + + ``livetime_counter`` is deliberately excluded - it is a clock-cycle count, + not a particle count, so Poisson statistics do not apply to it. + +Housekeeping at L1A +------------------- + +**[CODE]** ``hit_utils.process_housekeeping_data``, shared with L1B. It: + +* Drops the CCSDS header fields and the five ``hskp_spare*`` fields. +* Collapses ``leak_i_00`` .. ``leak_i_63`` into a single 2-D ``leak_i`` on a + new ``adc_channels`` (0-63) coordinate - ``concatenate_leak_variables``. +* Applies the CDF attributes, and manually sets ``DEPEND_0 = epoch`` on + ``sc_tick`` (which is a coordinate in the counts product but a variable + here). + +L1A reads the packet with ``use_derived_value=False``, so all values are raw +DN. L1B reads the same packet with ``True``. See :ref:`hit-l1b-hk`. diff --git a/docs/source/algorithm-code-documentation/hit/l1b.rst b/docs/source/algorithm-code-documentation/hit/l1b.rst new file mode 100644 index 0000000000..7f0861e43a --- /dev/null +++ b/docs/source/algorithm-code-documentation/hit/l1b.rst @@ -0,0 +1,356 @@ +.. _hit-l1b: + +L1A to L1B: Livetime Correction and Rates +========================================= + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**[DOC]** Section 6 of the algorithm document. + +* **Input**: L1A CDF files (science), or the L0 CCSDS file again (housekeeping) +* **Processing requirements**: instrument livetime, conversion tables for + housekeeping +* **Output**: L1B CDF files + +L1B is conceptually the simplest level: **divide counts by the fractional +livetime**. Two wrinkles make it more than that - the livetime counter needs a +nonlinear unpacking, and the sectored rates need livetime from a *different* +10 minutes. + +.. _hit-l1b-livetime: + +The livetime fraction +--------------------- + +**[DOC]** Section 6.2. The livetime counter (``ERATES[0]``, frame bytes 6-7) +counts **16 MHz FPGA clock cycles during which the front-end electronics were +waiting for a trigger**. A full 60-second frame at 100% livetime would be +**960,000,000 cycles**. + +That does not fit in the 16-bit compressed field. The document is explicit: +for nominal 60-second frames, **any livetime of 14% or more overflows and the +field rolls over**. Rather than fix the encoding, the conversion is defined as +a **three-segment piecewise linear function of the decompressed counter**, +valid from about 0.002% to 110% livetime: + +.. math:: + + \mathrm{Livetime\ Fraction} = + \begin{cases} + \mathrm{LIVE\_TIME} \times 1.04 \times 10^{-9} + & \mathrm{LIVE\_TIME} > 16000 \\ + \mathrm{LIVE\_TIME} \times 3.41 \times 10^{-5} + 0.14 + & 0 \le \mathrm{LIVE\_TIME} \le 4101 \\ + \mathrm{LIVE\_TIME} \times 6.827 \times 10^{-5} + & 4101 \le \mathrm{LIVE\_TIME} \le 16000 + \end{cases} + +Note the ``+ 0.14`` offset on the first segment - that is the rollover being +undone. Note also that the segments are written in a deliberately confusing +order in the document, and that 4101 and 16000 appear in two branches each. + +.. warning:: + + **[DOC]** The conversion is **wrong below LIVE_TIME = 16000 corresponding to + 0.001667% livetime**. The document says this situation "should be detectable + via other mnemonics, such as ``NUMTRIG`` (Table 12) or STIM event counts + (``PBUFRATES`` #29 and #30, Table 16)". **No such detection exists in the + code** - there is no quality flag for an out-of-range livetime. See + :ref:`hit-gap-livetime-range`. + +**[CODE]** ``hit_l1b.livetime_fraction_calculation`` implements exactly this, +resolving the overlapping bounds as ``<= 4101``, ``> 4101 and <= 16000``, +``> 16000``: + +.. code-block:: python + + livetime1 = livetime_counter <= 4101 + livetime2 = (livetime_counter > 4101) & (livetime_counter <= 16000) + livetime3 = livetime_counter > 16000 + + livetime_fraction[livetime1] = livetime_counter[livetime1] * 3.41e-5 + 0.14 + livetime_fraction[livetime2] = livetime_counter[livetime2] * 6.827e-5 + livetime_fraction[livetime3] = livetime_counter[livetime3] * 1.04e-9 + +.. note:: + + ``l1b/constants.py`` defines ``LIVESTIM_PULSES = 270`` with the comment + "Expected number of livestim pulses per integration time. This is used to + calculate the fractional livetime". **It is not used anywhere.** It is a + remnant of an earlier approach (the document's section 4.3.2.2 mentions + using livestim data to characterise the livetime error). Treat it as dead. + +Counts to rates +--------------- + +**[DOC]** Equation 9: + +.. math:: + + \mathrm{Count\ Rate} = \frac{\mathrm{Counts}}{\mathrm{Livetime\ Fraction}} + +Applied to **all** science rates (document Tables 13, 15-24). The units stay +"counts per integration time" - the values are simply no longer integers. The +uncertainties are divided by the same livetime. + +**[CODE]** ``hit_l1b.calculate_rates`` divides the value and both uncertainty +arrays by ``livetime`` and casts to ``float32``. + +L1B standard rates +------------------ + +**[CODE]** ``process_standard_rates_data``. It copies these twelve arrays from +L1A, along with their two uncertainty companions each, and divides all of them +by livetime: + +.. code-block:: text + + sngrates coinrates pbufrates + l2fgrates l2bgrates + l3fgrates l3bgrates + penfgrates penbgrates + ialirtrates l4fgrates l4bgrates + +Plus ``dynamic_threshold_state`` (renamed from ``hdr_dynamic_threshold_state``), +which L2 needs. + +Deliberately **not** included: ``sectorates`` (it becomes its own product) and +the event-processing counters ``nread`` .. ``nbadtags`` and the ``num_*`` +hazard/trigger counters from ERATES. + +.. note:: + + **[CODE]** ``initialize_l1b_dataset`` is given an explicit coordinate list + that omits ``l4fgrates_index`` and ``l4bgrates_index``. The dimensions get + created implicitly when the data arrays are assigned, so this works, but the + two L4 arrays end up without explicit index coordinates. Harmless today; + worth knowing if you add anything that keys off those coordinates. + +.. _hit-l1b-summed: + +L1B summed rates +---------------- + +**[DOC]** Section 6.3 and Table 28. Standard-rate energy bins are combined into +**wider bins with better counting statistics**, useful during quiet times. Two +things happen at once: + +* **Bins are merged across penetration ranges.** A single summed bin draws + from R2, R3 and R4 counters simultaneously. E.g. "H 1.8-3.6 MeV/nuc" sums + four R2 bins and two R3 bins. +* **Species are merged into groups.** ``he`` = He-3 + He-4 across all its + contributing bins. (The document also names CNO and NeMgSi as groups in this + section, but Table 28 does not actually define them - they only exist in the + sectored product.) + +Then divide by livetime. + +**[DOC]** *"The Lev1B Summed Rate uncertainties are calculated by summing the +upper and lower uncertainties from Lev1A and dividing by the livetime, just as +is done in calculating the Lev1B science variables."* + +.. important:: + + That is a **linear** sum of uncertainties, not a quadrature sum. It is what + the document specifies and what the code does. It is conservative (it + overestimates the combined uncertainty for independent bins). Do not + "correct" it to quadrature without asking the HIT team - and note that the + same linear summing is then used again at L2 for the standard intensity + product. + +**[CODE]** ``SUMMED_PARTICLE_ENERGY_RANGE_MAPPING`` in ``l1b/constants.py`` is +the machine-readable form of Table 28: for each species, a list of +``{"energy_min", "energy_max", "R2": [...], "R3": [...], "R4": [...]}`` where +the lists are **indices into ``l2fgrates`` / ``l3fgrates`` / ``penfgrates``** +(i.e. Particle IDs). + +.. list-table:: + :header-rows: 1 + :widths: 20 16 64 + + * - Species + - Bins + - Energy bins (MeV/nuc) + * - ``h`` + - 4 + - 1.8-3.6, 4-6, 6-10, 10-15 + * - ``he3`` + - 3 + - 4-6, 6-10, 10-15 + * - ``he4`` + - 4 + - 1.8-3.6, 4-6, 6-10, 10-15 + * - ``he`` + - 3 + - 4-6, 6-10, 10-15 (He-3 + He-4) + * - ``c``, ``n``, ``o``, ``ne``, ``mg`` + - 4 each + - 4-6, 6-10, 10-15, 15-27 + * - ``na`` + - 2 + - 10-15, 15-27 + * - ``al``, ``ni`` + - 3 each + - 6-10 (Al) / 10-15, 15-27, 27-40 + * - ``si``, ``s``, ``ar``, ``ca``, ``fe`` + - 5 each + - 4-6, 6-10, 10-15, 15-27, 27-40 + * - **Total** + - **67** + - Matches the 67 rows in each ``imap_hit_summed-dt-factors`` CSV. + +The mechanics live in ``hit_utils``: + +* ``initialize_particle_data_arrays`` creates zero-filled + ``(epoch, n_bins)`` arrays for the species and its two uncertainties. +* ``sum_particle_data`` does the actual + ``l2fgrates[:, R2].sum(axis=1) + l3fgrates[:, R3].sum(axis=1) + + penfgrates[:, R4].sum(axis=1)``, and the same for both uncertainty arrays. +* ``add_energy_variables`` writes ``_energy_mean`` and the two + deltas. +* ``add_summed_particle_data_to_dataset`` orchestrates the three. + +.. important:: + + **The same three helpers are reused at L2** to build the *standard* + intensity product from L1B standard rates. The only difference is which + mapping dict is passed in. If you change ``sum_particle_data`` you change + both products. + +.. _hit-l1b-energy-mean: + +Energy bin identification - a real deviation +-------------------------------------------- + +**[DOC]** Section 6.1 is unambiguous. At every level above L1A, an energy bin +is identified by its **geometric mean** and its edges: + +.. math:: + + E_{char} = \sqrt{E_{min} \cdot E_{max}} + +.. math:: + + \mathrm{delta\_plus} = E_{max} - E_{char}, \qquad + \mathrm{delta\_minus} = E_{char} - E_{min} + +The document explains why: the true mean energy of particles in a bin varies +with the spectrum (even during a single event), so the geometric mean is only +a **characteristic** label. The bin **limits** are the fundamental quantity. + +**[CODE]** ``hit_utils.add_energy_variables`` uses the **arithmetic** mean: + +.. code-block:: python + + energy_mean = np.round( + np.mean(np.array([energy_min_values, energy_max_values]), axis=0), 3 + ).astype(np.float32) + +This affects ``_energy_mean``, ``_energy_delta_plus`` and +``_energy_delta_minus`` in **every** L1A sectored, L1B summed, L1B sectored, +L2 standard, L2 summed and L2 macropixel product. It is the most widespread +doc/code deviation in HIT. See :ref:`hit-gap-geometric-mean`. + +.. _hit-l1b-sectored: + +L1B sectored rates +------------------ + +**[DOC]** Sections 4.2.2 and 6.2. Three things make this different from every +other rate: + +#. **The integration time is 10 minutes, not 1.** A complete sectored set is 10 + consecutive frames. +#. **Counts are transmitted 10 minutes after they are collected.** Block *n*'s + counts must be divided by block *n-1*'s livetime. The document's Figure 14 + is a timeline of exactly this. +#. **A factor of 15.** Each look direction only sees its inclination bin for + 1/15 of a rotation, so the raw counts are divided by 15 as well. + +.. math:: + + \mathrm{Livetime}_{sector} = \sum_{i=0}^{9} \mathrm{Livetime\ Fraction}_i + +.. math:: + + \mathrm{Count\ Rate}_{sector} = + \frac{\mathrm{Raw\ Counts}}{15 \times \mathrm{Livetime}_{sector}} + +**[CODE]** The 10-minute *shift* is done at L1A (``subset_livetime``, see +:ref:`hit-l1a`); L1B only has to do the *sum* and the division. + +* ``sum_livetime_10min`` sums the livetime fraction in non-overlapping + 10-element windows and ``np.repeat``\ s each sum 10 times, so the result has + the same shape as the input: + ``[5,5,5,5,5,5,5,5,5,5, 6,6,6,6,6,6,6,6,6,6, ...]``. +* ``process_sectored_rates_data`` then computes, for each + ``_sectored_counts`` array: + + .. code-block:: python + + rates = np.where( + counts != FILLVAL_INT64, + (counts / (SECTORS * livetime_10min_reshaped)).astype(np.float32), + FILLVAL_FLOAT32, + ) + + with ``SECTORS = 15`` and livetime reshaped to ``[:, None, None, None]`` to + broadcast over ``(energy, azimuth, zenith)``. +* The variables are renamed from ``_sectored_counts`` to plain + ````. + +Two notes on the implementation: + +* It drops out of xarray into NumPy deliberately - the comment explains that + the counts and the livetime live on **different epoch coordinates** + (``epoch`` vs ``epoch_livetime``), so xarray's automatic alignment would + otherwise produce an empty result. +* The fill value **changes type** here: int64 fill in, float32 fill + (``-1.00e31``) out. Both are defined in ``l1b/constants.py``. + +.. _hit-l1b-hk: + +L1B housekeeping +---------------- + +**[DOC]** Section 6.5, Tables 29 and 30. Raw DN are converted to volts and +temperatures: + +* **Preamp voltages** and the EBOX supply rails: simple linear + ``V = a * DN`` or ``V = a * DN + b``. E.g. + ``+5.7VA Ebox: V = 0.001835 * DN``, + ``-12VA Ebox: V = 0.004680 * DN - 13.95``, + ``L3/4A Bias: V = 0.06712 * DN - 2.77``. +* **Temperatures** (``TEMP0``-``TEMP3``, ``ANALOG_TEMP``, ``HVPS_TEMP``, + ``IDPU_TEMP``, ``LVPS_TEMP``): thermistor lookups from + HIT-ELEC-HDBK-0008, reproduced as Table 30: 191 rows covering + **-40 to +150 degrees C in 1-degree steps**, each giving the thermistor + resistance and the expected voltage and DN for both the Analog board and + the FEE board. +* Everything else is passed through unchanged. + +.. note:: + + Table 29 has a copy-paste error: the ``Preamp L1A``, ``Preamp L1B`` and + ``Preamp L234B`` rows all give the equation as + ``V = 0.00121 * Digital L234A Preamp``, naming the L234A input for all + four. The intent is clearly ``V = 0.00121 * ``. + +**[CODE]** **None of this is in Python.** The conversions live in the XTCE at +``imap_processing/hit/packet_definitions/hit_packet_definitions.xml`` as +``PolynomialCalibrator`` elements (156 of them), with the eight thermistor +channels using ``ContextCalibrator`` chains (20-22 context ranges each) to +express the piecewise lookup. + +``hit_l1b`` therefore does not compute anything for housekeeping - it just +re-reads the **L0 CCSDS file** with ``derived=True``: + +.. code-block:: python + + datasets_by_apid = get_datasets_by_apid(packet_file, derived=True) + l1b_dataset = process_housekeeping_data( + datasets_by_apid[HitAPID.HIT_HSKP], attr_mgr, "imap_hit_l1b_hk" + ) + +That is why the L1B housekeeping CLI branch takes an **L0** dependency, not an +L1A one. If a conversion is wrong, fix the XTCE. diff --git a/docs/source/algorithm-code-documentation/hit/l2.rst b/docs/source/algorithm-code-documentation/hit/l2.rst new file mode 100644 index 0000000000..98c13708ab --- /dev/null +++ b/docs/source/algorithm-code-documentation/hit/l2.rst @@ -0,0 +1,340 @@ +.. _hit-l2: + +L1B to L2: Conversion to Intensity +================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**[DOC]** Section 7 of the algorithm document. + +* **Input**: L1B CDF files +* **Processing requirements**: geometry factors, energy bin widths, + efficiencies, dynamic threshold state, instrument status table, instrument + status summary, spin pulse +* **Output**: L2 CDF files + +L2 turns livetime-corrected counts into **physical intensities** in +:math:`\mathrm{cm^{-2}\,s^{-1}\,sr^{-1}\,(MeV/nuc)^{-1}}`, independent of the +instrument. + +.. note:: + + The document's input list above includes "instrument status table, + instrument status summary, spin pulse". **None of those are used by the + code.** The only ancillary inputs are the four factor CSVs, and the only + state read from the data is ``dynamic_threshold_state``. + +The intensity equation +---------------------- + +**[DOC]** Equation 12, for standard rates: + +.. math:: + + j_i(N, X) = \frac{1}{60\,\Delta t\,\Delta E_i} + \frac{\sum_{j=2}^{4} q_{ij}(N,X) - b_{ij}(N,X)} + {\sum_{j=2}^{4} \sigma_{ij}(N,X)\,\lambda_{ij}(N,X)} + +with: + +.. list-table:: + :header-rows: 1 + :widths: 14 86 + + * - Symbol + - Meaning + * - :math:`N` + - ion species + * - :math:`i` + - energy bin + * - :math:`X` + - look direction + * - :math:`j` + - penetration range, 2 to 4 + * - :math:`q` + - particle counts (L1B livetime-corrected) + * - :math:`\Delta t` + - livetime, seconds + * - :math:`\Delta E` + - energy bin width, MeV + * - :math:`\sigma` + - geometry factor, cm\ :sup:`2` sr + * - :math:`\lambda` + - detection efficiency (**assumed 1 until determined otherwise**) + * - :math:`b` + - background counts (primarily GCR and ACR; **0 until measured in + flight**) + +The factor of **60** converts the L1B rate from "counts per integration time" +to counts per second. The sectored version (equation 14) uses **600** instead, +because its integration time is 10 minutes. The summed version (equation 16) +is the same as equation 12 with an extra sum over the bins that merge into +each summed bin. + +Uncertainties run through the same equation with :math:`q` replaced by the +L1B :math:`\delta_{u}` / :math:`\delta_{l}`. Systematic uncertainties are +**zero at launch**, and the total is + +.. math:: + + \delta_{full} = \sqrt{\delta_{stat}^2 + \delta_{sys}^2} + +**[CODE]** ``hit_l2.calculate_intensities``: + +.. code-block:: python + + intensity = ( + rates / (factors.delta_time * factors.delta_e + * factors.geometry_factor * factors.efficiency) + ) - factors.b + + intensity = xr.where(rates == FILLVAL_FLOAT32, FILLVAL_FLOAT32, intensity) + +where ``delta_time`` is ``SECONDS_PER_MIN = 60`` for standard and summed +products and ``SECONDS_PER_10_MIN = 600`` for the macropixel product. + +Two differences from the document worth understanding before you touch this: + +.. important:: + + **1. The document's** :math:`\Delta t` **is dropped.** Equation 12 contains + both a literal 60 *and* :math:`\Delta t` (defined as "livetime from + telemetry, units of s"). Applying both would double-count livetime, which + L1B has already divided out. The code treats the ``60`` *as* :math:`\Delta t` + - the integration time in seconds - which is the only reading that is + dimensionally consistent. The code's docstrings state this explicitly. + + **2.** :math:`b` **is subtracted after the division, not before.** The + document subtracts background *counts* from the numerator; the code + subtracts ``b`` from the finished *intensity*. These are only equivalent if + ``b`` is expressed in intensity units. **Today every ``b`` in the ancillary + CSVs is 0, so nothing differs numerically.** If the HIT team ever delivers a + non-zero background, establish which units it is in before trusting either + form. See :ref:`hit-gap-background`. + +Selecting factors by dynamic threshold state +-------------------------------------------- + +**[DOC]** *"To process these data ... the dynamic threshold state for the time +period in question must be determined using the dynamic threshold bit. The base +state is dynamic threshold level 0. The geometry factors and efficiencies can be +different for the different dynamic threshold states."* + +Algorithm document Tables 32-35 give the standard-rate factors for DT0, DT1, +DT2 and DT3 respectively; Table 36 gives the sectored factors and Table 37 the +summed factors. + +**[CODE]** The state is a **per-frame** value, so the factor arrays are built +per-frame too: + +#. ``load_ancillary_data`` reads one CSV per state **actually present** in the + dataset, selected by the filename substring ``dt-factors``, and + lowercases the column names and species values. +#. ``get_species_ancillary_data`` filters to one species and groups by + ``lower energy (mev)``, returning ``delta_e``, ``geometry_factor``, + ``efficiency`` and ``b`` as arrays. +#. ``calculate_intensities_for_a_species`` stacks those per-frame according to + ``dynamic_threshold_state``, giving a 3-D ``(epoch, energy, 1)`` array for + standard/summed products or ``(epoch, energy, zenith)`` for sectored. +#. For sectored products ``reshape_for_sectored`` repeats along the azimuth + axis to 4-D ``(epoch, energy, azimuth, zenith)`` - **the factors depend on + declination (zenith) but not on inclination (azimuth)**. For the others, + ``np.squeeze(..., axis=-1)`` drops the trailing size-1 axis. +#. ``build_ancillary_dataset`` wraps them in an ``xr.Dataset`` sharing the + species array's coordinates, so the division aligns on epoch. + +.. warning:: + + **[CODE]** Nothing validates that all four states are present, or that the + CSV family matches the product. ``load_ancillary_data`` uses + ``next(path for path in ancillary_files if f"dt{state}-factors" in str(path))`` + - a missing file raises a bare ``StopIteration``, and passing (say) the + ``summed`` CSVs to the macropixel path fails later with a shape mismatch. + +.. _hit-l2-where-summing-happens: + +Where the cross-range summing happens +------------------------------------- + +This asymmetry causes more confusion than anything else in HIT, so it is worth +stating plainly: + +.. list-table:: + :header-rows: 1 + :widths: 26 26 48 + + * - Product + - Summed at + - Mapping used + * - Summed intensity + - **L1B** + - ``SUMMED_PARTICLE_ENERGY_RANGE_MAPPING`` (``hit/l1b/constants.py``), + 67 bins + * - Standard intensity + - **L2** + - ``STANDARD_PARTICLE_ENERGY_RANGE_MAPPING`` (``hit/l2/constants.py``), + 204 bins + * - Macropixel intensity + - not summed + - the sectored counts are already per-species + +So ``imap_hit_l1b_summed-rates`` already has ``h``, ``he3``, ... variables +ready to be divided by factors, while ``imap_hit_l1b_standard-rates`` still has +raw ``l2fgrates``/``l3fgrates``/``penfgrates`` arrays that L2 must combine +first - using the **same** ``hit_utils.add_summed_particle_data_to_dataset`` +helper that L1B used for the summed product. + +The three L2 products +--------------------- + +Standard intensity +^^^^^^^^^^^^^^^^^^ + +**[CODE]** ``process_standard_intensity``. + +#. Start with an empty dataset carrying only ``epoch`` and + ``dynamic_threshold_state``. +#. For each of the 17 species in + ``STANDARD_PARTICLE_ENERGY_RANGE_MAPPING``, sum the L1B rates across R2/R3/R4 + into that species' energy bins (and sum the uncertainties linearly - see + :ref:`hit-l1b-summed`). +#. Convert everything to intensity. +#. Add zero systematic uncertainties and the quadrature total. +#. Rename ```` to ``_standard_intensity``. + +**204 energy bins total.** This exactly matches the 204 "Intensity" entries in +algorithm document Table 31 and the 204 data rows in each +``imap_hit_standard-dt-factors`` CSV. + +.. list-table:: + :header-rows: 1 + :widths: 20 12 68 + + * - Species + - Bins + - Notes + * - ``h`` + - 12 + - + * - ``he3`` + - 11 + - + * - ``he4`` + - 12 + - + * - ``he`` + - 11 + - He-3 + He-4 combined + * - ``c``, ``n``, ``o`` + - 12 each + - + * - ``ne`` + - 13 + - + * - ``na`` + - 8 + - R3 only - Na is not classified by the R2 or R4 matrices + * - ``mg``, ``si`` + - 14 each + - + * - ``al`` + - 9 + - + * - ``s``, ``ar``, ``ca`` + - 13 each + - + * - ``fe`` + - 16 + - the widest coverage, up to 52-70 MeV/nuc + * - ``ni`` + - 9 + - R3 only + * - **Total** + - **204** + - + +Summed intensity +^^^^^^^^^^^^^^^^ + +**[CODE]** ``process_summed_intensity``. Simplest of the three: deep-copy the +L1B summed rates, convert each of the 17 species to intensity, add +systematic/total uncertainties, rename to ``_summed_intensity``. +67 bins. + +Macropixel intensity +^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** ``process_macropixel_intensity``, then +``transform_to_10_minute_chunks``. + +The intensity calculation itself is the same, with ``delta_time = 600`` and 4-D +factor arrays. The interesting part is the reshaping afterwards. + +Recall from :ref:`hit-l1a` that the L1A/L1B sectored arrays are **9/10 fill**: +each 1-minute record holds one species/energy combination and fill everywhere +else. ``transform_to_10_minute_chunks`` collapses each run of 10 records into a +single dense record: + +#. Take every 10th record as the output template + (``isel(epoch=slice(None, None, 10))``). +#. Walk the species/energy order + ``[("h", 3), ("he4", 2), ("cno", 2), ("nemgsi", 2), ("fe", 1)]``, which is + exactly ``MOD_10_MAPPING``'s order, keeping a running offset ``species_i`` + from 0 to 9. +#. For each species/energy, pull the slice ``[species_i::10, energy_i]`` - the + records where *that* combination was actually transmitted - and write it + into the output at energy plane ``energy_i``. +#. Recompute the epoch: + + .. code-block:: python + + new_epochs = start + (end - start) // 2 - nanoseconds_per_10_min + + i.e. the centre of the 10-record transmission window, **shifted back 10 + minutes to the collection window**. This is the counterpart of the livetime + shift at L1A: the timestamp now refers to when the particles arrived, not + when the bytes came down. +#. Set ``epoch_delta`` to 5 minutes, so the record spans its full 10-minute + collection window. + +.. warning:: + + **[CODE]** This function assumes the record count is an exact multiple of + 10 (``reshape(-1, 10)`` will raise otherwise) and that the species order + within each group is exactly the ``MOD_10_MAPPING`` order. L1A's + ``find_complete_mod10_sets`` does guarantee both for data that passes + through the normal chain, but the function itself has no guard. + +Uncertainties at L2 +------------------- + +**[CODE]** Three layers, all present: + +* **Statistical** - ``calculate_intensities_for_all_species`` runs the + ``_stat_uncert_minus`` and ``_stat_uncert_plus`` arrays through the same + intensity equation as the data. +* **Systematic** - ``add_systematic_uncertainties`` writes ``_sys_err_minus`` + and ``_sys_err_plus`` as **zeros**, matching the document's "at launch, the + values for all systematic uncertainties will be 0". +* **Total** - ``add_total_uncertainties`` computes + :math:`\sqrt{\delta_{stat}^2 + \delta_{sys}^2}` into + ``_total_uncert_minus`` / ``_plus``. + +.. note:: + + The ``b`` subtraction is applied to the uncertainty arrays too, since they + go through the identical ``calculate_intensities`` call. With ``b = 0`` this + is a no-op. If ``b`` ever becomes non-zero, subtracting a background from an + *uncertainty* is wrong and will need a separate code path. + +CDF attribute handling +---------------------- + +**[CODE]** ``add_cdf_attributes`` has one wrinkle worth knowing: macropixel +uncertainty and systematic-error variables are 4-D while the standard and +summed ones are 2-D, so they need different ``DEPEND_*`` attributes. The YAML +holds both, with the macropixel versions carrying a ``_macropixel`` suffix +(``h_total_uncert_minus_macropixel`` vs ``h_total_uncert_minus``), and the +function picks based on whether ``"macropixel"`` is in the logical source. +The macropixel product also uses an ``epoch_macropixel`` attribute set, because +its cadence and ``epoch_delta`` differ. diff --git a/docs/source/algorithm-code-documentation/hit/l3-scope.rst b/docs/source/algorithm-code-documentation/hit/l3-scope.rst new file mode 100644 index 0000000000..9056da28d8 --- /dev/null +++ b/docs/source/algorithm-code-documentation/hit/l3-scope.rst @@ -0,0 +1,175 @@ +.. _hit-l3-scope: + +L3 Scope: What Is Not in This Repository +======================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**[DOC]** Section 9 of the algorithm document, PDF pages 149-169. + +.. warning:: + + **HIT L3 is produced by a separate repository, run closer to the science + team.** Nothing in ``imap_processing`` should compute a charge, a cosine + correction, an incident ion energy, or a pitch angle. This page exists so + that you can (a) recognise an L3 task when one is handed to you, and (b) + know what L1A and L2 are obliged to hand L3. + +There are **three** L3 products. Two of them consume L2 or I-ALiRT outputs +directly. The third, the PHA products, consumes **L1A direct events** - and +that is where this repository has an unfinished obligation. + +.. _hit-l3-pha: + +1. PHA products (section 9.1) +----------------------------- + +**[DOC]** For each individual detected particle, L3 produces: + +* time of detection +* calculated ion energy (MeV/nuc) +* calculated charge (Z) + +**Input: HIT L1A raw event data.** That is the ``imap_hit_l1a_direct-events`` +product from this repository. + +The processing chain is roughly: + +#. **Classify the event.** Table 40 is a 12-column truth table over "which + layer had the largest signal" (``L1A14``, ``L1A0``, ``L2A``, ``L3A``, + ``L3B``, ``L2B``, ``L1B0``, ``L1B14``, ``L4iA``, ``L4oA``, ``L4iB``, + ``L4oB``) giving a class (``L12A``, ``L123A``, ``PENA``, ``2TEL``, ``ILA``, + ``L142A``, ...) and a range (``2A``, ``3A``, ``4A``, ``2B``, ``3B``, ``4B``, + or ``NOCALC``). "Largest signal on a layer" means comparing all high-gain + values on that layer plus any non-zero low-gain values **multiplied by 20**. +#. **Convert ADC to MeV.** :math:`E[\mathrm{MeV}] = a \cdot \mathrm{PHA[ADC]} + b`, + with six coefficient pairs (L1/L2/L3 x low/high gain) in Table 42. The + coefficients derive from FSW ADC calibration data (J. Dumonthier, GSFC-672, + version 2024-05-23). Note the erratum: on 2026-05-14 all L1A/B14 low-gain + offsets were reduced by 8. +#. **Correct for the Kapton foils.** The apertures are shielded by dual foils + with an effective thickness of ~14.8 um silicon-equivalent. A multiplicative + correction is applied to the L1 signals from the ``WINCORR2`` (Range 2) or + ``WINCORR3`` (Range 3/4) arrays, stored as fixed-point integers scaled by + 256. +#. **Cosine-correct.** Both :math:`\Delta E` and :math:`E'` are multiplied by + :math:`K_\theta = L_\theta^{1/a}`, where :math:`L_\theta` is the ratio of + nominal to actual path length through the detector. Tables 43-45 give + :math:`K_\theta` for all 150 L1 x L2 segment combinations per range. The + power-law index :math:`a` was tuned for best He-3/He-4 discrimination: + **1.61 (R2), 1.68 (R3), 1.82 (R4)** for the science apertures and + **2.36 / 1.73 / 1.76** for the A0/B0 I-ALiRT apertures. +#. **Calculate the charge.** Each of 15 species (H, 4He, C, N, O, Ne, Na, Mg, + Al, Si, S, Ar, Ca, Fe, Ni - He-3 excluded) has a double-power-law fit to its + track: + + .. math:: + + f_t(E') = \left[\left(a_1 E'^{b_1}\right)^{\gamma} + + \left(a_2 E'^{b_2}\right)^{\gamma}\right]^{1/\gamma} + + Evaluate all 15 at the measured :math:`E'`, then interpolate the + (:math:`Z_t`, :math:`f_t`) graph at the measured :math:`\Delta E` under a + power-law assumption to get a **non-integer** Z: + + .. math:: + + Z(\Delta E) = A\,(\Delta E)^B, \quad + B = \frac{\log(Z_{t,2}/Z_{t,1})}{\log(f_t(E'_2)/f_t(E'_1))}, \quad + A = \frac{Z_{t,1}}{[f_t(E'_1)]^B} + + Above the Ni track, extrapolate with the Fe/Ni pair; below the H track, with + the H/4He pair. He-3 shows up as Z ~ 1.9 between the H and 4He tracks - that + is the point of using a decimal charge. +#. **Compute the incident energy** as the sum of the energy deposits in all + detectors the particle interacted with. Per-detector energies are also + reported. Valid :math:`\Delta E` and :math:`E'` bounds per range are in + Table 46. + +**Ancillary files L3 needs:** four ADC-to-MeV conversion sets (one per detector +type) and three Z lookup tables (one per range), all supplied by the HIT team. + +.. _hit-l3-sectored: + +2. Sectored / pitch angle products (section 9.2) +------------------------------------------------ + +**[DOC]** Combines: + +* ``imap_hit_l2_macropixel-intensity`` - the 120 look directions on a 10-minute + cadence +* **MAG L1D** magnetic field vectors in the despun frame, **averaged over the + same 10 minutes** + +For each of the 120 (declination, inclination) look directions, compute the +angle between the particle acceptance direction and **B**. Output: + +* an array of pitch angles matching the HIT sectored bins +* a second product carrying both pitch angle **and gyrophase** +* a 2-D "skymap" rebinned to **22.5 degrees in pitch angle x 24 degrees in + gyrophase** + +**No ancillary files are required for this product.** + +.. _hit-l3-electrons: + +3. Electron science products (section 9.3) +------------------------------------------ + +**[DOC]** Turns the 6 I-ALiRT electron rates (3 per side, see +:ref:`hit-ialirt`) into science-quality electron **intensities**. + +* Low and medium energy products are single-parameter - inner-L4 energy deposit + only. +* High energy products use L4 **and** L3. +* Requires **modelled response functions** to map measured deposit to incident + energy. +* Requires **ion contamination subtraction** using high-energy proton + measurements (range/bin TBD). Early in the mission the raw PHA data will also + be used to characterise the contamination. + +**Ancillary files L3 needs:** electron response matrices, one for the L4-only +products and one for the L3-vs-L4 products. + +What this repository owes L3 +---------------------------- + +.. list-table:: + :header-rows: 1 + :widths: 30 18 52 + + * - L3 needs + - From + - Status here + * - Decoded PHA events (per event: detector IDs, gain flags, ADC values, + Particle ID, priority buffer, STIM/HAZ flags, DEINDEX/EPINDEX) + - L1A + - **Missing.** ``imap_hit_l1a_direct-events`` contains only the raw + concatenated binary. See :ref:`hit-gap-events`. + * - ``imap_hit_l2_macropixel-intensity`` with correct look directions + - L2 + - **Produced**, but the spin-rate correction to the 15th inclination bin + is missing. See :ref:`hit-gap-spinrate`. + * - The 6 I-ALiRT electron rates + - I-ALiRT + - **Produced.** + * - High-energy proton rates for contamination subtraction + - I-ALiRT / L2 + - **Produced** (``hit_h_a_side_high_en``, ``hit_h_b_side_high_en``, and + the L2 standard H bins). + * - Correct energy bin identification (geometric mean + edges) + - L1B / L2 + - **Deviates** - the code uses the arithmetic mean. See + :ref:`hit-gap-geometric-mean`. + +.. important:: + + The **event record format is fully specified** in sections 4.2.5 - 4.2.8 of + the algorithm document (PDF pages 25-33): the 32-bit Event Record Header + (Table 4), the 20-bit ADC field layout (Table 5 and Figure 9), the byte + padding rules (Table 6), the detector group flags (Table 7), the + Extended Header Block, the STIM Information Block, and the detector address + table (Table 8). **Decoding that is a Level 1A job and belongs in this + repository** - it is the raw-to-reformatted step, not derived science. Only + the *physics* applied to the decoded events (energy, charge, cosine + correction) belongs to L3. diff --git a/docs/source/algorithm-code-documentation/hit/overview.rst b/docs/source/algorithm-code-documentation/hit/overview.rst new file mode 100644 index 0000000000..7538433bb4 --- /dev/null +++ b/docs/source/algorithm-code-documentation/hit/overview.rst @@ -0,0 +1,542 @@ +.. _hit-overview: + +Instrument and Measurement Concepts +=================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +Everything on this page is background. Very little code depends on it +directly, but almost every design decision in :ref:`hit-l1a`, +:ref:`hit-l1b` and :ref:`hit-l2` only makes sense once you have it. + +What HIT measures +----------------- + +**[DOC]** HIT measures **~2-40 MeV/nucleon ion (H to Ni) composition, energy +spectra, angular distributions and temporal variations**. It is one of five +in-situ instruments on IMAP; together with SWAPI and CoDICE it gives +continuous ion coverage from 0.1 keV to 40 MeV/nucleon. + +The design draws heavily on the **Low Energy Telescope (LET) on STEREO** and +on **EPI-Hi on Parker Solar Probe**. Most of the rate definitions, the onboard +matrices, the priority buffers and the rate compression scheme in this +document are literally the STEREO/LET ones. + +The primary science products are **energetic ion intensities** in +:math:`\mathrm{cm^{-2}\,s^{-1}\,sr^{-1}\,(MeV/nuc)^{-1}}` for 16 species in +roughly 12 energy bins each. + +The measurement technique +------------------------- + +**[DOC]** HIT uses the standard **"dE/dx vs residual E"** technique: + +* A particle passes *through* one detector, depositing :math:`\Delta E`. +* It then *stops* in a following detector, depositing :math:`E'`. +* The pair :math:`(\Delta E, E')` identifies the incident **species** and + **kinetic energy**. Different elements trace out separate hyperbola-like + tracks in that plane. +* Which detector **segments** fired identifies the **arrival direction**. + +This technique has extensive flight heritage (ACE/SIS, Voyager/CRS, +Voyager/LECP, STEREO/LET, STEREO/HET, PSP/EPI-Hi). + +.. important:: + + **The species identification happens on board, not on the ground.** The + flight software sorts each event through a lookup "matrix" and increments a + counter. What reaches the SDC is already binned by species and energy. + Ground processing at L1A/L1B/L2 is therefore bookkeeping, livetime + correction and unit conversion - **not** particle identification. The only + place ground code redoes the physics is L3, from the raw PHA events, and + that is a different repository (see :ref:`hit-l3-scope`). + +The sensor head +--------------- + +**[DOC]** 14 solid-state detectors (SSDs), each subdivided into segments, read +out by four custom **PHASIC** chips (16 channels each, dual gain). + +.. list-table:: + :header-rows: 1 + :widths: 12 16 16 56 + + * - Layer + - Thickness + - Active area + - Notes + * - **L1** + - 24 um + - 2 cm\ :sup:`2`, 3 segments each + - Sits in the outer region of all 10 entrance apertures. Segments are + named ``a``, ``b``, ``c`` (e.g. ``L1A2b``). + * - **L2** + - 50 um + - 10.2 cm\ :sup:`2`, 10 segments + - Two of them (``L2A0``-``L2A9``, ``L2B0``-``L2B9``), in the centre of + the sensor head. + * - **L3** + - 1000 um + - 15.6 cm\ :sup:`2`, 3 segments + - Two of them. Segments are referred to as inner/outer (``L3Ai``, + ``L3Ao``, ``L3Bi``, ``L3Bo``). + * - **L4** + - 1500 um + - 2 active areas (inner ``i`` / outer ``o``) + - **Only behind the 2 I-ALiRT apertures.** Optimised for energetic + electrons. + +**Apertures.** There are **10 entrance apertures** in two groups of five, +arranged along the arc of a circle: + +* **8 "science" apertures** - ``A1``-``A4`` and ``B1``-``B4``. These feed the + ion science rates and the sectored rates. +* **2 "I-ALiRT" apertures** - ``A0`` and ``B0``. These additionally have an + L4 detector behind L1, used for real-time electron measurements. + +The two-letter prefix throughout the telemetry is **side** (``A`` or ``B``) +followed by **aperture number** (0-4) and **segment** (``a``/``b``/``c``), +e.g. ``L1B3c`` = side B, layer 1, aperture 3, segment c. + +Penetration ranges +------------------ + +**[DOC]** Events are classified by how deep they got. This is the single most +important organising concept in the HIT data, because the counters are laid +out by range first. + +.. list-table:: + :header-rows: 1 + :widths: 12 22 22 44 + + * - Range + - Detectors hit + - Frame counters + - Meaning + * - **RNG2** / R2 + - L1 L2 + - ``l2fgrates`` (132), ``l2bgrates`` (12) + - Stopped in an L2 detector. Lowest energies. + * - **RNG3** / R3 + - L2 L3 + - ``l3fgrates`` (167), ``l3bgrates`` (12) + - Stopped in one L3 detector. + * - **RNG4** / PEN + - L3A L3B + - ``penfgrates`` (33), ``penbgrates`` (15) + - Went through both L3s and possibly beyond ("penetrating"). + * - **RNG2I** + - L1 L4 L2 + - ``l4fgrates`` (48), ``l4bgrates`` (24) + - New for HIT: ions through an **I-ALiRT** aperture, so they lose extra + energy in L4 and land at higher incident energy for the same range. + * - **RNG3I** + - L1 L4 L2 L3 + - (same arrays) + - As above. + * - **RNG4I** + - L1 L4 L2 L3 L3 + - (same arrays) + - As above. + +R2, R3 and R4 match STEREO/LET exactly. The three ``I`` ranges are new. + +.. note:: + + **[CODE]** ``l4fgrates`` and ``l4bgrates`` are decommutated at L1A and + carried through L1B, but **nothing at L2 uses them**. Only R2, R3 and R4 + feed the intensity products. See :ref:`hit-gap-l4rates`. + +The matrices, Particle IDs, FGRATES and BGRATES +----------------------------------------------- + +**[DOC]** Each penetration range has an onboard **matrix**: a lookup table +spanning **128 bins on the E' (x) axis and 400 bins on the dE (y) axis**, +covering Z = 1 to Z >= 40. The flight software computes the pair of indices +(**EPINDEX** 0-127 and **DEINDEX** 0-399) for each event, looks up which box +it lands in, and increments the corresponding counter. + +* **Foreground rates (FGRATES)** - counters for boxes that lie along an + element or isotope track. These are the real science: H, He-3, He-4, C, N, + O, Ne, Na, Mg, Al, Si, S, Ar, Ca, Fe, Ni, each split into energy bins. +* **Background rates (BGRATES)** - counters for broad regions that are *not* + on a track: the Li/Be/B region between He and C, the "backward moving + particle" corner, STIM regions. **All background events get Particle ID + 255.** + +**Particle ID** is simply the event's index into the FGRATES array for its +range. So Particle ID 0 in Range 2 is "H, 1.0-1.8 MeV/nuc". It is also +written into each PHA event record header, which is how you tie an event +back to a rate bin. + +Two traps: + +* **Particle IDs are not unique across ranges.** The same ID means different + things in R2, R3 and R4. The document is explicit that deduplicating them + on board was rejected as too expensive, and that the ground must handle it. +* **Some species exist in one range and not another.** Na, for example, is + classified by the Range 3 matrix but not by Range 2 or Range 4; in those it + falls into ID 255. This is why + ``STANDARD_PARTICLE_ENERGY_RANGE_MAPPING`` has empty ``"R2"``/``"R4"`` + lists for many entries - that is correct, not an oversight. + +Dynamic thresholds +------------------ + +**[DOC]** During a large SEP event the instrument would saturate: dead time +rises, chance coincidences rise, data quality falls. HIT mitigates this +automatically by **reducing its own geometry factor** - it disables high-gain +PHA channels on selected segments, which raises the effective energy +threshold. Light ions (H, He) stop being counted; heavy ions (Z >= 6), which +trigger low gain anyway, are largely unaffected. + +.. list-table:: + :header-rows: 1 + :widths: 10 90 + + * - State + - What is disabled + * - **DT0** + - Nothing. Nominal. All high gains functioning. + * - **DT1** + - High-gain PHA disabled on the **outer regions of all 16 L1 segments** + in the science apertures. Geometry factor drops to that of the inner + L1 detectors only. + * - **DT2** + - High-gain PHA disabled on **all science-aperture L1 detectors except + the two centre ones** (``L1A2``, ``L1B2``). + * - **DT3** + - High-gain PHA additionally disabled on **all L2 detectors except + ``L2A4``, ``L2A5``, ``L2B4``, ``L2B5``**, and on the **outer L3** + detectors. + +The transition up is driven by a commandable single-detector count rate +threshold; the transition back down happens at roughly half that rate +(hysteresis). **The I-ALiRT apertures are never affected** - they stay in +their nominal configuration in all four states. + +.. important:: + + **This is why L2 needs four ancillary tables per product.** The geometry + factor and efficiency depend on the dynamic threshold state, so the state + at the time of each science frame selects which table to use. The state is + telemetered in **frame byte 1, bits 0-1**, decommutated as + ``hdr_dynamic_threshold_state`` and carried into L1B and L2 as + ``dynamic_threshold_state``. + +Operating modes +--------------- + +**[DOC]** HIT's operating principle is "turn us on and leave us on" - it stays +on through spacecraft operations including thruster firings. + +.. list-table:: + :header-rows: 1 + :widths: 16 84 + + * - Mode + - What it is + * - **Science** + - SSD high voltage bias on. The normal state, and the only one producing + science frames. + * - **Safe / standby** + - SSD bias off. Entered by command or automatically on a fault such as + high SSD current. + * - **Boot** + - FSW table/image upload to SRAM, then commit to MRAM, then a + "Maintenance Status" packet, then a transition to Safe. + +.. note:: + + **[CODE]** Nothing in ``imap_processing/hit`` checks the operating mode. + There is no mode field in the science frame header; the closest proxies are + the housekeeping ``ENABLE_HVPS`` and ``FEE_RUNNING`` flags and the + ``hdr_code_ok`` bit in the science frame. + +The HIT Science Frame +--------------------- + +This is the structure that everything at L1A hangs off. + +**[DOC]** One **Science Frame is one minute of data**. It is transmitted as +**20 CCSDS packets** on **APID 1252**, 262 bytes of payload each, at 20 +packets per minute. + +.. code-block:: text + + Packets 0-4 Science Frame Header, counters and rates + Packet 5 remaining counters and rates, then the start of the Event Buffer + Packets 6-19 Event Buffer (raw pulse-height events) + +**[CODE]** ``decom_hit.assemble_science_frames`` splits at the packet +boundary: the **first 6 packets** (6 x 262 = 1572 bytes) are concatenated into +``count_rates_raw``, and the **last 14** into ``pha_raw``. That split is exact: +the fixed-format section is bytes 0-1571 and the Event Buffer header begins at +byte 1572. + +The frame header is 5 bytes: + +.. list-table:: + :header-rows: 1 + :widths: 10 20 70 + + * - Byte + - Field + - Meaning + * - 0 + - ``hdr_unit_num`` (bits 6-7), ``hdr_frame_version`` (bits 0-5) + - Unit: 00 = EM1, 01 = EM2, 10 = FM. Version is currently 11, modulo + 128 if it ever exceeds that. + * - 1 + - ``hdr_code_ok`` (bit 7), ``hdr_heater_duty_cycle`` (bits 3-6), + ``hdr_leak_conv`` (bit 2), ``hdr_dynamic_threshold_state`` (bits 0-1) + - The dynamic threshold state lives here. **L2 depends on it.** + * - 2 + - ``hdr_minute_cnt`` + - HIT internal minute counter. **Its value mod 10 selects which + species/energy the sectored rates in this frame belong to.** + * - 3-5 + - spare + - Three spare bytes, dropped by the code. + +.. note:: + + The document describes a separate 5-byte *frame* header (SCID/version, + SFLEN length, SFCHECK checksum) in section 5.3, and a *MISCBITS* block in + Table 11. **Table 11 is what the packets actually contain and what the code + parses.** Section 5.3's SFLEN/SFCHECK description also states the frame + never exceeds 4160 bytes, while Table 27 places the last event-buffer byte + at 5239. Treat the tables as authoritative; they agree with the code. + +Sectored rates: how the sky is divided +-------------------------------------- + +**[DOC]** Anisotropy comes from combining the **8 science apertures** with the +**spacecraft spin**: + +* **Declination** - 8 bins of 22.5 degrees each, covering the full 180 + degrees. Determined on board from *which* L1 segment and *which* L2 segment + fired (algorithm document Table 2). 0 degrees is the spin axis in the + sunward direction. +* **Inclination** - 15 bins of 24 degrees each, covering the 360 degrees of + spin. Determined purely **by timing**, incrementing every second, assuming + the nominal **4 rpm** spin. Zero inclination is the zero spin phase from the + spacecraft Time and Status message; a new spin-phase-zero resets the index. + +8 x 15 = **120 look directions** per species/energy combination. + +Because 120 directions would blow the telemetry budget, the frame carries +**only one species/energy combination per minute**, cycling through 10 of +them: + +.. list-table:: + :header-rows: 1 + :widths: 8 16 32 44 + + * - mod 10 + - Species + - Energy + - Notes + * - 0 + - H + - 1.8 - 3.6 MeV + - + * - 1 + - H + - 4.0 - 6.0 MeV + - + * - 2 + - H + - 6.0 - 10.0 MeV + - + * - 3 + - 4He + - 4.0 - 6.0 MeV/n + - + * - 4 + - 4He + - 6.0 - 12.0 MeV/n + - + * - 5 + - CNO + - 4.0 - 6.0 MeV/n + - Element **group**, not a single species. + * - 6 + - CNO + - 6.0 - 12.0 MeV/n + - + * - 7 + - NeMgSi + - 4.0 - 6.0 MeV/n + - Element group. + * - 8 + - NeMgSi + - 6.0 - 12.0 MeV/n + - + * - 9 + - Fe + - 4.0 - 12.0 MeV/n + - + +**[CODE]** This is ``MOD_10_MAPPING`` in ``hit/l0/constants.py``, keyed on +``hdr_minute_cnt % 10``. + +Two consequences that trip people up, both handled in the code: + +#. **A complete sectored set takes 10 minutes.** L1A only emits sectored data + for runs of 10 consecutive frames whose ``hdr_minute_cnt % 10`` is exactly + ``0,1,...,9``. +#. **Sectored counts are accumulated for 10 minutes and transmitted over the + next 10 minutes.** So block *n*'s counts must be divided by block *n-1*'s + livetime. See :ref:`hit-l1b-sectored`. + +.. warning:: + + **[DOC]** The inclination bins assume exactly 4 rpm. Real spin rates differ, + which makes the **fifteenth inclination bin** narrower or wider than 24 + degrees. The document states plainly: *"This needs to be corrected on the + ground. No onboard correction is planned."* **No such correction exists in + the code.** See :ref:`hit-gap-spinrate`. + +Naming: declination/inclination vs zenith/azimuth +------------------------------------------------- + +**[CODE]** The code does **not** use the document's names: + +.. list-table:: + :header-rows: 1 + :widths: 30 20 50 + + * - Document + - Code + - Values + * - Declination (8 bins of 22.5 deg) + - ``zenith`` + - ``ZENITH_ANGLES`` = 11.25, 33.75, 56.25, 78.75, 101.25, 123.75, + 146.25, 168.75 (bin centres) + * - Inclination (15 bins of 24 deg) + - ``azimuth`` + - ``AZIMUTH_ANGLES`` = 12, 36, 60, ..., 348 (bin centres) + +Also note the **transpose**: the frame stores ``sectorates`` as +``(8 declination, 15 inclination)``, and ``parse_count_rates`` transposes it to +``(epoch, azimuth, zenith)`` = ``(epoch, 15, 8)``. All downstream arrays are in +that order. + +Boresight and pointing +---------------------- + +**[DOC]** HIT is rotated **30 degrees counterclockwise from the spacecraft ++Y axis**. The boresight is taken along the spacecraft **+Z** (sun-spacecraft +line) between the L0B and L1B apertures. The instrument vector in the +spacecraft frame is :math:`(-0.5,\ 0.866025,\ 0)`; use the transpose of the +rotation matrix to go from spacecraft to instrument. + +**[CODE]** ``imap_processing/spice/geometry.py`` knows about HIT - +``SpiceFrame.IMAP_HIT = -43500``, a boresight lookup of ``[0, 1, 0]``, and a +spacecraft-to-instrument spin phase offset of ``119.6452/360`` (nominally +30 + 90 = 120 degrees). **No HIT processing code calls any of it.** Pointing +is needed at L3, not here. + +Uncertainties +------------- + +**[DOC]** Two families, handled at different levels: + +* **Statistical.** Asymmetric Poisson, per Gehrels 1986 at 1-sigma + (0.8413 confidence). Computed at L1A from raw counts, then carried through + by the same arithmetic that transforms the counts: + + .. math:: + + \delta_u = \sqrt{n+1} + 1, \qquad \delta_l = \sqrt{n} + + These populate ``DELTA_PLUS`` and ``DELTA_MINUS``. **Fill values must + propagate to fill values.** + +* **Systematic.** Two known sources, neither yet quantified: + + * **Chance coincidences** - two particles arriving close enough in time to + look like one multi-hit event. Segmenting the detectors mitigates it; + ground consistency checks are promised but unspecified. + * **Livetime errors** at very high rates - the livetime counter does not + account for the coincidence window opened by each trigger. The proposed + correction is + :math:`\mathrm{livetime_{corrected} = livetime} + (\Delta t \times N_{trig})`, + with :math:`\Delta t` to be determined from the onboard "livestim" pulser + data and accelerator calibration runs. **Neither the correction nor + :math:`\Delta t` is defined yet.** + + **[DOC]** At launch all systematic uncertainties are **zero**, to be updated + in flight. The total is the quadrature sum: + + .. math:: + + \delta_{full} = \sqrt{\delta_{stat}^2 + \delta_{sys}^2} + +**[CODE]** L1A computes the Gehrels values; L1B divides them by livetime; L2 +runs them through the intensity equation; ``add_systematic_uncertainties`` +writes zeros and ``add_total_uncertainties`` does the quadrature sum. The +chain is complete and honest - it just has zeros in the systematic slot, as +intended. + +Vocabulary +---------- + +.. list-table:: + :header-rows: 1 + :widths: 26 74 + + * - Term + - Meaning + * - **Science Frame** + - One minute of HIT data, 20 CCSDS packets on APID 1252. The L1A record. + * - **Aperture** + - One of 10 entrance windows. ``A1``-``A4``/``B1``-``B4`` are science; + ``A0``/``B0`` are I-ALiRT (they have an L4 detector). + * - **Range (R2/R3/R4)** + - Penetration depth class. Determines which matrix, which counter array, + and which Particle ID namespace applies. + * - **Matrix** + - The onboard 128 x 400 lookup in (E', dE) space that assigns each event + a species and energy bin. Drawn in Appendix C of the algorithm + document. + * - **FGRATES / BGRATES** + - Foreground (on an element track - real science) / background (broad + off-track regions, Particle ID 255) counter arrays. + * - **Particle ID** + - The index into a range's FGRATES array. 255 means unidentified or + background. **Not unique across ranges.** + * - **Standard rates** + - Full-instrument (no look direction), 1-minute, native energy bins. The + main product. + * - **Summed rates** + - Standard bins combined into wider bins, and species combined into + groups, for better statistics during quiet times. + * - **Sectored rates / macropixel** + - The 120-look-direction anisotropy product. 10-minute cadence, 10 + species/energy combinations, reduced energy resolution. "Macropixel" + is the name the L2 product uses. + * - **Dynamic threshold (DT0-DT3)** + - Automatic geometry-factor reduction during intense events. Selects + which ancillary factor table L2 uses. + * - **Livetime** + - Fraction of the minute the FEE spent waiting for a trigger. Telemetered + as a counter of 16 MHz clock cycles, compressed to 16 bits. + * - **STIM** + - Onboard pulser events. "Livetime STIM" pulses measure livetime + independently; "ADC STIM" events calibrate the ADCs. Tagged in the + event record header, counted in ``pbufrates`` 29-30. + * - **HAZ (hazard)** + - A trigger arriving within ~2.8 us of the previous one. Counted + separately, currently rejected from analysis, and does **not** increment + livetime when rejected. + * - **Priority buffer** + - One of 32 onboard queues that sample events for telemetry, weighted to + favour rare/interesting particle classes. ``pbufrates`` counts them. + * - **PHASIC** + - Pulse Height Analysis System Integrated Circuit. Four of them, 16 + dual-gain channels each. + * - **Event record / PHA word** + - The variable-length raw pulse-height data in the Event Buffer. **Not + decoded by this repository yet.** diff --git a/docs/source/algorithm-code-documentation/hit/reference-tables.rst b/docs/source/algorithm-code-documentation/hit/reference-tables.rst new file mode 100644 index 0000000000..6c41e23f0d --- /dev/null +++ b/docs/source/algorithm-code-documentation/hit/reference-tables.rst @@ -0,0 +1,429 @@ +.. _hit-reference-tables: + +Reference Tables - Where to Look Them Up +======================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +Rule of thumb +------------- + +Big tables are **not** reproduced in these pages. They go stale, and most of +them already exist in a machine-readable form that the code actually reads. +This page tells you which file, and which page of the algorithm document if you +have a copy. + +HIT is an extreme case: **roughly 90 of the algorithm document's 169 pages are +lookup tables.** Tables 17-27 alone (the frame byte map) run to 25 pages of +"frame byte N to frame byte N+1, H (4.5-5.0 MeV/nuc), Particle ID 7". Tables +32-37 (geometry factors) run to another 38. Loading any of that into an agent's +context is almost always waste - you need three numbers, not fifteen hundred +rows. + +Authority order, highest first: + +#. **The code and its ancillary files.** What runs. +#. **The XTCE** (``hit_packet_definitions.xml``). What is parsed, and where the + housekeeping conversions live. +#. **The algorithm document.** What the instrument team intends - and note it + is still marked **Draft**, with a revision as recent as June 2026 that + changed a packet table "to reflect what is actually in the packets". + +.. note:: + + If a table you need is genuinely required for development, add it as a new + RST file under this directory rather than inlining it into one of the + narrative pages. That keeps it out of the default context. + +Machine-readable tables in the repository +----------------------------------------- + +The frame byte map +^^^^^^^^^^^^^^^^^^ + +**This is the important one.** Algorithm document Tables 11-27 - 25 pages of +byte assignments - exist in full as +``COUNTS_DATA_STRUCTURE`` in ``imap_processing/hit/l0/constants.py``. It has +been verified byte-for-byte against the document. The per-field byte ranges are +summarised in :ref:`hit-l1a`; the ordered dict itself is the authority. + +What it does **not** contain is the *meaning* of each index within an array - +which Particle ID is which species and energy. For that you still need the +document, or the two mapping dicts below. + +Particle ID to species/energy mappings +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +.. list-table:: + :header-rows: 1 + :widths: 46 54 + + * - File + - Contains + * - ``imap_processing/hit/l2/constants.py`` + ``STANDARD_PARTICLE_ENERGY_RANGE_MAPPING`` + - 17 species, **204 energy bins**, each with the R2/R3/R4 index lists. + This is the machine-readable form of document Table 31 **and** of the + relevant subset of Tables 17, 19 and 21. + * - ``imap_processing/hit/l1b/constants.py`` + ``SUMMED_PARTICLE_ENERGY_RANGE_MAPPING`` + - 17 species, **67 energy bins**, with R2/R3/R4 index lists. The + machine-readable form of document Table 28. + * - ``imap_processing/hit/l0/constants.py`` ``MOD_10_MAPPING`` + - The 10 sectored species/energy combinations. Document Table 3. + +Ancillary CSVs +^^^^^^^^^^^^^^ + +``imap_processing/tests/hit/test_data/ancillary/`` holds the twelve +``imap_hit_{standard,summed,sectored}-dt{0,1,2,3}-factors_*.csv`` files - +the machine-readable form of document Tables 32-37. Format and quirks are in +:ref:`hit-ancillary`. + +Housekeeping conversions +^^^^^^^^^^^^^^^^^^^^^^^^ + +``imap_processing/hit/packet_definitions/hit_packet_definitions.xml``. Document +Table 29 (voltage conversions) is 156 ``PolynomialCalibrator`` elements; +document Table 30 (191 thermistor rows, -40 to +150 degrees C) is eight +``ContextCalibrator`` chains of 20-22 ranges each. **Edit the XTCE, not +Python.** + +Packet definitions +^^^^^^^^^^^^^^^^^^ + +.. list-table:: + :header-rows: 1 + :widths: 46 54 + + * - File + - Contains + * - ``imap_processing/hit/packet_definitions/hit_packet_definitions.xml`` + - ``HIT_HSKP`` (APID 1251) and ``HIT_SCIENCE`` (APID 1252). 133 + parameters. Note that the science packet's payload is a single opaque + ``science_data`` field - the internal structure is in + ``COUNTS_DATA_STRUCTURE``, not the XTCE. + * - ``imap_processing/ialirt/packet_definitions/ialirt_hit.xml`` + - The HIT I-ALiRT fields: ``HIT_MET``, ``HIT_SC_TICK``, ``HIT_STATUS``, + ``HIT_SUBCOM``, ``HIT_FAST_RATE_1``, ``HIT_FAST_RATE_2``, + ``HIT_SLOW_RATE``, ``HIT_EVENT_DATA_00`` .. ``_10``, ``HIT_SPARE``. + +APIDs +----- + +**[DOC]** Table 9. Only two of the six are parsed by the XTCE in this +repository. + +.. list-table:: + :header-rows: 1 + :widths: 10 10 26 14 14 26 + + * - Dec + - Hex + - Description + - Bytes + - Cadence + - Status here + * - 1250 + - 0x4e2 + - Autonomy / Aliveness + - 2 + - 1/sec + - Not parsed. + * - 1251 + - 0x4e3 + - Housekeeping + - 140 + - 1/min + - ``HitAPID.HIT_HSKP``. Feeds ``imap_hit_l1a_hk`` and + ``imap_hit_l1b_hk``. + * - 1252 + - 0x4e4 + - Science + - 262 + - 20/min + - ``HitAPID.HIT_SCIENCE``. Feeds everything else. + * - 1253 + - 0x4e5 + - I-ALiRT + - 54 + - 1/sec + - ``HitAPID.HIT_IALRT``. Defined in the enum but handled entirely by the + I-ALiRT machinery, not by ``hit_utils``. + * - 1254 + - 0x4e6 + - Message Log + - 0-360 + - as needed + - Not parsed. + * - 1255 + - 0x4e7 + - Memory Dump + - 360 + - on command + - Not parsed. + +**[DOC]** The science APID also carries a **packet number 0-19 in bits 0-4 of +the CCSDS subseconds field** (Table 10), overwriting those bits. The code does +not use it - it identifies frames from the grouping flags and sequence counters +instead. + +Index to the algorithm document +------------------------------- + +Page numbers are **PDF page numbers** in +``HIT_Algorithm_Document_v1p11p00_06_02_2026.pdf`` (169 pages). The printed page +number in the footer is one lower. + +Narrative sections - worth reading +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +.. list-table:: + :header-rows: 1 + :widths: 12 46 12 30 + + * - Section + - Title + - Pages + - Summarised in + * - 2 + - HIT Instrument Description + - 10-16 + - :ref:`hit-overview` + * - 3 + - Operating Modes (incl. dynamic thresholds) + - 16-18 + - :ref:`hit-overview` + * - 4.1 + - Data Product Level Definitions, data flow + - 19-21 + - :ref:`hit-data-products` + * - 4.2 + - Science Data (rates, sectors, event buffer, PHA word) + - 21-36 + - :ref:`hit-overview`, :ref:`hit-l1a`, :ref:`hit-l3-scope` + * - 4.3 + - Uncertainties + - 36-37 + - :ref:`hit-overview` + * - 4.4 + - I-ALiRT Data + - 37-38 + - :ref:`hit-ialirt` + * - 5.1 + - Rates Compression/Decompression Algorithm + - 38-40 + - :ref:`hit-l1a` + * - 5.3 + - HIT Science Frame Header + - 40-42 + - :ref:`hit-overview` + * - 5.5 + - Lev1A Uncertainty + - 78 + - :ref:`hit-l1a` + * - 6.1-6.2 + - Energy Bins, Livetime Corrected Rates + - 79-81 + - :ref:`hit-l1b` + * - 6.3 + - Summed Rates (prose) + - 81 + - :ref:`hit-l1b` + * - 7.1 + - Conversion to Intensity (prose) + - 98-99 + - :ref:`hit-l2` + * - 7.2-7.3 + - Sectored and Summed intensity (prose) + - 138, 142 + - :ref:`hit-l2` + * - 8 + - I-ALiRT Algorithms + - 145-150 + - :ref:`hit-ialirt` + * - 9 + - Lev3 Algorithms + - 150-169 + - :ref:`hit-l3-scope` + +Tables - look up, do not read +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +.. list-table:: + :header-rows: 1 + :widths: 10 50 14 26 + + * - Table + - Contents + - Pages + - Machine-readable at + * - 2 + - Sector directions (L1 x L2 segment to declination bin, 0-7) + - 24 + - -- (onboard only) + * - 3 + - Species/energy bins for sector rates + - 25 + - ``MOD_10_MAPPING`` + * - 4 + - Event Record Header bit layout + - 27 + - -- **needed for** :ref:`hit-gap-events` + * - 5, 6 + - ADC field bit allocations; event record bit padding + - 29 + - -- **needed for** :ref:`hit-gap-events` + * - 7 + - Detector group flags (Extended Header Block) + - 31 + - -- **needed for** :ref:`hit-gap-events` + * - 8 + - Detector name to address, 0-63 + - 32-34 + - -- **needed for** :ref:`hit-gap-events` + * - 9, 10 + - APIDs; science packet numbers + - 41 + - ``HitAPID`` + * - 11 + - MISCBITS (frame header) + - 43 + - ``COUNTS_DATA_STRUCTURE`` + * - 12 + - ERATES (livetime and trigger counters) + - 43 + - ``COUNTS_DATA_STRUCTURE`` + * - 13 + - SNGRATES (116 singles rates by detector address) + - 44-48 + - ``COUNTS_DATA_STRUCTURE`` + * - 14 + - EVPRATES (event processing counters) + - 49 + - ``COUNTS_DATA_STRUCTURE`` + * - 15 + - COINRATES (26 coincidence rates) + - 50 + - ``COUNTS_DATA_STRUCTURE`` + * - 16 + - PBUFRATES (32 priority buffers, with descriptions) + - 51-52 + - ``COUNTS_DATA_STRUCTURE`` + * - **17** + - **L2FGRATES - 132 Range 2 foreground rates with Particle IDs** + - **53-57** + - ``STANDARD_PARTICLE_ENERGY_RANGE_MAPPING`` + * - 18 + - L2BGRATES - 12 Range 2 background rates + - 57-58 + - -- + * - **19** + - **L3FGRATES - 167 Range 3 foreground rates with Particle IDs** + - **58-65** + - ``STANDARD_PARTICLE_ENERGY_RANGE_MAPPING`` + * - 20 + - L3BGRATES - 12 Range 3 background rates + - 65 + - -- + * - **21** + - **PENFGRATES - 33 Range 4 foreground rates with Particle IDs** + - **66-67** + - ``STANDARD_PARTICLE_ENERGY_RANGE_MAPPING`` + * - 22 + - PENBGRATES - 15 Range 4 background rates + - 67-68 + - -- + * - 23 + - IALIRTRATES - the 20 I-ALiRT rates in the science frame + - 69 + - :ref:`hit-ialirt` + * - 24 + - SECTORRATES - 120 look direction byte assignments + - 70-74 + - ``COUNTS_DATA_STRUCTURE`` + * - 25, 26 + - L4FGRATES (48) and L4BGRATES (24) - the I-ALiRT-aperture ion ranges + - 74-77 + - -- see :ref:`hit-gap-l4rates` + * - 27 + - Event Buffer byte range + - 77 + - ``COUNTS_DATA_STRUCTURE`` + * - **28** + - **Full table of summed Lev1B rates - which L1A bins feed which summed + bin** + - **81-89** + - ``SUMMED_PARTICLE_ENERGY_RANGE_MAPPING`` + * - 29 + - Housekeeping raw-to-EU conversion equations + - 89-91 + - the XTCE + * - 30 + - Thermistor conversion table (191 rows, -40 to +150 degrees C) + - 91-98 + - the XTCE ``ContextCalibrator`` chains + * - **31** + - **HIT Level 2 Standard Rate products - 204 entries with contributing + ranges** + - **99-107** + - ``STANDARD_PARTICLE_ENERGY_RANGE_MAPPING`` + * - 32-35 + - Geometry factors, bin widths, efficiencies for Standard Rates, DT0-DT3 + - 107-138 + - ``imap_hit_standard-dt-factors_*.csv`` + * - 36 + - Same for Sectored Rates, all 8 declination sectors + - 138-142 + - ``imap_hit_sectored-dt-factors_*.csv`` + * - 37 + - Same for Summed Rates + - 142-145 + - ``imap_hit_summed-dt-factors_*.csv`` + * - 38 + - I-ALiRT subcommutation map (60 slots x 3 rate types) + - 145-147 + - ``HIT_PREFIX_TO_RATE_TYPE`` - **partially stale**, see + :ref:`hit-gap-ialirt-slots` + * - 39 + - The 20 I-ALiRT rates, described + - 148 + - :ref:`hit-ialirt` + * - 40-52 + - **L3 only.** Event classification, WINCORR arrays, ADC-MeV + coefficients, cosine corrections, energy bounds, double-power-law fit + parameters + - 153-167 + - -- separate repository, see :ref:`hit-l3-scope` + +Appendices and figures +^^^^^^^^^^^^^^^^^^^^^^ + +.. list-table:: + :header-rows: 1 + :widths: 18 82 + + * - Item + - Note + * - **Appendix C** + - The HIT matrix maps - graphical renderings of the 128 x 400 dE vs E' + lookup for each range, with the foreground element tracks and + background regions drawn on. Referenced repeatedly by sections 4.2.1 + and 4.2.5. **Not present in the v1.11.00 PDF** despite being cited; + request it separately if you need it. + * - Figure 8 + - The data flow diagram, reproduced in :ref:`hit-data-products`. + * - Figure 13 + - Visualisation of the energy bins for all ion species. The quickest way + to see the shape of the product inventory. + * - Figure 14 + - The sectored livetime timeline - the clearest statement of the + 10-minute offset. + * - Figure 15 + - Simulated L4 energy-loss distributions with the I-ALiRT rate boxes + drawn on. See :ref:`hit-ialirt`. + * - Figures 17-21 + - L3 ion tracks, charge resolution, charge histograms. + * - Figures 22-23 + - Sectored rate geometry, and the definition of pitch angle and + gyrophase (after van den Berg et al. 2020). diff --git a/docs/source/algorithm-code-documentation/idex.rst b/docs/source/algorithm-code-documentation/idex.rst index 8cc0a8343e..72675a6e6a 100644 --- a/docs/source/algorithm-code-documentation/idex.rst +++ b/docs/source/algorithm-code-documentation/idex.rst @@ -21,4 +21,4 @@ Level 1 Processing Code: decode idex_l1b idex_l2a - idex_l2b + idex_l2b \ No newline at end of file diff --git a/docs/source/algorithm-code-documentation/idex/ancillary.rst b/docs/source/algorithm-code-documentation/idex/ancillary.rst new file mode 100644 index 0000000000..92982195f3 --- /dev/null +++ b/docs/source/algorithm-code-documentation/idex/ancillary.rst @@ -0,0 +1,350 @@ +.. _idex-ancillary: + +Ancillary Inputs and Dependencies +================================= + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**[DOC]** Section 4.3 organises dependencies by the first level that needs them. +This page does the same, and adds **[CODE]** the exact paths, readers and +failure modes. + +Two kinds of ancillary file +--------------------------- + +IDEX ancillary inputs split cleanly into two groups, and the distinction matters +operationally: + +**In-repository, versioned with the code.** The XTCE definitions, the 10-day +window table, the EU conversion table, the event-message dictionaries and the +atomic mass table all live under ``imap_processing/idex/`` and ship with the +package. Changing one is a code change and goes through review and CI. + +**Delivered as SDC ancillary files.** The two L2A calibration curves arrive as +versioned ``imap_idex_l2a-calibration-curve-*_YYYYMMDD_vNNN.csv`` files through +the normal dependency mechanism. **[DOC]** Section 4.8: "Calibration products +are external ancillary files, not hard-coded algorithm constants. New +calibration files will receive a new versioned filename and the processing +configuration should point to the newest file version." + +.. note:: + + The boundary is not where you might expect it. The waveform DN-to-engineering + -unit factors (``ConversionFactors``) are **hard-coded constants in + ``idex_constants.py``**, not an ancillary file - even though section 4.8 + explicitly says "L1B products are affected by DN-to-pC and engineering-unit + conversion updates" and lists them alongside the calibration curves. + Updating them is therefore a code release, not an ancillary delivery. This is + the most likely place for a future refactor. + +Product window and metadata +--------------------------- + +``idex_10_day_CDF_names.csv`` +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +.. list-table:: + :widths: 24 76 + + * - Path + - ``imap_processing/idex/idex_10_day_CDF_names.csv`` + (``idex_constants.IDEX_10_DAY_RANGES_PATH``) + * - Provided by + - The IDEX team. + * - Read by + - ``idex_utils.get_10_day_window_end_date()``, called from + ``idex_l1a()``. + * - Columns + - ``start_date``, ``end_date``, ``doy`` - all as strings + (``dtype=str``), dates as ``YYYYMMDD``. + * - Rows + - 444, covering 2025-01-01 through 2037-01-01. + * - Behavior + - Requires an **exact** ``start_date`` match. ``ValueError`` if no row + matches, and a second ``ValueError`` guard if more than one does. + Windows restart each 1 January, so the final window of each year is + 5-6 days rather than 10. + +This file is the single source of truth for "what is a valid IDEX L1A start +date". If the instrument team changes the product cadence, this is the first +file to change - and see the warning in :ref:`idex` about the other three places +a cadence lives. + +A 3-row test copy exists at +``imap_processing/tests/idex/test_data/test_idex_10_day_window.csv``. + +CDF attribute definitions +^^^^^^^^^^^^^^^^^^^^^^^^^ + +Six YAML files under ``imap_processing/cdf/config/``: +``imap_idex_global_cdf_attrs.yaml`` plus one +``imap_idex_l{1a,1b,2a,2b,2c}_variable_attrs.yaml``. Loaded through +``idex_utils.get_idex_attrs(data_level)``, which wraps +``ImapCdfAttributes.add_instrument_global_attrs("idex")`` and +``.add_instrument_variable_attrs("idex", data_level)``. + +These are where ``logical_source``, units, fill values, valid ranges and +``DEPEND_n`` relationships live. A missing entry surfaces as a ``KeyError`` from +``get_variable_attributes()`` at build time, or as a ``cdflib`` ``ISTPError`` at +write time. + +L0 dependencies +--------------- + +XTCE packet definitions +^^^^^^^^^^^^^^^^^^^^^^^ + +.. list-table:: + :header-rows: 1 + :widths: 40 14 46 + + * - File + - Size + - Defines + * - ``packet_definitions/idex_science_packet_definition.xml`` + - ~148 KB + - APID 1424. CCSDS headers, ``SHCOARSE``/``SHFINE``, the science routing + fields (``IDX__SCI0TYPE``, ``IDX__SCI0FRAGOFF``, ``IDX__SCI0EVTNUM``, + ``IDX__SCI0COMP``, ``IDX__SCI0PACK``, ``IDX__SCI0FRAG``, + ``IDX__SCI0AID``, ``IDX__SCI0CAT``), the whole ``IDX__TXHDR*`` metadata + header, the ``IDX__SCI0RAW`` waveform payload, sync and CRC fields. + **Contains branching logic**, which is why IDEX science cannot use + ``packet_file_to_datasets()``. + * - ``packet_definitions/idex_housekeeping_packet_definition.xml`` + - ~612 KB + - Housekeeping plus the event-message packet (APID 1418: + ``ELSEC_EVTPKT``, ``ELSSEC_EVTPKT``, ``ELID_EVTPKT``, + ``EL1PAR_EVTPKT`` … ``EL4PAR_EVTPKT``) and the catalog list (APID 1419: + ``IDX_CATLST.*``). + +**[DOC]** Both chapter 3 and section 3.6 are emphatic that these XML files are +**the authoritative source** for field names, bit widths, encodings, aliases and +enumerations, and that the document deliberately does not reproduce them field +by field. Appendix B shows only the first 100 lines of the science XML. Take the +document at its word here: if a field's definition is in question, read the XML. + +L1A dependencies +---------------- + +``idex_evt_msg_parsing_dictionaries.json`` +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +.. list-table:: + :widths: 24 76 + + * - Path + - ``imap_processing/idex/idex_evt_msg_parsing_dictionaries.json`` + * - Read by + - ``PacketParser._create_evt_msg_data()``, then + ``evt_msg_decode_utils.render_event_template()``. + * - Structure + - A dict of named dictionaries. Two are looked up by name: + ``eventMsgDictionary`` (event id → message template) and + ``logEntryIdDictionary`` (event id → short log-entry name). The rest are + value-lookup dictionaries referenced **by name from inside the + templates**, e.g. ``sciState16Dictionary``, ``opCodeLCDictionary``. + * - Gotcha + - JSON stringifies all object keys. The reader converts every key back to + ``int`` before use. If you hand-edit this file, keep keys numeric. + +The template grammar and the ``dictName(value)`` output wrapper are described in +:ref:`idex-l1`. Remember that ``idex_l1b.EventMessage`` compares **whole +rendered strings** for equality, so this file and that enum are coupled. + +L1B dependencies +---------------- + +``idex_variable_unpacking_and_eu_conversion.csv`` +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +.. list-table:: + :widths: 24 76 + + * - Path + - ``imap_processing/idex/idex_variable_unpacking_and_eu_conversion.csv`` + * - Read by + - Twice: ``idex_l1b.unpack_instrument_settings()`` for the bit fields, and + the shared ``imap_processing.utils.convert_raw_to_eu()`` for the + polynomial coefficients (with ``packet_name="IDEX_SCI"``). + * - Columns + - ``index``, ``mnemonic``, ``var_name``, ``starting_bit``, + ``nbits_padding_before``, ``unsigned_nbits``, ``unit``, ``c0``-``c7``, + ``convertAs``, ``packetName``. + * - Rows + - 31 settings. Rows are de-duplicated on ``mnemonic`` for the unpacking + pass, because a segmented-polynomial setting occupies several rows. + +The 31 settings, grouped by the packed telemetry word they come from - note that +each ADC readout word carries **two** settings: + +.. list-table:: + :header-rows: 1 + :widths: 34 44 22 + + * - Source variable + - Mnemonics + - Units + * - ``idx__txhdrprochkch01`` + - ``current_1v_pol``, ``current_1p9v_pol`` + - mA + * - ``idx__txhdrprochkch23`` + - ``temperature_1``, ``temperature_2`` + - °C + * - ``idx__txhdrprochkch45`` + - ``voltage_1v_bus``, ``fpga_temperature`` + - V, °C + * - ``idx__txhdrprochkch67`` + - ``voltage_1p9v_bus``, ``voltage_3p3v_bus`` + - V + * - ``idx__txhdrhvpshkch01`` + - ``detector_voltage``, ``sensor_voltage`` + - V + * - ``idx__txhdrhvpshkch23`` + - ``target_voltage``, ``rejection_voltage`` + - V + * - ``idx__txhdrhvpshkch45`` + - ``reflectron_voltage``, ``current_hvps_sensor`` + - V, mA + * - ``idx__txhdrhvpshkch67`` + - ``positive_current_hvps``, ``negative_current_hvps`` + - mA + * - ``idx__txhdrlvhk0ch01`` + - ``voltage_3p3_ref``, ``voltage_3p3_op_ref`` + - V + * - ``idx__txhdrlvhk0ch23`` + - ``voltage_neg6v_bus``, ``voltage_pos6v_bus`` + - V + * - ``idx__txhdrlvhk0ch45`` + - ``voltage_pos16v_bus``, ``voltage_pos3p3v_bus`` + - V + * - ``idx__txhdrlvhk0ch67`` + - ``voltage_neg5v_bus``, ``voltage_pos5v_bus`` + - V + * - ``idx__txhdrlvhk1ch01`` + - ``current_3p3v_bus``, ``current_16v_bus`` + - A + * - ``idx__txhdrlvhk1ch23`` + - ``current_6v_bus``, ``current_neg6v_bus`` + - A + * - ``idx__txhdrlvhk1ch45`` + - ``current_5v_bus``, ``current_neg5v_bus`` + - A + * - ``idx__txhdrlvhk1ch67`` + - ``current_2p5v_bus``, ``current_neg2p5v_bus`` + - A + +``target_voltage`` is the one to watch scientifically - it is the +3 kV target +bias that sets the ion acceleration, and therefore the TOF scale. + +SPICE, spin and ephemeris +^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** All accessed through ``imap_processing/spice/``, never directly: + +.. list-table:: + :header-rows: 1 + :widths: 32 68 + + * - Need + - Helper + * - Spacecraft state relative to the Sun + - ``spice.geometry.imap_state(et, observer=SpiceBody.SUN)`` + * - IDEX boresight in a target frame + - ``spice.geometry.instrument_pointing(et, SpiceFrame.IMAP_IDEX, + IDEX_EVENT_REFERENCE_FRAME, cartesian=True)`` + * - Cartesian → spherical + - ``spice.geometry.cartesian_to_spherical()`` + * - Solar longitude + - ``spice.geometry.solar_longitude(et, degrees=True)`` + * - Spin phase at an event + - ``spice.spin.get_spacecraft_spin_phase(query_met_times=met)`` then + ``spice.spin.get_spin_angle(..., degrees=True)`` + * - Time conversions + - ``spice.time.ttj2000ns_to_et``, ``et_to_met``, ``met_to_ttj2000ns``, + ``str_yyyymmdd_to_ttj2000ns``, ``epoch_to_doy``, ``et_to_datetime64`` + +Frame constants: ``SpiceFrame.IMAP_IDEX`` is NAIF id ``-43700``; boresight +``[0, 1, 0]``; spin-phase offset ``179.9229/360``. The configured event +reference frame is ``idex_constants.IDEX_EVENT_REFERENCE_FRAME = +SpiceFrame.ECLIPJ2000``, and it propagates all the way to the L2C map's +``Spice_reference_frame`` global attribute - change it in one place and the map +frame changes. + +The **spin table** is a separate mission product from the kernels, and is a +separate CLI dependency (hence 3 dependencies for L1B science: the L1A file, +the kernels, the spin data). + +.. note:: + + Tests mock ``idex_l1b.get_spice_data`` wholesale (see + ``imap_processing/tests/idex/conftest.py``), substituting ones for the + ephemeris arrays and uniform random values for ``spin_phase``, ``longitude`` + and ``latitude``. The L1B validation comparison explicitly skips variables + "known to differ because of time-system conventions or mocked SPICE + quantities". **There is no test that exercises real SPICE geometry for + IDEX.** Tests needing real kernels carry the ``external_kernel`` marker and + are excluded from the default selection. + +L2A dependencies +---------------- + +.. list-table:: + :header-rows: 1 + :widths: 42 58 + + * - File + - Contents + * - ``imap_idex_l2a-calibration-curve-t-rise_20250101_v002.csv`` + - Eight smooth-power-law parameters relating target rise time to impact + speed, plus an unread ninth error factor. Inverted by + ``invert_rise_time_to_velocity()``. + * - ``imap_idex_l2a-calibration-curve-yield-params_20250101_v001.csv`` + - Eight smooth-power-law parameters for charge yield (C/kg) vs impact + speed, plus an unread ninth error factor. Used by + ``calculate_mass_from_velocity()``. + * - ``imap_processing/idex/atomic_masses.csv`` + - 21 reference ion masses with names: H, H₂, C, O, Na, Mg24/25/26, + Si28/29/30, K39/41, Ca40/42/44, Fe54/56/57/58, Au197. Read by + ``time_to_mass()``. **In-repository**, not an SDC ancillary file. + +Both calibration CSVs are keyed in the ``ancillary_files`` dict by the +descriptor segment of their filename - ``cli.py`` does +``path.stem.split("_")[2]``, giving ``l2a-calibration-curve-t-rise`` and +``l2a-calibration-curve-yield-params``. A filename that does not follow the +``imap_idex___`` convention will produce the wrong +dict key and a ``KeyError`` in ``load_calibration_files()``. + +.. warning:: + + The ``atomic_masses.csv`` mass column has apparent off-by-one problems + relative to the isotope names it labels: ``22,Na`` (sodium-23), + ``23,Mg24``, ``24,Mg25``, ``25,Mg26``, ``27,Si28``, ``39,Ca40``, + ``53,Fe54``, ``196,Au197``. Either the masses are indices into some other + scale or the file is misaligned by one row. Because the whole TOF mass path + is currently NaN-filled this has no effect on published data, but it must be + settled with the IDEX team before the mass scale is released - every + ``time_to_mass()`` stretch factor is fitted against these numbers. + +Calibration maintenance +----------------------- + +**[DOC]** Section 4.8, worth reproducing because it defines the operational +contract: + +* New calibration files get a **new versioned filename**; the processing + configuration points at the newest version. Nothing is edited in place. +* If a calibration file is updated, existing products can be **regenerated by + rerunning the affected levels**. L2A products are affected by rise-time and + yield updates; L1B products by DN-to-pC and EU conversion updates. +* Validation tests are rerun after any calibration update. +* The instrument carries **on-board pulsers** that inject programmable known + charges into each CSA to test the engineering-unit conversions. Those + injections are reviewed manually and periodically to monitor each channel's + DN-to-pC conversion. + +**[CODE]** The pulser injections are visible to the pipeline in two places: the +``pulser_on`` state variable in the L1B message product, and the +``pulser_flag`` event classification at L1A (see +:ref:`idex-event-classification`). Nothing in this repository analyses them or +derives a conversion factor from them - that is the manual review the document +describes. diff --git a/docs/source/algorithm-code-documentation/idex/data-products.rst b/docs/source/algorithm-code-documentation/idex/data-products.rst new file mode 100644 index 0000000000..a988666659 --- /dev/null +++ b/docs/source/algorithm-code-documentation/idex/data-products.rst @@ -0,0 +1,332 @@ +.. _idex-data-products: + +Data Products and What Feeds What +================================= + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This is the "what goes into what" map. It is the page to read before touching +``cli.py``, adding a product, or changing a cadence. + +Product inventory +----------------- + +**[CODE]** Everything this repository produces for IDEX. The ``logical_source`` +strings come from ``imap_processing/cdf/config/imap_idex_global_cdf_attrs.yaml`` +and are what ``write_cdf()`` uses to name the output file. + +.. list-table:: + :header-rows: 1 + :widths: 34 8 10 48 + + * - ``logical_source`` + - Level + - Cadence + - Contents + * - ``imap_idex_l1a_sci-10days`` + - L1A + - 10 days + - One record per dust event. Six raw-DN waveform arrays, the two waveform + time axes, the full FPGA header telemetry as event metadata, and ten + event/saturation flags. + * - ``imap_idex_l1a_msg-10days`` + - L1A + - 10 days + - One record per instrument log entry. ``epoch``, ``elsec_evtpkt``, + ``elssec_evtpkt`` and a rendered ``messages`` string. + * - ``imap_idex_l1a_catlst-10days`` + - L1A + - 10 days + - Packet catalog summary (APID 1419), **raw** values, with ``epoch`` + added. + * - ``imap_idex_l1b_sci-10days`` + - L1B + - 10 days + - Same events, waveforms in engineering units, 31 unpacked instrument + settings, ``dead_time``, trigger mode / level / origin, and ten + SPICE-derived geometry variables. + * - ``imap_idex_l1b_msg-10days`` + - L1B + - 10 days + - Reduced state log: ``science_on`` and ``pulser_on`` only, and only at + epochs where one of them actually changes. + * - ``imap_idex_l1b_catlst-10days`` + - L1B + - 10 days + - Packet catalog summary, **derived** values. **[CODE]** Produced by the + *L1A* job (``PacketParser``), not by ``idex_l1b()``. See below. + * - ``imap_idex_l2a_sci-10days`` + - L2A + - 10 days + - Per-event fits for the three low-rate channels (fit parameters, impact + charge, modelled waveform, chi-square), velocity and mass estimates, and + the TOF mass-spectrum variables. **Many of these are deliberately NaN.** + * - ``imap_idex_l2b_sci-1mo`` + - L2B + - 1 month + - Daily dust counts and uptime-corrected count rates, binned by spin-phase + quadrant, optionally by impact charge or mass. **The mass- and + charge-binned variables are deliberately fill-valued.** + * - ``imap_idex_l2c_rectangular-map-1mo`` + - L2C + - 1 month + - The same daily counts and rates binned on a 6° rectangular sky map in + ECLIPJ2000 instead of by spin phase. Same deliberate fill block. + +The processing chain +-------------------- + +**[DOC]** Algorithm document Figure 4.1, annotated **[CODE]** with what the +repository actually does. + +.. code-block:: text + + CCSDS packets (imap_idex_l0_raw_YYYYMMDD_vNNN.pkts) + | + XTCE: idex_science_packet_definition.xml + idex_housekeeping_packet_definition.xml + | + idex_l0.decom_packets() + +----------------+----------------+ + | | | + science packets APID 1418 (EVT) APID 1419 (CATLST) + (APID 1424) | | + | | | + +---------v--------+ +----v----------+ +--v-------------+ + | l1a_sci-10days | | l1a_msg-10days| | l1a_catlst | + | fragment assembly| | template | | l1b_catlst | + | Rice decompress | | rendering | | (passthrough) | + | event flags | | | +----------------+ + +---------+--------+ +----+----------+ + | | + DN->EU CSV ----> | | + SPICE kernels -> | | + spin table ----> | | + +---------v--------+ +----v----------+ + | l1b_sci-10days | | l1b_msg-10days| + | pC / mA waveforms| | science_on | + | dead_time | | pulser_on | + | trigger decode | +----+----------+ + | ephemeris, spin | | + +---------+--------+ | + | | + t-rise cal CSV -> | | + yield cal CSV -> | | + atomic_masses.csv-> | | + +---------v--------+ | + | l2a_sci-10days | | + | target/IG fits | | + | charge, v, mass | | + | TOF mass scale | | + +---------+--------+ | + | | + +-------+--------+ + | + idex_l2b() (3 x l2a + msg) + | + +------------+------------+ + | | + +---------v--------+ +----------v-------------------+ + | l2b_sci-1mo | | l2c_rectangular-map-1mo | + | counts/rates vs | | counts/rates on a 6 deg | + | spin quadrant | | ECLIPJ2000 rectangular grid | + +------------------+ +------------------------------+ + +**[CODE]** Note the shape of the last step: ``idex_l2b()`` returns a +**two-element list**, ``[l2b_dataset, l2c_dataset]``. L2C is produced by the L2B +job because it needs no additional dependencies and is a rebinning of the same +counts. There is no ``l2c`` branch in ``cli.py``'s ``Idex.do_processing`` - +asking for ``--level l2c`` raises ``NotImplementedError`` even though ``"l2c"`` +is listed in ``PROCESSING_LEVELS`` in ``imap_processing/__init__.py``. + +The windowing model +------------------- + +IDEX does not produce daily files. It produces **10-day** files up to L2A and +**monthly** files at L2B/L2C. This is a direct consequence of the event rate: +**[DOC]** roughly 16 dust events per day, so a daily product would frequently +contain a handful of events or none at all. + +The 10-day windows +^^^^^^^^^^^^^^^^^^ + +**[CODE]** The windows are **not** computed. They come from a fixed lookup +table supplied by the IDEX team: +``imap_processing/idex/idex_10_day_CDF_names.csv``, 444 rows covering +2025-01-01 through 2037-01-01, with columns ``start_date``, ``end_date``, +``doy``. + +.. code-block:: text + + start_date,end_date,doy + 20250101,20250110,1 + 20250110,20250120,10 + ... + 20361225,20370101,360 + +Windows **restart at each new year**, so the last window of a year is 5-6 days +rather than 10. That is expected and is called out in a comment in +``idex_constants.py``. + +``idex_utils.get_10_day_window_end_date(start_date)`` is the only reader. It +requires an **exact** match on ``start_date`` and raises ``ValueError`` +otherwise. The practical consequence: **an IDEX L1A job can only be launched on +a start date that appears in that CSV.** The SDC scheduler, not this repository, +is responsible for knowing those dates. + +Window boundaries are compared as TT-J2000 nanoseconds via +``str_yyyymmdd_to_ttj2000ns()``, half-open: ``start <= epoch < end``. + +The monthly windows +^^^^^^^^^^^^^^^^^^^ + +**[CODE]** L2B/L2C take **three** 10-day L2A files (the CLI expects 3 or 4 +dependencies: three science files plus at least one message file) and produce +one monthly product. There is no monthly lookup table - the month is whatever +the three inputs span. Internally L2B works **per day of year**, so the monthly +product's ``epoch`` dimension is one entry per day that had any event, with the +value set to the **mean epoch of that day's events** (not midnight). + +.. warning:: + + ``epoch_to_doy`` is used as the grouping key, so day-of-year, not a full + date, is the identity of a daily bin. A monthly product that spans a New Year + boundary relies on ``dict.fromkeys`` to preserve encounter order (so DOY 365 + sorts before DOY 1), but a product spanning **more than one year** would + collide DOYs from different years into the same bin. At a one-month cadence + this cannot happen in practice; do not extend the cadence without fixing it. + +The L0 time-tagging problem +--------------------------- + +.. important:: + + **An L0 file's date does not tell you which events are inside it.** + +**[DOC]** Section 3.4 is explicit that the CCSDS secondary header time +(``SHCOARSE`` / ``SHFINE``) "describes when the packet was generated" and is +"distinct from the dust-impact event time, which is reconstructed from the FPGA +metadata header." + +Because IDEX stores events onboard and transmits them later in transmit mode, +the packet-generation time is effectively a **downlink** time. L0 files are +organized by that time. So: + +* Events from a single day can be spread across **several** L0 files. +* A single L0 file can contain events from **several different days**, possibly + far apart. +* An event's correct 10-day window may have been "closed" days before the packet + carrying it was ever generated. + +**[CODE]** Every level deals with this the same way - over-query the inputs, +then filter on the reconstructed event epoch: + +.. list-table:: + :header-rows: 1 + :widths: 14 86 + + * - Level + - How it copes + * - L1A + - ``idex_l1a(packet_files, window_start_date)`` accepts a **list** of L0 + files, parses all of them, concatenates per product type along ``epoch``, + sorts, drops duplicate epochs, and *then* masks to + ``[window_start, window_end)``. The input list is sorted first + (``sorted(packet_files)``) so that ``drop_duplicates("epoch", + keep="last")`` keeps the **highest file version** when the same event + arrives in two files. + * - L1B + - ``cli.py`` notes: *"since there may be events that occur before the start + date of the job, there is a buffer added to the upstream dependency + query. This means that there may be multiple l1a science files that are + returned but we only want to process the file with the same start date."* + It therefore filters ``science_files`` down to the one whose **filename** + contains ``self.start_date``, and raises ``ValueError`` if none matches. + * - L2A + - Takes ``science_files[0]`` - a single L1A file - plus two ancillary + calibration files. + * - L2B + - Selects inputs by descriptor (``sci-10days`` and ``msg-10days``), + de-duplicates the housekeeping files with ``set()``, and sorts **both** + lists by first epoch before concatenating. + +Do not simplify any of this to "one input file per job". The over-query is +deliberate and the filtering is what makes the products correct. + +.. note:: + + The *buffering* half of the fix lives outside this repository - it is the + SDC's dependency query that decides how many L0 or L1A files to hand a job. + This repository only implements the filtering half. If events start going + missing from products, the first question is whether the upstream query + buffer is wide enough, not whether ``idex_l1a()`` is filtering correctly. + +CLI wiring and dependency counts +-------------------------------- + +**[CODE]** ``imap_processing/cli.py``, ``class Idex``. The dependency-count +checks are strict and are the first thing to fail when the SDC query changes. + +.. list-table:: + :header-rows: 1 + :widths: 10 20 70 + + * - Level + - Dependencies + - Behavior + * - ``l1a`` + - at most 2 + - ``ValueError`` if ``len(dependency_list) > 2``. Passes **all** + ``source="idex"`` file paths plus ``self.start_date`` to ``idex_l1a()``, + which returns a **list** of datasets (science, message, catlst - whichever + were present and non-empty in the window). + * - ``l1b`` + - 3 for ``sci-10days``, 1 otherwise + - ``ValueError`` on a mismatch. The three for science are the L1A file, the + SPICE kernels and the spin data. Selects the L1A file by start date, then + calls ``idex_l1b(load_cdf(file), self.descriptor)``. + * - ``l2a`` + - exactly 3 + - One L1B science file plus the two calibration CSVs. Ancillary files are + keyed by ``path.stem.split("_")[2]``, i.e. the descriptor segment of + ``imap_idex_l2a-calibration-curve-t-rise_20250101_v002.csv`` becomes + ``l2a-calibration-curve-t-rise``. + * - ``l2b`` + - 3 or 4 + - Three ``sci-10days`` L2A files and one or more ``msg-10days`` L1B files. + Returns both the L2B and the L2C dataset. + * - ``l2c`` + - -- + - **No branch.** Falls through to ``NotImplementedError``. + +Example invocations:: + + imap_cli --instrument idex --level l1a --start-date 20260101 \ + --descriptor sci-10days + imap_cli --instrument idex --level l1b --start-date 20260101 \ + --descriptor sci-10days + imap_cli --instrument idex --level l2a --start-date 20260101 \ + --descriptor sci-10days + imap_cli --instrument idex --level l2b --start-date 20260101 \ + --descriptor sci-1mo + +The catalog-list oddity +----------------------- + +**[CODE]** ``PacketParser.__init__`` loops over ``{"l1a": raw_datasets_by_apid, +"l1b": derived_datasets_by_apid}``. Event messages are produced from the raw +dictionary only, guarded by ``level == "l1a"``. The catalog list has **no such +guard**, so a single L1A job emits *both* ``l1a_catlst-10days`` (raw DN) and +``l1b_catlst-10days`` (derived engineering units). + +Consequences worth knowing: + +* An L1B ``catlst`` product exists that ``idex_l1b()`` cannot produce - + ``idex_l1b()`` raises ``ValueError`` for any descriptor that is not + ``sci-10days`` or ``msg-10days``. +* The processing of ``catlst`` is a passthrough plus an ``epoch`` computed from + ``shcoarse`` / ``shfine`` - i.e. tagged by **packet creation time**, which for + a catalog summary is the right answer. +* The algorithm document does not mention the catalog-list product at all + (section 4.6.2 states L1A produces only ``sci`` and ``msg``), although + section 4.2 mentions "catalog-list products" in passing. diff --git a/docs/source/algorithm-code-documentation/idex/event-classification.rst b/docs/source/algorithm-code-documentation/idex/event-classification.rst new file mode 100644 index 0000000000..cf7ba7ae12 --- /dev/null +++ b/docs/source/algorithm-code-documentation/idex/event-classification.rst @@ -0,0 +1,260 @@ +.. _idex-event-classification: + +Event Classification and Saturation Flags +========================================= + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +.. important:: + + **Everything on this page is [CODE].** ``imap_processing/idex/idex_event_flags.py`` + (~480 lines) has no counterpart anywhere in the 1 June 2026 algorithm + document - not in the L1A section, not in the product description, not in + the flowchart. It was added after the document snapshot. + + It matters more than its absence from the document suggests: these flags are + what decide whether an L2A velocity and mass estimate is published at all, + and they are the sole filter on which events reach the L2B and L2C count + products. + +Why it exists +------------- + +IDEX triggers on things that are not dust. The onboard pulser injects known +charges to verify the DN-to-engineering-unit conversions; software and external +triggers capture noise baselines on demand; and a real impact can drive a +high-gain channel past its ADC ceiling, making the fitted charge meaningless. + +The flags answer three questions per event: + +1. **What kind of event is this?** science, pulser, or noise capture - mutually + exclusive. +2. **Does the TOF waveform actually look like a dust impact?** the dust-hit + flag. +3. **Which of the six channels are saturated?** one flag per channel. + +The ten flags +------------- + +All ten are ``uint8``, event-indexed, valued 0 or 1. They are created at +**L1A** (``RawDustEvent.process()`` calls ``classify_event_flags()``), copied +forward unchanged at **L1B** (``idex_l1b_science()``), and copied forward again +at **L2A**. + +``idex_event_flags.EVENT_FLAG_NAMES`` - mutually exclusive event type, plus the +dust-hit qualifier: + +.. list-table:: + :header-rows: 1 + :widths: 28 72 + + * - Flag + - Set when + * - ``noise_capture_flag`` + - No trigger channel is active at all, **or** a software/external trigger + is present and the only active channel is TOF High. + * - ``pulser_flag`` + - TOF High is the *only* active channel, its trigger mode is ``1`` + (Threshold), and its trigger threshold is exactly **1000 DN** + (``_PULSER_THRESHOLD_DN``). + * - ``science_event_flag`` + - Everything else. This is the default, not a positive test. + * - ``dust_hit_flag`` + - Only ever set on a science event, and only when the TOF waveform passes + the peak test described below. + +``idex_event_flags.SATURATION_FLAG_NAMES`` - one per waveform channel: + +``tof_high_saturation_flag``, ``tof_mid_saturation_flag``, +``tof_low_saturation_flag``, ``target_high_saturation_flag``, +``target_low_saturation_flag``, ``ion_grid_saturation_flag``. + +``ALL_FLAG_NAMES`` is the concatenation, and is what the L1B and L2A code +iterate over. + +Event type classification +------------------------- + +The inputs are raw header telemetry, not waveforms: + +.. code-block:: python + + trigger_id = telemetry["idx__txhdrtrigid"] + active = {channel for bit, channel in _TRIGGER_CHANNELS.items() + if trigger_id & (1 << bit)} + # plus any channel whose idx__txhdr{hg,mg,lg}trigmode != 0 + sw_or_ext = bool(trigger_id & ((1 << 4) | (1 << 5))) + hg_mode = telemetry["idx__txhdrhgtrigmode"] + hg_threshold= (telemetry["idx__txhdrhgtrigctrl1"] >> 22) & 0x3FF + +``_TRIGGER_CHANNELS`` maps bits 0-3 to ``"TOF H"``, ``"TOF L"``, ``"TOF M"``, +``"Target H"`` - the same bit assignment L1B uses for ``trigger_origin``, but +here bits 4 and 5 (software and external) are handled separately rather than +becoming channels. + +Note that the active-channel set is the **union** of two sources: the trigger-ID +bits, *and* any gain channel with a non-zero trigger mode. An armed-but-not- +firing channel therefore still counts as active, which is what keeps a +genuinely multi-channel event from being mistaken for a pulser event. + +The decision, in order: + +.. code-block:: text + + if not active or (sw_or_ext and active ⊆ {"TOF H"}): + -> noise_capture + elif active == {"TOF H"} and hg_mode == 1 and hg_threshold == 1000: + -> pulser + else: + -> science_event + +The pulser test is deliberately narrow - it recognises the *specific* +configuration the onboard stimulus sequence uses. A pulser run at any other +threshold would be classified as a science event. + +Dust-hit detection +------------------ + +``_has_dust_hit()``. A science event is only a dust hit if the **raw** TOF +waveform contains at least **two peaks** that each clear **7 baseline sigma** +and have a FWHM of at least **20 ns**. + +The parameters: + +.. list-table:: + :header-rows: 1 + :widths: 34 16 50 + + * - Constant + - Value + - Role + * - ``_BASELINE_WINDOW_US`` + - 3.0 + - Baseline is estimated from the first 3 µs of the record. + * - ``_PEAK_THRESHOLD_SIGMA`` + - 7.0 + - Minimum peak height above baseline, in sigma. + * - ``_MIN_PEAK_WIDTH_US`` + - 0.020 + - Minimum FWHM, i.e. 20 ns. + * - ``_MIN_PEAK_COUNT`` + - 2 + - Peaks required. + * - ``_MIN_PEAK_DISTANCE_US`` + - 0.030 + - Minimum separation between peaks, 30 ns, converted to samples via the + median sample spacing. + +Baseline and noise (``_baseline_corrected``) use **robust** statistics, not the +mean and standard deviation: + +.. math:: + + \mathrm{baseline} = \mathrm{median}(x_{\text{first }3\mu s}), + \qquad + \sigma = 1.4826 \times \mathrm{MAD} + +with a fallback to ``nanstd`` if the MAD is zero or non-finite. The 1.4826 +factor is the standard MAD-to-sigma conversion for Gaussian noise; it is used +because a real TOF record has large outliers by construction and a plain +standard deviation would be dominated by the signal it is trying to detect. + +Peak candidates are always found on **TOF High**, because it has the best +sensitivity. But a saturated peak has a flat top and no meaningful FWHM. So +``_saturation_aware_width()`` does the following per peak: + +1. If the TOF High sample at the peak is not saturated, measure the FWHM on + TOF High. +2. Otherwise, walk down the gain stages - **Mid, then Low** - find the sample + nearest in *time* to the peak, and if that sample is not saturated, measure + the FWHM on that channel instead. +3. If all three are saturated, return NaN and the peak does not qualify. + +``_fwhm()`` finds the half-maximum crossings by walking outward from the peak +and then **linearly interpolating** the crossing time between the two bracketing +samples (``_crossing_time``). It requires both sides to be properly bracketed - +a peak that runs off the end of the record returns NaN rather than a truncated +width. + +The result is a detector that finds two distinct ion arrivals - which is what a +mass spectrum looks like and what electronic noise generally does not. + +Saturation flags +---------------- + +``classify_saturation_flags()``. A channel is saturated if **any finite sample** +reaches 95 % of the channel's full scale: + +.. list-table:: + :header-rows: 1 + :widths: 30 20 50 + + * - Channels + - Full scale + - Threshold at 95 % + * - The three TOF (10-bit) + - ``_TOF_MAX_DN`` = 1023 + - 971.85 DN + * - The three low-rate (12-bit) + - ``_LOW_RATE_MAX_DN`` = 4095 + - 3890.25 DN + +``_SATURATION_FRACTION = 0.95``. The comparison is on the **raw DN** waveform at +L1A, before any engineering-unit conversion - which is the only place the ADC +ceiling is meaningful. + +The three low-rate waveforms are optional arguments (defaulting to ``None``) so +that the classification API can be called with TOF waveforms alone; a ``None`` +channel gets a flag of 0, not a fill value. + +What downstream code does with them +----------------------------------- + +This is the part worth memorising, because it is where the flags become +scientifically load-bearing. + +**L2A - ``_mask_saturated_derived_estimates()``.** For each of ``target_low``, +``target_high``, ``ion_grid``, if the channel's saturation flag is 1, then +``{channel}_impact_charge``, ``{channel}_velocity_estimate`` and +``{channel}_dust_mass_estimate`` are set to NaN. The **fit parameters and fit +results survive** - they stay available for diagnostics. This function raises +``KeyError`` if any of the three saturation flags is missing from the dataset, +so L2A cannot run on an L1B file produced before the flags existed. + +**L2A - ``_mask_non_science_derived_estimates()``.** If +``science_event_flag != 1``, the six velocity and mass estimates are set to NaN. +Again the fits survive: the code comment states that "fits and fitted charges +remain available for diagnostics in all instrument modes." Unlike the saturation +masking, this one is tolerant - a missing ``science_event_flag`` logs a debug +message and skips. + +**L2A - ``calculate_ion_grid_velocity_and_mass()``.** Consumes three saturation +flags directly to pick which target channel to use as the denominator of the +charge ratio: Target High if unsaturated, else Target Low if unsaturated, else +give up. Returns ``(NaN, NaN)`` immediately if Ion Grid itself is saturated. + +**L2B / L2C - ``_get_dust_hit_indices()``.** Every count in every L2B and L2C +product is filtered to ``dust_hit_flag == 1``: + +.. code-block:: python + + if "dust_hit_flag" not in l2a_dataset: + return np.array([], dtype=int) + dust_hit = l2a_dataset["dust_hit_flag"].data[current_day_indices] == 1 + return current_day_indices[dust_hit] + +.. warning:: + + Note the failure mode: if ``dust_hit_flag`` is absent from the L2A input, + this returns an **empty index array** rather than raising. Every count in + the monthly product would then be zero, every rate would be zero, and + ``rate_calculation_quality_flags`` would still read 1 (it only reports + uptime problems). An L2B product built from pre-flag L2A files would look + like a month in which IDEX detected no dust at all. + +Testing +------- + +``imap_processing/tests/idex/test_idex_event_flags.py`` (175 lines) covers the +classification branches and the saturation thresholds with synthetic waveforms, +and does not require external test data. diff --git a/docs/source/algorithm-code-documentation/idex/implementation-status.rst b/docs/source/algorithm-code-documentation/idex/implementation-status.rst new file mode 100644 index 0000000000..1c757fc193 --- /dev/null +++ b/docs/source/algorithm-code-documentation/idex/implementation-status.rst @@ -0,0 +1,392 @@ +.. _idex-implementation-status: + +Implementation Status and Known Gaps +==================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This page is the honest accounting of where the code stands against the +algorithm document. **Read it before proposing or estimating work.** + +Accurate as of the most recent survey of ``imap_processing/idex`` against the +IDEX Algorithms Document dated 1 June 2026. If you change something material, +update this page in the same commit. + +Summary +------- + +.. list-table:: + :header-rows: 1 + :widths: 16 22 62 + + * - Level + - State + - Notes + * - L0 + - **Complete** + - 54 lines, one function. Science and housekeeping decommutated + separately because the science XTCE branches. + * - L1A science + - **Mature** + - Fragment assembly, retransmit handling, both waveform decode paths, + time-axis reconstruction and incomplete-event filtering are all done and + tested. The event-flag classification is an undocumented addition. + * - L1A messages / catlst + - **Complete, partly undocumented** + - Template rendering works and degrades gracefully. The ``catlst`` + products are not in the document at all, and an ``l1b_catlst`` is + emitted by the *L1A* job. + * - L1B science + - **Complete, diverges from the document** + - All six operations from section 4.7.1 are implemented. The TOF + conversion factors and their **units** differ from Table 4.2; the code + is newer. + * - L1B messages + - **Complete, diverges from the document** + - ``pulser_on`` uses paired-transition logic the document does not + describe. Extremely brittle exact-string matching. + * - L2A + - **Written, largely withheld** + - Target and ion-grid fits, velocity and mass are published. The entire + TOF mass-spectrum path is computed and then NaN-filled. Ion-grid + velocity/mass is implemented despite being listed as future work. + * - L2B / L2C + - **Written, partly withheld; undocumented** + - Dust-hit counts and uptime-corrected rates are published; everything + mass- or charge-resolved is fill-valued. ~840 lines that the document + covers with two flowchart boxes. + * - L3 + - **Does not exist** + - Not a gap. See :ref:`idex-l3-scope`. + * - Quicklook + - **Not started** + - No IDEX quicklook code, and the document does not specify any. + +Where the code has moved past the document +------------------------------------------ + +The 1 June 2026 document was written against a snapshot of this repository and +is unusually faithful to it. These are the places it has since fallen behind. +**In all of them the code is authoritative.** + +.. list-table:: + :header-rows: 1 + :widths: 26 74 + + * - Topic + - Divergence + * - **Event and saturation flags** + - ``idex_event_flags.py`` (~480 lines, ten flags) appears nowhere in the + document. It gates L2A's derived estimates and is the sole filter on + L2B/L2C counts. See :ref:`idex-event-classification`. + * - **TOF conversion factors** + - Document Table 4.2: TOF High/Mid/Low = 2.89e-4 / 1.13e-2 / 5.14e-1 + **pC/DN**. Code ``ConversionFactors``: 7.50e-5 / 2.93e-3 / 1.34e-1 + **mA/DN**. The three low-rate channels agree exactly. The L1B variable + attrs say ``mA``, and ``test_idex_l1b.py`` rescales the IDEX team's own + validation file from "the legacy pC factors" before comparing. + * - **Ion-grid velocity and mass** + - Document section 4.7.9 lists it as future work and section 4.7.7 lists + ``ion_grid_velocity_estimate`` and ``ion_grid_dust_mass_estimate`` in + the NaN block. Both are **implemented and published**, masked only by + saturation and science-event state. + * - **L2B / L2C** + - The document describes them in one sentence each plus a flowchart box. + ``idex_l2b.py`` implements daily binning, four bin schemes, an uptime + model derived from the message log, rate quality flags, and a + rectangular sky map. + * - **``pulser_on`` logic** + - Document section 4.7.2 describes unconditional exact-match assignment. + Code requires a ``PULSER_ON`` message **immediately followed** by a + ``PULSER_OFF`` message. + * - **Event message strings** + - The document quotes rendered strings without the dictionary wrapper + (``SCI state change: ACQSETUP ==> ACQ``). The code's renderer emits + ``sciState16Dictionary(ACQSETUP)``, and ``EventMessage`` matches the + wrapped form. + * - **Catalog list products** + - Section 4.6.2 states L1A produces only ``sci`` and ``msg``. The code + also produces ``l1a_catlst-10days`` and ``l1b_catlst-10days``. + * - **Trigger origin labels** + - Document: "software trigger". Code: ``"SW trigger"``. Cosmetic, but the + code's spelling is what appears in the CDF. + +Deliberately withheld products +------------------------------ + +Not bugs. The variables are created so the CDF schema stays stable and then +overwritten so unvalidated numbers are not mistaken for science. The test suite +asserts this behavior. + +**L2A** - filled with ``np.nan`` unconditionally: + +``tof_peak_area_under_fit``, ``tof_peak_chi_square``, +``tof_peak_fit_parameters``, ``tof_peak_kappa``, +``tof_peak_reduced_chi_square``, ``tof_snr``, ``mass``, ``mass_scale``. + +**L2B** - filled with ``np.iinfo(np.int64).min`` (counts) or ``np.nan`` (rates): + +``counts_by_mass``, ``counts_by_charge``, ``rate_by_mass``, ``rate_by_charge``. + +**L2C** - same: + +``counts_by_mass_map``, ``counts_by_charge_map``, ``rate_by_mass_map``, +``rate_by_charge_map``. + +What a consumer can actually use today: per-event target and ion-grid fits, +impact charges, velocity and mass estimates; the per-event SPICE context; and +monthly dust-hit counts and uptime-corrected rates by spin quadrant and by 6° +sky pixel. **No composition, no mass spectrum, no mass- or charge-resolved +rates.** + +Suspected defects +----------------- + +Each of these was found by reading the code against the document. None is +confirmed by the IDEX team. Confirm before changing behavior. + +.. list-table:: + :header-rows: 1 + :widths: 26 74 + + * - Where + - Issue + * - ``idex_l2b.compute_counts_by_charge_and_mass`` + - **Unit mismatch on dust mass.** Applies ``FG_TO_KG`` (×1e-15) to + ``target_low_dust_mass_estimate``, which L2A computes as C ÷ (C/kg) = + **kilograms** and declares as ``UNITS: Kg``. Against mass bin edges of + 6.31e-17 to 1.00e-14 kg this drives every value below the lowest edge, + where the clip puts it in the bottom bin. Currently invisible because + the mass-binned outputs are fill-valued. **Must be resolved before those + products are released.** Either L2A should report femtograms (and its + attrs updated) or L2B should drop the conversion. + * - ``idex_l2a`` + - **Dimension name mismatch on the TOF peak fits.** + ``tof_peak_fit_parameters`` is produced with core dims + ``["mass_index", "peak_fit_parameters_index"]`` (plural), but the index + and label variables, and the ``DEPEND_1`` / ``DEPEND_2`` entries in + ``imap_idex_l2a_variable_attrs.yaml``, all use + ``peak_fit_parameter_index`` (singular). The labels therefore do not + attach to the array's third dimension. + * - ``idex_l2b._get_dust_hit_indices`` + - **Silent empty result.** If ``dust_hit_flag`` is absent from the L2A + input, returns an empty index array rather than raising. Every count and + rate in the month becomes zero while + ``rate_calculation_quality_flags`` still reads 1, because that flag only + reports uptime problems. A month built from pre-flag L2A files would + look like a month with no dust. + * - ``idex_l1a.PacketParser._create_science_dataset`` + - **No guard for an all-bad file.** If every event is skipped + (conflicting fragments or wrong waveform lengths), + ``xr.concat(processed_dust_impact_list, dim="epoch")`` is called on an + empty list and raises. + * - ``idex_l1a._create_science_dataset`` + - **Fatal on out-of-order packets.** A waveform packet arriving before its + header raises ``KeyError`` and aborts the entire file. Everywhere else + L1A is defensive and skips the affected event; this one case is not. + * - ``idex_l1a`` waveform decoders + - **Unasserted length agreement.** The uncompressed path drops the last 4 + samples, the compressed path drops the last 3. Both are expected to land + on 8189 high-rate samples so that a window containing both kinds of + event concatenates cleanly. Nothing checks it. + * - ``idex_l2a.calculate_kappa`` + - **Signed mean where the docstring says magnitude.** Computes + ``mean(mass[peaks] - round(mass[peaks]))``, which ranges over (−0.5, + 0.5] and cancels, so a spectrum with peaks equally split above and below + integer mass scores near-perfect. The docstring claims a 0-1 range where + "closer to zero indicates better accuracy". An absolute value or RMS + would match the stated intent. + * - ``idex_l2a.time_to_mass`` + - **Hardcoded, unvalidated search window.** Stretch factors are + ``linspace(1400, 1500, 10)``. Nothing checks whether the winning stretch + landed on an edge of that bracket, and no diagnostic is emitted if it + does. + * - ``idex/atomic_masses.csv`` + - **Apparent off-by-one between mass and isotope name**: ``22,Na``, + ``23,Mg24``, ``27,Si28``, ``39,Ca40``, ``53,Fe54``, ``196,Au197``. + Every ``time_to_mass`` stretch factor is fitted against these numbers. + * - ``idex_l2a.load_calibration_files`` + - **No validation.** ``.values.flatten()[:8]`` on a CSV read with + ``skiprows=1, header=None``. A reordered or differently-shaped + calibration file yields eight wrong numbers rather than an error. + * - ``idex_l2b.compute_rates_*`` + - **Inconsistent no-data sentinel.** Charge/mass-binned rate arrays are + initialised to ``-1.0``; the agnostic rate arrays are initialised to + ``np.nan``. Both mean "no uptime data". + * - ``idex_l2b`` + - **DOY is the daily key.** ``epoch_to_doy`` groups by day of year, so a + product spanning more than one calendar year would collide days from + different years. Safe at a monthly cadence; unsafe if the cadence grows. + +Dead code +--------- + +``idex_l2a.remove_signal_noise()``, ``sine_fit()`` and +``butter_lowpass_filter()`` implement a three-stage noise filter (linear +detrend, sine-wave background subtraction at +``TARGET_NOISE_FREQUENCY = 7000``, then a second-order Butterworth low-pass at +``TARGET_HIGH_FREQUENCY_CUTOFF = 100``). **Nothing in the pipeline calls them.** +``estimate_dust_mass()`` has a ``remove_noise: bool = False`` parameter that +only logs "remove_noise is ignored for this fit path" when set - the fit always +runs on the raw low-rate waveform. + +They are exercised by ``test_idex_l2a.py`` and are presumably retained for a +future filtered-fit path. Do not delete them without asking the IDEX team, and +do not assume the fit is filtered when reading the L2A output. + +Not implemented at all +---------------------- + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Item + - Status + * - Combined target-channel fit + - **[DOC]** 4.7.8. Not started. The saturation flags it needs already + exist. + * - Combined TOF-channel fit + - **[DOC]** 4.7.10. Not started. Would also supply the fit-validity flag + both future-work sections promise. + * - Dead-time correction to rates + - ``dead_time`` is computed at L1B and never consumed. L2B corrects for + acquisition uptime only. + * - Pulser-based conversion monitoring + - **[DOC]** 4.8 describes periodic **manual** review of pulser + injections. The pipeline flags pulser events but derives nothing from + them. Arguably correct as-is. + * - Decontamination-cycle handling + - **[DOC]** chapter 2 describes monthly 8-hour 120 °C bakeouts. Visible + only as event-message log entries; nothing models or flags them. + * - CRC / checksum verification + - The science XTCE defines ``CHECKSUM``. Nothing verifies it. + * - ``l2c`` CLI branch + - ``"l2c"`` is in ``PROCESSING_LEVELS`` but ``Idex.do_processing`` has no + branch for it - it raises ``NotImplementedError``. L2C is produced by + the ``l2b`` job. Harmless, but confusing. + +Hard failures in the code +------------------------- + +Explicit exceptions, so you know what a bad input looks like: + +.. list-table:: + :header-rows: 1 + :widths: 44 56 + + * - Location + - Condition + * - ``cli.py`` ``Idex.do_processing`` + - ``NotImplementedError`` for any level other than l1a/l1b/l2a/l2b. + * - ``cli.py`` ``Idex.do_processing`` + - ``ValueError`` if dependency counts are wrong: >2 (l1a), ≠3 for + ``sci-10days`` or ≠1 otherwise (l1b), ≠3 (l2a), <3 or >4 (l2b). + * - ``cli.py`` ``Idex.do_processing`` + - ``ValueError`` if no ``source="idex"`` science file is found for L1B, or + if none has ``self.start_date`` in its filename. + * - ``idex_utils.get_10_day_window_end_date`` + - ``ValueError`` if the start date is not a row in + ``idex_10_day_CDF_names.csv``, or if more than one row matches. + * - ``idex_l1a._create_science_dataset`` + - ``KeyError`` if a waveform packet arrives before its header packet. + * - ``idex_l1b.idex_l1b`` + - ``ValueError`` for any descriptor not starting with ``sci-10days`` or + ``msg-10days``. + * - ``idex_l2a._mask_saturated_derived_estimates`` + - ``KeyError`` if any of the three low-rate saturation flags is missing + from the L2A dataset. + +Soft failures worth knowing +--------------------------- + +These return ``None`` or NaN instead of raising, which means they show up as +missing data rather than a failed job: + +* ``idex_l1a()`` skips (with a warning) any product type with no events inside + the 10-day window, and can return an empty list. +* ``RawDustEvent.process()`` returns ``None`` for conflicting fragments or a + waveform-length mismatch. +* ``idex_l1b_msg()`` returns ``None`` if no science or pulser transition is + present - so **no L1B message file is written**, and L2B then has no uptime + information for that period and falls back to ``rate_calculation_quality_flags + = 0``. +* ``estimate_dust_mass()`` returns all-NaN on a ``curve_fit`` ``RuntimeError``. +* ``invert_rise_time_to_velocity()`` returns NaN for a non-finite or + non-positive rise time, or if no root exists in 0.1-100 km/s. +* ``calculate_snr()`` returns all-NaN if the −7 to −5 µs baseline window is + empty. + +Testing +------- + +**[DOC]** Section 4.9 sets a project-level requirement of **at least 90 % +automated test coverage** for the IDEX pipeline, consistent with the repo-wide +Codecov patch-coverage gate. + +**[CODE]** ``imap_processing/tests/idex/``, ~3000 lines across six test modules +plus a 216-line ``conftest.py``. + +.. list-table:: + :header-rows: 1 + :widths: 36 64 + + * - Module + - Covers + * - ``test_idex_l0.py`` + - Decommutation, packet counts. + * - ``test_idex_l1a.py`` + - Event assembly, duplicate and conflicting fragments, incomplete events, + compressed vs uncompressed decode consistency, event counts, comparison + against the IDEX team's validation HDF5. + * - ``test_idex_l1b.py`` + - EU conversion, waveform units, setting unpacking, trigger decode, dead + time, SPICE attachment (mocked), message-state reduction, CDF writing. + * - ``test_idex_l2a.py`` + - Calibration loading, time-to-mass, SNR, peak fits, smooth power law, + rise-time inversion, waveform fits, mass estimation, and explicit + assertions that the NaN block is filled. + * - ``test_idex_l2b.py`` + - Counts, rates, uptime percentage, spin binning, map binning, fill block. + * - ``test_idex_event_flags.py`` + - Classification branches and saturation thresholds, synthetic waveforms + only. + +Fixtures worth knowing about (``conftest.py``): + +* Three L0 ``.pkts`` files in ``test_data/`` - one science (2023-12-18), one + event-message (2025-01-08), one catalog-list (2024-12-06) - plus a matched + compressed / non-compressed pair from 2023 day 102. +* ``l1a_example_data`` and ``l1b_example_data`` load **IDEX-team-produced HDF5 + validation files** via ``xr.open_datatree``, one group per event. +* ``ancillary_files`` points at the two calibration CSVs kept in ``test_data/``. +* ``get_spice_data`` is **mocked wholesale** - ones for ephemeris, uniform + random for spin phase, longitude and latitude. + +.. important:: + + **Some IDEX tests need the SDC test-data download.** The repository's default + selection is ``-m "not external_kernel and not external_test_data"``. Seven + IDEX tests carry an explicit ``@pytest.mark.external_test_data`` decorator - + and they are the ones that compare against the IDEX team's validation HDF5 + files, i.e. the only tests that check the numbers rather than the plumbing. + To run them:: + + poetry run pytest imap_processing/tests/idex -vvv -m "external_test_data" + + ``imap_processing/tests/idex/conftest.py`` additionally sets + ``pytestmark = pytest.mark.external_test_data`` at module scope. It is the + only conftest in the repository that does so, and whether pytest propagates a + conftest-level ``pytestmark`` to collected tests is version-dependent - so + the effective count of skipped-by-default IDEX tests may be 7 or may be all + 100. Check with ``pytest --collect-only -m external_test_data`` before + relying on a green local run. Either way, the ``l1a_example_data`` and + ``l1b_example_data`` fixtures request ``_download_test_data`` directly, so + they attempt a network fetch regardless of markers. + + Separately: **no IDEX test exercises real SPICE geometry.** + ``test_get_spice_data`` mocks the SPICE functions and uses a fake spin table + (the ``furnish_kernels`` call is commented out), so it verifies array shapes + and names but not values; the ``l1b_dataset`` fixture mocks + ``get_spice_data`` wholesale. Ephemeris, boresight pointing, solar longitude + and spin phase are structurally tested and numerically untested. diff --git a/docs/source/algorithm-code-documentation/idex/index.rst b/docs/source/algorithm-code-documentation/idex/index.rst new file mode 100644 index 0000000000..6bb3de5a3c --- /dev/null +++ b/docs/source/algorithm-code-documentation/idex/index.rst @@ -0,0 +1,265 @@ +:orphan: + +.. _idex-index: + +IDEX +==== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +.. currentmodule:: imap_processing.idex + +This is the IDEX (Interstellar Dust Experiment) instrument module, which contains +the code for processing data from the IDEX instrument. + +Purpose of these pages +---------------------- + +These pages are a **condensed, self-contained working reference** for the IDEX +processing algorithms, written so that a developer (human or AI agent) can get +productive without reading the full algorithm document. + +They are a summary of the source document below plus what the code in +``imap_processing/idex`` actually does. Where the two disagree, that is called +out explicitly in :ref:`idex-implementation-status`. + +.. _idex-source-documents: + +Source documents +---------------- + +**None of these are redistributed in this repository.** This is an open-source +repository and the mission documents are not ours to publish. Request them from +the IDEX instrument team at LASP / CU Boulder or the SDC document store. + +.. list-table:: + :header-rows: 1 + :widths: 22 78 + + * - Short name + - Reference + * - **Algorithm document** + - *IDEX Algorithms Document*, dated 1 June 2026. Ethan Ayari, Alex Doner, + Jamey Szalay, Scott Knappmiller, Mihály Horányi. 39 numbered pages + (41 PDF pages). The primary source for these pages. Referred to below + as "the algorithm document". + * - **Science Data Management Plan (SDMP)** + - Defines the IDEX product inventory and the L1A/L1B/L2A/L2B/L2C split + that the algorithm document's Figure 4.1 reproduces. Not read by any + code here. + * - **XTCE packet definitions** + - ``imap_processing/idex/packet_definitions/idex_science_packet_definition.xml`` + and ``idex_housekeeping_packet_definition.xml``. The algorithm document + explicitly defers to these as authoritative for field names, bit widths, + encodings and enumerations, and reproduces only their first 100 lines + in Appendix B. + * - **Heritage instruments** + - SUDA (electronics heritage), and LDEX on LADEE for the impact-charge + pulse model. Horányi et al. (2014), *The Lunar Dust Experiment (LDEX)*, + Space Sci. Rev. 185(1-4), 93-113, doi:10.1007/s11214-014-0118-7 is cited + directly in ``idex_l2a.fit_impact``. + +.. tip:: + + If you hold a copy of the algorithm document, put it in ``docs/reference/``. + That directory is gitignored, so it will never be committed, and the section + index in :ref:`idex-reference-tables` is written against that location. + +.. important:: + + Two conventions used throughout these pages: + + * **[DOC]** marks a statement taken from the algorithm document. It describes + the intended behavior, which may not be what the code does yet. + * **[CODE]** marks a statement verified against ``imap_processing/idex``. + + When those conflict, the code is what runs and the document is what the + instrument team expects. Both matter; do not silently "fix" one to match the + other without asking. + +.. warning:: + + **IDEX is moving faster than its algorithm document.** The 1 June 2026 + document is unusually code-aware - it was clearly written against a snapshot + of this repository - but the code has since moved on in ways that matter: + + * **Event classification and saturation flags do not appear in the document + at all.** ``imap_processing/idex/idex_event_flags.py`` (~480 lines) adds + ten per-event flags at L1A that gate what L2A and L2B are allowed to + publish. See :ref:`idex-event-classification`. + * **The TOF DN-to-engineering-unit factors changed, and so did their + units.** The document's Table 4.2 gives pC/DN for all six channels; the + code converts the three TOF channels to **mA**, with different numbers. + See :ref:`idex-l1`. + * **Ion-grid velocity and mass are implemented**, though the document lists + them under "Future Work" (section 4.7.9). + * **L2B and L2C are implemented** (``idex_l2b.py``, ~840 lines) and are + barely more than a flowchart box in the document. + + Treat the document as the statement of intent for L0-L2A, and this page set + plus the code as the statement of fact. + +.. warning:: + + **Product cadences are 10 days and 1 month, and they have moved before.** + L1A, L1B and L2A are **10-day** products cut on a fixed calendar table + (``idex_10_day_CDF_names.csv``); L2B and L2C are **monthly**. If the + instrument team changes the cadence again, the changes land in that CSV, + in ``get_10_day_window_end_date()``, in the ``logical_source`` strings, and + in the ``descriptor`` strings the CLI branches on - four places, none of + which validate each other. See :ref:`idex-data-products`. + +.. warning:: + + **L0 files are not tagged by event time.** A dust event's epoch comes from + the FPGA metadata header (when the impact happened), but the L0 ``.pkts`` + file is organized by the packet's own downlink/creation time. Events from + one day can therefore be spread across several L0 files, and one L0 file can + contain events from several days. Every IDEX level deals with this by + over-querying its inputs and then filtering on the reconstructed event + epoch. Do not "optimize" that away. See :ref:`idex-data-products`. + +Which page to read +------------------ + +Read only what you need. Each page is designed to be loaded on its own. + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Page + - Read it when you need to know... + * - :ref:`idex-overview` + - What IDEX physically is, how one dust impact becomes six waveforms, and + the vocabulary (event, block, fragment, science type, high/low rate, + dead time). **Start here if you are new.** + * - :ref:`idex-data-products` + - The full product inventory, exact ``logical_source`` strings, the 10-day + and monthly windowing model, the L0 time-tagging problem, and how the + CLI is wired. **The "what goes into what" map.** + * - :ref:`idex-l1` + - Packet decommutation, event assembly from fragments, event time + reconstruction, waveform decoding and Rice-Golomb decompression, time + axes, event-message rendering, then the L1B DN conversions, instrument + setting unpacking, dead time, trigger decoding and SPICE geometry. + * - :ref:`idex-event-classification` + - The ten L1A event and saturation flags: how an event is labelled + science / pulser / noise-capture, how a dust hit is detected, and what + downstream code refuses to publish because of them. **Not in the + algorithm document at all.** + * - :ref:`idex-l2` + - L2A waveform fits, impact charge, the rise-time-to-velocity inversion, + the charge-yield mass estimate, the TOF mass scale and peak fits, and + the deliberate NaN block. Then L2B/L2C daily counts, uptime-corrected + rates and rectangular maps. + * - :ref:`idex-ancillary` + - The 10-day window table, the EU conversion table, the two L2A + calibration curves, the atomic mass table, the event-message + dictionaries, and the SPICE dependencies. + * - :ref:`idex-l3-scope` + - Why there is no IDEX L3 here, and what sits downstream of L2C. + * - :ref:`idex-implementation-status` + - What is implemented, what is deliberately withheld, where the code + deviates from the document, and the suspected defects. + **Read before proposing work.** + * - :ref:`idex-reference-tables` + - Where the big tables live (XTCE, ancillary CSVs, CDF attribute YAML, + PDF page ranges). Deliberately *not* reproduced inline. + +.. toctree:: + :maxdepth: 1 + + overview + data-products + l1 + event-classification + l2 + ancillary + l3-scope + implementation-status + reference-tables + +Ten-second orientation +---------------------- + +* IDEX is a **high-resolution impact-ionization time-of-flight mass + spectrometer** for interstellar dust (ISD) and interplanetary dust particles + (IDP). A dust grain hits a +3 kV target, the impact ionizes it, reflectron + ion optics focus the ions onto a central detector, and the flight times give + a mass spectrum. +* IDEX is **event-driven, not a scanner.** There is no sweep, no spin binning + at the instrument level, no accumulation interval. Nothing happens until a + grain arrives. **[DOC]** The mission expectation is roughly **16 events per + day**, which is why the products are 10 days and a month long rather than a + day. +* Every event produces **six waveforms**: ``TOF_High``, ``TOF_Mid``, + ``TOF_Low`` (three gain stages of the same TOF signal, 260 MHz), and + ``Target_High``, ``Target_Low``, ``Ion_Grid`` (charge-sensitive amplifiers, + 4.0625 MHz). +* One event is **many CCSDS packets**: one FPGA metadata header packet followed + by waveform fragments, routed by ``IDX__SCI0TYPE`` and ordered by + ``IDX__SCI0FRAGOFF``. Reassembling them is the single most fiddly part of + IDEX L1A. +* Processing chain in this repository:: + + CCSDS .pkts + -> L1A event-level, raw DN waveforms + metadata + event flags (10 days) + -> L1B engineering units, dead time, trigger decode, SPICE geometry (10 days) + -> L2A waveform fits, impact charge, velocity, mass, TOF mass scale (10 days) + -> L2B daily counts and uptime-corrected rates vs spin phase (1 month) + -> L2C the same counts and rates as rectangular sky maps (1 month) + + There is also a parallel **event-message** chain + (``l1a_msg-10days -> l1b_msg-10days``) that turns instrument log entries into + ``science_on`` / ``pulser_on`` state, and a **catalog-list** (``catlst``) + passthrough product. +* **A large fraction of the L2A and L2B product is deliberately NaN or + fill-valued.** The variables exist so the CDF schema is stable; the science + team has not validated the TOF mass spectrum or the derived mass, so those + values are overwritten just before the dataset is returned. This is on + purpose, it is tested, and it is not a bug. See :ref:`idex-l2`. +* **This repository goes all the way to the map product.** Unlike most IMAP + instruments, IDEX has no separate L3 repository to hand off to. + See :ref:`idex-l3-scope`. + +Where the code lives +-------------------- + +.. code-block:: text + + imap_processing/idex/ + idex_constants.py IDEXAPID, ConversionFactors, DT_BLOCK, + IDEX_10_DAY_RANGES_PATH, SPICE_ARRAYS, + ION_GRID_VELOCITY_*, IDEX_SPACING_DEG, + IDEX_EVENT_REFERENCE_FRAME + idex_utils.py get_idex_attrs, setup_dataset, + get_10_day_window_end_date + idex_l0.py decom_packets() - science vs housekeeping split + idex_l1a.py idex_l1a(), PacketParser, RawDustEvent, Scitype + decode.py rice_decode() - Rice-Golomb decompression + evt_msg_decode_utils.py render_event_template() + idex_event_flags.py classify_event_flags(), saturation flags + idex_l1b.py idex_l1b_science(), idex_l1b_msg(), TriggerMode, + TriggerOrigin, EventMessage, get_spice_data() + idex_l2a.py idex_l2a(), estimate_dust_mass(), fit_impact(), + time_to_mass(), analyze_peaks(), calibration curves + idex_l2b.py idex_l2b() - produces BOTH L2B and L2C + + packet_definitions/ + idex_science_packet_definition.xml APID 1424 + idex_housekeeping_packet_definition.xml APIDs incl. 1418 (EVT), 1419 (CATLST) + + idex_10_day_CDF_names.csv 444 product windows, 2025-2036 + idex_variable_unpacking_and_eu_conversion.csv 31 instrument settings + idex_evt_msg_parsing_dictionaries.json event-message templates + atomic_masses.csv 21 reference ion masses + + imap_processing/cdf/config/imap_idex_global_cdf_attrs.yaml + imap_processing/cdf/config/imap_idex_l1a_variable_attrs.yaml + imap_processing/cdf/config/imap_idex_l1b_variable_attrs.yaml + imap_processing/cdf/config/imap_idex_l2a_variable_attrs.yaml + imap_processing/cdf/config/imap_idex_l2b_variable_attrs.yaml + imap_processing/cdf/config/imap_idex_l2c_variable_attrs.yaml + imap_processing/cli.py (class Idex) dependency wiring per level + imap_processing/tests/idex/ tests, L0 test data, calibration CSVs \ No newline at end of file diff --git a/docs/source/algorithm-code-documentation/idex/l1.rst b/docs/source/algorithm-code-documentation/idex/l1.rst new file mode 100644 index 0000000000..a0fa47fdeb --- /dev/null +++ b/docs/source/algorithm-code-documentation/idex/l1.rst @@ -0,0 +1,704 @@ +.. _idex-l1: + +L0, L1A and L1B Processing +========================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This page covers packet decommutation through the L1B products: how six +waveforms and a pile of metadata are recovered from a stream of CCSDS packets, +and how those become engineering units with geometry attached. + +L0: decommutation only +---------------------- + +**[CODE]** ``imap_processing/idex/idex_l0.py``, 54 lines, one function. + +``decom_packets(packet_file)`` returns a **three-tuple**: + +1. ``list`` of decommutated science packets, +2. ``dict[apid, Dataset]`` of **raw** housekeeping datasets, +3. ``dict[apid, Dataset]`` of **derived** housekeeping datasets. + +Science and housekeeping are decommutated separately and differently: + +.. code-block:: python + + science_decom_packet_list = list(packet_generator(packet_file, science_xtce_file)) + raw_datasets_by_apid = packet_file_to_datasets(packet_file, hk_xtce_file, + use_derived_value=False) + derived_datasets_by_apid = packet_file_to_datasets(packet_file, hk_xtce_file, + use_derived_value=True) + +.. important:: + + **IDEX cannot use ``packet_file_to_datasets()`` for science packets.** That + helper requires every packet of a given APID to have the same field set, and + the IDEX science XTCE has *branching logic* - a header packet and a waveform + packet are both APID 1424 but carry different fields. So science packets come + back as a flat list of ``SpacePacket`` objects and all the structuring happens + in L1A. ``imap_processing/utils.py`` line ~255 names IDEX as the motivating + counter-example in its own docstring. + +There are no separate L0 files per content type. One +``imap_idex_l0_raw_YYYYMMDD_vNNN.pkts`` file may contain science, event-message +and catalog-list packets; the split happens here, by APID and XTCE. + +Relevant APIDs (``idex_constants.IDEXAPID``): + +.. list-table:: + :header-rows: 1 + :widths: 14 16 70 + + * - APID + - Name + - Content + * - 1424 + - ``IDEX_SCIENCE`` + - Dust-event header and waveform packets. + * - 1418 + - ``IDEX_EVT`` + - Event-message (instrument log) packets. + * - 1419 + - ``IDEX_CATLST`` + - Packet catalog summary. + +L1A science: assembling an event +-------------------------------- + +**[CODE]** ``imap_processing/idex/idex_l1a.py``. Two classes do the work: +``PacketParser`` (per L0 file) and ``RawDustEvent`` (per dust impact). + +Packet classification +^^^^^^^^^^^^^^^^^^^^^ + +Science packets are routed by ``IDX__SCI0TYPE``. ``class Scitype(IntEnum)``: + +.. list-table:: + :header-rows: 1 + :widths: 12 26 62 + + * - Value + - Name + - Role + * - 1 + - ``FIRST_PACKET`` + - FPGA metadata header. Starts a new event. + * - 2 + - ``TOF_HIGH`` + - TOF high-gain waveform fragment + * - 4 + - ``TOF_LOW`` + - TOF low-gain waveform fragment + * - 8 + - ``TOF_MID`` + - TOF mid-gain waveform fragment + * - 16 + - ``TARGET_LOW`` + - Target low-gain waveform fragment + * - 32 + - ``TARGET_HIGH`` + - Target high-gain waveform fragment + * - 64 + - ``ION_GRID`` + - Ion-grid waveform fragment + +Note that the values are a one-hot bit pattern but are compared for equality, +not masked. Note also the ordering trap: ``TOF_LOW`` is 4 and ``TOF_MID`` is 8, +so the numeric order is High, Low, Mid - not High, Mid, Low. + +Event identity: two different event numbers +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +This is subtle and worth reading twice. **[CODE]** ``_create_science_dataset`` +maintains two dictionaries: + +* ``dust_events: dict[_EventKey, RawDustEvent]`` - the events themselves, keyed + by a **stable** key. +* ``active_event_keys: dict[int, _EventKey]`` - a mapping from the wire-level + event number to the stable key, used to route incoming fragments. + +The stable key is a ``NamedTuple``: + +.. code-block:: python + + _EventKey(coarse_upper = packet["IDX__TXHDRTIMESEC1"], + coarse_lower = packet["IDX__TXHDRTIMESEC2"], + fine_subseconds= packet["IDX__TXHDRTIMESUBS"], + event_number = packet["IDX__TXHDREVTNUM"]) + +The routing key is ``packet["IDX__SCI0EVTNUM"]``, a 16-bit field present on +*every* science packet. So the key is built from ``TXHDREVTNUM`` (the FPGA's +event number, in the metadata header) plus the full event timestamp, while +routing uses ``SCI0EVTNUM`` (the transport-level event number). **[DOC]** +Section 4.6.3 says the same thing: "The stable event key is built from the +transmit timestamp fields and event number, rather than from the event number +alone." + +The reason is that the wire-level event number wraps and is reused. Keying +events on it alone would merge two genuinely different impacts. Keying on the +timestamp as well makes the key effectively unique. + +Failure modes here: + +* A waveform packet whose ``SCI0EVTNUM`` has no active header raises + ``KeyError("Have not receive header information from event number ...")`` and + **aborts the whole file**. This is the one place IDEX L1A is not defensive. +* A duplicate header packet for an already-seen ``_EventKey`` logs a warning, + re-points ``active_event_keys``, and continues. +* Any packet without an ``IDX__SCI0TYPE`` field logs ``"Unhandled packet + received"`` and is dropped. + +Event time reconstruction +^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** section 4.6.4 = **[CODE]** ``calculate_idex_event_time()``: + +.. math:: + + t_{MET} = \left(2^{16}\,\mathtt{IDX\_\_TXHDRTIMESEC1} + + \mathtt{IDX\_\_TXHDRTIMESEC2}\right) + + 20\times10^{-6}\,\mathtt{IDX\_\_TXHDRTIMESUBS} + +then ``met_to_ttj2000ns(t_MET)`` gives the CDF ``epoch``. The 32-bit seconds +counter is split across two 16-bit telemetry words, so the upper word must be +shifted left by 16. MET counts seconds from 2010-01-01. Subsecond resolution is +**20 µs**, which is coarse compared to the 3.85 ns waveform sampling - the +waveform time axis, not ``epoch``, is what carries fine timing within an event. + +The same function is reused for event messages (``elsec_evtpkt`` / +``elssec_evtpkt``) and for catalog-list packets (``shcoarse`` / ``shfine``), +because all three use the same seconds-plus-20 µs encoding. + +Waveform time axes +^^^^^^^^^^^^^^^^^^ + +**[CODE]** ``RawDustEvent._set_sample_trigger_times()``. This turns the header's +pre-trigger *block* counts into a time offset in microseconds. + +Two packed header fields are unpacked: + +``IDX__TXHDRBLOCKS`` (32-bit) + * bits 6-11: number of **low**-rate pre-trigger blocks, + ``(n_blocks >> 6) & 0b111111`` + * bits 16-19: number of **high**-rate pre-trigger blocks, + ``(n_blocks >> 16) & 0b1111`` + * (bits 20-23 and 24-29 are the dead-time fields, unpacked later at L1B) + +``IDX__TXHDRSAMPDELAY`` (32-bit) + * bits 0-9: high-gain TOF delay + * bits 10-19: mid-gain TOF delay + * bits 20-29: low-gain TOF delay + +Which delay applies is chosen from the low 10 bits of ``IDX__TXHDRTRIGID``, +checked in the order **bit 0 → HG, bit 1 → LG, bit 2 → MG**, defaulting to HG. + +.. math:: + + t_{LS,\mathrm{trig}} &= \Delta t_{LS}\,(N_{pre,LS}+1)\cdot 8 \\ + t_{HS,\mathrm{trig}} &= \Delta t_{HS}\,(N_{pre,HS}+1)\cdot 512 + - \Delta t_{HS}\,(N_{delay}-1) + +and the axes themselves: + +.. math:: + + t_{LS}[i] = i\,\Delta t_{LS} - t_{LS,\mathrm{trig}}, + \qquad + t_{HS}[i] = i\,\Delta t_{HS} - t_{HS,\mathrm{trig}} + +The ``+1`` on the block count and the ``−1`` on the delay are the FPGA's +off-by-one conventions, not arithmetic errors. **[DOC]** section 4.6.5 gives the +same equations. + +Fragment assembly and duplicate handling +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** ``_populate_bit_strings()``. Fragments are stored in +``fragments_by_scitype: dict[Scitype, dict[int, bytes]]``, keyed by +``IDX__SCI0FRAGOFF``, rather than being concatenated on arrival. That choice is +what makes retransmission handling possible. The resolution rules, in order: + +.. list-table:: + :header-rows: 1 + :widths: 40 60 + + * - Situation + - Action + * - Slot empty + - Store the fragment. + * - New bytes identical to stored bytes + - Warn, discard the duplicate. + * - New fragment **longer** than stored + - Warn, **replace** - a longer retransmit is assumed more complete. + * - New fragment **shorter** than stored + - Warn, discard the shorter copy. + * - Same length, different bytes + - Warn, record the slot in ``conflicting_fragment_slots``. + +Any event with a non-empty ``conflicting_fragment_slots`` set is **skipped +entirely** by ``process()`` - it returns ``None`` and the event never reaches the +CDF. This is intentional: a silently corrupt waveform is worse than a missing +event. + +``_assemble_bits(scitype)`` then joins the stored fragments in ascending +fragment-offset order into a single binary string. + +Waveform decoding +^^^^^^^^^^^^^^^^^ + +Two paths, chosen by the per-event ``IDX__SCI0COMP`` flag (stored as +``self.compressed``, tested as ``self.compressed.raw_value == 1``). + +**Uncompressed** - ``_read_waveform_bits()`` slices fixed bit patterns out of +32-bit words: + +.. list-table:: + :header-rows: 1 + :widths: 18 34 48 + + * - Rate + - 32-bit word layout + - Post-processing + * - High + - 2 pad + 10 + 10 + 10 + - **The last 4 samples are dropped** - the code comment says "the very last + four numbers are usually bad." + * - Low + - 8 pad + 12 + 12 + - none + +**Compressed** - ``decode.rice_decode()`` (241 lines, originally by Corinne +Wuerthner). Rice-Golomb with per-subframe predictor selection: + +.. list-table:: + :header-rows: 1 + :widths: 12 88 + + * - Bits + - Predictor + * - ``00`` + - Constant. Every sample equals the first; only the first is stored. + * - ``01`` + - Verbatim. Store every sample at full NBIT width - the escape hatch for + uncorrelated data where Rice coding would expand rather than compress. + * - ``10`` + - Linear #1: :math:`X(n) = X(n-1)`. Needs one uncompressed warm-up sample. + * - ``11`` + - Linear #2: :math:`X(n) = 2X(n-1) - X(n-2)`. + +A compressed frame starts on a byte boundary with the marker ``0xF5``. +Subframes are 64 samples (``SUB_FRAME_SIZE``) and are **not** aligned to any +byte or word boundary. The Rice parameter ``k`` is ``ceil(log2(NBITS))`` bits +wide - 4 bits for 10-bit data - and is omitted entirely for the ``00`` and +``01`` predictors. + +The compressed path requests ``sample_count = MAX_BLOCKS * SAMPLES_PER_BLOCK`` +(8192 high, 512 low) and then trims the **last 3** high-rate samples. + +.. note:: + + The two paths trim different amounts (4 uncompressed, 3 compressed) and must + still produce equal-length arrays, because ``xr.concat`` over ``epoch`` in + ``_create_science_dataset`` would otherwise broadcast or fail when a 10-day + window contains both compressed and uncompressed events. In practice both + land on **8189** high-rate samples, which is the length assumed in + ``idex_l2a.time_to_mass``'s docstring. **Nothing asserts this.** If you touch + either decoder, check the other. + +Metadata preservation and the AID substitution +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** Every header telemetry item past the first 7 (the CCSDS header +fields) is carried forward to the L1A dataset as a lowercased, event-indexed +variable - ``idx__txhdrblocks``, ``idx__txhdrtrigid``, ``idx__txhdrhgtrigmode`` +and so on. Two exceptions: + +* ``idx__sci0aid`` is **skipped**. **[DOC]** section 4.6.7: "SCI0AID is skipped + because it is not updated properly by the flight software." +* ``idx__txhdrfswaidcopy`` is **renamed to** ``aid`` and is the acquisition + identifier the products use. + +Incomplete event filtering +^^^^^^^^^^^^^^^^^^^^^^^^^^ + +After decoding, ``process()`` checks each waveform's length against the length +of its time coordinate. Any mismatch means a dropped packet somewhere; the event +is logged (``"Missing packet for event number %s. Skipping event.."``) and +returns ``None``. **[DOC]** Section 4.6.8 records that this behavior was +requested by the IDEX team: warn, but still produce the CDF from the events that +are complete. + +.. warning:: + + If **every** event in a file is skipped, ``xr.concat`` is called on an empty + list and raises. There is no guard for the all-events-bad case. + +Finally the dataset is sorted by ``epoch`` and the index and label coordinates +(``time_{low,high}_sample_rate_index``, ``..._label``) are attached. + +Event classification at L1A +^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** Before returning, ``process()`` calls ``classify_event_flags()`` and +attaches ten ``uint8`` flags to the event. **This has no counterpart in the +algorithm document.** It is covered on its own page: +:ref:`idex-event-classification`. + +L1A event messages +------------------ + +**[CODE]** ``PacketParser._create_evt_msg_data()``, from the **raw** APID 1418 +dataset. + +The epoch comes from ``elsec_evtpkt + 20e-6 * elssec_evtpkt``. Only three +time-related variables are carried into the output (``epoch``, +``elsec_evtpkt``, ``elssec_evtpkt``); the event identifier and parameter bytes +are consumed to build the message string and are **not** preserved as separate +variables. + +Message rendering uses ``idex_evt_msg_parsing_dictionaries.json``, which +contains at least ``eventMsgDictionary`` (id → template) and +``logEntryIdDictionary`` (id → short name), plus value-lookup dictionaries. JSON +stringifies object keys, so the code converts every key back to ``int`` before +use. + +Each message has an id (``elid_evtpkt``) and four parameter bytes +(``el1par_evtpkt`` … ``el4par_evtpkt``) stacked into an ``(n_events, 4)`` +array. ``evt_msg_decode_utils.render_event_template()`` substitutes them into +the template. The placeholder grammar is:: + + {p} one parameter byte, rendered as hex + {p+} n bytes combined big-endian, rendered as hex + {p|} looked up in dictName, rendered as dictName(value) + {p+|} both + +Unresolved lookups fall back to the raw hex value; an event id with no template +falls back to ``" (0xAA, 0xBB, 0xCC, 0xDD)"``; a template that +raises falls back to a string containing the exception. The pipeline never drops +a message for being unrenderable. + +.. important:: + + **The rendered strings keep the dictionary-name wrapper.** A lookup for + dictionary ``sciState16Dictionary`` renders as + ``sciState16Dictionary(ACQ)``, not ``ACQ``. This is why the constants in + ``idex_l1b.EventMessage`` look the way they do, and why they do **not** + match the strings quoted in the algorithm document (section 4.7.2). Any + change to ``render_event_template``'s output format silently breaks the L1B + state reduction, which compares whole strings for equality. + +L1B: engineering units and geometry +----------------------------------- + +**[CODE]** ``imap_processing/idex/idex_l1b.py``. ``idex_l1b(l1a_dataset, +descriptor)`` dispatches on the descriptor prefix: ``sci-10days`` → science +path, ``msg-10days`` → message path, anything else → ``ValueError``. + +Waveform conversion +^^^^^^^^^^^^^^^^^^^ + +A single multiply per channel, :math:`Q = \mathrm{DN} \times C`, using +``idex_constants.ConversionFactors``. + +.. important:: + + **The code and the algorithm document disagree here, on both the numbers and + the units, for the three TOF channels.** + +.. list-table:: + :header-rows: 1 + :widths: 16 12 20 20 32 + + * - Channel + - Rate (MHz) + - **[CODE]** factor + - **[CODE]** unit + - **[DOC]** Table 4.2 factor (pC/DN) + * - ``TOF_High`` + - 260 + - 7.50e-5 + - mA + - 2.89e-4 + * - ``TOF_Mid`` + - 260 + - 2.93e-3 + - mA + - 1.13e-2 + * - ``TOF_Low`` + - 260 + - 1.34e-1 + - mA + - 5.14e-1 + * - ``Target_High`` + - 4.0625 + - 1.63e-1 + - pC + - 1.63e-1 + * - ``Target_Low`` + - 4.0625 + - 1.58e+1 + - pC + - 1.58e+1 + * - ``Ion_Grid`` + - 4.0625 + - 7.46e-4 + - pC + - 7.46e-4 + +The three low-rate channels agree exactly. The three TOF channels were +**reinterpreted as a current, not a charge**, and rescaled by a factor of about +0.26 in the process. The code is the newer of the two: the ``UNITS`` in +``imap_idex_l1b_variable_attrs.yaml`` say ``mA`` for the TOF channels and ``pC`` +for the low-rate channels, and ``test_idex_l1b.py`` explicitly rescales the IDEX +team's own validation file, whose TOF arrays "use the legacy pC factors", before +comparing. Do not revert the code to Table 4.2. + +Instrument setting unpacking and EU conversion +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Two mechanisms, in sequence, driven by +``idex_variable_unpacking_and_eu_conversion.csv`` (31 settings). + +**Step 1 - bit unpacking** (``unpack_instrument_settings``). Many settings are +packed two-to-a-word into ADC readout telemetry (``idx__txhdrprochkch01`` +carries both ``current_1v_pol`` and ``current_1p9v_pol``, and so on). For each +row: + +.. math:: + + M = 2^{N_{bits}} - 1, + \qquad + x_{unpacked} = \left(x_{raw} \gg (b_{start} - b_{pad})\right)\ \&\ M + +using the CSV's ``unsigned_nbits``, ``starting_bit`` and +``nbits_padding_before``. Rows are de-duplicated on ``mnemonic`` first, because +segmented-polynomial conversions occupy several rows per setting and each +setting must be unpacked exactly once. + +**Step 2 - polynomial conversion** (``convert_raw_to_eu`` from +``imap_processing/utils.py``, the shared helper). Same CSV, now read for the +``c0``-``c7`` coefficients: + +.. math:: + + P(x) = \sum_{i=0}^{7} C_i x^i + +Both ``UNSEGMENTED_POLY`` and segmented conversions (different coefficient sets +over different raw-DN ranges) are supported. + +The 31 settings are three groups: processor housekeeping (bus currents, board +temperatures, FPGA temperature), HVPS housekeeping (detector, sensor, target, +rejection and reflectron voltages and HVPS currents), and low-voltage +housekeeping (±5 V, ±6 V, 16 V, 3.3 V, 2.5 V bus voltages and currents). See +:ref:`idex-ancillary`. + +Dead time +^^^^^^^^^ + +**[CODE]** ``get_event_dead_time()``, from the same packed +``idx__txhdrblocks`` word L1A used for pre-trigger blocks: + +.. math:: + + N_{shift} = (\mathtt{blocks} \gg 20)\ \&\ \mathtt{0b1111}, + \qquad + N_{base} = (\mathtt{blocks} \gg 24)\ \&\ \mathtt{0b111111} + +.. math:: + + t_{dead} = N_{base}\, 2^{N_{shift}}\, \Delta t_{block}, + \qquad + \Delta t_{block} = \frac{8}{4.0625\times 10^{6}} \approx 1.96923\ \mu\mathrm{s} + +Stored as the event-indexed variable ``dead_time`` in seconds. **[DOC]** section +4.7.1 gives the identical formula. + +Trigger mode and threshold level +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** ``get_trigger_mode_and_level()``, run for each of the three TOF gain +channels (``lg``, ``mg``, ``hg``), producing six variables: +``trigger_mode_{lg,mg,hg}`` and ``trigger_level_{lg,mg,hg}``. + +For each channel, ``idx__txhdr{ch}trigmode``: + +* ``0`` → that channel did not arm a trigger for this event; both outputs are + ``None`` / ``NaN``. +* ``1`` → ``Threshold``, ``2`` → ``SinglePulse``, ``3`` → ``DoublePulse``. + The label written out is the channel prefix plus the mode name, e.g. + ``"HGThreshold"``. + +The threshold level comes from the **upper 10 bits** of +``idx__txhdr{ch}trigctrl1``: + +.. math:: + + L_{trig} = (\mathtt{trigctrl1} \gg 22)\ \&\ \mathtt{0b1111111111} + +and is then multiplied by that channel's TOF conversion factor, so it is +reported in the same units as the converted waveform (**mA**, per the note +above). + +.. note:: + + ``compute_trigger_values`` is applied through ``xr.apply_ufunc(..., + vectorize=True)`` and the mode array is then explicitly rebuilt as an + ``object`` array with ``pd.isna`` values forced back to ``None``. That dance + exists because pandas 3 string inference converts the no-trigger ``None`` + values to ``NaN``, which is not writable as a CDF string. Don't simplify it. + +Trigger origin +^^^^^^^^^^^^^^ + +**[CODE]** ``get_trigger_origin()``, from the **lower 10 bits** of +``idx__txhdrtrigid``. Bits 0-5 are checked: + +.. list-table:: + :header-rows: 1 + :widths: 8 30 62 + + * - Bit + - ``TriggerOrigin`` + - Label + * - 0 + - ``HS_ADC0I_TOF_HG`` + - ``HS ADC0I trigger (TOF HG)`` + * - 1 + - ``HS_ADC0Q_TOF_LG`` + - ``HS ADC0Q trigger (TOF LG)`` + * - 2 + - ``HS_ADC1Q_TOF_MG`` + - ``HS ADC1Q trigger (TOF MG)`` + * - 3 + - ``LS_ADC1_TARGET_HG`` + - ``LS ADC1 trigger (Target HG / low range)`` + * - 4 + - ``SW_TRIGGER`` + - ``SW trigger`` + * - 5 + - ``EXTERNAL_TRIGGER`` + - ``external trigger`` + +**Multiple bits can be set.** The labels are joined with ``", "`` into a single +string variable ``trigger_origin``. No bits set yields +``"Unknown trigger origin"``. + +.. note:: + + The algorithm document writes the bit-4 label as "software trigger" and + bit-5 as "external trigger"; the code writes ``"SW trigger"`` and + ``"external trigger"``. The strings are compared nowhere, so this is + cosmetic - but if you ever filter on them, use the code's spelling. + + Note also that the same ``idx__txhdrtrigid`` field is read three times for + three different purposes: to select the TOF sample delay (L1A, bits 0-2 in a + priority order), to label the trigger origin (L1B, bits 0-5 as a set), and + to classify the event type (L1A flags, bits 0-3 plus 4-5). The three readings + are consistent but independent. + +SPICE geometry, pointing and spin phase +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** ``get_spice_data()``. Ten variables, all event-indexed, all from the +shared ``imap_processing/spice/`` helpers: + +.. code-block:: python + + et = ttj2000ns_to_et(l1a_dataset["epoch"].data) + met = et_to_met(et) + + spin_phase = get_spin_angle(get_spacecraft_spin_phase(query_met_times=met), + degrees=True) + ephemeris = imap_state(et, observer=SpiceBody.SUN) # ECLIPJ2000 + idex_pointing= instrument_pointing(et, SpiceFrame.IMAP_IDEX, + IDEX_EVENT_REFERENCE_FRAME, cartesian=True) + solar_lon = solar_longitude(et, degrees=True) + lon, lat = cartesian_to_spherical(idex_pointing)[:, 1:].T + +.. list-table:: + :header-rows: 1 + :widths: 34 66 + + * - Variable(s) + - Meaning + * - ``ephemeris_position_{x,y,z}`` + - IMAP position relative to the **Sun**, ECLIPJ2000. + * - ``ephemeris_velocity_{x,y,z}`` + - IMAP velocity relative to the Sun, ECLIPJ2000. + * - ``longitude``, ``latitude`` + - The **IDEX boresight** direction at the event epoch, transformed from + the ``IMAP_IDEX`` frame into ``IDEX_EVENT_REFERENCE_FRAME`` + (= ``SpiceFrame.ECLIPJ2000``) and converted to spherical. These are the + two coordinates L2C bins its sky maps on - they are a *pointing* + direction, not a dust arrival direction. + * - ``solar_longitude`` + - Solar longitude at the event epoch, degrees. + * - ``spin_phase`` + - Spacecraft spin phase at the event MET, converted from fractional phase + to **degrees**. Binned into quadrants by L2B. + +The IDEX frame is ``SpiceFrame.IMAP_IDEX`` (NAIF id ``-43700``), with boresight +``[0, 1, 0]`` in ``imap_processing/spice/geometry.py`` and a spin-phase offset +of ``179.9229 / 360``. + +**[DOC]** Section 4.7.1 stresses that these are attached for event context and +downstream interpretation only - they do not alter any L1B waveform +conversion. **[CODE]** The code comment says the same: "Spice data is not used +for calculations yet but are saved in the CDF for reference." That stops being +true at L2B/L2C, which bin on ``spin_phase``, ``longitude`` and ``latitude``. + +L1B event messages +------------------ + +**[CODE]** ``idex_l1b_msg()``. Reduces the L1A message log to two state +variables by **exact string comparison** against ``class EventMessage``: + +.. list-table:: + :header-rows: 1 + :widths: 22 78 + + * - Constant + - String + * - ``SCIENCE_ON`` + - ``SCI state change: sciState16Dictionary(ACQSETUP) ==> sciState16Dictionary(ACQ)`` + * - ``SCIENCE_OFF`` + - ``SCI state change: sciState16Dictionary(ACQ) ==> sciState16Dictionary(CHILL)`` + * - ``PULSER_ON`` + - ``SEQ success (len=0x0580, opCodeLCDictionary(enstim))`` + * - ``PULSER_OFF`` + - ``UPK stim pulser operation completed, , PulserSel=0x00000007`` + +``science_on`` is a straightforward 1 / 0 / 255 mapping on those two strings. + +``pulser_on`` is **not**. The code only marks a pulser transition when a +``PULSER_ON`` message is **immediately followed** by a ``PULSER_OFF`` message: + +.. code-block:: python + + consecutive = np.where((messages[:-1] == PULSER_ON) & (messages[1:] == PULSER_OFF))[0] + pulser_on = np.full(len(messages), 255) + pulser_on[consecutive] = 1 + pulser_on[consecutive + 1] = 0 + +An unpaired ``PULSER_ON``, or a ``PULSER_OFF`` whose predecessor was something +else, leaves the state at 255 (unchanged). **[DOC]** Section 4.7.2 describes the +simple unconditional exact-match behavior instead, so the document is out of +date on this point. The code's own comment explains the intent: "susprel but +previous message was NOT enstim → pulser_on stays whatever it was." + +Finally, rows where **both** flags are 255 are dropped. The L1B message product +is therefore a **state-change log, not a copy of the L1A event log**. If no +transitions are present, ``idex_l1b_msg()`` returns ``None`` and no file is +written - which means L2B can legitimately find no uptime information for a +period. + +.. warning:: + + All four of these are literal, whole-string, exact comparisons against + strings produced by ``render_event_template()``. Any change to the + event-message JSON dictionaries, to the template grammar, or to the + ``dictName(value)`` wrapper format will silently turn every message into + 255, empty the L1B message product, and make every L2B rate fall back to + ``-1`` / NaN with ``rate_calculation_quality_flags = 0``. This is the most + brittle coupling in the IDEX pipeline. diff --git a/docs/source/algorithm-code-documentation/idex/l2.rst b/docs/source/algorithm-code-documentation/idex/l2.rst new file mode 100644 index 0000000000..5005e0aef8 --- /dev/null +++ b/docs/source/algorithm-code-documentation/idex/l2.rst @@ -0,0 +1,575 @@ +.. _idex-l2: + +L2A, L2B and L2C Processing +=========================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +L2A turns six waveforms into physical quantities per dust impact. L2B and L2C +turn a month of those impacts into count rates and sky maps. + +.. important:: + + **A large part of what L2A and L2B compute is deliberately overwritten before + the dataset is returned.** The TOF mass spectrum and the mass-binned rate + products are computed, tested, and then filled with NaN (L2A) or integer fill + (L2B/L2C), because the science team has not validated them. The variables + remain in the product so the CDF schema is stable across releases. + + This is intentional, it is asserted by the test suite, and it is documented + in **[DOC]** sections 4.7.7 and 4.7.11. Do not "fix" it by deleting the NaN + block. The withheld variables are listed in full below. + +L2A: per-event science +---------------------- + +**[CODE]** ``imap_processing/idex/idex_l2a.py``, ~1330 lines. +``idex_l2a(l1b_dataset, ancillary_files)``. Input is one L1B 10-day science +file plus two calibration CSVs; output is one L2A 10-day file with the same +``epoch`` records. + +Calibration inputs +^^^^^^^^^^^^^^^^^^ + +``load_calibration_files()`` reads two single-row CSVs, skipping the header row +and taking the **first 8 values** of each: + +.. list-table:: + :header-rows: 1 + :widths: 40 60 + + * - Ancillary key + - Purpose + * - ``l2a-calibration-curve-t-rise`` + - Relates target rise time to impact speed. Inverted to get velocity. + * - ``l2a-calibration-curve-yield-params`` + - Charge yield (C/kg) as a function of impact speed. Used to get mass. + +**[DOC]** Table 4.3 gives the current parameter values: + +.. list-table:: + :header-rows: 1 + :widths: 16 10 10 10 10 12 12 10 10 + + * - Calibration + - log A + - a1 + - a2 + - a3 + - vb (km/s) + - vc (km/s) + - k + - σ + * - Rise time + - 1.27 + - −0.2 + - −2.1 + - −0.37 + - 5.3 + - 13.3 + - 13.3 + - 0.28 + * - Charge yield + - 0.06 + - 2.8 + - 5.9 + - 4.1 + - 13.0 + - 22.7 + - 8.2 + - 0.40 + +A ninth value (Δ = 1.33 and 1.47 respectively) is present in each file as a +reported error factor and is **not read** - hence the ``[:8]`` slice. The 8th +value (σ, unpacked as ``_m`` in ``log_smooth_powerlaw``) is read but also unused +in the current formula. + +.. warning:: + + ``load_calibration_files`` uses ``.values.flatten()[:8]`` with no validation. + A calibration file with a different column order, an extra leading column, or + a different number of header rows will silently produce eight wrong numbers + rather than an error. + +The waveform fit +^^^^^^^^^^^^^^^^ + +**[CODE]** ``estimate_dust_mass()`` + ``fit_impact()``, applied independently to +``Target_Low``, ``Target_High`` and ``Ion_Grid`` against the low-rate time axis. +The model is the LDEX impact-charge pulse: + +.. math:: + + y(t) = C_0 + H(t - t_0)\,A + \left[1 - \exp\!\left(-\frac{t-t_0}{\tau_r}\right)\right] + \exp\!\left(-\frac{t-t_0}{\tau_d}\right) + +with :math:`H` the Heaviside step. The five fitted parameters, in the order they +appear in ``{channel}_fit_parameters`` and in +``target_fit_parameter_labels``: + +``time_of_impact`` (:math:`t_0`), ``constant_offset`` (:math:`C_0`), +``amplitude`` (:math:`A`), ``rise_time`` (:math:`\tau_r`), +``discharge_time`` (:math:`\tau_d`). + +Initial guesses and bounds: + +.. list-table:: + :header-rows: 1 + :widths: 24 36 40 + + * - Parameter + - Initial guess + - Bounds + * - :math:`t_0` + - 0 µs + - unbounded + * - :math:`C_0` + - mean of the first **5 µs** of the record (the pre-trigger baseline) + - unbounded + * - :math:`A` + - the largest absolute excursion from the baseline + - **strictly positive** for target channels; for ``Ion_Grid``, strictly + positive *or* strictly negative depending on the sign of the initial + guess, because the ion-grid signal polarity varies + * - :math:`\tau_r` + - 0.371 µs + - strictly positive + * - :math:`\tau_d` + - 37.1 µs + - strictly positive + +Fit via ``scipy.optimize.curve_fit`` with ``maxfev=100_000``. A ``RuntimeError`` +(fit did not converge) is caught: the function logs a warning and returns all +NaN for that event and channel rather than raising. + +Impact charge from the fit +^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** The reported charge is **not** the fitted amplitude parameter. The +code evaluates the analytic maximum of the fitted pulse, + +.. math:: + + t_{max} = t_0 + \tau_r \ln\!\left(\frac{\tau_d}{\tau_r} + 1\right) + +evaluates the model there, and takes + +.. math:: + + Q_{impact} = \left| y(t_{max}) - C_0 \right| + +The comment explains why: evaluating the analytic extremum avoids being limited +by the discrete 246 ns sample grid. ``chi_square()`` then returns the sum of +squared residuals and the reduced chi-square, following lmfit's +``_calculate_statistics()`` convention (``redchi = chisqr / max(1, ndata - +nparams)``). + +Per channel, L2A writes seven variables: +``{channel}_fit_parameters``, ``{channel}_impact_charge``, +``{channel}_velocity_estimate``, ``{channel}_dust_mass_estimate``, +``{channel}_chi_squared``, ``{channel}_reduced_chi_squared``, +``{channel}_fit_results`` (the modelled waveform). + +Velocity from rise time +^^^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** ``invert_rise_time_to_velocity()``. Both IDEX calibration curves +share one functional form, a **smoothly transitioning triple power law in log +space** (``log_smooth_powerlaw``): + +.. math:: + + \log_{10} y = \log_{10} A + a_1 \log_{10} v + + \log_{10}\left[ + \left(1 + \left(\tfrac{v}{v_b}\right)^{k}\right)^{(a_2-a_1)/k} + \left(1 + \left(\tfrac{v}{v_c}\right)^{k}\right)^{(a_3-a_2)/k} + \right] + +:math:`a_1, a_2, a_3` are the slopes of the low-, mid- and high-velocity +segments; :math:`v_b` and :math:`v_c` are the transition speeds; :math:`k` sets +the sharpness of both transitions. For the rise-time curve :math:`y = \tau_r` +in µs; for the yield curve :math:`y = Y(v)` in C/kg. + +Because the calibration is defined forward (:math:`\tau_r = f(v)`), velocity is +recovered by a scalar root solve in :math:`\log_{10} v` using Brent's method +over the bracket + +.. math:: + + -1 \le \log_{10} v \le 2 + \qquad\Longleftrightarrow\qquad + 0.1 \le v \le 100\ \mathrm{km\,s^{-1}} + +A non-finite or non-positive rise time, or a root that does not exist inside +that bracket, returns NaN (logged at ``ERROR`` level) and the mass estimate +becomes NaN too. + +Mass from charge and velocity +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** ``calculate_mass_from_velocity()``. Evaluate the yield curve at the +estimated velocity, convert the fitted charge to coulombs, divide: + +.. math:: + + m_{dust} = \frac{Q_{impact} \times 10^{-12}}{Y(v)} + +With :math:`Y` in C/kg and :math:`Q` in C, the result is in **kilograms**, which +is what ``UNITS: Kg`` in ``imap_idex_l2a_variable_attrs.yaml`` declares. Keep +that in mind when reading the L2B section below. + +Ion-grid velocity and mass +^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Section 4.7.9 lists this under "Future Work". **[CODE]** It is +implemented, in ``calculate_ion_grid_velocity_and_mass()``, using exactly the +relation the document specifies: + +.. math:: + + v(Q_I/Q_T) = c \left(\frac{Q_I}{Q_T}\right)^{p} + v_0, + \qquad c = 55\ \mathrm{km/s},\quad p = -3.2,\quad v_0 = 1.5\ \mathrm{km/s} + +The constants are ``ION_GRID_VELOCITY_SCALE``, ``ION_GRID_VELOCITY_EXPONENT`` +and ``ION_GRID_VELOCITY_OFFSET`` in ``idex_constants.py``. + +The target charge :math:`Q_T` is chosen by saturation state: **Target High if +unsaturated, else Target Low if unsaturated, else give up**. Both charges are +taken as absolute values. Zero or non-finite charges, or a saturated Ion Grid, +return ``(NaN, NaN)``. + +The resulting mass is computed from the **target** charge at the ion-grid +velocity, not from the ion-grid charge: + +.. code-block:: python + + mass_estimate = calculate_mass_from_velocity(target_charge, velocity_estimate, + yield_params) + +which matches the document's statement that ":math:`Q_I` and :math:`v_{IG}` will +then be fed into equations 4.27 and 4.30" only loosely - the document's wording +suggests :math:`Q_I`, the code uses :math:`Q_T`. Worth confirming with the IDEX +team; the charge-yield curve is calibrated against target charge, so the code's +choice is the physically defensible one. + +.. note:: + + The document also states that "for the current release, the ion-grid-derived + velocity and mass estimates are explicitly replaced with NaN". **That is no + longer true.** ``ion_grid_velocity_estimate`` and + ``ion_grid_dust_mass_estimate`` are **not** in the current NaN block; they + are masked only by the saturation and science-event rules described in + :ref:`idex-event-classification`. + +The TOF mass scale +^^^^^^^^^^^^^^^^^^ + +**[CODE]** ``time_to_mass()``. Uses ``TOF_High`` only. The goal is to find the +affine map from flight time to :math:`\sqrt{m}`: + +.. math:: + + t_i = t_{\text{offset}} + A\sqrt{m_i} + \qquad\Longrightarrow\qquad + m(t) = \left(\frac{t - t_{\text{shift}}}{A}\right)^{2} + +The algorithm is a brute-force cross-correlation search: + +1. Load 21 reference ion masses from ``atomic_masses.csv`` (H, H₂, C, O, Na, Mg + and Si isotopes, K, Ca, several Fe isotopes, Au). +2. Build a comb of 10 candidate stretch factors, ``np.linspace(1400, 1500, 10)`` + in nanoseconds. +3. For each stretch, build a synthetic spike train ``t_i``: zeros everywhere, + ``1`` at the sample index closest to each predicted peak time. +4. Cross-correlate that spike train with the measured TOF waveform + (``np.correlate(..., mode="full")``). The lag of the maximum is the best + ``t_shift`` for that stretch. +5. Pick the stretch with the highest correlation, convert shift to seconds via + ``FM_SAMPLING_RATE`` and stretch to seconds via ``NS_TO_S``, and apply the + squared map above to get a per-sample mass scale in amu. + +.. warning:: + + The stretch search window is hardcoded: ``min_stretch = 1400``, span 100 ns, + 10 steps - a 10 ns grid. There is no check that the best stretch landed + inside the window rather than on its edge, and no diagnostic is emitted if + it does. If the instrument's true stretch factor is outside 1400-1500 ns, + every mass scale will be wrong in the same direction with no warning. This + is one reason the mass scale is currently withheld. + +Peak detection and EMG fits +^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** Peaks are found with ``scipy.signal.find_peaks(tof, prominence=0.01)`` +on the converted TOF High waveform. Around each peak, ``analyze_peaks()`` takes +a ±5-sample slice and fits an **exponentially modified Gaussian** +(``scipy.stats.exponnorm.pdf``) parameterised by shape ``k``, location ``mu`` +and scale ``sigma``, with :math:`\lambda = 1/(k\sigma)`. The three reported +parameters, per ``peak_fit_parameter_labels``, are ``mu``, ``sigma``, +``lambda``. + +Results are written into a fixed **500-slot** array indexed by rounded integer +mass. If two peaks round to the same integer mass, the second is pushed to the +next free slot upward; anything that cannot find a slot below 500 is discarded +with a warning. ``calculate_area_under_emg()`` integrates the fitted PDF over +the slice with ``scipy.integrate.quad``. + +Two quality metrics accompany the peaks: + +``tof_snr`` (``calculate_snr``) + Baseline window **−7 µs ≤ t ≤ −5 µs** on the high-rate axis - pre-impact, by + construction. :math:`\mathrm{SNR} = (\max(\text{TOF}) - \overline{b}) / + \sigma_b`. Returns all-NaN with a warning if the window is empty. + +``tof_peak_kappa`` (``calculate_kappa``) + The mean signed distance between each detected peak's assigned mass and the + nearest integer. Near zero means the mass scale lines up with real atomic + masses; large means it does not. A cheap self-consistency check on + ``time_to_mass``. + +.. note:: + + The docstring says kappa "ranges between zero and one" and that "closer to + zero indicates better accuracy", but the implementation takes a **signed** + mean of ``mass - round(mass)``, which ranges over (−0.5, 0.5] and can + cancel: a spectrum whose peaks are equally split high and low of integer + would score a near-perfect kappa. An absolute value or RMS would match the + documented intent. + +The L2A NaN block +^^^^^^^^^^^^^^^^^ + +**[CODE]** Three separate masking stages run at the end of ``idex_l2a()``, in +this order: + +1. ``_mask_saturated_derived_estimates()`` - per-channel, per-event, driven by + the saturation flags. Raises ``KeyError`` if a saturation flag is missing. +2. ``_mask_non_science_derived_estimates()`` - per-event, driven by + ``science_event_flag``. Silently skips if the flag is absent. +3. The **unconditional NaN block** - these ten variables are overwritten with + NaN for every event, no exceptions: + + * ``tof_peak_area_under_fit`` + * ``tof_peak_chi_square`` + * ``tof_peak_fit_parameters`` + * ``tof_peak_kappa`` + * ``tof_peak_reduced_chi_square`` + * ``tof_snr`` + * ``mass`` + * ``mass_scale`` + + **[DOC]** Section 4.7.7 lists the same set plus + ``ion_grid_dust_mass_estimate`` and ``ion_grid_velocity_estimate``, which are + no longer in the code's block. + +What **is** published from L2A: the fit parameters, impact charges, fit results +and chi-squares for all three low-rate channels, and the velocity and mass +estimates for all three (subject to the saturation and science-event masks), +plus the L1B time axes and the ten SPICE context variables copied forward. + +Future work at L2A +^^^^^^^^^^^^^^^^^^ + +**[DOC]** Sections 4.7.8 and 4.7.10 describe two planned changes, neither +started in the code: + +* **Combined target fits.** Fit ``Target_High`` and ``Target_Low`` as one + waveform - high gain where it is unsaturated, low gain where it is not - after + baseline-correcting and converting both to common charge units. One best + estimate of charge, rise time and decay time per event instead of two, plus a + fit-quality flag. +* **Combined TOF fits.** The same idea across TOF High / Mid / Low, to get a + more robust time-to-mass conversion, better peak detection across saturated + and low-amplitude regions, peak areas, and quality metrics. + +Note that the saturation infrastructure both of these need **already exists** +(see :ref:`idex-event-classification`) - the per-channel saturation flags and +the gain-walking logic in ``_saturation_aware_width()`` are the same pattern. + +L2B and L2C: counts, rates and maps +----------------------------------- + +**[CODE]** ``imap_processing/idex/idex_l2b.py``, ~840 lines. +``idex_l2b(l2a_datasets, msg_data_l1b) -> [l2b_dataset, l2c_dataset]``. + +Inputs are **three** 10-day L2A science files and one or more L1B message files. +Both input lists are concatenated along ``epoch`` and sorted; the message +datasets are additionally de-duplicated. + +Daily binning +^^^^^^^^^^^^^ + +Everything is per **day of year**, derived with ``epoch_to_doy()``. Unique DOYs +are extracted with ``dict.fromkeys`` to preserve encounter order across a New +Year boundary. Each day's ``epoch`` value in the output is the **mean epoch of +that day's events**, not midnight. + +Only events with ``dust_hit_flag == 1`` are counted +(``_get_dust_hit_indices()``). + +The bin edges +^^^^^^^^^^^^^ + +**[CODE]** Hardcoded at module level in ``idex_l2b.py``: + +.. list-table:: + :header-rows: 1 + :widths: 22 78 + + * - Dimension + - Edges + * - ``MASS_BIN_EDGES`` + - 10 log-spaced bins from ``6.31e-17`` to ``1.00e-14`` **kg**. + * - ``CHARGE_BIN_EDGES`` + - 10 log-spaced bins from ``1.00e-01`` to ``1.00e+04`` **pC**. + * - ``SPIN_PHASE_BIN_EDGES`` + - ``[0, 90, 180, 270, 360]`` - four quadrants. + * - Sky grid + - ``AzElSkyGrid(IDEX_SPACING_DEG)`` with ``IDEX_SPACING_DEG = 6``, i.e. a + 60 × 30 rectangular grid, from the shared + ``imap_processing/ena_maps/`` machinery. + +Reported bin centres are the **geometric** means of the log-spaced edges +(``sqrt(edge[:-1] * edge[1:])``) and the arithmetic midpoints for spin phase. + +``bin_spin_phases()`` shifts by +45° before digitising, so the four quadrants +are centred on 0/90/180/270 rather than starting there: **315-45, 45-135, +135-225, 225-315**. Values outside [0, 360) log a warning. + +Which L2A variables feed the bins +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** Only the **Target Low** channel: + +.. code-block:: python + + mass_vals = l2a_dataset["target_low_dust_mass_estimate"] + charge_vals = l2a_dataset["target_low_impact_charge"] + +Target High and Ion Grid are not used at L2B at all. Values are clipped to the +first and last bin edge before histogramming, so out-of-range events land in the +end bins rather than being dropped. + +.. warning:: + + **Suspected unit bug.** L2B applies ``mass_vals = FG_TO_KG * mass_vals`` + (femtograms → kg, i.e. ×1e-15) to a variable that L2A produces in + **kilograms** and declares as ``UNITS: Kg``. Against mass bin edges of + 6.31e-17 to 1.00e-14 kg, this pushes every physically plausible mass far + below the lowest edge, where the clip sends it into the bottom bin. + + The effect is currently invisible because the mass-binned outputs are + fill-valued (below), which is probably why it has survived. It must be + resolved before those products are released - either L2A should report + femtograms, or L2B should drop the conversion. See + :ref:`idex-implementation-status`. + +Count rates +^^^^^^^^^^^ + +**[CODE]** Rates are corrected for **science-acquisition uptime only**, derived +entirely from the L1B message product's ``science_on`` variable: + +.. math:: + + \mathrm{rate} = \frac{\mathrm{counts}} + {0.01 \times P_{on}(\mathrm{doy}) \times 86400} + +where :math:`P_{on}` is the percentage of that day the instrument was +acquiring. ``get_science_acquisition_on_percentage()`` walks the on/off event +list, splits each interval at day boundaries, and accumulates on-time and +total-time per DOY. Two conventions worth knowing: + +* The state before the first message is **assumed to be the opposite** of the + first message's state, and a synthetic event is inserted at midnight of the + first day. +* The last interval is extended to the end of its day. + +Days with no uptime information get :math:`P_{on} = -1`; days with +:math:`P_{on} \le 0` get ``rate_calculation_quality_flags = 0`` and their rates +stay at the initialised value (``-1`` for the charge/mass-binned arrays, ``NaN`` +for the agnostic ones - note the inconsistency). Missing DOYs are logged. + +.. note:: + + ``dead_time`` is computed at L1B and is **not** used here. The rate is + corrected for acquisition uptime but not for per-event dead time. At ~16 + events per day the correction is negligible, but the omission is + undocumented. + +The L2B / L2C variable sets +^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Shared between both products: ``on_off_times``, ``on_off_events``, +``impact_day_of_year``, ``impact_charge``, ``mass`` and their label arrays. + +**L2B** (``imap_idex_l2b_sci-1mo``), binned on ``spin_phase``: + +.. list-table:: + :header-rows: 1 + :widths: 36 20 44 + + * - Variable + - Dimensions + - Published? + * - ``counts`` + - (epoch, spin_phase) + - **yes** + * - ``rate`` + - (epoch, spin_phase) + - **yes** + * - ``counts_by_charge`` + - (epoch, impact_charge, spin_phase) + - no - integer fill + * - ``counts_by_mass`` + - (epoch, mass, spin_phase) + - no - integer fill + * - ``rate_by_charge`` + - (epoch, impact_charge, spin_phase) + - no - NaN + * - ``rate_by_mass`` + - (epoch, mass, spin_phase) + - no - NaN + * - ``rate_calculation_quality_flags`` + - (epoch,) + - **yes** - 1 = good, 0 = insufficient uptime data + +**L2C** (``imap_idex_l2c_rectangular-map-1mo``), the same quantities binned on +``rectangular_lon_pixel`` × ``rectangular_lat_pixel`` instead of spin phase: +``counts_map`` and ``rate_map`` are **published**; ``counts_by_charge_map``, +``counts_by_mass_map``, ``rate_by_charge_map`` and ``rate_by_mass_map`` are +fill-valued. + +The L2C dataset carries three extra global attributes: ``sky_tiling_type`` = +``RECTANGULAR``, ``Spacing_degrees`` = ``6``, ``Spice_reference_frame`` = +``ECLIPJ2000``. + +The L2B / L2C fill block +^^^^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** Exactly as at L2A, the mass- and charge-binned arrays are computed +and then overwritten immediately before return. The code comment states the +reason: *"Keep the mass/charge computations above for validation and future +work, but withhold those products from publication until the fitting routines +and derived values are validated. The agnostic products are intentionally left +untouched."* + +Integer arrays are filled with ``IDEX_INT_FILLVAL = np.iinfo(np.int64).min``; +float arrays with ``np.nan``. + +So the honest summary of what a consumer can use from a monthly IDEX product +today is: **dust-hit counts and uptime-corrected rates, by spin quadrant and by +6° sky pixel, per day, with a per-day quality flag.** Everything mass- or +charge-resolved is a placeholder. + +The geometry caveat +^^^^^^^^^^^^^^^^^^^ + +The L2C map bins on ``longitude`` and ``latitude``, which L1B derives from the +**IDEX boresight pointing direction** at the event epoch. It is where the +instrument was looking, not a reconstructed dust arrival direction - IDEX has no +per-event direction-of-arrival measurement. Longitudes are wrapped with +``np.mod(..., 360)`` and latitudes clipped to [−90, 90]; non-finite geometry is +dropped by the agnostic path but *not* by the charge/mass path, which clips +instead. diff --git a/docs/source/algorithm-code-documentation/idex/l3-scope.rst b/docs/source/algorithm-code-documentation/idex/l3-scope.rst new file mode 100644 index 0000000000..0406ff917e --- /dev/null +++ b/docs/source/algorithm-code-documentation/idex/l3-scope.rst @@ -0,0 +1,95 @@ +.. _idex-l3-scope: + +L3 Scope - IDEX Does Not Have One Here +====================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +Short page, because the answer is short: **there is no IDEX L3 in this +repository, and as far as the algorithm document and the code are concerned, +there is no IDEX L3 anywhere.** + +Where IDEX stops +---------------- + +For most IMAP instruments, ``imap-processing`` stops at L2 and a separate +repository run closer to the science team produces L3. SWE's electron moments, +SWAPI's solar-wind proton parameters and IMAP-Lo's L3 maps all live outside this +repository, and those instruments' pages carry an "L3 scope" section warning you +off writing fitting or moment code here. + +**IDEX is different.** It does not hand off at L2. + +**[CODE]** ``imap_processing/__init__.py`` lists +``"idex": ["l1a", "l1b", "l2a", "l2b", "l2c"]`` - the highest level IDEX defines +is **L2C**, and L2C is already the mapped, aggregated, publication-shaped +product: monthly count-rate maps on a 6° rectangular ECLIPJ2000 grid. The +document's Figure 4.1 puts "Publication Level Products" immediately downstream of +L2B and L2C, as an output of the pipeline rather than another processing level. + +**[DOC]** The acronym list in chapter 0 defines L0, L1A, L1B, L2A, L2B and L2C. +It does not define L3. No chapter mentions an IDEX L3 product, an L3 dependency, +or an L3 hand-off. + +So the practical answer to "is IDEX L3 missing from this repository?" is: as far +as anything available here says, **no - it does not exist as a product level.** + +.. note:: + + This is a statement about the algorithm document and the code as of the + 1 June 2026 revision, not about the mission's Science Data Management Plan, + which this repository does not contain. If the SDMP defines an IDEX L3, this + page is the place to record it. Confirm with the IDEX team rather than + assuming either way. + +What the mapping infrastructure implies +--------------------------------------- + +L2C uses the **shared ENA-map machinery** +(``imap_processing/ena_maps/ena_maps.py``, ``AzElSkyGrid``, ``SkyTilingType``) +that IMAP-Lo, Hi and Ultra use for their sky maps. In those instruments that +machinery is L2-and-up territory; IDEX uses only the rectangular tiling and only +for binning counts, not for any of the ENA-specific pointing-set or +exposure-weighting logic. + +``imap_processing/ena_maps/utils/naming.py`` treats IDEX (and GLOWS) as special +cases in two places - both instruments are excluded from parts of the standard +map-naming convention. If you are extending IDEX's map products, read those two +branches first. + +What a real "L3" would need +--------------------------- + +If the IDEX team does eventually specify a higher-level product, the honest +statement of what L2 can hand it today is: + +**Available and trustworthy:** + +* Per-event impact charge, fitted pulse parameters and fit quality for + ``Target_Low``, ``Target_High`` and ``Ion_Grid`` (L2A), masked where the + channel saturated or the event was not a science event. +* Per-event velocity and mass estimates from all three low-rate channels (L2A), + same masking. Note the mass unit question flagged in :ref:`idex-l2`. +* Per-event SPICE context: spacecraft position and velocity relative to the Sun, + IDEX boresight longitude and latitude, solar longitude, spin phase (L1B, + copied forward through L2A). +* Per-event classification and saturation flags (L1A, copied forward). +* Daily dust-hit counts and uptime-corrected rates by spin quadrant and by 6° + sky pixel (L2B/L2C). + +**Not available yet, and blocking any compositional science:** + +* **The TOF mass spectrum.** ``mass``, ``mass_scale``, ``tof_snr``, + ``tof_peak_kappa`` and every ``tof_peak_*`` variable are computed and then + NaN-filled. Until the time-to-mass conversion and the EMG peak fits are + validated, IDEX publishes no elemental composition at all - which is the + instrument's headline measurement. +* **Mass- and charge-resolved rates.** ``counts_by_mass``, ``rate_by_mass``, + ``counts_by_charge``, ``rate_by_charge`` and their map equivalents are + fill-valued at L2B/L2C. +* **Combined-gain fits** for both the target pair and the TOF triple + (**[DOC]** sections 4.7.8 and 4.7.10). + +In other words, the gap in IDEX is not a missing processing level. It is that +the mass-spectrum half of the existing levels is written but withheld. See +:ref:`idex-implementation-status`. diff --git a/docs/source/algorithm-code-documentation/idex/overview.rst b/docs/source/algorithm-code-documentation/idex/overview.rst new file mode 100644 index 0000000000..e81ba0e013 --- /dev/null +++ b/docs/source/algorithm-code-documentation/idex/overview.rst @@ -0,0 +1,243 @@ +.. _idex-overview: + +Instrument Overview and Vocabulary +================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +Everything on this page is **[DOC]** unless marked otherwise - it is the +physical and telemetry context from chapters 2 and 3 of the algorithm document, +condensed. Read it once; the rest of the pages assume this vocabulary. + +What IDEX measures +------------------ + +The Interstellar Dust Experiment is a **high-resolution impact-ionization +time-of-flight (TOF) mass spectrometer**. It provides the elemental composition +and mass distribution of interstellar dust (ISD) and interplanetary dust +particles (IDP). Scientifically it is the solid-phase counterpart to the gas +and pickup-ion composition measured by IMAP-Lo, CoDICE and SWAPI. + +The measurement principle is **impact ionization**: + +1. A dust grain enters the aperture through a set of grid electrodes and strikes + the target. The effective target area is **600 cm²**. +2. The target is biased at **+3 kV**, accelerating the impact-generated positive + ions away from it. +3. A shaped electrostatic field (biased rings plus a curved grid electrode) + provides spatial and temporal focusing of those ions onto a centrally-located + detector, with **reflectron-type** ion optics so that ions of the same mass + but different initial energy arrive together. +4. The impact-generated **negative** charge is collected on the target itself. +5. Flight time from impact to detector scales as :math:`\sqrt{m}`, so the TOF + waveform is a mass spectrum once it is calibrated. That calibration is what + L2A's ``time_to_mass()`` attempts. + +Target cleanliness is a first-order concern - condensed volatiles on the target +change the ion yield. IDEX has a one-time deployable door, and its flight +operations include a monthly-ish decontamination cycle that raises the target to +120 °C for 8 hours. **[CODE]** Nothing in this repository models, corrects for, +or flags decontamination cycles; they appear only as event-message log entries. + +The six waveform channels +------------------------- + +Every dust impact is recorded on **six** channels, from three physical signal +sources. The three TOF channels are three gain stages of the *same* detector +signal, present so that a single event can be measured across a wide dynamic +range without saturating. + +.. list-table:: + :header-rows: 1 + :widths: 16 14 14 18 38 + + * - Channel + - Source + - Rate + - Digitization + - What it is + * - ``TOF_High`` + - TOF detector + - high (260 MHz) + - 10-bit + - Highest-gain TOF stage. The primary mass-spectrum input at L2A; + saturates first on large impacts. + * - ``TOF_Mid`` + - TOF detector + - high (260 MHz) + - 10-bit + - Mid-gain TOF stage. + * - ``TOF_Low`` + - TOF detector + - high (260 MHz) + - 10-bit + - Low-gain TOF stage. Survives the largest impacts. + * - ``Target_High`` + - Target CSA + - low (4.0625 MHz) + - 12-bit + - High-gain target charge-sensitive amplifier. The preferred impact-charge + channel when unsaturated. + * - ``Target_Low`` + - Target CSA + - low (4.0625 MHz) + - 12-bit + - Low-gain target CSA. Used when Target High saturates. **[CODE]** It is + also the channel L2B bins mass and charge on. + * - ``Ion_Grid`` + - Ion grid CSA + - low (4.0625 MHz) + - 12-bit + - Ion-grid charge-sensitive amplifier. Its signal may be positive or + negative depending on polarity; L2A explicitly allows a negative fitted + amplitude on this channel only. + +The two sampling cadences: + +.. math:: + + \Delta t_{HS} = \frac{1}{260}\ \mu\mathrm{s} \approx 3.846\ \mathrm{ns}, + \qquad + \Delta t_{LS} = \frac{1}{4.0625}\ \mu\mathrm{s} \approx 246.15\ \mathrm{ns}. + +**[CODE]** These live as ``RawDustEvent.HIGH_SAMPLE_RATE`` and +``LOW_SAMPLE_RATE`` in ``idex_l1a.py``, expressed in **microseconds per +sample**. Note the naming trap: they are named "rate" but hold a *period*. + +A separate constant, ``idex_constants.FM_SAMPLING_RATE = +0.0038466235767167234e-6`` seconds, is the flight-model quartz-oscillator +period used only by ``time_to_mass()``. It is the same 260 MHz cadence carried +to more digits and in seconds rather than microseconds. + +How an event is captured +------------------------ + +IDEX has several instrument modes (boot, idle, science, transmit, plus +decontamination and door actuation). Only two matter to the pipeline: + +**Science mode.** The instrument continuously samples the CSA channels and keeps +the six waveforms in a rolling buffer. Nothing is recorded as an event until an +impact-like **trigger** fires. On trigger, the flight software freezes the +pre-trigger and post-trigger portions of the buffer and stores the event to +onboard memory (an 8 GB NAND flash). + +**Transmit mode.** Stored events are packetized into CCSDS telemetry: one +metadata/header packet per event, then waveform fragments. + +The important consequence of the rolling buffer is that **the waveform starts +before the impact**. The number of pre-trigger "blocks" retained is in the +header, and L1A uses it to build a time axis whose zero is the trigger, with +negative times before the impact. The L2A baseline-noise windows (the first +5 µs of the low-rate record, and −7 µs to −5 µs on the high-rate record) exist +because of this pre-trigger data. + +Blocks, samples and fragments +----------------------------- + +Three different units of "chunk" appear in IDEX and they are easy to confuse. + +.. list-table:: + :header-rows: 1 + :widths: 18 82 + + * - Term + - Meaning + * - **Sample** + - One digitized value from one ADC. The natural unit of the waveform + arrays. + * - **Block** + - The FPGA's unit of buffer bookkeeping, ~1.969 µs of collection time. + **[CODE]** A low-rate block is **8 samples**; a high-rate block is + **512 samples** (8/4.0625 MHz = 512/260 MHz). The header reports + pre-trigger counts in *blocks*, never in samples. ``MAX_HIGH_BLOCKS = + 16`` and ``MAX_LOW_BLOCKS = 64``, so a full event is nominally 8192 + high-rate and 512 low-rate samples per channel. ``DT_BLOCK = 8 / + 4.0625e6 ≈ 1.96923 µs`` in ``idex_constants.py`` is the block duration + used for the dead-time calculation. + * - **Fragment** + - A telemetry-level chunk. One waveform channel does not fit in one CCSDS + packet, so it arrives as several packets carrying ``IDX__SCI0RAW`` + payloads, ordered by ``IDX__SCI0FRAGOFF``. Fragments are a packetization + artifact with no physical meaning, and L1A's job is to make them + disappear. + +Trigger and dead time +--------------------- + +The trigger does not return the instrument instantly to a ready state. After +the post-trigger samples are collected, the instrument observes a configured +**dead time** before it can accept another event. That interval is encoded in +the FPGA header and reconstructed at L1B as the ``dead_time`` variable. + +Dead time matters for rate interpretation - it is time during which a real dust +impact could not have been recorded. **[CODE]** L1B computes and stores +``dead_time``, but **L2B's rate calculation does not use it**. L2B corrects only +for science-acquisition uptime derived from the event-message log. At ~16 events +per day the correction would be negligible, but the omission is undocumented. +See :ref:`idex-implementation-status`. + +Trigger *origin* (which channel fired) and trigger *mode* (threshold, single +pulse, double pulse) are both decoded at L1B, and both feed the event +classification at L1A. See :ref:`idex-event-classification`. + +Two telemetry streams +--------------------- + +IDEX produces two kinds of telemetry the pipeline cares about, and they are +processed on completely separate paths that only rejoin at L2B. + +**Science telemetry (APID 1424).** The dust events. Header packet plus waveform +fragments, decommutated from ``idex_science_packet_definition.xml``. This is the +path that produces ``sci-10days`` at every level. + +**Event-message telemetry (APID 1418).** Timestamped instrument log entries - +state changes, pulser activity, command responses. Not dust events. These are +decommutated from the housekeeping XTCE and rendered into human-readable strings +at L1A, then reduced at L1B to two state variables, ``science_on`` and +``pulser_on``. **[CODE]** L2B needs ``science_on`` to know what fraction of each +day the instrument was actually acquiring, which is the denominator of every +count rate it publishes. This is the only place the two streams meet. + +A third stream, the **catalog list (APID 1419)**, is a packet-catalog summary +that the pipeline passes through with only a time conversion applied. It is not +described in the algorithm document. + +Time systems +------------ + +.. list-table:: + :header-rows: 1 + :widths: 26 74 + + * - Quantity + - Definition + * - **Packet creation time** + - ``SHCOARSE`` / ``SHFINE`` in the CCSDS secondary header. When the *packet* + was generated, i.e. roughly when the event was downlinked from onboard + storage. **This is not the event time, and this is the source of the L0 + file-grouping problem described in** :ref:`idex-data-products`. + * - **Event time (MET)** + - Reconstructed from the FPGA metadata header, not from the CCSDS header: + + .. math:: + + t_{MET} = 2^{16}\,\mathtt{TXHDRTIMESEC1} + + \mathtt{TXHDRTIMESEC2} + + 20\times10^{-6}\,\mathtt{TXHDRTIMESUBS} + + The 32-bit seconds counter is split across two 16-bit telemetry words; + the subsecond field is in units of 20 µs. **[CODE]** + ``calculate_idex_event_time()`` in ``idex_l1a.py``. Mission elapsed time + counts from 2010-01-01. + * - **epoch** + - **[CODE]** The CDF time coordinate on every IDEX product: + **TT-J2000 nanoseconds**, produced by ``met_to_ttj2000ns()``. UTC is a + derived, human-readable view of this and is never the stored basis. + * - **Waveform time axes** + - Per-event 1-D arrays ``time_low_sample_rate`` and + ``time_high_sample_rate``, in **microseconds relative to the trigger**, + so pre-trigger samples are negative. Stored as real variables on each + event, not as global coordinates, because the pre-trigger offset varies + per event. The integer index coordinates + ``time_{low,high}_sample_rate_index`` are what the CDF dimensions + actually depend on. diff --git a/docs/source/algorithm-code-documentation/idex/reference-tables.rst b/docs/source/algorithm-code-documentation/idex/reference-tables.rst new file mode 100644 index 0000000000..dacaa1a521 --- /dev/null +++ b/docs/source/algorithm-code-documentation/idex/reference-tables.rst @@ -0,0 +1,324 @@ +.. _idex-reference-tables: + +Reference Tables - Where to Look Them Up +======================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +The algorithm document's large tables are **deliberately not reproduced** in +these pages: they are long, they go stale, and in almost every case a +machine-readable version already exists in the repository that the code actually +reads. The document itself says the same thing about its own packet tables +(section 3.6): "This document avoids reproducing the full XML tables because +doing so would duplicate information already maintained by the packet-definition +files." + +The document is **not in this repository** - see :ref:`idex-source-documents`. +This page tells you where to look instead, and gives a page index for the parts +that only exist in the PDF. + +Rule of thumb +------------- + +.. list-table:: + :header-rows: 1 + :widths: 42 58 + + * - If you need... + - Go to + * - A science packet field's name, bit offset, width, type or enumeration + - ``imap_processing/idex/packet_definitions/idex_science_packet_definition.xml`` + * - An event-message or catalog-list packet field + - ``imap_processing/idex/packet_definitions/idex_housekeeping_packet_definition.xml`` + * - Which 10-day window a date belongs to + - ``imap_processing/idex/idex_10_day_CDF_names.csv`` + * - An instrument setting's bit layout or EU polynomial + - ``imap_processing/idex/idex_variable_unpacking_and_eu_conversion.csv`` + * - An event-message template or value dictionary + - ``imap_processing/idex/idex_evt_msg_parsing_dictionaries.json`` + * - The reference ion masses used by the TOF mass scale + - ``imap_processing/idex/atomic_masses.csv`` + * - A rise-time or charge-yield calibration coefficient + - the two ``imap_idex_l2a-calibration-curve-*`` ancillary CSVs + (copies in ``imap_processing/tests/idex/test_data/``) + * - A DN-to-engineering-unit waveform factor + - ``imap_processing/idex/idex_constants.py``, ``class ConversionFactors`` + - **not** an ancillary file, and **not** document Table 4.2 + * - An APID + - ``imap_processing/idex/idex_constants.py``, ``class IDEXAPID`` + * - A science packet type value + - ``imap_processing/idex/idex_l1a.py``, ``class Scitype`` + * - A trigger mode or origin label + - ``imap_processing/idex/idex_l1b.py``, ``class TriggerMode``, + ``class TriggerOrigin``, ``TRIGGER_LABELS`` + * - An event-message string the L1B reduction matches on + - ``imap_processing/idex/idex_l1b.py``, ``class EventMessage`` + * - An event or saturation flag name, threshold or rule + - ``imap_processing/idex/idex_event_flags.py`` + * - An L2B bin edge (mass, charge, spin phase, sky grid) + - ``imap_processing/idex/idex_l2b.py``, module level + * - A CDF variable's units, fill value, valid range or ``DEPEND_n`` + - ``imap_processing/cdf/config/imap_idex_l{1a,1b,2a,2b,2c}_variable_attrs.yaml`` + * - A ``logical_source`` string or global attribute + - ``imap_processing/cdf/config/imap_idex_global_cdf_attrs.yaml`` + * - A SPICE frame id, boresight or spin offset + - ``imap_processing/spice/geometry.py`` + * - An equation from L1A, L1B, L2A or L2B + - :ref:`idex-l1` / :ref:`idex-l2` - the load-bearing ones are transcribed + * - Anything else + - the algorithm document, using the page index below + +Machine-readable tables in the repository +----------------------------------------- + +Packet definitions +^^^^^^^^^^^^^^^^^^ + +The two XTCE XML files are the authoritative field definitions and, unlike the +document, cannot be out of date with respect to processing - they are what the +code parses. The document's Table 3.2 lists a *representative subset* of fields +(``SHCOARSE``, ``SHFINE``, ``IDX__SCI0TYPE``, ``IDX__SCI0FRAGOFF``, +``IDX__SCI0RAW``, ``IDX__TXHDRTIMESEC1/2``, ``IDX__TXHDRTIMESUBS``, +``IDX__TXHDRTRIGID``, ``IDX__TXHDRBLOCKS``, ``IDX__TXHDRSAMPDELAY``, +``ELSEC_EVTPKT``, ``ELSSEC_EVTPKT``, ``ELID_EVTPKT``, +``EL1PAR_EVTPKT``-``EL4PAR_EVTPKT``) and explicitly declines to reproduce the +rest. + +Useful greps:: + + # Every science field name + grep -o 'name="IDX__[A-Z0-9]*"' \ + imap_processing/idex/packet_definitions/idex_science_packet_definition.xml \ + | sort -u + + # Every catalog-list field + grep -o 'name="IDX_CATLST\.[A-Z0-9]*"' \ + imap_processing/idex/packet_definitions/idex_housekeeping_packet_definition.xml + + # Every event-message field + grep -o 'name="EL[0-9A-Z]*_EVTPKT"' \ + imap_processing/idex/packet_definitions/idex_housekeeping_packet_definition.xml + +Packed-field bit layouts +^^^^^^^^^^^^^^^^^^^^^^^^ + +Several header fields carry more than one quantity. These are the ones the +pipeline unpacks, gathered in one place because they are scattered across three +modules in the code: + +.. list-table:: + :header-rows: 1 + :widths: 30 16 22 32 + + * - Field + - Bits + - Quantity + - Unpacked in + * - ``IDX__TXHDRBLOCKS`` + - 6-11 + - low-rate pre-trigger blocks + - ``idex_l1a._set_sample_trigger_times`` + * - ``IDX__TXHDRBLOCKS`` + - 16-19 + - high-rate pre-trigger blocks + - ``idex_l1a._set_sample_trigger_times`` + * - ``IDX__TXHDRBLOCKS`` + - 20-23 + - dead-time shift + - ``idex_l1b.get_event_dead_time`` + * - ``IDX__TXHDRBLOCKS`` + - 24-29 + - dead-time base + - ``idex_l1b.get_event_dead_time`` + * - ``IDX__TXHDRSAMPDELAY`` + - 0-9 + - high-gain TOF delay + - ``idex_l1a._set_sample_trigger_times`` + * - ``IDX__TXHDRSAMPDELAY`` + - 10-19 + - mid-gain TOF delay + - ``idex_l1a._set_sample_trigger_times`` + * - ``IDX__TXHDRSAMPDELAY`` + - 20-29 + - low-gain TOF delay + - ``idex_l1a._set_sample_trigger_times`` + * - ``IDX__TXHDRTRIGID`` + - 0-2 + - delay channel select (HG/LG/MG priority) + - ``idex_l1a._set_sample_trigger_times`` + * - ``IDX__TXHDRTRIGID`` + - 0-5 + - trigger origin bit set + - ``idex_l1b.get_trigger_origin`` + * - ``IDX__TXHDRTRIGID`` + - 0-3, 4-5 + - active channels; software/external trigger + - ``idex_event_flags.classify_event_flags`` + * - ``IDX__TXHDR{HG,MG,LG}TRIGCTRL1`` + - 22-31 + - trigger threshold level + - ``idex_l1b.get_trigger_mode_and_level``, + ``idex_event_flags.classify_event_flags`` + * - ``idx__txhdr{prochk,hvpshk,lvhk0,lvhk1}ch*`` + - varies + - two instrument settings each + - ``idex_l1b.unpack_instrument_settings`` (driven by the EU CSV) + +Constants at a glance +^^^^^^^^^^^^^^^^^^^^^ + +.. list-table:: + :header-rows: 1 + :widths: 40 20 40 + + * - Constant + - Value + - Where + * - High-rate sample period + - 1/260 µs + - ``idex_l1a.RawDustEvent.HIGH_SAMPLE_RATE`` + * - Low-rate sample period + - 1/4.0625 µs + - ``idex_l1a.RawDustEvent.LOW_SAMPLE_RATE`` + * - FM quartz period (seconds) + - 3.8466235767e-9 + - ``idex_constants.FM_SAMPLING_RATE`` + * - Samples per low-rate block + - 8 + - ``idex_constants.SAMPLES_PER_BLOCK`` + * - Samples per high-rate block + - 512 + - ``idex_l1a.RawDustEvent.NUMBER_SAMPLES_PER_HIGH_SAMPLE_BLOCK`` + * - Max blocks (high / low) + - 16 / 64 + - ``idex_l1a.RawDustEvent.MAX_{HIGH,LOW}_BLOCKS`` + * - Block duration + - ≈1.96923 µs + - ``idex_constants.DT_BLOCK`` + * - Rice subframe size + - 64 + - ``decode.SUB_FRAME_SIZE`` + * - Saturation fraction + - 0.95 + - ``idex_event_flags._SATURATION_FRACTION`` + * - TOF / low-rate full scale + - 1023 / 4095 DN + - ``idex_event_flags._TOF_MAX_DN``, ``_LOW_RATE_MAX_DN`` + * - Pulser threshold + - 1000 DN + - ``idex_event_flags._PULSER_THRESHOLD_DN`` + * - Dust-hit peak threshold + - 7 σ + - ``idex_event_flags._PEAK_THRESHOLD_SIGMA`` + * - Dust-hit minimum FWHM + - 20 ns + - ``idex_event_flags._MIN_PEAK_WIDTH_US`` + * - Fit initial rise / decay + - 0.371 / 37.1 µs + - ``idex_l2a.estimate_dust_mass`` + * - Velocity inversion bracket + - 0.1-100 km/s + - ``idex_l2a.invert_rise_time_to_velocity`` + * - TOF SNR baseline window + - −7 to −5 µs + - ``idex_l2a.BaselineNoiseTime`` + * - Mass-scale stretch search + - 1400-1500 ns, 10 steps + - ``idex_l2a.time_to_mass`` + * - Ion-grid V(R) constants + - c=55, p=−3.2, v₀=1.5 + - ``idex_constants.ION_GRID_VELOCITY_*`` + * - Sky map spacing + - 6° + - ``idex_constants.IDEX_SPACING_DEG`` + * - Event reference frame + - ECLIPJ2000 + - ``idex_constants.IDEX_EVENT_REFERENCE_FRAME`` + +Document page index +------------------- + +For the parts that genuinely only exist in the PDF. Page numbers are **PDF +pages** (what your reader shows), which run two ahead of the document's own +printed page numbers. + +.. list-table:: + :header-rows: 1 + :widths: 14 24 62 + + * - PDF pages + - Section + - Worth opening for + * - 5-6 + - 0. Nomenclature + - Acronym list and definitions. Note that L3 is not among them. + * - 8-10 + - 2. Instrument description + - **Figure 2.1** (instrument outline with the three signal types labelled), + **Figure 2.2** (sensor head), **Figure 2.3** (electronics block + diagram). The figures are the reason to open this chapter; the prose is + summarised in :ref:`idex-overview`. + * - 11-14 + - 3.1-3.3 Modes and measurement + - **Figure 3.1** (mode state diagram), **Figure 3.2** (high/low-rate + packing), **Figure 3.3** (instrument timeline around a trigger). + Figure 3.3 is the clearest explanation of pre-trigger blocks anywhere. + * - 15-17 + - 3.4-3.6 Telemetry + - **Table 3.1** (science packet types - reproduced in :ref:`idex-l1`), + **Table 3.2** (representative XTCE fields). Both are short; the XML is + authoritative. + * - 19 + - 4.1 Overview + - **Figure 4.1**, the full pipeline flowchart with ancillary inputs on + both sides. The single most useful page in the document. Reproduced in + text form in :ref:`idex-data-products`. + * - 22-23 + - 4.4 Data volume + - **Figure 4.2** (expected ISD/IDP counts over the mission) and + **Table 4.1** (weekly data volume by product level). The source of the + "~16 events per day" figure that justifies the 10-day cadence. Not + reproduced here - neither is read by any code. + * - 23-27 + - 4.5-4.6 L0 and L1A + - Matches the code closely. See :ref:`idex-l1`. + * - 27-31 + - 4.7.1-4.7.2 L1B + - **Table 4.2** (waveform conversion factors) - **superseded by the code**, + see :ref:`idex-l1`. + * - 31-34 + - 4.7.3-4.7.6 L2A + - **Table 4.3** (calibration parameters, reproduced in :ref:`idex-l2`) and + the fit, inversion and mass-scale equations. + * - 35 + - 4.7.7-4.7.10 + - The NaN placeholder rationale and the two future-work sections. Short + and worth reading verbatim before touching L2A. + * - 36-37 + - 4.8-4.9 + - Calibration maintenance policy and the testing requirement. The + calibration policy in 4.8 is the operational contract for ancillary file + versioning; see :ref:`idex-ancillary`. + * - 38 + - Appendix A + - A code fragment of ``parser.py`` from the ``space_packet_parser`` + library. Nothing IDEX-specific; ignore. + * - 39-41 + - Appendix B + - The first 100 lines of the science XTCE. Read the XML instead. + +Notable absences from the document +---------------------------------- + +If you go looking for these in the PDF, save yourself the time - they are not +there: + +* Event classification and saturation flags (``idex_event_flags.py``). +* L2B and L2C algorithms beyond a flowchart box - binning, uptime model, rate + quality flags, the sky map. +* The catalog-list (``catlst``) products. +* Any IDEX L3. +* Quicklook products. +* Dead-time correction of count rates (dead time is *computed* per 4.7.1, but + nothing says what consumes it). +* Checksum or CRC verification. diff --git a/docs/source/algorithm-code-documentation/lo.rst b/docs/source/algorithm-code-documentation/lo.rst index a06acd39c0..024e8838f4 100644 --- a/docs/source/algorithm-code-documentation/lo.rst +++ b/docs/source/algorithm-code-documentation/lo.rst @@ -15,4 +15,4 @@ The L0 code to decommutate the CCSDS packet data can be found below. :template: autosummary.rst :recursive: - l0.utils + l0.utils \ No newline at end of file diff --git a/docs/source/algorithm-code-documentation/mag.rst b/docs/source/algorithm-code-documentation/mag.rst index b9cfe41216..82075e6c1e 100644 --- a/docs/source/algorithm-code-documentation/mag.rst +++ b/docs/source/algorithm-code-documentation/mag.rst @@ -26,4 +26,4 @@ Level 1A Processing Code: :recursive: l1a.mag_l1a_data - l1a.mag_l1a + l1a.mag_l1a \ No newline at end of file diff --git a/docs/source/algorithm-code-documentation/mag/ancillary.rst b/docs/source/algorithm-code-documentation/mag/ancillary.rst new file mode 100644 index 0000000000..8bf6d9ead6 --- /dev/null +++ b/docs/source/algorithm-code-documentation/mag/ancillary.rst @@ -0,0 +1,255 @@ +.. _mag-ancillary: + +Ancillary and Calibration Files +=============================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**Document:** section 7.2. + +Everything the MAG team delivers to the SDC arrives as a **CDF ancillary file**. +The document is explicit that **the filename must never be needed to determine +any property of the data inside** - all validity information lives in the file's +data or metadata. It is also explicit that **if two calibration files are valid +for the same data point, the most recently generated file wins**. + +The IMAP magnetometer requires dynamic calibration to remove the magnetic field +of the spacecraft. The calibration matrices should be applied based on sensor +and range and the offsets added based on sensor and range. The gradiometer factor +should be used to calculate time varying offsets via gradiometry during the +calibration process, and the spin averaging factors should be used to calculate +time varying offsets. + +How ancillary files are loaded +------------------------------ + +**[CODE]** ``MagAncillaryCombiner`` in +``imap_processing/ancillary/ancillary_dataset_combiner.py`` - a thin subclass of +``AncillaryCombiner`` with no MAG-specific behaviour. + +The base class assumes ancillary CDFs have **no time-varying variables inside +them**: a file is valid from the first second of its start date to the last +second of its end date. It takes a collection of files (different dates, +different versions), works out the total span, and produces a **single combined +dataset with one ``epoch`` entry per day**, where each day points at the values +from the file that is valid for that day. That is why every MAG level does +``calibration_dataset.sel(epoch=day)``. + +Two details worth remembering: + +* Ancillary files can have **no end date**, so the constructor requires an + ``expected_end_date``. ``cli.py`` passes ``start_date + 3 days`` for MAG. +* Validity dates come from the ``AncillaryFilePath`` filename, and version + ordering resolves overlaps. This is the one place the filename *is* used, and + it is a repository-wide convention rather than a MAG choice. + +.. _mag-eng-calibration: + +Engineering (ENG) calibration - ``l1b-calibration`` +--------------------------------------------------- + +Consumed by :ref:`mag-l1b` and by the I-ALiRT path. + +Derived from **ground calibration** at the Magnetsrode facility of TU +Braunschweig, folding together the nominal scale factor, the gain (sigma) and +orthogonality (omega) matrix, and the Measurement Frame -> Unit Reference Frame +rotation. + +.. list-table:: + :header-rows: 1 + :widths: 20 22 58 + + * - Variable + - Shape + - Meaning + * - ``MFOTOURFO`` + - ``(epoch, 3, 3, 4)`` + - MAGo measurement frame -> unit reference frame, per range. + * - ``MFITOURFI`` + - ``(epoch, 3, 3, 4)`` + - MAGi measurement frame -> unit reference frame, per range. + * - ``OTS`` + - ``(epoch,)`` + - MAGo time shift, **seconds**. + * - ``ITS`` + - ``(epoch,)`` + - MAGi time shift, seconds. + +**[DOC]** the validity of an ENG calibration file is specified in +**pre-timeshift** values, and multiple ENG calibrations may apply to +non-overlapping ranges within one processing window. + +A fallback copy, ``imap_mag_l1b-calibration_20240229_v002.cdf``, is bundled in +``imap_processing/mag/l1b/`` and used only when ``mag_l1b`` is called with +``calibration_dataset=None``. Do not rely on it outside tests. + +Calibration matrices (in-flight) - ``l2-calibration`` +------------------------------------------------------ + +Consumed by :ref:`mag-l2`. **[DOC]** generated approximately **monthly**; +nothing inside varies with time. + +Document contents: + +* Start and end time of validity +* Generation time +* **Time shift, in seconds, for each sensor** +* For each sensor, for each range, a dimensionless **3x3 matrix T** transforming + from the L1A data frame to the (spinning) spacecraft frame +* Additional metadata (TBD) + +**[CODE]** variables actually read: + +.. list-table:: + :header-rows: 1 + :widths: 22 22 56 + + * - Variable + - Shape + - Meaning + * - ``URFTOORFO`` + - ``(epoch, 3, 3, 4)`` + - MAGo URF -> ORF, per range. + * - ``URFTOORFI`` + - ``(epoch, 3, 3, 4)`` + - MAGi URF -> ORF, per range. + +The per-sensor time shift described in the document is **not** read from this +file at L2; the per-vector ``timedeltas`` in the offsets file serves that role +instead. + +Offsets - ``l2-norm-offsets`` and ``l2-burst-offsets`` +------------------------------------------------------- + +Consumed by :ref:`mag-l2`. **[DOC]** generated **daily**. These are the output of +the MAG team's spacecraft-field removal and offset determination. How Imperial +produces them is described in :ref:`mag-cmad`. + +Document contents, **for each sensor and for every timestamped vector in the L0 +data**: + +* A **3x1 offset matrix H** in nT to be removed. May be NaN when no good offset + can be determined; in the CDF, invalid values use ``FILLVAL`` placed outside + ``VALIDMIN``/``VALIDMAX`` so the intent is unambiguous. +* A **quality flag** (see :ref:`mag-quality-flags`). +* A **quality bitmask**, with some bits reserved for in-flight calibration. + Bit definitions are in :ref:`mag-quality-flags`. +* A **Delta-T** in +/- milliseconds adjusting the vector timestamp. +* Additional metadata (TBD) that may include time-varying information about how + the calibration was determined - for example when interference was occurring + and was removed. **This metadata should be copied into the final science + file**, so that the MAG team can annotate products with calibration steps that + are not yet defined. + +**[CODE]** variables read: ``epoch``, ``offsets`` ``(n, 3)``, ``timedeltas`` +``(n,)`` in seconds, ``quality_flag`` ``(n,)``, ``quality_bitmask`` ``(n,)``. + +.. important:: + + The offsets file also carries a ``Parents`` global attribute naming the + **exact L1B/L1C files** the offsets were generated against. + ``retrieve_mag_l1_inputs_from_l2_offsets`` downloads those files and L2 uses + them, ignoring anything passed in as a dependency. Epochs must match exactly + or ``mag_l2`` raises. + +L1D calibration - ``l1d-calibration`` +-------------------------------------- + +Consumed by :ref:`mag-l1d` and by the I-ALiRT path. + +**[DOC]** a **single file** whose offsets must be estimated *before the fact*, +making them considerably less precise than the L2 offsets. Contents: + +* Start and end time of validity; generation time +* For each sensor, for each range, a dimensionless **3x3 matrix T** (L1A frame -> + spinning spacecraft frame) +* For each sensor, for each range, a **3x1 offset matrix H** in nT +* A **gradiometer factor K** +* Additional metadata (TBD) + +**[CODE]** ``MagL1dConfiguration`` reads: + +.. list-table:: + :header-rows: 1 + :widths: 34 16 50 + + * - Variable + - Shape + - Meaning + * - ``URFTOORFO`` + - ``(3, 3, 4)`` + - MAGo URF -> ORF per range. Read with the shared + ``retrieve_matrix_from_l2_calibration``. + * - ``URFTOORFI`` + - ``(3, 3, 4)`` + - MAGi URF -> ORF per range. + * - ``offsets`` + - ``(2, 4, 3)`` + - ``[sensor, range, axis]``; sensor ``0 = MAGo``, ``1 = MAGi``. nT, ORF. + * - ``number_of_spins`` + - scalar + - Spin Count Calibration Value. Nominally 240 spins (~1 hour). + * - ``spin_average_application_factor`` + - scalar + - How much of the computed spin offset to apply, in ``[-1, 1]``. + * - ``quality_flag_threshold`` + - scalar + - Gradiometer offset magnitude above which data is flagged. + * - ``gradiometer_factor`` + - ``(3, 3)`` + - Kappa. + +.. note:: + + **Kappa is a 3x3 matrix, not a scalar.** The document's section 7.2.1 still + writes the gradiometer equation with a scalar :math:`\mathrm{K}`, but Issue 5 + Revision 1 changed L1D specifically to a matrix so each axis can be handled + independently and a small rotation of the measured gradiated offset can be + absorbed. Section 7.3.5 and the code both use the matrix form. + + It is possible that :math:`\mathrm{K} = 0` will be used in flight, which + disables gradiometry without a code change. + +SDC configuration (not an ancillary file) +------------------------------------------ + +**[CODE]** ``imap_processing/mag/imap_mag_sdc_configuration_v001.py``: + +.. code-block:: python + + L1C_INTERPOLATION_METHOD = "linear_filtered" + ALWAYS_OUTPUT_MAGO = True + +The algorithm document is explicit that the interpolation method is a **software +configuration variable and not a MAG-provided input file**, which is why this is +a Python module in this repository rather than an ancillary CDF. Changing +``ALWAYS_OUTPUT_MAGO`` also requires dependency-system changes so MAGi files +become upstream dependencies of L2. + +Ancillary files produced *by* this repository +---------------------------------------------- + +**[CODE]** L1D emits three ancillary CDFs alongside its science products. They +are written by ``Mag.post_processing`` in ``cli.py`` with ``istp=False`` and +``terminate_on_warning=False``, bypassing ``write_cdf``; any failure is logged +and swallowed. + +.. list-table:: + :header-rows: 1 + :widths: 40 60 + + * - ``Logical_source`` + - Contents + * - ``imap_mag_l1d_spin-offsets`` + - ``epoch``, ``x_offset``, ``y_offset``, ``validity_start_time``, + ``validity_end_time``, ``start_spin_counter``, ``end_spin_counter``. One + row per spin-averaging chunk (nominally ~1 hour). + * - ``imap_mag_l1d_gradiometry-offsets-norm`` + - ``epoch``, ``gradiometer_offsets`` ``(n, 3)``, + ``gradiometer_offset_magnitude``, ``quality_flags``. One row per MAGo + vector. + * - ``imap_mag_l1d_gradiometry-offsets-burst`` + - Same, for burst mode. + +**[DOC]** section 7.3.5 expects **two** spin-offset files, one per sensor. +Only one is produced. See :ref:`mag-implementation-status`. diff --git a/docs/source/algorithm-code-documentation/mag/cmad.rst b/docs/source/algorithm-code-documentation/mag/cmad.rst new file mode 100644 index 0000000000..17cbe7f16c --- /dev/null +++ b/docs/source/algorithm-code-documentation/mag/cmad.rst @@ -0,0 +1,436 @@ +.. _mag-cmad: + +Upstream Calibration and Cleaning (CMAD) +======================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This page summarises what the public **IMAP Calibration and Measurement +Algorithms Document (CMAD)** says about MAG, and how that relates to the +algorithm document the rest of these pages are built on. + +Read it when you need to know **where the L2 offsets and calibration matrices +come from**, what the quality bitmask bits mean, or what artifacts remain in +released L2 data. None of the processing described here runs in this +repository; it runs at Imperial College London and its results reach the SDC as +the ``l2-calibration`` and ``l2-{norm,burst}-offsets`` ancillary files (see +:ref:`mag-ancillary`). + +The algorithm document is not in the CMAD +----------------------------------------- + +For SWAPI, CoDICE, HIT, SWE, IDEX and GLOWS, the CMAD embeds the instrument's +own SDC-facing algorithm document, sometimes at a newer revision. **MAG is +different.** IMAP-MAG-SW-009 (the *MAG Science Algorithm Document*) is not +embedded or cited anywhere in the CMAD. Its L0 to L1D and I-ALiRT procedures, +the compression appendix, and the calibration and offset file formats do not +appear in it at all. (IMAP-Lo is partly similar: the CMAD embeds only its +Mapping Algorithms appendix, not the main Data Product Algorithms document.) + +Instead, CMAD section 4.4 (*Magnetic Fields*) embeds a different Imperial College +technical note, **IMAP-OPS-TN-ICL-013, *IMAP MAG Data Cleaning Processes***. Its +change log says Issue 3 was a "focused version to be used as input for CMAD" and +Issue 4 was "removal of non-relevant sections for CMAD", so the substitution was +deliberate on the MAG team's part. + +The CMAD does not say why. The most likely reason is that the two documents +serve different readers: + +* SW-009 exists "to completely recreate the data processing pipeline". It is an + implementation specification for the SDC. +* The CMAD exists to "enable any user to fully understand the scientific data + they are using". For MAG, what determines L2 science quality is the cleaning + and offset determination done upstream at Imperial. The SDC's part is to apply + the result. + +The section 4.4 heading still reads "MAG data product definitions and processing +algorithms are embedded below", the boilerplate used for every instrument. The +CMAD is also marked *preliminary* and has open placeholders (for example "Include +details of I-ALiRT data here"). SW-009 may be added in a later version, but that +is speculation. + +.. important:: + + **These pages remain the only description of the SDC pipeline that is tied + to the code.** The CMAD adds context from upstream of the pipeline; it + neither replaces nor contradicts the L1A to L1D procedures in SW-009. Where + the two documents disagree about L2, it is listed in + :ref:`mag-cmad-vs-sw009` below. + +Where MAG appears in the CMAD +----------------------------- + +``IMAP_CMAD_20260722.pdf`` (version 1.1, preliminary). Printed page numbers are +one less than the PDF page index; PDF pages are given in parentheses. + +.. list-table:: + :header-rows: 1 + :widths: 28 18 54 + + * - CMAD section + - Printed (PDF) pages + - Content + * - 2.5 Instrument description + - 6 (7) + - One paragraph: two fluxgates on a 2.5 m boom for gradiometry, magnetic + cleanliness programme. Defers to Horbury et al. [2025]. + * - 3.5.1 Strategy & Approach + - 151 (152) + - Ground calibration at Magnetsrode; in-flight calibration using the spin; + cross-calibration with ACE, Wind, DSCOVR, Aditya-L1 and a co-launched + mission. + * - 3.5.2 Pre-Flight Instrument Calibration + - 152-163 (153-164) + - Embeds **IMAP-OPS-TN-ICL-017, *IMAP MAG Calibration Inputs + Description***, Issue 2, 4 June 2026. Despite the section heading, this + is entirely **in-flight** calibration. Section 3.5.3 (In-Flight) is an + empty placeholder, and there is no pre-flight calibration report. + * - 4.1.1 CDF data file contents + - ~378-380, ~502-516 + - Variable listings for the ten L2 products, + ``imap_mag_l2_{burst,norm}-{dsrf,gse,gsm,rtn,srf}``. No L1 products. + * - 4.4 Magnetic Fields + - 912-929 (913-930) + - Embeds **IMAP-OPS-TN-ICL-013, *IMAP MAG Data Cleaning Processes***, + Issue 4, 4 June 2026. + * - 5.4.5 Data usage caveats, MAG + - 1163-1166 (1164-1167) + - **The authoritative quality flag and bitmask definitions**, plus + descriptions of the remaining artifacts and their science impacts. + +Both technical notes apply to **Imperial calibration code v2.2.0** +(``ImperialCollegeLondon/IMAP_MAG_Calibration`` on GitHub) and to **Data Release +1**, covering 1 January to 29 April 2026, released August 2026. Numbers quoted +below are for that release and **will change** in later ones. + +.. _mag-cmad-chain: + +The L2 offset production chain +------------------------------ + +**[CMAD]** ICL-013 section 2. The Imperial wrapper script +``calibrate_L2_offsets.m`` loads housekeeping and configuration, then runs, in +order: + +1. **Apply the calibration matrices** (see :ref:`mag-cmad-matrices`). +2. **Remove the IMAP-Lo pivot platform** signal. +3. **Remove the Ultra decontamination heater and Hi-45/Hi-90** signals. +4. **Apply the spin-plane offset** correction (a baseline plus a + spin-tone-reducing optimisation). +5. **Apply the spin-axis offset** correction (a solar-wind technique). +6. **Clean thruster and pre-thruster** signals. +7. **Output one offset per vector** of the input L1C (normal) or L1B (burst) + file. This is the ``l2-{norm,burst}-offsets`` file that :ref:`mag-l2` + consumes. + +Steps 2, 3 and 6 are *cleaning*: they remove signals generated by the spacecraft +and other instruments. Steps 4 and 5 are what the MAG team calls *calibration*: +they estimate the sensor and static spacecraft offsets. Everything is folded +into the single per-vector offset, so the SDC cannot separate the contributions. + +.. note:: + + Because the offset is the sum of all of these corrections, a single L2 + ``FILLVAL`` vector can come from any of them. The one **known** source is a + bug in burst-mode thruster cleaning (see :ref:`mag-cmad-thrusters`). The + SDC's ``MagL2.apply_offsets`` is doing the right thing when it turns these + vectors into ``FILLVAL``. + +.. _mag-cmad-matrices: + +Calibration matrices +-------------------- + +**[CMAD]** ICL-017 section 3. + +Release 1 uses **CalibrationMatricesV9**, defined relative to the +``IMAP_MAG_BASE`` SPICE frame. That matches the design in :ref:`mag-overview`: +the calibration matrix takes each sensor into one idealised frame. The +parameterisation per sensor is three polar angles (Theta 1-3), three azimuths +(Phi 1-3) and two relative gains (Gain 1-2). This is presumably what arrives +here as ``URFTOORFO``/``URFTOORFI`` in ``l2-calibration``, but the CMAD does not +describe the delivered file. + +How the angles were determined: + +* The **19-20 January 2026 CME** provided a long high-field interval, which + separates the angular contribution to spin tone from the offset contribution. +* **Theta 3 and Phi 3** (spin-axis alignment) minimise the spin tone in the DSRF + Z axis after applying the matrix and despinning. +* **Theta 1 and Theta 2** minimise the spin tone in the DSRF spin plane, and are + checked by recomputing the angles with the Kepko method (agreement within + 0.1 degrees overall, 0.01 degrees in the quietest window). +* **Phi 2** (the X-Y angle) minimises the **second harmonic** of the spin tone + over 3-minute windows during the CME, as a weighted mean. +* MAGi angles are then further tuned to **co-align MAGi with MAGo** over the + CME, so that gradiometry works. + +.. note:: + + The V9 gains are **not exactly 1** (they differ from unity by ~0.1%). SW-009 + and :ref:`mag-overview` say the in-flight gain and orthogonality corrections + stay at the identity "until there is evidence to justify a change". That + evidence has now arrived. + +.. _mag-cmad-offsets: + +Offsets +------- + +**[CMAD]** ICL-017 sections 4-5. There are two techniques, one per axis group. +Both work in the sensor frame after the calibration matrix is applied, with Z +along the spin axis. + +Spin-plane offsets (X, Y): Kepko method +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +1. **Baseline:** rolling Kepko offsets over an **8-hour window**, with the angles + fixed from the calibration matrix, computed **after** pivot-platform removal. +2. **Optimisation:** for each day, offsets are fitted on 3-hour intervals to + **minimise spin tone in** :math:`|B|`, applied by linear interpolation, and + checked hourly. Where spin tone is still too high, the fit is repeated on + progressively shorter intervals, **down to 6 minutes**. This runs after + pivot-platform removal, Hi/Ultra removal and baseline application. +3. **Alternative optimiser:** on a few days, minimising spin tone in :math:`|B|` removed + real solar-wind power at the spin frequency and injected spin tone into the + components. On those days the optimiser minimises **component-level** spin + tone over longer intervals instead. + +Inputs are version-controlled CSVs, one per day per sensor, listed in +``calibration_input_release1_v001.json``. + +Spin-axis offset (Z): Leinweber method +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +The Leinweber solar-wind technique produces **one offset per day** plus the +fraction of the day that was usable. Days with less than 5% usable data are +dropped, outliers beyond 1.5 standard deviations of a moving mean are +excluded, and the final value comes from a **linear fit** across the period. +Four CSVs per sensor cover Release 1. + +.. important:: + + SW-009 (and older text in these pages) says the offsets are determined "with + the Leinweber method". In practice **Leinweber is used only for the spin + axis**; the spin plane uses Kepko plus spin-tone optimisation. + +Typical magnitudes +^^^^^^^^^^^^^^^^^^ + +These are useful as a sanity check when an offsets file looks wrong. They are +Release 1 means over 1 January to 29 April 2026 (ICL-017 tables 1-2); the standard +deviations are all 0.21 nT or less. + +.. list-table:: + :header-rows: 1 + :widths: 20 20 20 20 + + * - Sensor + - X (spin plane) + - Y (spin plane) + - Z (spin axis) + * - MAGo + - +0.83 nT + - -5.69 nT + - +0.97 nT + * - MAGi + - +6.88 nT + - -8.60 nT + - +1.99 nT + +The larger MAGi values are expected: MAGi is closer to the spacecraft. + +Cleaning processes +------------------ + +**[CMAD]** ICL-013 section 3. All three use **housekeeping from the spacecraft +or other instruments** as the trigger, and most of them use **gradiometry** +(scaling the MAGi - MAGo difference and subtracting it from MAGo). + +.. warning:: + + The gradiometer factors below are **Imperial's L2 cleaning parameters**. They + are unrelated to the kappa matrix in the ``l1d-calibration`` file that this + repository applies at :ref:`mag-l1d`, even though both are called "kappa" or + "gradiometer factor". + +IMAP-Lo pivot platform +^^^^^^^^^^^^^^^^^^^^^^ + +The IMAP-Lo pivot platform had three set points (75, 90 and 105 degrees) and +moved up to once a day. Each position produces a different static field at both +sensors, larger at MAGi. + +* **Cleaning:** per-axis, per-sensor deltas relative to the 90 degree position + (zero by construction), derived from March-April 2026 movements, are applied + according to IMAP-Lo housekeeping ``ILOGLOBAL.PPM_NHK_POT_PRI``. Between set + points the delta is interpolated linearly. The deltas are up to ~0.25 nT at + MAGo and ~0.6 nT at MAGi. +* **Not cleaned:** the oscillations *during* motion (about 6 minutes, generally + near 10:35 UT). These are flagged instead (see :ref:`mag-cmad-quality`). + +Ultra decontamination heaters and Hi-45/Hi-90 +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Added in Imperial code v2.1.0. A **superposed epoch analysis** over one month +of data (1-30 November) builds an average gradiometer profile of each signal, triggered by: + +* ``U45_DECON_HTR_CURR`` / ``U90_DECON_HTR_CURR`` for the Ultra decontamination + heaters, which switch on every 10 minutes; +* ``H45_CURR`` / ``H90_CURR`` for the Hi signal. + +A scaled profile is then subtracted from MAGo at each trigger. The gradiometer +factors are **0.65** for the heaters (from a separate MAGo/MAGi epoch analysis, +possible because the signal is strong on the spin axis) and **1.5** for Hi (from +minimising spin-plane spin tone, because the Hi signal lies mostly in the spin +plane). + +.. _mag-cmad-thrusters: + +Thrusters and the pre-thruster signal +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +The daily repointing produces a train of large spikes at thruster firing +(around 10:00 UT). About **one hour earlier** (around 09:00 UT) a short DC signal +of unknown origin appears, which the team calls the *pre-thruster signal*. + +* **Trigger:** a jump of more than 0.05 mA between consecutive samples of + ``TAC_THRUST_BUS_GRP_1_CURR`` or ``TAC_THRUST_BUS_GRP_2_CURR`` (spacecraft + packet X285). The current rises an hour before firing, so the same trigger + locates both signals. +* **Method:** select a short segment (about 2 minutes for the pre-thruster + signal), transform to a despun frame, interpolate MAGi onto the MAGo timeline + (more accurate when despun, because the field is nearly constant), and subtract + a diagonal gradiometer factor times the difference from MAGo. This is applied + wherever MAGi - MAGo exceeds 0.1 nT in any axis, plus about 2 seconds either + side to avoid residual edge spikes. +* **Factors:** one diagonal matrix per signal, measured by hand per axis in the + orthogonal MAG frame and rotated into the despun frame before use: + pre-thruster diag(0.718, 0.963, 0.455), thruster diag(0.965, 0.845, 0.965). + +.. warning:: + + **Known bug (CMAD):** burst-mode thruster cleaning introduces ``FILLVAL`` + (``-1e31``) gaps of a few seconds, within about 2 minutes of the firing. + Burst-mode cleaning also frequently leaves residual thruster spikes. Normal + mode is cleaned effectively. The periods are flagged, and fixing this is a + stated target for future Imperial releases. + +Not cleaned +^^^^^^^^^^^ + +* **Trajectory correction manoeuvres (TCMs)**, every one to two months: long sequences of + large pulses, sometimes with precursors outside the main sequence. Flagged for + 3-6 hours around each TCM. +* **Pivot platform motion** (above). + +.. _mag-cmad-quality: + +Quality flag and bitmask +------------------------ + +**[CMAD]** section 5.4.5. **This supersedes the list in SW-009 section 7.2** and +matches the ``VAR_NOTES`` on ``qf_bitmask`` in +``imap_processing/cdf/config/imap_mag_l2_variable_attrs.yaml``. + +``quality_flags``: ``0`` good data, ``1`` bad data (not suitable for science). +A raised flag is **always** accompanied by a non-zero bitmask. + +``quality_bitmask`` (bit 0 is the least significant): + +.. list-table:: + :header-rows: 1 + :widths: 10 34 56 + + * - Bit + - Meaning + - Raised for + * - 0 + - Data sourced from the secondary sensor + - Output is MAGi rather than MAGo. + * - 1 + - Thruster firing signals have been removed + - The daily thruster firing (around 10:00 UT) and the pre-thruster activity + (around 09:00 UT). Cleaning may be imperfect, especially in burst mode. + * - 2 + - Spacecraft interference impacts these data + - Uncleaned spacecraft signals: 3-6 hours around TCMs, and IMAP-Lo pivot + platform motion (around 10:35 UT, about 6 minutes). + * - 3 + - Instrument signals have been removed + - Signals from other instruments removed from the data. + * - 4-7 + - Reserved for in-flight calibration + - Currently unused. + +SW-009's ``SCTONES`` and ``PIVOTPLATFORMINTERFERENCE`` bits no longer exist. +Pivot platform motion is now reported through bit 2. Spin tone has no bit; it +is handled by the spin-plane offset optimisation instead. + +.. note:: + + **[CODE]** Nothing changes in this repository: ``quality_flags`` and + ``quality_bitmask`` are still copied through opaquely from the offsets file. + MAG still has no enum in ``imap_processing/quality_flags.py``; see + :ref:`mag-implementation-status`. + +Remaining artifacts in L2 +------------------------- + +**[CMAD]** section 5.4.5 tells users to expect the following in released L2 data. +They matter here mainly for triaging "is this a pipeline bug?" reports. + +* **Burst-mode thruster residuals** that can mimic solitons, mirror modes or + other ion-scale structures. +* **Spin-axis offset uncertainty** of ~0.1 nT, at times up to ~0.5 nT. The spin + axis is roughly GSE/GSM X and RTN R. The error shows up as a near-DC shift in + that component, which biases :math:`|B|` and can shift field direction by up to ~10 + degrees for typical IMF strengths. +* **Spin tone** at the ~15 s spin period, in the spin-plane components (roughly + GSE/GSM Y and Z, RTN T and N) and in :math:`|B|`. Sources are spin-plane offset and + calibration-matrix errors. It is sometimes significant, including during CMEs + (from gain and angle uncertainty), and can be mistaken for alpha-particle or + other heavy-ion cyclotron waves. + +.. _mag-cmad-vs-sw009: + +Differences from SW-009 +----------------------- + +Only the L2 and in-flight calibration material overlaps. Where the two disagree: + +.. list-table:: + :header-rows: 1 + :widths: 22 39 39 + + * - Topic + - SW-009 + - CMAD (ICL-013, ICL-017, 5.4.5) + * - Quality bitmask + - Eight named bits, ``SEC_SENS`` last, includes ``SCTONES`` and + ``PIVOTPLATFORMINTERFERENCE``. + - Four named bits, secondary sensor at bit 0. **Matches the code's YAML.** + * - Offset method + - "Leinweber method" for the offsets. + - Leinweber for the spin axis only; Kepko plus spin-tone optimisation for + the spin plane. + * - In-flight gain/orthogonality + - Identity until justified. + - V9 matrices with fitted angles and gains that are not exactly 1. + * - Spacecraft-field removal + - Generic "several processes" for 0-64 Hz and DC steps. + - Named processes for the pivot platform, Hi/Ultra and thrusters, with + triggers and factors; TCMs and pivot motion flagged, not cleaned. + * - L2 frames + - DSRF, SRF, RTN, GSE. + - Also GSM. The code already produces it; see + :ref:`mag-implementation-status`. + * - Provenance + - Offsets file names the L1 file it applies to. + - ICL-017 adds that the calibration input versions are "captured in the + parent metadata field of the released L2 science data files". See the + open question in :ref:`mag-implementation-status`. + +Everything SW-009 specifies for **L0 to L1D, compression, I-ALiRT and the +ancillary file formats** is absent from the CMAD, so the rest of these pages keep +SW-009 as their primary source. diff --git a/docs/source/algorithm-code-documentation/mag/data-products.rst b/docs/source/algorithm-code-documentation/mag/data-products.rst new file mode 100644 index 0000000000..479e13bfe3 --- /dev/null +++ b/docs/source/algorithm-code-documentation/mag/data-products.rst @@ -0,0 +1,328 @@ +.. _mag-data-products: + +Data Products and Pipeline +========================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This page is the map of **what exists, what feeds what, and what it is called**. +Use it to find the right module and the right ``Logical_source`` before diving +into an algorithm page. + +Level definitions +----------------- + +.. list-table:: + :header-rows: 1 + :widths: 8 30 62 + + * - Level + - Frame / units + - Meaning for MAG + * - L0 + - MFO, MFI. Engineering units. + - Raw CCSDS packets, NM and BM unseparated. Not produced by this + repository. + * - L1A + - MFO, MFI. Engineering units. + - Packets separated by mode and by **physical sensor**, decommutated and + decompressed into per-vector time series. No calibration. + * - L1B + - URFO, URFI. **nT**. + - Engineering (ground) calibration applied; compression rescaling applied; + per-sensor time shift applied. + * - L1C + - URFO, URFI. nT. + - **Normal mode only.** Gaps caused by burst-mode operation are filled with + burst data interpolated onto a synthesised normal-mode timeline. Carries + a per-sample flag for measured vs. interpolated. + * - L1D + - DSRF, SRF, RTN, GSE. nT. + - **Preliminarily calibrated data for other instrument teams.** Uses + predicted offsets, spin averaging and gradiometry rather than the MAG + team's after-the-fact offsets, so it is available quickly but is less + precise than L2. Formerly called "L2Pre". + * - L2 + - DSRF, SRF, RTN, GSE, GSM. nT. + - **Fully calibrated data for release.** Uses MAG-team per-vector offsets, + time deltas and quality flags. + * - L3 + - -- + - **Does not exist for MAG.** L2 is the final released product. + +Source packets +-------------- + +**[CODE]** ``imap_processing/mag/l0/mag_l0_data.py``, ``Mode`` IntEnum. Fields +are defined in ``imap_processing/mag/packet_definitions/MAG_SCI_COMBINED.xml``. + +.. list-table:: + :header-rows: 1 + :widths: 12 20 68 + + * - APID + - Name + - Contents + * - 1052 + - ``Mode.NORMAL`` + - Normal rate science telemetry. Up to 32 vectors/s per sensor. + * - 1068 + - ``Mode.BURST`` + - Burst rate science telemetry. Up to 128 vectors/s per sensor. + * - -- + - ``MAG_SCI_IALIRT`` + - Real-time stream, defined in + ``imap_processing/ialirt/packet_definitions/ialirt_mag.xml`` and handled + entirely outside ``imap_processing/mag`` (see :ref:`mag-ialirt`). + +Housekeeping APIDs are not processed by this repository. + +The ``MagL0`` header fields you will actually use +------------------------------------------------- + +**[CODE]** ``MagL0`` in ``mag/l0/mag_l0_data.py``. The load-bearing fields: + +.. list-table:: + :header-rows: 1 + :widths: 24 76 + + * - Field + - Meaning + * - ``SHCOARSE`` + - Packet mission elapsed time (whole seconds). + * - ``PUS_SSUBTYPE`` + - **Seconds of data in the packet, minus one.** ``seconds_per_packet = + PUS_SSUBTYPE + 1``. + * - ``COMPRESSION`` + - 1 if the vector block is Fibonacci/zig-zag compressed. + * - ``MAGO_ACT`` / ``MAGI_ACT`` + - Whether each sensor is active. Also referred to as FOB / FIB. + * - ``PRI_SENS`` + - ``0`` = MAGo is PRIMARY, ``1`` = MAGi is PRIMARY + (``PrimarySensor`` enum). + * - ``PRI_VECSEC`` / ``SEC_VECSEC`` + - Encoded rate 0-7. ``MagL0.__post_init__`` converts these **in place** to + actual rates via ``2 ** value``, giving 1, 2, 4, ..., 128. Do not decode + them a second time. + * - ``PRI_COARSETM`` / ``PRI_FNTM`` + - MET seconds and 16-bit sub-second counter for the **first PRIMARY + vector**. ``SEC_COARSETM``/``SEC_FNTM`` likewise for SECONDARY. + * - ``VECTORS`` + - The raw bit-packed vector block, converted to a big-endian ``uint8`` + numpy array on construction. + +``MagL0`` defines ``__eq__``/``__hash__`` on ``(SHCOARSE, APID, SRC_SEQ_CTR)``, +which is how ``decom_packets`` **de-duplicates** repeated packets. + +Products produced by this repository +------------------------------------ + +**[CODE]** These are the exact ``Logical_source`` strings, taken from +``imap_processing/cdf/config/imap_mag_global_cdf_attrs.yaml``. Anything not in +these tables does not exist. + +.. note:: + + The document writes product names as ``imap_mag_l1a_raw_normal_...``. The + code follows the IMAP filename convention instead, where the descriptor is a + single hyphenated token: ``imap_mag_l1a_norm-raw``. **Use the code's + strings.** + +L1A +^^^ + +Produced by ``mag_l1a.mag_l1a(packet_file)`` - one call handles both APIDs in +the file and emits between 2 and 6 datasets. + +.. list-table:: + :header-rows: 1 + :widths: 34 66 + + * - ``Logical_source`` + - Contents + * - ``imap_mag_l1a_norm-raw`` + - One row **per packet**. ``raw_vectors`` is the undecoded byte block, + zero-padded to the longest packet in the file, plus every header field as + its own variable. Epoch is ``SHCOARSE``. + * - ``imap_mag_l1a_burst-raw`` + - Same, for burst packets. + * - ``imap_mag_l1a_norm-mago`` + - One row **per vector**: ``vectors`` (x, y, z, range) and + ``compression_flags`` (is_compressed, compression_width). + * - ``imap_mag_l1a_norm-magi`` + - Same, MAGi. + * - ``imap_mag_l1a_burst-mago`` + - Same, burst MAGo. + * - ``imap_mag_l1a_burst-magi`` + - Same, burst MAGi. + +L1B +^^^ + +``mag_l1b.mag_l1b(input_dataset, day_to_process, calibration_dataset)`` - **one +L1A input, one L1B output**. Raw L1A files raise ``ValueError``. + +* ``imap_mag_l1b_norm-mago`` +* ``imap_mag_l1b_norm-magi`` +* ``imap_mag_l1b_burst-mago`` +* ``imap_mag_l1b_burst-magi`` + +L1C +^^^ + +``mag_l1c.mag_l1c(first, day, second, previous_day_dataset)``. **Normal mode +only** - burst data is an *input* used to fill gaps, not an output. + +* ``imap_mag_l1c_norm-mago`` +* ``imap_mag_l1c_norm-magi`` + +L1D +^^^ + +``mag_l1d.mag_l1d(science_data, calibration_dataset, day)`` produces **all +frames and both modes in a single call**, plus ancillary offset files. + +Science products (8 when burst data is available, 4 otherwise): + +* ``imap_mag_l1d_norm-srf``, ``imap_mag_l1d_norm-dsrf``, + ``imap_mag_l1d_norm-gse``, ``imap_mag_l1d_norm-rtn`` +* ``imap_mag_l1d_burst-srf``, ``imap_mag_l1d_burst-dsrf``, + ``imap_mag_l1d_burst-gse``, ``imap_mag_l1d_burst-rtn`` + +Ancillary products, written by ``Mag.post_processing`` in ``cli.py`` with +``istp=False`` (they bypass ``write_cdf``): + +* ``imap_mag_l1d_spin-offsets`` +* ``imap_mag_l1d_gradiometry-offsets-norm`` +* ``imap_mag_l1d_gradiometry-offsets-burst`` + +.. note:: + + There is **no** ``imap_mag_l1d_*-gsm``. L1D outputs SRF, DSRF, GSE and RTN + only; GSM appears at L2 and in I-ALiRT. + +L2 +^^ + +``mag_l2.mag_l2(calibration, offsets, input_data, day, mode, frames)``. One call +produces all frames **for a single mode**; the mode comes from the CLI +descriptor. + +* ``imap_mag_l2_norm-srf``, ``imap_mag_l2_norm-gse``, ``imap_mag_l2_norm-gsm``, + ``imap_mag_l2_norm-rtn``, ``imap_mag_l2_norm-dsrf`` +* ``imap_mag_l2_burst-srf``, ``imap_mag_l2_burst-gse``, + ``imap_mag_l2_burst-gsm``, ``imap_mag_l2_burst-rtn``, + ``imap_mag_l2_burst-dsrf`` + +``DEFAULT_L2_FRAMES`` orders DSRF **last** deliberately: some vectors may become +NaN/FILLVAL after that rotation, and ``rotate_frame`` mutates the dataclass in +place. + +Dependency graph +---------------- + +.. code-block:: text + + L0 packets (APID 1052 + 1068, 25 h) + | + +-- mag_l1a --> norm-raw, burst-raw + | norm-mago, norm-magi, burst-mago, burst-magi + | + +-- mag_l1b (+ l1b-calibration ancillary) + | each L1A sensor/mode file -> matching L1B file + | + +-- mag_l1c (norm L1B + burst L1B + previous day's L1C) + | -> norm-mago, norm-magi [normal mode only] + | + +-- mag_l1d (L1C norm mago+magi REQUIRED, + | L1B burst mago+magi OPTIONAL, + | + l1d-calibration ancillary, + SPICE) + | -> 4 or 8 science files + spin/gradiometry ancillary files + | + +-- mag_l2 (L1C norm OR L1B burst, chosen via the offsets file's Parents, + + l2-calibration ancillary + + l2-{norm,burst}-offsets ancillary, + SPICE) + -> 5 science files for the requested mode + +.. important:: + + **L1D and L2 are siblings, not sequential.** Both consume L1B burst and L1C + normal data directly. L2 does **not** consume L1D. L1D exists purely to get a + usable product out fast. + +CLI wiring +---------- + +**[CODE]** ``class Mag(ProcessInstrument)`` in ``imap_processing/cli.py``, and +``PROCESSING_LEVELS["mag"] = ["l1a", "l1b", "l1c", "l1d", "l2"]`` in +``imap_processing/__init__.py``. + +.. list-table:: + :header-rows: 1 + :widths: 10 44 46 + + * - Level + - Science dependencies + - Ancillary dependencies + * - ``l1a`` + - Exactly one ``mag`` ``l0`` file. + - None. + * - ``l1b`` + - Exactly one ``mag`` ``l1a`` file. + - Exactly one ``l1b-calibration``. + * - ``l1c`` + - One or two ``mag`` ``l1b`` files valid for the start date, plus + optionally the previous day's ``l1c`` file. + - None. + * - ``l1d`` + - All ``mag`` ``l1c`` files plus all ``mag`` ``l1b`` files. + - ``l1d-calibration``. + * - ``l2`` + - Retrieved from the offsets file's ``Parents``, **not** from the + dependency list (falls back to passed-in L1B/L1C with a warning). + - Exactly one ``l2-{norm,burst}-offsets`` **and** one ``l2-calibration``. + +A few CLI behaviours that surprise people: + +* ``day_buffer = start_date + 3 days`` is passed to ``MagAncillaryCombiner`` so + that open-ended calibration files get a synthetic end date at least three days + past the processing day. +* At L2 the descriptor is split on ``-``; the first token + (``norm``/``burst``) selects both the offsets descriptor and the + ``DataMode``. +* After L2, ``Parents`` is rewritten to record the L1 file that was actually + pulled from the offsets file, dropping any passed-in ``imap_mag_l1b_`` / + ``imap_mag_l1c_`` names, so provenance matches the data. +* Every MAG dataset is checked for monotonically increasing epochs (warning + only) and for epochs within +/-24 h of the processing day (hard error). + +Global attributes carried through the pipeline +----------------------------------------------- + +**[CODE]** These are set at L1A and propagated (sometimes lossily) forward. They +are not in the algorithm document but the pipeline depends on them. + +.. list-table:: + :header-rows: 1 + :widths: 28 72 + + * - Attribute + - Meaning and lifetime + * - ``is_mago`` + - ``"True"``/``"False"`` string. L1A -> L1B -> L1C. + * - ``is_active`` + - Whether the sensor was active. L1A -> L1B -> L1C. + * - ``all_vectors_primary`` + - True when the sensor was the PRIMARY sensor in **every** packet. L1D uses + this to decide whether gradiometry may be applied at all. + * - ``vectors_per_second`` + - String of the form ``"ttj2000ns:rate,ttj2000ns:rate"``, recording every + rate change and when it happened. Parsed by + ``constants.vectors_per_second_from_string``. **This is how L1C knows the + expected cadence**, so it is load-bearing, and L1B shifts the embedded + timestamps when it applies the time shift. + * - ``missing_sequences`` + - List of missing CCSDS source sequence counters, or ``"None"``. Empty + arrays are dropped by cdflib, hence the string. + * - ``interpolation_method`` + - Set at L1C to the ``InterpolationFunction`` name used. diff --git a/docs/source/algorithm-code-documentation/mag/ialirt.rst b/docs/source/algorithm-code-documentation/mag/ialirt.rst new file mode 100644 index 0000000000..d337b120ed --- /dev/null +++ b/docs/source/algorithm-code-documentation/mag/ialirt.rst @@ -0,0 +1,280 @@ +.. _mag-ialirt: + +I-ALiRT - Real-Time Space Weather Stream +======================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**Code:** ``imap_processing/ialirt/l0/parse_mag.py`` (the whole algorithm), +``imap_processing/ialirt/l0/mag_l0_ialirt_data.py`` (status bit decoders), +``imap_processing/ialirt/packet_definitions/ialirt_mag.xml``. + +**Document:** section 7.4. + +.. note:: + + The MAG I-ALiRT code does **not** live under ``imap_processing/mag/``. It + lives with the other instruments' I-ALiRT parsers, and imports the pieces it + needs from the MAG modules (``TimeTuple``, ``calibrate_vector``, + ``shift_time``, ``MagL1d.calculate_gradiometry_offsets``, + ``MagL1d.apply_gradiometry_offsets``, ``ValidFrames``). Its output is a + **list of dicts destined for the I-ALiRT database, not a CDF.** + +What it is +---------- + +**[DOC]** IMAP provides near-real-time interplanetary magnetic field +measurements from upstream of the Earth for space weather forecasting. MAG +contributes a fixed **1 vector per sensor every 4 seconds**. The rate is not +configurable. + +The algorithm is: decommutate the I-ALiRT-specific packet format, then reuse the +main science steps to get to an L1D-equivalent product. + +.. code-block:: text + + 1. Decommutate I-ALiRT data (one vector per sensor spread over 4 packets) + 2. Output a raw I-ALiRT data product + 3. Apply L1A steps (gaps, convert sample times to MET/J2000) + 4. Apply L1B steps (ENG calibration, time shift) + 5. Apply L1C steps (gaps) <- the document notes there are none + 6. Apply L1D steps (calibration, truncation, offsets, gradiometry, magnitude) + +Packet format +------------- + +**[DOC]** ``MAG_SCI_IALIRT``: + +.. list-table:: + :header-rows: 1 + :widths: 26 16 16 42 + + * - Mnemonic + - Bits + - Start bit + - Notes + * - CCSDS primary header + - 48 + - 0 + - ``PHVERNO``, ``PHTYPE``, ``PHSHF``, ``PHAPID``, ``PHGROUPF``, + ``PHSEQCNT``, ``PHDLEN`` + * - ``SHCOARSE`` + - 32 + - 48 + - + * - ``ACQ_TM_COARSE`` + - 32 + - 80 + - Science acquisition time, whole seconds + * - ``ACQ_TM_FINE`` + - 16 + - 112 + - Science acquisition sub-second counter + * - ``STATUS`` + - 24 + - 128 + - Mixed status and science; layout depends on the packet ID + * - ``DATA`` + - 24 + - 152 + - Science payload + * - ``CHECKSUM`` + - 16 + - 176 + - + +**One science sample is spread over four sequential packets.** The leading two +bits of ``STATUS`` are a 0-3 packet ID identifying which of the four structures +this packet carries. + +.. code-block:: text + + Primary (MAGo) components: xx, yy, zz + Secondary (MAGi) components: aa, bb, cc + + P1 = xx y + P2 = y zz + P3 = aa b + P4 = b cc + +Timestamps come from packets 0 and 2: + +* Packet ID **0** carries ``ACQ_TM_COARSE``/``ACQ_TM_FINE`` for the + **primary** sensor. +* Packet ID **2** carries them for the **secondary** sensor. + +STATUS field layout +^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Table 7-2. Each packet ID unpacks to a different set of fields. + +.. list-table:: + :header-rows: 1 + :widths: 12 88 + + * - Packet + - Fields (after the 2-bit packet ID and 1 validity bit) + * - 0 + - ``p1V5V``, ``p1V5C``, ``p1V8V``, ``p1V8C`` (2 bits each), ICU temp + (7 bits), MAGo saturation flag, MAGi saturation flag, mode (4 bits) + * - 1 + - ``p2V5V``, ``p2V5C`` (2 bits each), ``p3V3V`` (8 bits), ``p3V3C`` + (9 bits) + * - 2 + - ``p8V5V``, ``p8V5C`` (2 bits each), ``n8V5V`` (8 bits), ``n8V5C`` + (9 bits) + * - 3 + - MAGo temp (8 bits), MAGi temp (8 bits), MAGo range (2 bits), MAGi range + (2 bits), multiple-bit error status + +**[CODE]** These are decoded by ``Packet0`` .. ``Packet3`` in +``mag_l0_ialirt_data.py``. Several 2-bit fields in the document are decoded as +individual warning/danger flags in the code (``hk1v5_warn``, +``hk1v5_danger``, ...), and the 7/8/9-bit engineering values are left-shifted +back into a 16-bit scale (``icu_temp = ((status >> 6) & 0x7F) << 5``, and so on). + +Validity +^^^^^^^^ + +**[DOC]** ``PRI_ISVALID = 1`` only if the validity bit is set in **both** +packets 0 and 1; ``SEC_ISVALID = 1`` only if set in **both** packets 2 and 3. + +**[CODE]** ``pri_isvalid`` is read only from ``Packet1`` bit 21 and +``sec_isvalid`` only from ``Packet3`` bit 21. The document's AND across two +packets is not implemented. See :ref:`mag-implementation-status`. + +Processing +---------- + +**[CODE]** ``process_packet(accumulated_data, engineering_calibration_dataset, +l1d_calibration_dataset)`` runs on roughly one minute of accumulated packets. + +1. Group packets +^^^^^^^^^^^^^^^^ + +``get_pkt_counter`` extracts the 2-bit packet ID from ``mag_status >> 22``. +``find_groups`` collects runs of ``pkt_counter`` 0..3. A group whose counters are +not exactly ``[0, 1, 2, 3]`` is skipped and logged as an "incomplete group". +Groups with both ``pri_isvalid`` and ``sec_isvalid`` zero are dropped. + +.. note:: + + **[DOC]** says a missing packet within a group should NaN only the fields it + would have carried. **[CODE]** drops the whole group. + +2. Extract the vectors +^^^^^^^^^^^^^^^^^^^^^^ + +``extract_magnetic_vectors`` reassembles six 16-bit signed integers from the +four 24-bit ``DATA`` fields using the ``xxy | yzz | aab | bcc`` layout above. + +3. Timestamps +^^^^^^^^^^^^^ + +``get_time`` builds ``TimeTuple(coarse, fine)`` for each sensor and converts with +``to_j2000ns()``, then applies the per-sensor ENG time shift via +``mag.l1b.mag_l1b.shift_time``. This is the L1A + L1B timing step reused +verbatim. + +.. warning:: + + The document says a second is split into **1/65535** fine-time units for + I-ALiRT; ``TimeTuple`` uses ``MAX_FINE_TIME = 65536``. See + :ref:`mag-implementation-status`. + +4. L1B equivalent +^^^^^^^^^^^^^^^^^ + +``calculate_l1b`` uses ``retrieve_matrix_from_single_l1b_calibration`` to pull +``MFOTOURFO``/``OTS`` and ``MFITOURFI``/``ITS`` from a **single** engineering +calibration dataset (no ``MagAncillaryCombiner`` day selection), then calls +``mag.l1b.mag_l1b.calibrate_vector`` with the range from ``fob_range`` / +``fib_range`` in the STATUS field. + +Invalid sensors are filled with ``-32768``. + +5. L1D equivalent +^^^^^^^^^^^^^^^^^ + +``calibrate_and_offset_vectors`` applies ``URFTOORFO``/``URFTOORFI`` and the +per-(sensor, range) ``offsets`` from the L1D calibration file. + +Despinning is then done **without SPICE attitude**, using the alternative the +document explicitly permits: + +* ``sc_spin_phase`` (uint16 -> radians via ``2*pi/65535``), + ``sc_inertial_right`` (``0.0055 deg`` per count) and ``sc_inertial_decline`` + (``0.0027 deg`` per count) come from the **spacecraft** I-ALiRT packet. +* ``interpolate_spherical`` converts RA/Dec into a unit Cartesian vector, + ``CubicSpline`` interpolates each component to the vector's acquisition time + (chosen over linear interpolation because the vector moves along a curved arc + on the unit sphere), then converts back. Spin phase is unwrapped before + ``np.interp`` and rewrapped modulo 360. +* ``transform_instrument_vectors_to_inertial`` produces an **ECLIPJ2000** vector. + +``apply_gradiometry_correction`` then reuses +``MagL1d.calculate_gradiometry_offsets`` and +``MagL1d.apply_gradiometry_offsets`` with the ``gradiometer_factor`` from the +L1D calibration file - applied in ECLIPJ2000 rather than DSRF, since that is the +despun frame this path produces. + +``transform_to_frames`` uses SPICE ``frame_transform`` from ``ECLIPJ2000`` to +``IMAP_GSE``, ``IMAP_GSM`` and ``IMAP_RTN``. + +6. Clock angles +^^^^^^^^^^^^^^^ + +**[DOC]** section 7.4.5, added in Issue 5 Revision 2: + +.. math:: + + \theta_B = \arctan\!\left(\frac{B_z}{\sqrt{B_x^2 + B_y^2}}\right) + +:math:`\theta_B = 0` in the x-y plane, :math:`+90` degrees along :math:`+z`, GSE +coordinates. + +.. math:: + + \varphi_B = \operatorname{atan2}(B_y,\, B_x) + +Angles in the x-y plane, counterclockwise from the Earth-Sun line +(:math:`+x` axis), GSE coordinates. + +**[CODE]** ``cartesian_to_spherical`` returns ``(r, azimuth, elevation)``. The +elevation is ``arcsin(z / |v|)``, which is mathematically identical to the +document's :math:`\theta_B`. The azimuth is ``arctan2(y, x)`` wrapped to +**[0, 360)** degrees, whereas the document's MATLAB ``atan2`` convention gives +**[-180, 180]**. Downstream consumers need to know which they are getting. + +Angles are computed for **both** GSE and GSM. + +Output +------ + +**[CODE]** A list of dicts, one per complete group (the first group is always +skipped because its attitude interpolation is extrapolated): + +.. code-block:: text + + instrument "mag" + mag_epoch TTJ2000 ns of the MAGo vector + mag_B_GSE [Bx, By, Bz] to 3 decimal places + mag_B_GSM [Bx, By, Bz] + mag_B_RTN [Br, Bt, Bn] + mag_B_magnitude + mag_phi_B_GSM azimuth, degrees + mag_theta_B_GSM elevation, degrees + mag_phi_B_GSE azimuth, degrees + mag_theta_B_GSE elevation, degrees + mag_hk_status dict of ~30 decoded STATUS fields + +Values are ``Decimal`` for database storage. + +.. note:: + + **[DOC]** step 2 of section 7.4 asks for a **raw I-ALiRT data product** to be + written, named "MAG I-ALiRT RAW", with headers for the first/last sequence + counter and any sequence gaps. **[CODE]** no raw product is produced; the + pipeline goes straight to the L1D-equivalent dict. Sequence gaps + (``PHSEQCNT``) are not tracked either. See + :ref:`mag-implementation-status`. diff --git a/docs/source/algorithm-code-documentation/mag/implementation-status.rst b/docs/source/algorithm-code-documentation/mag/implementation-status.rst new file mode 100644 index 0000000000..f9b10aa390 --- /dev/null +++ b/docs/source/algorithm-code-documentation/mag/implementation-status.rst @@ -0,0 +1,425 @@ +.. _mag-implementation-status: + +Implementation Status and Known Gaps +==================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This page is the honest accounting of where the code stands against the +algorithm document (SW-009) and, where it overlaps, the public CMAD (see +:ref:`mag-cmad`). **Read it before proposing or estimating work.** + +Accurate as of the most recent survey of ``imap_processing/mag`` and +``imap_processing/ialirt/l0/parse_mag.py``. If you change something material, +update this page in the same commit. + +Summary +------- + +.. list-table:: + :header-rows: 1 + :widths: 14 22 64 + + * - Level + - State + - Notes + * - L0 / L1A + - **Mature** + - Both APIDs, uncompressed and compressed vector paths, all eight + validation cases pass. Sequence-gap bookkeeping is incomplete. + * - L1B + - **Complete but simplified** + - All three algorithmic steps implemented and validated. Only one + calibration is applied per day where the document allows several. + * - L1C + - **Complete, different construction** + - Gap filling and all six interpolation methods work and are validated. + Timeline construction takes a different (defensible) approach to the + document, and the output gap header is missing. + * - L1D + - **Substantially complete** + - All major steps implemented. Missing the per-sensor spin-offset split, + the gradiometer quality flag propagation, and provenance headers. + * - L2 + - **Complete for the nominal path** + - Everything the document asks for except metadata pass-through, the + sensor-attribute cross-check, and provenance headers. Adds GSM. + * - L3 + - **Out of scope** + - MAG has no L3. L2 is the final released product. + * - I-ALiRT + - **Works end to end** + - Produces GSE/GSM/RTN vectors, magnitude and clock angles. No raw product, + simplified validity logic, whole groups dropped on packet loss. + +There are **no** ``NotImplementedError`` raises anywhere in the MAG code, apart +from the generic unknown-data-level branch in ``Mag.do_processing``. + +Suspected bugs +-------------- + +These need confirmation with the MAG team or a test before being changed. They +are ordered by how likely they are to affect released data. + +.. list-table:: + :header-rows: 1 + :widths: 26 74 + + * - Location + - Issue + * - ``mag_l1c.py``, ``vector_magnitude`` + - ``np.linalg.norm(x[:4])`` is applied to the ``direction`` core dimension, + which has **length 4 (x, y, z, range)**. Slicing ``[:4]`` keeps all four, + so **the sensor range is included in the magnitude**. It almost certainly + should be ``x[:3]``. Nothing downstream reads L1C ``vector_magnitude`` + (L1D and L2 recompute it from three components), so the impact is limited + to the L1C product itself. + * - ``mag_l1d_data.py``, ``apply_spin_offsets`` + - The loop is ``for index in range(n_chunks - 1)``. With **exactly one** + spin-offset chunk the loop body never runs, so the x and y components are + left at ``FILLVAL`` while z is copied through. This is reachable on short + or heavily gapped days (fewer than ``2 * number_of_spins`` valid spins). + * - ``mag_l1a_data.py``, ``append_vectors`` + - ``missing_sequences`` is built with + ``range(most_recent + 1, vector_sequence)``, which produces an **empty + range** when the 14-bit CCSDS source sequence counter wraps 16383 -> 0. + Real gaps across a rollover are silently lost. The document explicitly + warns the counter is a rolling 14-bit unsigned integer. + * - ``mag_l1d_data.py``, ``generate_dataset`` + - When ``ALWAYS_OUTPUT_MAGO`` is ``False``, ``vectors``, ``epoch`` and + ``range`` are swapped to MAGi but ``magnitude``, ``quality_flags`` and + ``quality_bitmask`` are not. Since MAGi generally has a different sample + count, this will raise a shape error rather than silently mislabel data - + but it means the MAGi path is untested. + * - ``mag_l1d.py``, gradiometry gate + - ``if not input_mago_norm.attrs.get("all_vectors_primary", 1)``. L1A writes + ``is_mago``/``is_active`` as the **strings** ``"True"``/``"False"`` but + ``all_vectors_primary`` as a **bool**. If that attribute round-trips + through CDF as the string ``"False"``, ``not "False"`` is ``False`` and + gradiometry stays enabled when it should be disabled. Worth pinning down + with a round-trip test. + * - ``mag_l1d_data.py``, ``apply_gradiometry_offsets`` + - ``np.apply_along_axis(np.dot, 1, offsets, K)`` computes + :math:`\mathbf{o}^{T}\mathrm{K}`, i.e. :math:`\mathrm{K}^{T}\mathbf{o}`, + not :math:`\mathrm{K}\mathbf{o}`. Harmless for a symmetric or diagonal + kappa, wrong for a general one. Confirm the intended convention before + any off-diagonal kappa is delivered. + * - ``mag_l1c.py``, ``interpolate_gaps`` + - Range and compression flags for an interpolated sample are taken from + ``burst_vectors[burst_gap_start + index]``, where ``index`` counts + positions in the **output** timeline, not the burst timeline. The two + only coincide when the burst and normal rates are equal. The document + says "all sample properties (such as sensor range) should be interpolated + and output". + +Deviations from the algorithm document +-------------------------------------- + +These are design decisions, not bugs, but they will surprise anyone reading the +document first. + +L1C timeline construction +^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Document** (7.3.4 step 3): find the burst time ``tC`` closest to the last +pre-gap normal time ``tA``, decimate the burst *timestamps* to the normal +cadence such that ``tC`` is included, then subtract ``tA - tC`` from every +decimated time so the bridging series has zero jitter relative to the real +normal-mode samples. + +**Code:** generates a regular grid with ``np.arange(gap_start, gap_end, +1e9 // rate)`` and interpolates burst data onto it. Cross-day phase continuity +is handled instead by the previous-day L1C input. + +The intent is the same and the code's approach arguably produces a cleaner +timeline, but it is not the document's algorithm and the two will not produce +identical timestamps. + +The CIC decimation factor +^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Document:** ``decimation_factor = INPUT_SAMPLES_PER_SECONDS / 2``. + +**Code:** ``decimation_factor = input_rate / output_rate``. + +The document hardcodes a 2 Hz output, correct only for the default ``N_2_2`` +normal mode. **The code is right**; the document's template is a simplification. + +Product naming +^^^^^^^^^^^^^^ + +**Document:** ``imap_mag_l1a_raw_normal_[date]_[version].cdf``. + +**Code:** ``imap_mag_l1a_norm-raw``, following the IMAP filename convention where +the descriptor is one hyphenated token. All code strings are the authority. + +GSM at L2 +^^^^^^^^^ + +The document lists DSRF, SRF, RTN and GSE for L1D and L2, and GSM only for +I-ALiRT. The code additionally produces ``imap_mag_l2_{norm,burst}-gsm``, with +matching global attributes. This is an intentional extension. L1D does **not** +produce GSM. The public CMAD (section 4.1.1) lists the ``gsm`` L2 products +alongside the other frames, so the project documentation now agrees with the +code. + +Fine time divisor +^^^^^^^^^^^^^^^^^ + +``TimeTuple`` uses ``MAX_FINE_TIME = 65536``. Section 7.4.2 of the document says +"One second is equal to **65535** fine time units" for I-ALiRT. The main science +sections do not state a divisor. The difference is ~15 microseconds - well below +the tens-of-milliseconds time shift the pipeline already corrects for, but it is +a real discrepancy that should be settled with the MAG team. + +Not implemented +--------------- + +Sequence-counter provenance headers +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +The document asks for "first and last sequence counter in the data file" and a +list of sequence gaps at **L1A raw, L1A per-sensor, L1D normal mode, and L2 +normal mode**. + +* **L1A:** only the gap list, as the ``missing_sequences`` global attribute, and + it does not handle rollover. First/last counters are absent. +* **L1B:** propagates ``missing_sequences``. +* **L1C:** sets ``missing_sequences`` to ``""``. ``# TODO merge missing + sequences? replace?`` at ``mag_l1c.py:137``. +* **L1D and L2:** pass ``global_attributes={}``, so **no** sequence provenance + reaches either product. + +L1C gap header (the 1.1 second rule) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Document 7.3.4 step 5: walk the completed vector list, treat any spacing greater +than **1.1 seconds** as a gap, and write the gap start/end times as a header in +the output file. Not implemented anywhere. The code's gap detection is a +*relative* 7.5% tolerance used for *input* gap finding, which is a different +thing for a different purpose. + +Multiple ENG calibrations within one day +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Document 7.3.3 step 1: several ENG calibration files may be valid for +non-overlapping ranges within the processing window, and the latest generated +file wins per vector. The code applies one matrix and one time shift for the +whole day. ``# TODO: Check validity of time range for calibration`` at +``mag_l1b.py:73``. + +Per-sensor spin-average offsets +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Document 7.3.5 step 6: "Process each sensor's data independently", and the +outputs list "2 x CDF files for Spin Average offsets in NM for MAGo and MAGi". + +The code computes spin offsets from **MAGo only** (``self.vectors`` inside +``MagL1d``) and applies the same offsets to both sensors, emitting a single +``imap_mag_l1d_spin-offsets`` file. + +Gradiometer quality flag and magnitude at L1D +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Document 7.3.5 step 11: set a quality flag when the gradiometer offset magnitude +exceeds ``quality_flag_threshold``, and **add the gradiometer offset magnitude +as a data point per vector** in the science product. + +The code computes both inside ``calculate_gradiometry_offsets`` and writes them +to the ``gradiometry-offsets`` **ancillary** dataset only. The L1D science +product's ``quality_flags`` remains hardcoded ``np.zeros`` and there is no +per-vector gradiometer magnitude variable. + +L2 metadata pass-through +^^^^^^^^^^^^^^^^^^^^^^^^ + +Document 7.2: the offsets file may carry additional metadata (for instance, +periods when interference was occurring and was removed) and "this metadata +should be copied into the final science file allowing the MAG team to apply +additional metadata based on currently unknown calibration steps". + +The code copies only ``quality_flag`` and ``quality_bitmask``. Any other +metadata in the offsets file is dropped. + +L2 sensor-attribute cross-check +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Document 7.3.6 step 3: "Ensure that the sensor attributes in the offset and +calibration file match MAGo (or vice versa) and fail if they do not match." Not +implemented. There is a ``# TODO Check that the input file matches the offsets +file`` at ``mag_l2.py:97``. Epoch equality **is** enforced, which catches the +most likely mismatch. + +Quality bitmask definitions +^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Bit assignment: resolved.** SW-009 section 7.2 and the ``VAR_NOTES`` on +``qf_bitmask`` in ``imap_mag_l2_variable_attrs.yaml`` used to disagree: SW-009 +lists eight named bits including ``SCTONES`` and ``PIVOTPLATFORMINTERFERENCE``, +with ``SEC_SENS`` last. The public CMAD (section 5.4.5) now defines the bitmask +explicitly, and it **matches the YAML**: + +* Bit 0: data sourced from the secondary sensor +* Bit 1: thruster firing signals removed +* Bit 2: spacecraft interference (TCMs, IMAP-Lo pivot platform motion) +* Bit 3: instrument signals removed +* Bits 4-7: reserved for in-flight calibration + +Treat SW-009's list as superseded. See :ref:`mag-cmad-quality`. + +**Still not done:** MAG does not use ``imap_processing/quality_flags.py``; there +is no ``MagQualityFlags`` enum, and the bitmask is copied through opaquely from +the offsets file. The YAML ``VAR_NOTES`` is the only in-repo definition. + +L2 provenance of Imperial calibration inputs (unconfirmed) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +CMAD ICL-017 section 2 says the Imperial calibration input files (matrix +version, spin-plane and spin-axis offset CSVs, listed in +``calibration_input_release1_v001.json``) "are captured in the parent metadata +field of the released L2 science data files". + +**[CODE]** ``Mag.do_processing`` in ``cli.py`` sets L2 ``Parents`` to the SDC +dependency file names (calibration and offsets files) plus the one L1B/L1C file +actually used. The offsets file's own ``Parents`` is used only to **locate** that +L1 file: ``retrieve_mag_l1_inputs_from_l2_offsets`` downloads every entry and L2 +uses the first. It is never copied to the output. + +Whether this matters depends on what Imperial puts in the offsets file, which +has not been checked against a real delivered file. If the offsets ``Parents`` +lists anything besides the L1 file (such as the calibration-input JSON): + +* those entries do not reach L2 ``Parents``, contrary to the CMAD statement; +* ``retrieve_mag_l1_inputs_from_l2_offsets`` would try to ``download()`` them + from the SDC, and would fail if they are not SDC-hosted files. + +If instead the calibration-input names live only inside the offsets file, the +CMAD statement is satisfied indirectly, because the offsets file is itself a +parent of the L2 product. + +I-ALiRT gaps +^^^^^^^^^^^^ + +.. list-table:: + :header-rows: 1 + :widths: 40 60 + + * - Document requirement (7.4) + - Status + * - Step 2: output a raw I-ALiRT data product named "MAG I-ALiRT RAW" + - **Not implemented.** The pipeline goes straight to the L1D-equivalent + dict. + * - Step 2: use ``PHSEQCNT`` to identify non-sequential packets and record + gap start/end times as headers + - **Not implemented.** + * - Step 3: if any one of the four packets is missing, NaN only the + respective fields + - **Not implemented.** The whole group is dropped and logged. + * - Step 6: ``PRI_ISVALID`` requires the validity bit in **both** packets 0 + and 1; ``SEC_ISVALID`` in both packets 2 and 3 + - **Not implemented.** ``pri_isvalid`` comes from packet 1 only, + ``sec_isvalid`` from packet 3 only. + * - Step 8c: RTN output "maybe, TBC based on discussion with space weather + users" + - **Implemented** - RTN is produced. + +The I-ALiRT path also uses a **single** engineering calibration dataset via +``retrieve_matrix_from_single_l1b_calibration``, with no day selection or +multi-file combining. + +Outstanding TODOs in the code +------------------------------ + +.. list-table:: + :header-rows: 1 + :widths: 40 60 + + * - Location + - TODO + * - ``l0/decom_mag.py:84`` + - "Correct CDF attributes from email" + * - ``l0/decom_mag.py:137`` + - Epoch on the raw product is the **packet** start time; confirm this is + what MAG expects and fix ``CATDESC`` if so. + * - ``l0/decom_mag.py:145`` + - Units for ``raw_vectors`` are undefined. + * - ``l1b/mag_l1b.py:73`` + - Check validity of the time range for the calibration. + * - ``l1c/mag_l1c.py:57`` + - Find missing sequences and output them; handle a missing burst file by + passing the norm file through. + * - ``l1c/mag_l1c.py:572`` + - ``generate_empty_norm_array`` fills with ``np.zeros`` instead of + ``FILLVAL``. Rows that are never filled are dropped by + ``remove_missing_data``, so this is currently benign. + * - ``l1c/mag_l1c.py:665`` + - "we need extra data at the beginning and end of the gap" - the CIC buffer + heuristic may be insufficient at gap edges. + * - ``l1c/mag_l1c.py:842`` + - When falling back to the previous file, also retrieve the expected + vectors per second. + * - ``l1d/mag_l1d.py:137`` + - Frame-specific CDF attributes may be required for L1D. + * - ``l1d/mag_l1d_data.py:725`` + - Should gradiometry extrapolate, or should non-overlapping data be + removed? + * - ``l2/mag_l2.py:63, 97`` + - Retrieve the input file from the offsets dataset in ``cli.py`` (this is + now done); check that the input file matches the offsets file. + +Testing notes +------------- + +* Unit tests: ``imap_processing/tests/mag/test_mag_decom.py``, + ``test_mag_l1a.py``, ``test_mag_l1b.py``, ``test_mag_l1c.py``, + ``test_mag_l1d.py``, ``test_mag_l2.py``. +* Validation tests against MAG-team reference outputs live in + ``test_mag_validation.py`` with data in + ``imap_processing/tests/mag/validation/``: + + .. list-table:: + :header-rows: 1 + :widths: 18 22 60 + + * - Level + - Cases + - Marked ``external_test_data``? + * - L1A + - T001 - T008 + - No - these run by default. + * - L1B + - T009 - T012 + - No. + * - L1C + - T013 - T016, T024 + - **Yes** - excluded from the default selection. + * - L2 + - T021 (burst), T022 (norm) + - **Yes.** + +* There are **no validation cases for L1D** and none for I-ALiRT MAG in the MAG + validation harness. +* L1D and L2 tests need SPICE. Fixtures come from + ``imap_processing/tests/conftest.py`` plus + ``imap_processing/tests/mag/validation/calibration/spice/fake_mag_spin_data.csv``. + +Where to start if you are picking up work +------------------------------------------ + +Roughly in order of value per unit effort: + +1. **Confirm the L1C ``vector_magnitude`` slice** and the ``apply_spin_offsets`` + single-chunk case. Both are small, both are testable, both affect data. +2. **Fix the 14-bit sequence-counter rollover** and add the first/last sequence + counter attributes. This is the most-repeated missing requirement in the + document and it is provenance data that cannot be reconstructed later. +3. **Put the quality bitmask in** ``quality_flags.py`` as a real enum and make + the YAML derive from it. The bit assignment is now settled by CMAD section + 5.4.5 (see :ref:`mag-cmad-quality`). While there, check what a real delivered + offsets file carries in ``Parents`` (see the provenance note above). +4. **Propagate the gradiometer quality flag and magnitude into the L1D science + product.** The values are already computed. +5. **Split spin-average offsets per sensor.** Currently MAGo offsets are applied + to MAGi. +6. **Add L1D validation cases.** L1D is the second-largest body of MAG + algorithm code and has no reference-output tests. diff --git a/docs/source/algorithm-code-documentation/mag/index.rst b/docs/source/algorithm-code-documentation/mag/index.rst new file mode 100644 index 0000000000..0d683ac7a8 --- /dev/null +++ b/docs/source/algorithm-code-documentation/mag/index.rst @@ -0,0 +1,231 @@ +:orphan: + +.. _mag-index: + +MAG +=== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +.. currentmodule:: imap_processing.mag + +This is the MAG (magnetometer) instrument module, which contains the code for +processing data from the MAG instrument. + +Purpose of these pages +---------------------- + +These pages are a **condensed, self-contained working reference** for the MAG +processing algorithms, written so that a developer (human or AI agent) can get +productive without reading the full algorithm document. + +They are a summary of the algorithm document below plus what the code in +``imap_processing/mag`` actually does. Where the two disagree, that is called +out explicitly in :ref:`mag-implementation-status`. The MAG team's upstream +calibration and cleaning, which the public CMAD documents instead, is summarised +in :ref:`mag-cmad`. + +.. _mag-source-documents: + +Source documents +---------------- + +**None of these are redistributed in this repository.** This is an open-source +repository and the mission documents are not ours to publish. Request them from +the MAG instrument team at Imperial College London or the SDC document store. + +.. list-table:: + :header-rows: 1 + :widths: 22 78 + + * - Short name + - Reference + * - **Algorithm document** + - IMAP-MAG-SW-009-01B, *MAG Science Algorithm Document*, Issue 5 + Revision 2, 01 June 2026. Prepared by Alastair Crabtree, approved by + Tim Horbury, Imperial College London. 44 pages. The primary source for + these pages. + + Unlike most instruments' algorithm documents, this one is **not** + embedded in the public CMAD (below). + * - **CMAD** + - *Calibration and Measurement Algorithms Document for NASA's IMAP + Mission*, version 1.1 (preliminary), ``IMAP_CMAD_20260722.pdf``. Public. + For MAG it embeds two Imperial technical notes instead of the algorithm + document: + + * IMAP-OPS-TN-ICL-017, *IMAP MAG Calibration Inputs Description*, + Issue 2, 4 June 2026 (CMAD section 3.5.2, printed pages 152-163); + * IMAP-OPS-TN-ICL-013, *IMAP MAG Data Cleaning Processes*, Issue 4, + 4 June 2026 (CMAD section 4.4, printed pages 912-929). + + Section 5.4.5 (printed pages 1163-1166) holds the authoritative L2 quality + flag and bitmask definitions. All of this is summarised in + :ref:`mag-cmad`. + * - **TLM_MAG** ([RD01]) + - MAG telemetry definition spreadsheet. Superseded in practice by + ``imap_processing/mag/packet_definitions/MAG_SCI_COMBINED.xml``, which + is what the code parses. + * - **MAG TMTC / SDMP** + - Referenced by the algorithm document for exhaustive packet and product + structure. Not needed for any code in this repository. + +.. tip:: + + If you hold a copy of the algorithm document, put it in ``docs/reference/``. + That directory is gitignored, so it will never be committed, and the section + index in :ref:`mag-reference-tables` is written against that location. + Even though the CMAD is public, do not commit it: it is ~128 MB and far over + the repository's file-size limit. + +.. important:: + + Two conventions used throughout these pages: + + * **[DOC]** marks a statement taken from the algorithm document. It describes + the intended behavior, which may not be what the code does yet. + * **[CODE]** marks a statement verified against ``imap_processing/mag``. + + When those conflict, the code is what runs and the document is what the + instrument team expects. Both matter; do not silently "fix" one to match the + other without asking. + +.. note:: + + **This repository stops at L2.** For MAG that is not a limitation - L2 *is* + the final released science product in the algorithm document. There is no MAG + L3. The one thing to be careful about is that MAG's **L1D** looks like a + near-L2 product (it is calibrated, despun and delivered in science frames) + but it is a Level 1 product produced here, on purpose, for other instrument + teams to use before the MAG team's full offset determination is available. + +Which page to read +------------------ + +Read only what you need. Each page is designed to be loaded on its own. + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Page + - Read it when you need to know... + * - :ref:`mag-overview` + - What MAG physically is, the two sensors, ranging, sample rates, science + modes and the reference-frame chain. **Start here if you are new.** + * - :ref:`mag-data-products` + - The full product inventory, exact ``Logical_source`` strings, what feeds + what, and how the CLI is wired. **The "what goes into what" map.** + * - :ref:`mag-l1a` + - Packet decommutation, vector unpacking, and the Fibonacci/zig-zag + decompression algorithm. + * - :ref:`mag-l1b` + - Compression rescaling, the measurement-frame to unit-reference-frame + rotation, and the per-sensor time shift. + * - :ref:`mag-l1c` + - Gap filling: reconstructing a continuous normal-mode timeline from burst + mode data, and the six interpolation methods. + * - :ref:`mag-l1d` + - The rapid near-L2 product: spin-average offsets, gradiometry, despinning + and multi-frame output. + * - :ref:`mag-l2` + - The released science product: MAG-team offsets, calibration matrices, + quality flags and frame transforms. + * - :ref:`mag-ialirt` + - The real-time space-weather stream, which reuses L1A-L1D steps on a + four-packet-per-sample telemetry format. + * - :ref:`mag-ancillary` + - Every calibration and offset file: who delivers it, what variables it + contains, and which level consumes it. + * - :ref:`mag-cmad` + - How the MAG team produces the L2 offsets and matrices (cleaning and + calibration at Imperial), the **authoritative quality bitmask**, and the + artifacts left in released L2. Summarises the public CMAD. + * - :ref:`mag-implementation-status` + - What is implemented, where the code deviates from the document, known + and suspected bugs, and what is not written at all. **Read before + proposing work.** + * - :ref:`mag-reference-tables` + - Where the big tables live (XTCE, CDF attribute YAML, validation data). + Deliberately *not* reproduced inline. + +.. toctree:: + :maxdepth: 1 + + overview + data-products + l1a + l1b + l1c + l1d + l2 + ialirt + ancillary + cmad + implementation-status + reference-tables + +Ten-second orientation +---------------------- + +* MAG is a **conventional dual fluxgate magnetometer**. Two sensors on the + spacecraft boom: **MAGo** (outboard, far from the spacecraft, the primary + science sensor) and **MAGi** (inboard, closer to the spacecraft, used to + characterise and remove spacecraft-generated fields). +* Every measurement is a 3-component vector plus a **range** (0-3) plus a + timestamp. Calibration is different for every (sensor, range) pair, so range + is carried all the way through the pipeline. +* Two science telemetry streams: **normal mode (NM)**, APID 1052, nominally + 2 vectors/s from each sensor for ~23 h/day; and **burst mode (BM)**, + APID 1068, nominally 64 vectors/s from MAGo and 8 from MAGi for ~1 h/day. + Only one is transmitted at a time. +* Processing is **vector-by-vector**, on UTC-day windows with a **30 minute + buffer on each side** (a 25 hour file). The buffer is stripped at L1D and L2. +* Processing chain:: + + CCSDS packets + -> L1A raw + per-sensor per-mode timeseries, measurement frame (MFO/MFI) + -> L1B engineering calibration, unit reference frame (URFO/URFI), nT + -> L1C normal-mode gaps filled from burst mode (norm only) + -> L1D rapid near-L2 quality: SPICE frames, spin offsets, gradiometry + -> L2 released science: MAG-team offsets, SRF/DSRF/RTN/GSE/GSM + +* A separate **I-ALiRT** path takes a fixed 1 vector / 4 s real-time stream all + the way to an L1D-equivalent product for space weather forecasting. + +Where the code lives +-------------------- + +.. code-block:: text + + imap_processing/mag/ + constants.py DataMode, Sensor, VecSec, FIBONACCI_SEQUENCE, tolerances + imap_mag_sdc_configuration_v001.py SDC config: interpolation method, ALWAYS_OUTPUT_MAGO + packet_definitions/MAG_SCI_COMBINED.xml XTCE for APIDs 1052 and 1068 + l0/ + mag_l0_data.py MagL0 dataclass, Mode (APID) IntEnum + decom_mag.py packets -> MagL0 list, raw L1A dataset + l1a/ + mag_l1a.py MagL0 -> MAGo/MAGi datasets + mag_l1a_data.py the big one; vector unpacking + decompression + l1b/ + mag_l1b.py rescale, calibrate to URF, time shift + imap_mag_l1b-calibration_20240229_v002.cdf bundled fallback calibration + l1c/ + mag_l1c.py gap finding, timeline generation, gap filling + interpolation_methods.py 6 interpolation methods + CIC filter + l1d/ + mag_l1d.py orchestration, frame loop, ancillary output + mag_l1d_data.py MagL1d + MagL1dConfiguration; spin/gradiometry + l2/ + mag_l2.py orchestration, calibration matrix selection + mag_l2_data.py MagL2 + MagL2L1dBase + ValidFrames + + imap_processing/ialirt/l0/parse_mag.py the entire I-ALiRT MAG algorithm + imap_processing/ialirt/l0/mag_l0_ialirt_data.py I-ALiRT L0 dataclass + imap_processing/ialirt/packet_definitions/ialirt_mag.xml + + imap_processing/ancillary/ancillary_dataset_combiner.py MagAncillaryCombiner + imap_processing/cdf/config/imap_mag_*.yaml CDF attributes + imap_processing/tests/mag/ tests + validation data + imap_processing/cli.py (class Mag) dependency wiring per level \ No newline at end of file diff --git a/docs/source/algorithm-code-documentation/mag/l1a.rst b/docs/source/algorithm-code-documentation/mag/l1a.rst new file mode 100644 index 0000000000..db46c1b18e --- /dev/null +++ b/docs/source/algorithm-code-documentation/mag/l1a.rst @@ -0,0 +1,301 @@ +.. _mag-l1a: + +L1A - Decommutation and Decompression +===================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**Code:** ``imap_processing/mag/l0/decom_mag.py``, +``imap_processing/mag/l1a/mag_l1a.py``, +``imap_processing/mag/l1a/mag_l1a_data.py`` (the large one). + +**Document:** section 7.3.2 and Appendix 1. + +L1A takes a file of CCSDS packets and produces two things: a near-verbatim +"raw" record of each packet, and a per-vector time series split by physical +sensor. **No calibration and no unit conversion happen here.** + +Inputs and outputs +------------------ + +**[DOC]** Inputs are 25 hours of CCSDS packets (23:30 on the previous day to +00:30 on the next), the MAG telemetry spreadsheet, and the SPICE clock kernel. + +**[CODE]** The CLI passes a single L0 file path. The XTCE at +``mag/packet_definitions/MAG_SCI_COMBINED.xml`` replaces the spreadsheet, and +``imap_processing.spice.time.met_to_ttj2000ns`` replaces manual clock kernel +handling. + +The processing sequence +----------------------- + +1. Split and de-duplicate +^^^^^^^^^^^^^^^^^^^^^^^^^ + +``decom_mag.decom_packets`` iterates the packet file, keeps APIDs 1052 and 1068, +and builds a ``MagL0`` per packet. Packets are collected in a ``dict`` keyed on +the ``MagL0`` object itself, whose hash is ``(SHCOARSE, APID, SRC_SEQ_CTR)`` - +this **silently drops duplicate downlinks**. Returns +``{"norm": [...], "burst": [...]}``. + +2. Export raw +^^^^^^^^^^^^^ + +``decom_mag.generate_dataset`` writes one row per packet: + +* ``epoch`` = ``met_to_ttj2000ns(SHCOARSE)`` - the **packet** time, not a vector + time. +* ``raw_vectors`` = the undecoded byte block, zero-padded so all packets share + the widest length in the file. +* Every other header field becomes its own ``epoch``-dimensioned variable, + except ``SHCOARSE``, ``VECTORS`` and the PUS/spare fields. + +3. Decommutate vectors +^^^^^^^^^^^^^^^^^^^^^^ + +``mag_l1a.process_packets`` loops over packets. Per packet: + +.. code-block:: text + + primary_start_time = TimeTuple(PRI_COARSETM, PRI_FNTM) + secondary_start_time = TimeTuple(SEC_COARSETM, SEC_FNTM) + mago_is_primary = (PRI_SENS == PrimarySensor.MAGO.value) + + seconds_per_packet = PUS_SSUBTYPE + 1 + total_vectors = seconds_per_packet * vectors_per_second + +``MagL1aPacketProperties.__post_init__`` computes ``seconds_per_packet``, +``total_vectors`` and, for compressed packets, ``compression_width`` from the +first six bits of the first vector byte. + +Vectors are unpacked (see below), timestamped by +``MagL1a.calculate_vector_time`` -- first vector at the header time, each +subsequent vector ``+ 1/vectors_per_second`` -- and appended to a ``MagL1a`` +object for MAGo and one for MAGi, chosen by ``mago_is_primary``. + +``MagL1a.append_vectors`` also tracks CCSDS sequence-counter gaps into +``missing_sequences``. + +.. warning:: + + ``TimeTuple`` uses ``MAX_FINE_TIME = 65536`` when converting the fine-time + counter to seconds. The algorithm document says a second is split into + **1/65535** fractions for I-ALiRT (section 7.4.2). The two conventions differ + by one part in 65536 (~15 microseconds). See + :ref:`mag-implementation-status`. + +4. Export per sensor +^^^^^^^^^^^^^^^^^^^^ + +``mag_l1a.generate_dataset`` writes ``vectors`` (``epoch`` x ``direction`` of +size 4: x, y, z, range) and ``compression_flags`` (``epoch`` x ``compression`` of +size 2: is_compressed, compression_width), plus the global attributes described +in :ref:`mag-data-products`. + +Uncompressed vector unpacking +----------------------------- + +``MagL1a.process_uncompressed_vectors``. Each sample is 50 bits - three 16-bit +signed components and a 2-bit unsigned range - and is **not byte aligned**, so +the pattern repeats every 4 vectors (200 bits = 25 bytes). + +The implementation is an explicit four-case ``i % 4`` bit-shuffle over +``uint8``/``int32`` data, written by the MAG instrument team, followed by +``to_signed16``. Vectors ``0 .. primary_count-1`` go to the primary list, the +rest to the secondary list. + +.. tip:: + + Do not "clean up" this function. It is a direct transcription of the + instrument team's reference implementation and is covered by the eight L1A + validation cases (T001-T008) in ``imap_processing/tests/mag/validation/L1a/``. + +.. _mag-compression: + +Compression and decompression +----------------------------- + +**[DOC]** Appendix 1 and section 7.3.2.1. The compression is **lossless** and +applies only to the vector data block. It differs from the uncompressed format +in four ways: + +1. **Deltas**, not absolute vectors. MAG fields are large but vary slowly. +2. **Zig-zag encoding** maps signed deltas onto non-negative integers. +3. **Fibonacci encoding** gives those non-negative integers variable bit widths. +4. **Range data is only included if a range change occurs** within the packet. + +Compressed block layout +^^^^^^^^^^^^^^^^^^^^^^^ + +.. code-block:: text + + [ 6 bits ] COMPRESSION_WIDTH 0-20, width in bits. Usually 16. + [ 1 bit ] HAS_RANGE_DATA_SECTION 1 = per-vector range section present + [ 1 bit ] spare + + Primary vectors section + P1_X, P1_Y, P1_Z COMPRESSION_WIDTH bits each, signed (full vector) + P1_RNG 2 bits unsigned + P2..PN 3 variable-width Fibonacci/zig-zag deltas each + + Secondary vectors section + S1_X, S1_Y, S1_Z, S1_RNG same structure as P1 + S2..SM deltas + + [ 0-7 bits padding to the next byte boundary ] + + Primary range data section (only if HAS_RANGE_DATA_SECTION == 1) + P2_RNG .. PN_RNG 2 bits each, (N - 1) values + Secondary range data section (only if HAS_RANGE_DATA_SECTION == 1) + S2_RNG .. SM_RNG 2 bits each, (M - 1) values + + [ 0-7 bits padding ] + +Expected counts: + +.. code-block:: text + + Expected Primary Vectors = PRIMARY_RATE * (PUS_SSUBTYPE + 1) + Expected Secondary Vectors = SECONDARY_RATE * (PUS_SSUBTYPE + 1) + +where the encoded ``PRI_VECSEC``/``SEC_VECSEC`` values 0-7 map to rates +1, 2, 4, 8, 16, 32, 64, 128. **[CODE]** ``MagL0.__post_init__`` already applies +``2 ** value``. + +If ``HAS_RANGE_DATA_SECTION == 0``, every vector in that section shares the +first vector's range. + +Zig-zag encoding +^^^^^^^^^^^^^^^^ + +Interleaves signed integers onto non-negative ones so that small magnitudes +(regardless of sign) need few bits. + +.. code-block:: text + + encode: C = (n << 1) ^ (n >> 31) # arithmetic shift extracts the sign bit + decode: n = (C >> 1) ^ -(C & 1) + +Fibonacci encoding +^^^^^^^^^^^^^^^^^^ + +By Zeckendorf's theorem, every positive integer is a unique sum of +non-consecutive Fibonacci numbers. Set a bit for each Fibonacci number used +(least significant bit first), then **append a terminal 1**. Because the +representation never has two consecutive set bits, the terminal 1 creates the +only ``0b11`` in the code and makes it self-delimiting. + +Example, encoding 19: :math:`19 = 13 + 5 + 1` -> ``101001`` -> append the +terminator -> ``1101001``. + +Decoding, with the 40-element ``FIBONACCI_SEQUENCE`` in ``mag/constants.py`` +(1, 2, 3, 5, 8, ..., 165580141): + +.. code-block:: python + + def decode_fib_zig_zag(code): # code is an array of bits ending in 1, 1 + code = code[:-1] # drop the terminator + value = sum(FIBONACCI_SEQUENCE[: len(code)] * code) - 1 # Fibonacci decode + return int((value >> 1) ^ (-(value & 1))) # zig-zag decode + +Note the ``- 1``: the encoder biases by one so that zero is representable. + +Worked example from the document +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Raw vectors, ``COMPRESSION_WIDTH = 16``, all in range 3: + +.. code-block:: text + + P1(0,0,0,3) P2(-5,-5,-5,3) P3(-15,-15,-15,3) P4(5,-15,-20,3) + +As deltas: + +.. code-block:: text + + P1(0,0,0,3) P2(-5,-5,-5) P3(-10,-10,-10) P4(+20,0,-5) + +Encoded, with ``-5 = 0b010011`` (6 bits), ``-10 = 0b0101011`` (7 bits), +``+20 = 0b010100011`` (9 bits), ``0 = 0b11`` (2 bits): + +.. code-block:: text + + P1 = 0b00000000000000000000000000000000000000000000000011 (48 bits + 2 range) + P2 = 0b010011010011010011 + P3 = 0b010101101010110101011 + P4 = 0b01010001111010011 + +The high dynamic field escape hatch +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** If a compressed vector's three axes together exceed **60 bits**, +compression has failed because the field is too variable. From that point on, +**every remaining vector in that section is written uncompressed** at +``COMPRESSION_WIDTH`` bits per axis, with no range data. Range data is never in +the vector section after the first vector, compressed or not. + +**[CODE]** ``constants.MAX_COMPRESSED_VECTOR_BITS = 60``. + +How the code implements it +^^^^^^^^^^^^^^^^^^^^^^^^^^ + +``MagL1a.process_compressed_vectors``. The strategy is vectorised rather than a +bit-by-bit loop: + +1. ``np.unpackbits`` the whole block into a bit array. +2. Read the 8-bit header; unpack the first primary vector with + ``unpack_one_vector``. +3. Find **every** ``0b11`` in one shot by summing the bit array with a + ``np.roll`` of itself and looking for the value 2. Those indices are the + candidate Fibonacci terminators. +4. Walk the terminators to build ``primary_boundaries`` and + ``secondary_boundaries`` (three boundaries per vector), watching for the + 60-bit condition to switch into uncompressed handling and for the boundary + between the primary and secondary sections. +5. ``np.split`` on those boundaries, ``decode_fib_zig_zag`` each piece, and + accumulate deltas with ``convert_diffs_to_vectors``. +6. If a range data section is present, ``process_range_data_section`` overwrites + the range of vectors 2..N in each section. + +Supporting helpers worth knowing: + +.. list-table:: + :header-rows: 1 + :widths: 34 66 + + * - Function + - Purpose + * - ``unpack_one_vector(bits, width, has_range)`` + - Unpacks a full (uncompressed) vector. Pads to a byte boundary, + ``np.packbits``, then ``twos_complement`` per axis. + * - ``twos_complement(value, bits)`` + - Sign-extends a big-endian byte array of arbitrary bit width. + * - ``convert_diffs_to_vectors(first, diffs, count)`` + - Cumulative sum of deltas; sets every vector's range to the first + vector's range (later overwritten by the range data section if present). + * - ``process_range_data_section(range_bits, vectors)`` + - Two bits per vector, excluding the first. Raises if the length is wrong. + * - ``_process_vector_section(...)`` + - Shared primary/secondary handling, including the uncompressed tail after + a 60-bit overflow. + +.. warning:: + + ``process_compressed_vectors`` is the most intricate function in the MAG + codebase and its boundary arithmetic (the ``+1``/``-8``/``+7 // 8 * 8`` + adjustments) is tuned against the validation cases. If you change it, run + ``imap_processing/tests/mag/test_mag_validation.py::test_mag_l1a_validation`` + for all of T001-T008 before anything else. + +Sequence gaps +------------- + +**[DOC]** L1A files should carry a header giving the **first and last sequence +counter** in the file, and a header listing any gaps, remembering the counter is +a rolling 14-bit unsigned integer. + +**[CODE]** Only the gap list is implemented, as the ``missing_sequences`` global +attribute, and it does **not** handle the 14-bit rollover: ``append_vectors`` +does ``range(most_recent + 1, vector_sequence)``, which produces an empty range +when the counter wraps from 16383 to 0. First/last sequence counters are not +recorded at all. See :ref:`mag-implementation-status`. diff --git a/docs/source/algorithm-code-documentation/mag/l1b.rst b/docs/source/algorithm-code-documentation/mag/l1b.rst new file mode 100644 index 0000000000..8e0ed891a0 --- /dev/null +++ b/docs/source/algorithm-code-documentation/mag/l1b.rst @@ -0,0 +1,179 @@ +.. _mag-l1b: + +L1B - Engineering Calibration +============================= + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**Code:** ``imap_processing/mag/l1b/mag_l1b.py``. + +**Document:** section 7.3.3. + +L1B is the smallest step in the pipeline. **One L1A product maps to exactly one +L1B product**, and the four sensor/mode combinations are processed +independently. Only the vectors and the timestamps change; every other variable +and attribute is carried through unmodified. + +Inputs +------ + +**[DOC/CODE]** The **engineering (ENG) calibration** file supplies: + +* **8 x 3x3 matrices** - one per (sensor, range) - transforming from the + measurement frames (MFO, MFI) to the unit reference frames (URFO, URFI). +* **2 time shifts** - one per sensor, in seconds. + +The time validity of the calibration file is specified as **pre-timeshift** +values. + +**[CODE]** Variable names in the calibration CDF: + +.. list-table:: + :header-rows: 1 + :widths: 22 24 54 + + * - Variable + - Shape + - Meaning + * - ``MFOTOURFO`` + - ``(epoch, 3, 3, 4)`` + - MAGo MF -> URF, indexed ``[:, :, range]`` after day selection. + * - ``MFITOURFI`` + - ``(epoch, 3, 3, 4)`` + - MAGi MF -> URF. + * - ``OTS`` + - ``(epoch,)`` + - MAGo time shift, seconds. + * - ``ITS`` + - ``(epoch,)`` + - MAGi time shift, seconds. + +The file is combined across multiple ancillary inputs by +``MagAncillaryCombiner`` so that every day in the range has an entry, then +``retrieve_matrix_from_l1b_calibration`` does +``calibration_dataset.sel(epoch=day)``. + +**[CODE]** If ``calibration_dataset`` is ``None``, ``mag_l1b`` falls back to the +bundled ``imap_processing/mag/l1b/imap_mag_l1b-calibration_20240229_v002.cdf`` +and logs "Using default test calibration file." This is a **test convenience**; +the CLI always supplies a real file. + +Processing steps +---------------- + +1. Select the calibration +^^^^^^^^^^^^^^^^^^^^^^^^^ + +Sensor is determined from the ``Logical_source`` string (``"mago"`` / +``"magi"``); a raw L1A file raises ``ValueError``. The output logical source is +the input with ``l1a`` replaced by ``l1b``. + +.. warning:: + + **[DOC]** says the valid ENG calibration may span **multiple files applying + to non-overlapping time ranges within the processing window**, and that when + two files are valid for the same vector the most recently generated one wins. + + **[CODE]** applies exactly one calibration matrix and one time shift to the + whole day, chosen by ``sel(epoch=day)``. There is a + ``# TODO: Check validity of time range for calibration`` at + ``mag_l1b.py:73``. See :ref:`mag-implementation-status`. + +2. Rescale compressed vectors +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +``rescale_vector``. Vectors that came from compressed packets with a +``COMPRESSION_WIDTH`` other than 16 bits must be rescaled to 16-bit-equivalent +engineering values: + +.. math:: + + M = 2^{16 - \text{width}}, \qquad \mathbf{v}' = M \cdot \mathbf{v} + +.. list-table:: + :header-rows: 1 + :widths: 20 20 60 + + * - Width + - M + - Example + * - 14 + - 4 + - + * - 15 + - 2 + - ``[32766, -2, 1]`` -> ``[65532, -4, 2]`` + * - 16 + - 1 + - unchanged + * - 18 + - 1/4 + - ``[10, -2000, 0]`` -> ``[2.5, -500, 0]`` + +The result **must be floating point** - widths above 16 produce fractional +values, widths below 16 can overflow 16-bit integers. The code casts to +``np.float64`` and uses ``np.float_power``. + +Uncompressed vectors (``compression_flags[0] == 0``) are returned unchanged. + +3. Transform to URF +^^^^^^^^^^^^^^^^^^^ + +``calibrate_vector``. For each vector, take the **range** from the fourth +component, select ``calibration_matrix[:, :, range]``, and + +.. math:: + + \mathbf{v}_{URF} = T_{\text{sensor},\,\text{range}} \; \mathbf{v}_{MF} + +Data is now in **nT**. The range component is preserved untouched in position 3. +A non-integer range raises ``ValueError``. + +Steps 2 and 3 are combined in ``update_vector`` and applied with +``xr.apply_ufunc(..., vectorize=True)`` over the ``direction`` and +``compression`` core dimensions. + +4. Apply the time shift +^^^^^^^^^^^^^^^^^^^^^^^ + +``shift_time``. One value per sensor, in seconds, applied to every vector for +the whole validity period: + +.. code-block:: python + + time_shift_ns = np.int64(round(time_shift.item() * 1e9)) + shifted = epoch_times + time_shift_ns + +Positive shifts move times **forward**, negative shifts backward, zero is a +no-op. A time shift with more than one element raises ``ValueError``. + +This can move vectors across the day boundary, which is precisely why the L1 +files carry the 30-minute buffer on each side. + +``timeshift_vectors_per_second`` applies the same shift to the timestamps +embedded in the ``vectors_per_second`` global attribute string, so L1C's cadence +lookup stays aligned with the shifted epochs. + +5. Export +^^^^^^^^^ + +Output variables are ``vectors`` (unchanged shape, now nT + range) and +``compression_flags`` (copied verbatim). Global attributes ``is_mago``, +``is_active``, ``all_vectors_primary``, ``vectors_per_second`` and +``missing_sequences`` are propagated; a missing one is logged at INFO level and +skipped rather than raising. + +.. note:: + + **[DOC]** requires the ENG calibration file to appear in the output CDF's + ``Parents`` header. **[CODE]** this is handled generically by + ``ProcessInstrument.post_processing`` in ``cli.py`` from the dependency list, + not inside ``mag_l1b``. + +Validation +---------- + +``imap_processing/tests/mag/test_mag_validation.py::test_mag_l1b_validation`` +covers cases T009-T012 in +``imap_processing/tests/mag/validation/L1b/``. Each case has an input CSV, a +per-sensor expected output CSV, and (for T012) a bespoke calibration CDF. diff --git a/docs/source/algorithm-code-documentation/mag/l1c.rst b/docs/source/algorithm-code-documentation/mag/l1c.rst new file mode 100644 index 0000000000..9d037bedab --- /dev/null +++ b/docs/source/algorithm-code-documentation/mag/l1c.rst @@ -0,0 +1,283 @@ +.. _mag-l1c: + +L1C - Gap Filling from Burst Mode +================================= + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**Code:** ``imap_processing/mag/l1c/mag_l1c.py``, +``imap_processing/mag/l1c/interpolation_methods.py``. + +**Document:** section 7.3.4. + +Why L1C exists +-------------- + +MAG transmits either normal-mode or burst-mode telemetry, never both. So +whenever the instrument is in burst mode (nominally ~1 hour a day), the +normal-mode stream has a **hole**. The MAG science team wants a **continuous +normal-mode L2 product**, so the ground software synthesises the missing +normal-mode samples by filtering and interpolating burst data. + +L1C is therefore **normal mode only**. Burst data is an input, not an output. +The steps below are applied independently to MAGo and MAGi. + +Inputs +------ + +* L1B normal-mode data for one sensor (may be absent). +* L1B burst-mode data for the same sensor (may be absent). +* **[CODE]** Optionally, **the previous day's L1C file** for the same sensor. +* SDC configuration: ``L1C_INTERPOLATION_METHOD``. + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Available inputs + - Behaviour **[CODE]** + * - norm + burst + - Full ``process_mag_l1c`` path. + * - norm only + - ``fill_normal_data`` - pass through, no interpolation. + * - burst only + - Full path with a synthetic empty normal timeline. A trailing-edge + correction shortens the usable burst range by one output cadence, because + CIC delay compensation eats the end of the filtered series. + * - neither + - ``ValueError``. + +If a second dataset is supplied it must be the opposite mode for the same +sensor, otherwise ``select_datasets`` raises ``RuntimeError``. + +The previous-day input **[CODE]** +--------------------------------- + +Not in the algorithm document, but the document does ask for timestamps that are +"regular and consistent, across boundaries between days". The code implements +this by accepting the previous day's L1C file, delivered by sds-data-manager +orchestration. + +``_get_last_timestamp_and_rate_from_previous_day_in_ns`` takes the last sample +before midnight and the spacing that precedes it, matches that spacing against +the known ``VecSec`` rates, and uses it to seed the phase of any gap that opens +the current day. Without it, a leading gap is simply counted from the window +boundary. + +``_validated_previous_day`` **raises** if the file is not L1C for the same +sensor or has no epochs - a wrong file is treated as an orchestration error, not +something to work around. + +Configuration +------------- + +**[CODE]** ``imap_processing/mag/imap_mag_sdc_configuration_v001.py``: + +.. code-block:: python + + L1C_INTERPOLATION_METHOD = "linear_filtered" + ALWAYS_OUTPUT_MAGO = True + +The document is internally inconsistent here: its input list for section 7.3.4 +says ``L1C_INTERPOLATION_METHOD`` defaults to ``LINEAR``, while step 4 of the +same section says "with Linear Filtered being the default". **The code uses +``linear_filtered``**, which matches the later statement and is the physically +correct choice, since the flight software uses a CIC filter to produce the +telemetered cadence in the first place. + +The document is also explicit that the method is a **software configuration +setting**, not something the MAG team delivers in a file. That is why it lives +in a Python module in this repository rather than in an ancillary CDF. + +Processing steps +---------------- + +1. Mark measured samples +^^^^^^^^^^^^^^^^^^^^^^^^ + +Every sample carries a **generated flag**. ``ModeFlags`` in ``mag/constants.py``: + +.. code-block:: text + + ModeFlags.NORM = 0 directly measured normal-mode sample + ModeFlags.BURST = 1 synthesised from interpolated burst data + ModeFlags.MISSING = -1 placeholder; removed before output + +Internally the timeline is an ``(n, 8)`` float array: + +.. code-block:: text + + column 0 epoch (TTJ2000 ns) + columns 1-4 vector x, y, z, range + column 5 generated flag + columns 6-7 compression flags (is_compressed, compression_width) + +2. Find gaps +^^^^^^^^^^^^ + +``find_all_gaps`` / ``find_gaps``. A gap is a spacing that exceeds the expected +cadence by more than a tolerance: + +.. code-block:: python + + expected_gap = 1 / vectors_per_second * 1e9 # ns + is_gap = (diff - expected_gap) > expected_gap * L1C_TIMESTAMP_GAP_TOLERANCE + +``constants.L1C_TIMESTAMP_GAP_TOLERANCE = 0.075`` (7.5%), which allows for clock +drift: 75, 37.5, 18.75 or 9.375 ms at 1, 2, 4 or 8 Hz respectively. + +The expected cadence comes from the ``vectors_per_second`` global attribute. +``_find_rate_segments`` splits the day into contiguous rate segments, walking +each declared transition **backward** while the observed cadence already matches +the new rate. This stops a delayed Config-mode boundary from producing a +spurious one-sample micro-gap. + +If ``day_to_process`` is supplied, gaps are also added from the start of the +25-hour window to the first sample and from the last sample to the end of the +window. + +Gaps are returned as ``(start, end, vectors_per_second)`` where ``start`` and +``end`` are both real timestamps. + +3. Generate a new timeline +^^^^^^^^^^^^^^^^^^^^^^^^^^ + +``generate_timeline`` / ``generate_missing_timestamps``. For each gap, timestamps +are generated with ``np.arange(start, end, 1e9 // rate)`` at the rate declared +for that gap, in integer nanoseconds. Timestamps that already exist are removed. + +.. important:: + + **This is a deliberate deviation from the document.** + + **[DOC]** section 7.3.4 step 3 describes: find the burst time ``tC`` nearest + to the last pre-gap normal time ``tA``; decimate the burst *timestamps* by + ``M`` to the normal cadence such that ``tC`` is included; then subtract + ``tA - tC`` from every decimated time, so the bridged series has **zero + jitter** relative to the real normal-mode samples. + + **[CODE]** generates a regular grid from the gap start at the declared + cadence and interpolates burst data onto it. The intent (a regular, + jitter-free bridging series) is the same, and the previous-day mechanism + handles phase continuity across day boundaries, but the construction is not + the document's. + + ``generate_missing_timestamps`` raises if gap bounds are not integers - + float64 cannot represent TTJ2000 nanoseconds exactly. + +4. Interpolate burst data into the gaps +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +``interpolate_gaps``. Per gap: + +* Determine the burst rate at the gap start from the burst file's + ``vectors_per_second`` attribute, and the normal rate from the gap tuple. +* Take a **buffer** of extra burst samples on each side: + ``burst_buffer = (2 / norm_rate) * burst_rate`` samples. The CIC filter needs + roughly two normal-mode cadences of extra data because it destroys the + beginning and end of its output. +* Clip the gap timeline to the span where burst data actually exists. +* Call the configured interpolation function with the burst vectors (x, y, z + only), burst epochs, target timestamps, and both rates. +* Write the interpolated vectors, set the generated flag to ``BURST``, and copy + the range and compression flags from burst. + +5. Identify remaining gaps and export +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +``remove_missing_data`` drops every row still flagged ``MISSING``. + +**[DOC]** step 5 additionally requires: walk the completed vector list, treat any +spacing greater than **1.1 seconds** as a gap, and write the list of gap start +and end times as a header in the output file. + +**[CODE]** this is **not implemented**. ``global_attributes["missing_sequences"]`` +is set to ``""``, with a ``# TODO merge missing sequences? replace?`` at +``mag_l1c.py:137``, and the 1.1 second rule appears nowhere. See +:ref:`mag-implementation-status`. + +Output variables: + +* ``vectors`` - (x, y, z, range) +* ``vector_magnitude`` +* ``compression_flags`` + +* ``generated_flag`` - the ``ModeFlags`` value per sample +* Global attribute ``interpolation_method`` + +.. _mag-interpolation: + +Interpolation methods +--------------------- + +**[DOC]** six methods, all implemented in ``interpolation_methods.py`` and +exposed through the ``InterpolationFunction`` enum, which is callable: + +.. list-table:: + :header-rows: 1 + :widths: 26 74 + + * - Method + - Implementation + * - ``linear`` + - ``make_interp_spline(..., k=1)``. The document suggests ``numpy.interp`` + or ``InterpolatedUnivariateSpline(k=1)``. + * - ``quadratic`` + - ``k=2`` + * - ``cubic`` + - ``k=3`` + * - ``linear_filtered`` + - CIC filter, then ``linear``. **Default.** + * - ``quadratic_filtered`` + - CIC filter, then ``quadratic`` + * - ``cubic_filtered`` + - CIC filter, then ``cubic`` + +``remove_invalid_output_timestamps`` clips output timestamps to the span of the +input timestamps unless ``extrapolate=True``, so the pipeline never invents +science data outside the burst timeline. + +The CIC filter +^^^^^^^^^^^^^^ + +**[DOC]** A cascaded integrator-comb filter is what the MAG flight software uses +to decimate 1920 raw samples/second down to the telemetered rate, so replicating +it on the ground makes a synthesised normal-mode sample match what the +instrument would have produced. A CIC filter depends on the ratio of input to +output sample rate and introduces a **constant time delay**. + +The document's reference implementation: + +.. code-block:: python + + decimation_factor = INPUT_SAMPLES_PER_SECONDS / 2 + CIC1 = ones(decimation_factor) / decimation_factor + CIC2 = convolve(CIC1, CIC1) + delay = (len(CIC2) - 1) // 2 + + S_filtered = S[:-delay] + A_filtered = lfilter(CIC2, 1, A, axis=0)[delay:] + B = IUS(S_filtered, A_filtered, k=1)(T) + +**[CODE]** ``cic_filter`` generalises this correctly: + +.. code-block:: python + + decimation_factor = int(input_rate.value / output_rate.value) + +The document's ``/ 2`` hardcodes a 2 Hz output, which is only right for the +default ``N_2_2`` normal mode. The code is right for ``N_4_1`` and ``N_4_4`` too. +``cic_filter`` raises ``ValueError`` if the burst input rate is not strictly +greater than the normal output rate. + +``estimate_rate`` infers a rate from timestamp spacing by snapping to the nearest +value in ``POSSIBLE_RATES``, used only when a rate is not supplied. + +Validation +---------- + +``test_mag_validation.py::test_mag_l1c_validation`` covers T013, T014, T015, +T016 and T024 for both sensors in +``imap_processing/tests/mag/validation/L1c/``. These are marked +``@pytest.mark.external_test_data``, so they are excluded from the default test +selection. diff --git a/docs/source/algorithm-code-documentation/mag/l1d.rst b/docs/source/algorithm-code-documentation/mag/l1d.rst new file mode 100644 index 0000000000..b947be928c --- /dev/null +++ b/docs/source/algorithm-code-documentation/mag/l1d.rst @@ -0,0 +1,286 @@ +.. _mag-l1d: + +L1D - Rapid Near-L2 Product +=========================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**Code:** ``imap_processing/mag/l1d/mag_l1d.py``, +``imap_processing/mag/l1d/mag_l1d_data.py``, with the shared machinery in +``imap_processing/mag/l2/mag_l2_data.py``. + +**Document:** section 7.3.5. *Note: L1D was formerly called L2Pre, and the +workflow diagram in the document still labels it L2P.* + +Why L1D exists +-------------- + +The MAG team's L2 offsets are determined **after the fact** from a full day of +data using the Leinweber method, and are delivered as a hand-checked ancillary +file. That takes time. Other instrument teams need a magnetic field product +*quickly*. + +L1D produces a near-L2-quality product using only information available at +processing time: predicted per-range offsets from a slowly-varying calibration +file, plus two offset-estimation techniques computed on the fly - **spin +averaging** and **gradiometry**. It uses more processing steps than L2, not +fewer. + +.. important:: + + L1D is a **Level 1** product, produced in this repository, on purpose. It is + not a stand-in for L2 and it is not a partial L2. L2 does **not** consume it. + +Inputs +------ + +**[DOC/CODE]** + +.. list-table:: + :header-rows: 1 + :widths: 34 66 + + * - Input + - Notes + * - L1C normal-mode MAGo and MAGi + - **Required.** ``mag_l1d`` raises ``ValueError`` without both. + * - L1B burst-mode MAGo and MAGi + - Optional. Both or neither; burst output is skipped if absent. + * - ``l1d-calibration`` ancillary + - See :ref:`mag-ancillary`. + * - SPICE kernels + - For the SRF/DSRF/GSE/RTN transforms and the spin data. + * - SDC config ``ALWAYS_OUTPUT_MAGO`` + - Default ``True``. + +Datasets are routed by the last token of ``Logical_source`` +(``norm-magi``, ``norm-mago``, ``burst-magi``, ``burst-mago``); anything else +raises. + +Configuration read from the calibration file +-------------------------------------------- + +**[CODE]** ``MagL1dConfiguration`` (``mag_l1d_data.py``) selects the day with +``sel(epoch=day)`` and exposes: + +.. list-table:: + :header-rows: 1 + :widths: 34 16 50 + + * - Attribute + - Shape + - Meaning + * - ``mago_calibration`` + - ``(3, 3, 4)`` + - ``URFTOORFO`` - URF -> ORF per range, MAGo. + * - ``magi_calibration`` + - ``(3, 3, 4)`` + - ``URFTOORFI`` - URF -> ORF per range, MAGi. + * - ``calibration_offsets`` + - ``(2, 4, 3)`` + - ``offsets``, indexed ``[sensor, range, axis]`` where sensor + ``0 = MAGo``, ``1 = MAGi``. In nT, in ORF. + * - ``spin_count_calibration`` + - scalar + - ``number_of_spins`` - spins per averaging chunk. Nominally **240** + (~1 hour). + * - ``spin_average_application_factor`` + - scalar + - How much of the computed spin offset to apply, in ``[-1, 1]``. + * - ``quality_flag_threshold`` + - scalar + - Gradiometer offset magnitude above which data is flagged. + * - ``gradiometer_factor`` + - ``(3, 3)`` + - Kappa. **A matrix, not a scalar** - Issue 5 Revision 1 of the document + changed this specifically so each axis can be treated independently and + a small rotation of the gradiated offset can be absorbed. + * - ``apply_gradiometry`` + - bool + - Set ``False`` by ``mag_l1d`` when the MAGo L1C file's + ``all_vectors_primary`` attribute is falsy. + +Processing steps +---------------- + +**[CODE]** All of this happens in ``MagL1d.__post_init__``, so constructing the +dataclass *is* running the algorithm. + +.. code-block:: text + + frame = MAGO + truncate_to_24h(day) # strip the 30-minute buffers + calibrate + apply per-range offsets # URF -> ORF + rotate_frame(SRF) # SPICE + calculate_spin_offsets() # NORM only; reused for BURST + apply_spin_offsets() # to MAGo and MAGi + rotate_frame(DSRF) # SPICE despin + calculate_gradiometry_offsets() # if apply_gradiometry + apply_gradiometry_offsets() + magnitude = |B| + +1. Truncate to 24 hours +^^^^^^^^^^^^^^^^^^^^^^^ + +``MagL2L1dBase.truncate_to_24h``. Removes the 30-minute buffers. Raises if +nothing remains. + +2. Calibration and per-range offsets +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +``_calibrate_and_offset_vectors``. Range is re-attached as a fourth component, +``MagL2L1dBase.apply_calibration`` applies the per-range 3x3 matrix (URF -> +ORF), then ``apply_calibration_offset_single_vector`` **adds** +``offsets[sensor, range, :]``. Applied to both sensors. + +Quality flags for the L1D science product are hardcoded to ``0``, which the +document asks for explicitly so that L1D and L2 have the same shape. + +3. Rotate to SRF +^^^^^^^^^^^^^^^^ + +``MagL1d.rotate_frame`` overrides the base implementation to rotate **both** +sensors. It is careful about one thing: ``self.frame`` always describes the MAGo +data, so when the starting frame is ``MAGO``, the MAGi vectors are rotated from +``MAGI`` instead. Setting ``self.frame = MAGI`` raises. + +FILLVAL and NaN entries survive rotation - they are masked back to FILLVAL after +``frame_transform``. ``allow_spice_noframeconnect=True`` lets the transform +degrade gracefully when a frame chain is unavailable. + +.. _mag-spin-averaging: + +4. Spin-average offsets +^^^^^^^^^^^^^^^^^^^^^^^ + +``calculate_spin_offsets``. **Only meaningful in SRF and only on normal-mode +data.** + +The physical idea: in a spin-aligned frame the spacecraft-generated field is +static in the two spin-plane axes while the ambient field rotates, so averaging +over a whole number of spins leaves the spacecraft contribution behind. The +third (spin-aligned) axis gets no such benefit and is left alone. + +Implementation: + +* ``spice.spin.get_spacecraft_spin_phase(met)`` gives per-vector spin phase. + Vectors where the phase is NaN are marked NaN. +* Spin starts are where the phase decreases (wrap through zero) **plus** every + NaN-to-number transition, so gaps do not merge two spins. +* For a NaN gap longer than one median spin period (from + ``spice.spin.get_spin_data()``), synthetic spin starts are inserted so the + spin count stays honest across the gap. +* Spins are grouped into chunks of ``spin_count_calibration``. The mean of the + x and y components is taken per chunk with ``np.nanmean``. +* A chunk is **rejected** and the previous chunk's averages reused if more than + half of x or y is NaN, or if it contains fewer than half the expected samples. + +The result is an ``xr.Dataset`` with ``epoch``, ``x_offset``, ``y_offset``, +``validity_start_time``, ``validity_end_time``, ``start_spin_counter``, +``end_spin_counter`` - written out as the ``imap_mag_l1d_spin-offsets`` +ancillary product. + +5. Apply spin offsets +^^^^^^^^^^^^^^^^^^^^^ + +``apply_spin_offsets``. For each chunk interval, subtract +``offset * spin_average_application_factor`` from the x and y components of +every vector in that interval; z is copied through unchanged. The first chunk +catches everything before its own start time and the last chunk catches +everything after, so no vector is left unassigned. + +The same offsets object computed from normal mode is passed into the burst-mode +``MagL1d``, which is why ``mag_l1d`` constructs norm first. + +6. Despin +^^^^^^^^^ + +``rotate_frame(ValidFrames.DSRF)`` - SPICE ``IMAP_DPS``. + +.. _mag-gradiometry: + +7. Gradiometry +^^^^^^^^^^^^^^ + +**[DOC]** section 7.2.1. Far from any magnetic source the field looks like a +dipole, so a sensor further from a time-varying source sees a smaller variation. +MAGi is closer to every spacecraft source than MAGo, so the difference between +them estimates the spacecraft contribution: + +.. math:: + + \mathbf{B}_O(t) \;\leftarrow\; \mathbf{B}_O(t) - \mathrm{K}\, + \bigl(\mathbf{B}_I(t) - \mathbf{B}_O(t)\bigr) + +MAGi and MAGo are not sampled simultaneously and MAGo usually has the higher +cadence, so **MAGi must be interpolated onto the MAGo timeline first**. + +It is possible that a value of :math:`\mathrm{K} = 0` will be used in flight, +which disables the correction without a code change. + +**[CODE]** ``calculate_gradiometry_offsets``: + +* Uses ``interpolation_methods.linear`` with ``extrapolate=True`` to put MAGi on + the MAGo epochs. There is a ``# TODO: should this extrapolate or should + non-overlapping data be removed?`` at ``mag_l1d_data.py:725``. +* ``diff = aligned_magi - mago`` per axis. +* Records the magnitude of ``diff`` and a quality flag + ``magnitude > quality_flag_threshold``. +* Emits the ``imap_mag_l1d_gradiometry-offsets-{norm,burst}`` ancillary + datasets with ``gradiometer_offsets``, ``gradiometer_offset_magnitude`` and + ``quality_flags``. + +``apply_gradiometry_offsets`` computes ``np.dot(offset, K)`` per vector and +subtracts it. + +.. warning:: + + ``np.apply_along_axis(np.dot, 1, offset_value, gradiometer_factor)`` computes + the **row-vector product** :math:`\mathbf{o}^{T} \mathrm{K}`, i.e. + :math:`\mathrm{K}^{T}\mathbf{o}`, not :math:`\mathrm{K}\mathbf{o}`. If + ``gradiometer_factor`` is symmetric or diagonal this is immaterial, but the + convention needs confirming with the MAG team before any off-diagonal kappa + is delivered. See :ref:`mag-implementation-status`. + +Gradiometry is skipped entirely when MAGo was not the primary sensor for all +vectors - the document requires this, and ``mag_l1d`` enforces it via +``all_vectors_primary``. + +8. Magnitude and output +^^^^^^^^^^^^^^^^^^^^^^^ + +``magnitude = np.linalg.norm(vectors, axis=1)`` over the three components. + +``mag_l1d`` then walks the frames, calling ``rotate_frame`` then +``generate_dataset`` for SRF, DSRF, GSE and RTN in that order. **This mutates +the dataclass in place**, so each rotation starts from the previous frame and +the order is not arbitrary. + +``MagL1d.generate_dataset`` overrides the base method to swap in the MAGi +vectors, epochs and ranges when ``ALWAYS_OUTPUT_MAGO`` is ``False``, then +restores them. + +Output variables per science file: ``b_srf`` / ``b_dsrf`` / ``b_gse`` / +``b_rtn`` (from ``ValidFrames.var_name``), ``quality_flags``, +``quality_bitmask``, ``range``, ``magnitude``. Range is a **separate time +series**, not a fourth vector component, as the document requires. + +Outputs +------- + +**[DOC]** expects: + +* 2 spin-average offset CDFs (MAGo and MAGi), normal mode +* 2 gradiometer offset CDFs (normal and burst) +* L1D burst in DSRF, SRF, RTN, GSE +* L1D normal in DSRF, SRF, RTN, GSE + +**[CODE]** produces all of these except that **only one spin-offsets file is +produced**, computed from MAGo and applied to both sensors. See +:ref:`mag-implementation-status`. + +Ancillary datasets are written by ``Mag.post_processing`` in ``cli.py``, which +intercepts the three ancillary ``Logical_source`` values, generates an +``AncillaryFilePath`` filename, and calls ``xarray_to_cdf`` with +``istp=False`` and ``terminate_on_warning=False``. Failures are logged and +swallowed - an ancillary file will never fail the run. diff --git a/docs/source/algorithm-code-documentation/mag/l2.rst b/docs/source/algorithm-code-documentation/mag/l2.rst new file mode 100644 index 0000000000..783c80dfd9 --- /dev/null +++ b/docs/source/algorithm-code-documentation/mag/l2.rst @@ -0,0 +1,277 @@ +.. _mag-l2: + +L2 - Released Science Product +============================= + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**Code:** ``imap_processing/mag/l2/mag_l2.py``, +``imap_processing/mag/l2/mag_l2_data.py``. + +**Document:** section 7.3.6. + +L2 is the **final MAG product**. There is no MAG L3, so nothing downstream of +this repository refines it further. + +The idea is simple: take the L1B burst or L1C normal vectors, apply the MAG +team's per-vector offsets and time deltas, rotate into the science frames, and +publish. All the difficult work - removing spacecraft and instrument fields, +thruster interference and the IMAP-Lo pivot platform, and determining the +spin-plane (Kepko) and spin-axis (Leinweber) offsets - happens at Imperial +College London and arrives as numbers in an ancillary file. That upstream work is +described in the public CMAD and summarised in :ref:`mag-cmad`. + +Inputs +------ + +**[DOC/CODE]** Four input sources: + +.. list-table:: + :header-rows: 1 + :widths: 28 72 + + * - Input + - Notes + * - ``l2-calibration`` + - Slowly varying URF -> ORF matrices, one 3x3 per (sensor, range). + ``URFTOORFO`` and ``URFTOORFI``. May span multiple files; combined by + ``MagAncillaryCombiner``. Highest version wins for any given day. + * - ``l2-{norm,burst}-offsets`` + - **One hand-created file that must correspond exactly to one L1B or L1C + science file.** Per-vector offsets, time deltas, quality flag and quality + bitmask. + * - L1B burst or L1C normal science data + - **Retrieved from the offsets file's** ``Parents`` **attribute**, not from + the dependency list. + * - SDC config ``ALWAYS_OUTPUT_MAGO`` + - Default ``True``. + +.. important:: + + **The offsets file drives the input selection.** ``cli.py`` calls + ``retrieve_mag_l1_inputs_from_l2_offsets`` (in ``imap_processing/utils.py``), + which reads ``Parents`` from the offsets CDF and downloads exactly those L1 + files. Any L1B/L1C dependency passed in on the command line is ignored. If + the offsets file has no ``Parents``, the code logs a warning and falls back + to the passed-in dependency. + + After processing, ``Parents`` on the output datasets is rewritten to name the + L1 file that was actually used, so provenance always matches the data. + +The offsets file +---------------- + +**[DOC]** Generated **daily** by the MAG team. Contains, for every timestamped +vector in the corresponding L1 file: + +* A **3x1 offset** ``H`` in nT to be removed from the data. May be NaN when a + good offset cannot be determined, in which case **the corresponding L2 data + point must also become NaN**. In the delivered CDFs, invalid values use + ``FILLVAL`` outside ``VALIDMIN``/``VALIDMAX``. +* A **quality flag** to be copied into the L2 CDF. +* A **quality bitmask** to be copied into the L2 CDF. +* A **Delta-T** in +/- milliseconds adjusting the vector timestamp. +* Possibly additional metadata describing how the calibration was determined - + interference removal, and so on. This metadata is to be copied into the final + science file so the MAG team can annotate products with calibration steps that + are not yet defined. + +**[CODE]** Variables read: ``offsets``, ``timedeltas``, ``quality_flag``, +``quality_bitmask``, ``epoch``. + +.. _mag-quality-flags: + +Quality flags +------------- + +Two variables travel together. Both are produced by the MAG team and arrive in +the offsets file. + +``quality_flag`` is a single value: ``0`` for good data, ``1`` for bad data not +suitable for science. **[CMAD]** A raised flag is always accompanied by a +non-zero bitmask. + +``quality_bitmask`` is a bitfield. **[CMAD]** section 5.4.5 is the authoritative +definition (bit 0 is the least significant): + +.. list-table:: + :header-rows: 1 + :widths: 12 88 + + * - Bit + - Meaning + * - 0 + - Data is sourced from the secondary sensor. + * - 1 + - Thruster firing signals have been removed (daily firing around 10:00 UT + and the pre-thruster activity an hour earlier). + * - 2 + - Spacecraft interference impacts these data (TCMs, IMAP-Lo pivot platform + motion). + * - 3 + - Instrument signals have been removed. + * - 4-7 + - Reserved for in-flight calibration. + +This matches the ``VAR_NOTES`` on ``qf_bitmask`` in +``imap_processing/cdf/config/imap_mag_l2_variable_attrs.yaml``. See +:ref:`mag-cmad-quality` for what triggers each bit. + +.. note:: + + **[DOC]** SW-009 section 7.2 has an older, different list: eight named bits + (``THRUSTERINTERFERENCE``, ``SCINTERFERENCE``, ``SCTONES``, + ``INSTRUMENTINTERFERENCE``, ``PIVOTPLATFORMINTERFERENCE``, ``RESERVE6``, + ``RESERVE7``, ``SEC_SENS``) with the secondary-sensor bit last. **It is + superseded by the CMAD.** + +.. warning:: + + **[CODE]** MAG does **not** use ``imap_processing/quality_flags.py``. There is + no ``MagQualityFlags`` enum; the bitmask is copied through opaquely from the + offsets file as a ``uint16``. See :ref:`mag-implementation-status`. + +Processing steps +---------------- + +1. Select calibration +^^^^^^^^^^^^^^^^^^^^^ + +``retrieve_matrix_from_l2_calibration(calibration_dataset, day, use_mago)`` +selects ``URFTOORFO`` or ``URFTOORFI`` for the day. + +2. Select and validate offsets +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** ``mag_l2`` raises ``ValueError`` unless +``np.array_equal(input_data["epoch"], offsets_dataset["epoch"])``. This is the +document's "fail if these do not match" requirement, enforced as strict epoch +equality rather than a filename comparison. + +.. note:: + + The document also requires checking that the **sensor attributes** in the + offsets and calibration files match the sensor being produced, and failing if + they do not. This is not implemented; there is a + ``# TODO Check that the input file matches the offsets file`` at + ``mag_l2.py:97``. + +3. Select the primary vectors +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +``ALWAYS_OUTPUT_MAGO`` decides whether MAGo or MAGi is produced. It is ``True`` +and is expected to stay that way unless a sensor fails. + +.. warning:: + + The docstring on ``mag_l2`` notes that flipping this to ``False`` also + requires updating the dependency system so MAGi files become an upstream + dependency. It is not a one-line change. + +4. Apply the calibration matrix +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +``MagL2.apply_calibration`` reuses ``mag_l1b.calibrate_vector`` per vector, +selecting the 3x3 matrix by the range in component 3. The range component is +then dropped from the vectors and carried as a separate ``range`` time series. + +5. Apply offsets, time deltas and flags +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +``MagL2.__post_init__``: + +* ``apply_offsets`` **adds** the offset vector. Any vector whose offset contains + ``FILLVAL`` in any axis has **all three components** set to ``FILLVAL`` - + this is the code's realisation of the document's "if the offset is NaN, the + vector becomes NaN". +* ``shift_timestamps`` adds ``timedeltas * 1e9`` nanoseconds to the epochs. + Lengths must match or it raises. +* ``quality_flags`` and ``quality_bitmask`` are copied straight from the offsets + file into the output. + +``constants.FILLVAL = -1e31``, matching the ``FILLVAL`` in the vector CDF +attributes. + +6. Truncate to 24 hours +^^^^^^^^^^^^^^^^^^^^^^^ + +``truncate_to_24h`` removes the 30-minute buffers. **This must happen after the +offsets and time deltas are applied**, since a time delta can move a vector +across the boundary. ``mag_l2`` calls it explicitly, and +``generate_dataset`` calls it again per frame (harmlessly, since the second call +is a no-op once the data is already trimmed). + +7. Magnitude +^^^^^^^^^^^^ + +``calculate_magnitude`` - ``np.linalg.norm(vectors, axis=1)`` over the three +components. + +8. Transform to output frames +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +``rotate_frame`` per frame in ``DEFAULT_L2_FRAMES``: + +.. code-block:: python + + DEFAULT_L2_FRAMES = [ + ValidFrames.SRF, + ValidFrames.GSE, + ValidFrames.GSM, + ValidFrames.RTN, + ValidFrames.DSRF, # last: some vectors may become NaN + ] + +``rotate_frame`` mutates the dataclass in place, so each rotation chains from the +previous frame. FILLVAL and NaN values are re-masked after each transform. + +.. note:: + + **[DOC]** lists DSRF, SRF, RTN and GSE for L2. **[CODE]** additionally + produces **GSM**, which the document only mentions for I-ALiRT. Global + attributes exist for ``imap_mag_l2_{norm,burst}-gsm``, so this is an + intentional extension, not an accident. + +9. Output +^^^^^^^^^ + +Per frame: ``b_srf`` / ``b_gse`` / ``b_gsm`` / ``b_rtn`` / ``b_dsrf``, +``quality_flags``, ``quality_bitmask`` (as ``uint16``), ``range``, +``magnitude``. ``direction_label`` is ``["B_R", "B_T", "B_N"]`` for RTN and +``["Bx", "By", "Bz"]`` otherwise. + +.. note:: + + **[DOC]** step 9 also requires the normal-mode stream to carry a header + listing the first and last vector packet sequence count and any missing + sequence counters. **[CODE]** ``mag_l2`` passes ``global_attributes={}`` into + ``MagL2``, so no sequence-counter provenance reaches L2 at all. See + :ref:`mag-implementation-status`. + +Shared machinery +---------------- + +``MagL2L1dBase`` in ``mag_l2_data.py`` is the common base for ``MagL2`` and +``MagL1d``, because the two levels produce structurally identical files. It +owns: + +* ``generate_dataset`` - builds the ``xr.Dataset``, including calling + ``truncate_to_24h`` and constructing + ``imap_mag_{level}_{mode}-{frame}`` as the logical source ID. +* ``truncate_to_24h`` +* ``calculate_magnitude`` +* ``apply_calibration`` +* ``shift_timestamps`` +* ``rotate_frame`` +* ``ValidFrames`` - the frame enum described in :ref:`mag-overview`. + +Its docstring notes it "may also be extended for I-ALiRT"; today the I-ALiRT +path only borrows ``ValidFrames`` and a few static methods. + +Validation +---------- + +``test_mag_validation.py::test_mag_l2_validation`` covers T021 (burst) and T022 +(norm) in ``imap_processing/tests/mag/validation/L2/``, using real calibration +and offsets CDFs and an expected-output CSV. Marked +``@pytest.mark.external_test_data``. diff --git a/docs/source/algorithm-code-documentation/mag/overview.rst b/docs/source/algorithm-code-documentation/mag/overview.rst new file mode 100644 index 0000000000..c6aece5cd1 --- /dev/null +++ b/docs/source/algorithm-code-documentation/mag/overview.rst @@ -0,0 +1,363 @@ +.. _mag-overview: + +Instrument and Mission Concepts +=============================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +Everything on this page is background needed to read the algorithm pages. It is +mostly **[DOC]** (algorithm document sections 4 and 7.1). + +What MAG measures +----------------- + +MAG is a **conventional dual fluxgate magnetometer**. It measures the vector +interplanetary magnetic field at the spacecraft, continuously, from the L1 +Lagrangian point. The magnetic field underpins nearly every other IMAP +measurement: it carries energy, supports waves and turbulence, and controls the +propagation of charged particles, so MAG data is used by SWAPI, SWE, CoDICE and +HIT analyses as well as on its own. + +MAG also feeds the **I-ALiRT** near-real-time stream used for space weather +forecasting. + +Hardware +-------- + +.. list-table:: + :header-rows: 1 + :widths: 16 84 + + * - Element + - What it is + * - **MAGo** + - The **outboard** sensor - furthest from the spacecraft body on the boom, + so it sees the least spacecraft-generated field. **The primary source of + released science data.** + * - **MAGi** + - The **inboard** sensor - closer to the spacecraft. Its value is mostly in + *characterising* the spacecraft field so it can be removed from MAGo + (see :ref:`gradiometry `). + * - **FOB** + - Front End Electronics board connected to MAGo. Appears in telemetry field + names (``FOB_TEMP``, ``FOB_RANGE``, ``FOB_SATURATED``) - read "FOB" as + "MAGo". + * - **FIB** + - Front End Electronics board connected to MAGi. Read "FIB" as "MAGi". + * - **ICU** + - Instrument Control Unit. Runs MAG boot software (BSW) and application + software (ASW); generates all telemetry. + * - **PCU / ELB** + - Power Control Unit; the Electronics Box housing ICU, FOB, FIB and PCU on + the spacecraft platform (not on the boom). + +Ranging +------- + +MAG must cover sub-nT solar wind fields and the full Earth field (~40,000 nT) +for ground testing. Each sensor **autonomously changes range**. **[DOC]** + +.. list-table:: + :header-rows: 1 + :widths: 10 24 24 42 + + * - Range + - Approx. coverage + - Approx. resolution + - Notes + * - 0 + - +/-60,000 nT + - 256 pT + - Covers the Earth field; used for ground testing and at instrument start. + * - 1 + - +/-2,048 nT + - 64 pT + - + * - 2 + - +/-512 nT + - 16 pT + - Where MAGi may sit if spacecraft fields are large; MAGo may enter this + range during a large CME. + * - 3 + - +/-128 nT + - 4 pT + - Most sensitive. **Nominal range for MAGo at all times.** + +.. important:: + + **Every calibration parameter - offsets, gains, alignment - is different for + each sensor AND each range.** This is why every calibration input in this + pipeline is shaped ``(3, 3, 4)`` or ``(2, 4, 3)``: the trailing/leading axis + is range 0-3. The range is carried as the fourth component of the L1A/L1B/L1C + ``vectors`` array and as a separate ``range`` time series at L1D/L2. + + The two sensors range **independently**, and a range change can happen in the + middle of a science packet - which is why the compression format has an + optional per-vector range data section (see :ref:`mag-l1a`). + +Cadence, science modes and operating modes +------------------------------------------ + +Sample rate is independently configurable per sensor. Permitted rates are +**1, 2, 4, 8, 16, 32, 64, 128 vectors/second** (``VecSec`` in +``mag/constants.py``). Packet duration is configurable from 1 to 256 seconds, +limited to 512 vectors per packet. **[DOC]** + +.. list-table:: MAG science modes **[DOC]** + :header-rows: 1 + :widths: 20 16 16 20 28 + + * - Mode + - Primary vec/s + - Secondary vec/s + - Default packet cadence (s) + - Comment + * - ``N_2_2`` + - 2 + - 2 + - 8 + - **Default normal mode** + * - ``N_4_1`` + - 4 + - 1 + - 8 + - + * - ``N_4_4`` + - 4 + - 4 + - 8 + - + * - ``B_64_8`` + - 64 + - 8 + - 4 + - **Default burst mode** + * - ``B_64_64`` + - 64 + - 64 + - 4 + - + * - ``B_128_128`` + - 128 + - 128 + - 2 + - + +.. warning:: + + **Do not hardcode these six modes.** The document is explicit: the number of + vectors and their cadence **must be inferred from the packet headers**. The + table exists only so you recognise what you are looking at. The code follows + this - ``PRI_VECSEC``/``SEC_VECSEC`` and ``PUS_SSUBTYPE`` are read per packet. + **[CODE]** + +Operating modes **[DOC]**: + +* **Standby** - boot software only. Reduced housekeeping, no science, no I-ALiRT. +* **Config** - application software, no science telemetry. Used between science + mode transitions. +* **Normal (NM)** - normal rate science telemetry, APID **1052**. ~23 h/day. +* **Burst (BM)** - burst rate science telemetry, APID **1068**. ~1 h/day. Entered + by telecommand with a duration; exits back to NM by timeout, telecommand, or + spacecraft DSN event flagging. + +.. note:: + + **NM and BM are mutually exclusive**, except at transitions where the last + packet of one mode and the first packet of the other generally **overlap in + sample time**. With certain rate/duration combinations, both packet types can + be transmitted at the same time. This overlap is the reason L1C exists and is + the reason L1C gap-filling has to tolerate duplicate timestamps. + +The MAG science team wants a **continuous L2 normal-mode product**: real NM data +where it exists, and a synthesised low-cadence product built on the ground from +BM data where it does not. That synthesis is L1C. **[DOC]** + +Science telemetry layout +------------------------ + +**[DOC]** Both NM and BM packets share the same header parameter structure, so +header decoding is a common operation. After the CCSDS header and the MAG data +field headers, the payload is: + +.. code-block:: text + + [ PRIMARY vectors: X(16b signed) Y(16b) Z(16b) range(2b) ] x N + [ SECONDARY vectors: X(16b signed) Y(16b) Z(16b) range(2b) ] x M + +Each sample is therefore **50 bits** and is *not* byte aligned. + +**PRIMARY and SECONDARY are software labels, not sensors.** A boolean header +(``PRI_SENS``) is true when MAGo is PRIMARY. Nominally MAGo is PRIMARY, but this +**cannot be assumed** and can in principle change within a day. Every product +downstream of L0 is organised by **MAGo/MAGi**, never by primary/secondary. + +Samples carry **no individual timestamps**. The data field header holds one +instrument time for the first PRIMARY vector and one for the first SECONDARY +vector; every other vector time is derived by adding +``1 / vectors_per_second``. + +A note on timing +---------------- + +**[DOC]** IMAP is a spinning platform, so converting from the sensor frame to an +inertial frame needs a good measurement time. MAG timestamps vectors, but there +is a small systematic shift between the timestamp and the true measurement time. +The MAG team supplies a **time shift per sensor** (expected to be tens of +milliseconds at most, e.g. ``0.005`` s) that must be applied before any attitude +conversion. Positive shifts move times forward. + +This is applied at **L1B** (:ref:`mag-l1b`), and again per-vector as a +``timedeltas`` correction at **L2** (:ref:`mag-l2`). + +Reference frames +---------------- + +This chain is the spine of the whole pipeline. Each level's job is essentially +"advance one frame". + +.. list-table:: + :header-rows: 1 + :widths: 14 20 66 + + * - Frame + - Where it appears + - Meaning + * - **MFO / MFI** + - L0, L1A + - **Measurement Frame** for MAGo / MAGi. Raw engineering units, three + nearly-orthogonal sensor axes. + * - **URFO / URFI** + - L1B, L1C + - **Unit Reference Frame** per sensor. Reached by applying the ground + (engineering) calibration matrix. Data is now in **nT** and orthogonal. + * - **ORFO / ORFI** + - L1D, L2 (intermediate) + - **Orthogonal Reference Frame**. Reached by applying the in-flight + calibration matrix, which folds in gain/orthogonality corrections and the + boom-derived mounting. + * - **SRF** + - L1D, L2 + - **Spacecraft Reference Frame** - spinning, spin axis aligned. Spin + averaging is only possible here, in the two spin-plane axes. + * - **DSRF** + - L1D, L2 + - **De-spun Spacecraft Reference Frame**. Gradiometry is done here. + * - **RTN, GSE, GSM** + - L1D, L2, I-ALiRT + - Inertial science frames for release. + +**[CODE]** Frames are realised through SPICE. ``ValidFrames`` in +``mag/l2/mag_l2_data.py`` maps each name to an +``imap_processing.spice.geometry.SpiceFrame``: + +.. list-table:: + :header-rows: 1 + :widths: 26 34 40 + + * - ``ValidFrames`` member + - SPICE frame + - Output variable name + * - ``MAGO`` / ``MAGI`` + - ``IMAP_MAG_BASE`` + - ``vectors`` + * - ``MAGO_GROUND_CAL`` + - ``IMAP_MAG_O`` + - ``vectors`` + * - ``MAGI_GROUND_CAL`` + - ``IMAP_MAG_I`` + - ``vectors`` + * - ``SRF`` + - ``IMAP_SPACECRAFT`` + - ``b_srf`` + * - ``DSRF`` + - ``IMAP_DPS`` + - ``b_dsrf`` + * - ``GSE`` + - ``IMAP_GSE`` + - ``b_gse`` + * - ``GSM`` + - ``IMAP_GSM`` + - ``b_gsm`` + * - ``RTN`` + - ``IMAP_RTN`` + - ``b_rtn`` + +.. note:: + + **MAGO and MAGI both map to ``IMAP_MAG_BASE``.** This is deliberate: the MAG + team's in-flight calibration matrix transforms from each sensor's real + mechanical mount into a single *idealised* frame, so by the time SPICE is + involved both sensors share a frame. ``IMAP_MAG_O`` / ``IMAP_MAG_I`` + (``*_GROUND_CAL``) are retained for reference to the as-assessed ground + mounting and are not used in the nominal path. + +Calibration concept +------------------- + +**[DOC]** Two distinct calibration campaigns feed two distinct file families. + +**Ground calibration** (Magnetsrode facility, TU Braunschweig) characterises the +intrinsic sensor properties and produces what the pipeline calls the +**engineering (ENG) calibration**: + +1. A nominal **scale factor** yielding nT per component, folded together with + the **gain (sensitivity, sigma)** and **orthogonality (misalignment, omega)** + matrix. +2. A **rotation from measurement frame (MF) to unit reference frame (URF)**, + :math:`R_{MF \to URF}`. + +These are combined into a single 3x3 matrix per (sensor, range) and applied at +L1B. + +**In-flight calibration** (performed at Imperial College London) produces the +L1D and L2 inputs: + +1. **URF to spacecraft frame rotation**, provided by the project after + integration from the boom orientation. Possibly refined during commissioning, + then fixed. +2. **Gain and orthogonality corrections** to the ground values. *Until there is + evidence to justify a change these are the identity matrix.* Updated at + roughly **monthly** cadence. +3. **Removal of spacecraft-generated fields** - several MAG-team processes that + detect and remove low-frequency (0-64 Hz) and DC step changes caused by the + spacecraft and other instruments. +4. **Offsets in spacecraft coordinates**, determined with the **Leinweber + method** for magnetometer offsets in the solar wind. These combine the sensor + offset and the spacecraft offset, are **time varying**, and are delivered + **per vector, per day**. Data releases may include refined offsets later. + +Only steps 1, 2 and 4 land in this repository, as delivered numbers in +calibration files. Step 3 happens entirely at Imperial College; its results +arrive as the per-vector offsets and the quality bitmask. + +.. note:: + + **[CMAD]** The public CMAD describes how this worked out in flight for Data + Release 1, and it refines SW-009 on two points: + + * **Step 2:** the in-flight matrices (CalibrationMatricesV9) use fitted + angles and gains that are *not* exactly identity. + * **Step 4:** Leinweber is used only for the **spin-axis** offset. The + spin-plane offsets use the Kepko method plus a spin-tone optimisation. + + Step 3 is now a set of named cleaning processes (IMAP-Lo pivot platform, + Hi/Ultra heaters, thrusters). See :ref:`mag-cmad`. + +Processing windows and the 30-minute buffer +-------------------------------------------- + +**[DOC/CODE]** MAG processes UTC-day windows. Because vectors are packetised +across midnight, L0, L1A, L1B and L1C files carry **an extra 30 minutes on each +side** (midnight - 30 min to midnight + 24 h + 30 min, i.e. a 25 hour file). + +The buffer is **stripped to exactly 24 hours at L1D and L2**, after offsets and +time shifts have been applied - the order matters, because a time shift can move +a vector across the boundary. + +**[CODE]** ``MagL2L1dBase.truncate_to_24h`` does the trimming; ``cli.py`` calls +``check_epochs_within_day_offsets`` on every MAG output, which raises if any +epoch is more than 24 hours outside the processing day. + +Processing must also be **resilient to partial days**: dropped packets, MAG off +for part of the day, or partial downlink must not fail the run. diff --git a/docs/source/algorithm-code-documentation/mag/reference-tables.rst b/docs/source/algorithm-code-documentation/mag/reference-tables.rst new file mode 100644 index 0000000000..1ceaf73f62 --- /dev/null +++ b/docs/source/algorithm-code-documentation/mag/reference-tables.rst @@ -0,0 +1,243 @@ +.. _mag-reference-tables: + +Reference Tables - Where to Look Them Up +======================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +The algorithm document's large tables are **deliberately not reproduced** here: +they go stale, and in almost every case a machine-readable version already +exists in the repository that the code actually reads. + +The document itself is **not in this repository** - see +:ref:`mag-source-documents`. MAG is unusually self-contained: the packet +definition, the compression tables (there are none - the algorithm is closed +form), and every calibration number arrive either as XTCE or as delivered CDFs. +The only things you genuinely need the document for are listed at the bottom of +this page. + +Rule of thumb +------------- + +.. list-table:: + :header-rows: 1 + :widths: 42 58 + + * - If you need... + - Go to + * - A packet field's name, bit offset, width or type + - ``imap_processing/mag/packet_definitions/MAG_SCI_COMBINED.xml`` + * - The I-ALiRT packet layout + - ``imap_processing/ialirt/packet_definitions/ialirt_mag.xml`` and the + ``Packet0``-``Packet3`` classes in + ``imap_processing/ialirt/l0/mag_l0_ialirt_data.py`` + * - The Fibonacci sequence or any tunable constant + - ``imap_processing/mag/constants.py`` + * - Which interpolation method or sensor is configured + - ``imap_processing/mag/imap_mag_sdc_configuration_v001.py`` + * - A CDF variable's units, fill value, valid range or description + - ``imap_processing/cdf/config/imap_mag_l1{a,b,c}_variable_attrs.yaml`` and + ``imap_mag_l2_variable_attrs.yaml`` + * - The complete list of MAG products + - ``imap_processing/cdf/config/imap_mag_global_cdf_attrs.yaml`` + * - A calibration variable name or shape + - :ref:`mag-ancillary`, or ``cdflib.CDF(path).cdf_info()`` on a file in + ``imap_processing/tests/mag/validation/calibration/`` + * - The decompression algorithm + - :ref:`mag-compression` - fully transcribed, including the worked example + * - The gradiometry or spin-averaging equations + - :ref:`mag-l1d` - transcribed + * - The clock-angle formulae + - :ref:`mag-ialirt` - transcribed + * - How the L2 offsets and matrices were derived, or the quality bitmask bits + - :ref:`mag-cmad`, summarising the public CMAD (not SW-009) + * - Anything else + - the algorithm document, using the section index below + +Machine-readable tables in the repository +------------------------------------------ + +Packet definitions +^^^^^^^^^^^^^^^^^^ + +``imap_processing/mag/packet_definitions/MAG_SCI_COMBINED.xml`` is the +authoritative field definition for APIDs 1052 and 1068. It supersedes +``TLM_MAG`` ([RD01]) for anything the pipeline does, and unlike the spreadsheet +it cannot be out of date with respect to processing - it is what +``packet_generator`` parses. + +Both APIDs are in the one file; ``decom_mag.decom_packets`` filters on +``PKT_APID``. + +.. code-block:: bash + + grep -o 'name="[^"]*"' imap_processing/mag/packet_definitions/MAG_SCI_COMBINED.xml | sort -u + +CDF metadata +^^^^^^^^^^^^ + +``imap_processing/cdf/config/``: + +* ``imap_mag_global_cdf_attrs.yaml`` - **the definitive product list.** Every + ``Logical_source`` MAG can emit has an entry here; if a string is not in this + file, ``get_global_attributes`` will raise. +* ``imap_mag_l1a_variable_attrs.yaml`` +* ``imap_mag_l1b_variable_attrs.yaml`` +* ``imap_mag_l1c_variable_attrs.yaml`` +* ``imap_mag_l2_variable_attrs.yaml`` - **also used by L1D**; + ``mag_l1d`` calls ``add_instrument_variable_attrs("mag", "l2")``. + +.. note:: + + There is no ``imap_mag_l1d_variable_attrs.yaml``. L1D and L2 share the L2 + variable attributes because they produce structurally identical files. The + L1D ancillary outputs (spin offsets, gradiometry offsets) have **no** + attribute definitions at all - they are written with ``istp=False``. + +Delivered calibration files +^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Real examples live under ``imap_processing/tests/mag/validation/``: + +.. code-block:: text + + calibration/imap_mag_l1b-calibration_20240229_v001.cdf + calibration/imap_mag_l1b-calibration_20240229_v002.cdf + calibration/imap_mag_l1d-calibration_20000101_v003.cdf + calibration/imap_mag_l2-calibration_20251017_v004.cdf + calibration/imap_mag_l2-norm-offsets_20251017_20251017_v001.cdf + L2/T021/imap_mag_l2-calibration-matrices_20250506_v004.cdf + L2/T021/imap_mag_l2-calibration-matrices_20250506_v005.cdf + L2/T022/imap_mag_l2_norm-offsets_20250506_v006.cdf + +A copy of the L1B calibration is also bundled at +``imap_processing/mag/l1b/imap_mag_l1b-calibration_20240229_v002.cdf`` as a test +fallback. + +Validation data +^^^^^^^^^^^^^^^ + +``imap_processing/tests/mag/validation/`` is organised by level and test number. +Each case pairs an input (binary packets or CSV) with MAG-team reference output +CSVs: + +.. code-block:: text + + L1a/T001..T008/ mag-l0-l1a-tNNN-in.bin, mag-l0-l1a-tNNN-out.csv + L1b/T009..T012/ mag-l1a-l1b-tNNN-in.csv, ...-mago-out.csv, ...-magi-out.csv + L1c/T013..T016,T024/ mag-l1b-l1c-tNNN-{mago,magi}-{normal,burst}-{in,out}.csv + L2/T021,T022/ calibration + offsets CDFs, expected-output CSV + +The descriptive text files alongside them (``field_like.txt``, +``all_p_ones.txt``, ``hdr_field_and_range_change.txt``, ...) name what each case +is exercising, which is the fastest way to find the case that covers a +behaviour you are changing. + +Document section index +---------------------- + +For when you do have a copy of IMAP-MAG-SW-009-01B in ``docs/reference/``. +44 pages, Issue 5 Revision 2. + +SW-009 is not part of the public CMAD. For where MAG material *does* appear in +the CMAD, see the location table in :ref:`mag-cmad`. + +.. list-table:: + :header-rows: 1 + :widths: 14 16 70 + + * - Section + - Pages + - Contents + * - 4.1-4.3 + - 7-8 + - Hardware description, science objectives, boot/application software. + * - 4.4 + - 8-9 + - **Ranging.** Table 4-1 (sensor ranges), autoranging behaviour. + * - 4.4.1 + - 9-10 + - **Cadence.** Table 4-2 (the six science modes), NM/BM transition + overlap. + * - 4.5 + - 10-11 + - Table 4-3 (operating modes: Standby, Config, Normal, Burst). + * - 4.6 + - 11-12 + - **Science telemetry layout.** APIDs, primary/secondary labelling, the + 50-bit sample structure, timestamp derivation. + * - 4.7 + - 12-13 + - **Calibration.** Ground (Magnetsrode) and in-flight (Imperial College), + including the Leinweber offset method. *The CMAD gives the as-flown + method; see* :ref:`mag-cmad`. + * - 5 + - 13-15 + - Product overview and the **product summary table** (inputs, outputs, + frames, modes per level). + * - 6 + - 15 + - Background - Solar Orbiter heritage code. + * - 7.1 + - 16 + - General principle and the note on timing. + * - 7.2 + - 16-18 + - **Calibration file contents** and the **quality flag / bitmask table**. + *The bitmask table is superseded by CMAD section 5.4.5; see* + :ref:`mag-quality-flags`. + * - 7.2.1 + - 18-19 + - **Gradiometer mode** and the kappa equation. + * - 7.3.1 + - 20 + - Figure 1, the full pipeline workflow diagram. *L2P in the diagram is now + L1D.* + * - 7.3.2 + - 21-23 + - **L0 to L1A**, step by step. + * - 7.3.2.1 + - 23-27 + - **Decompression**, including the compressed block layout table, the + Python decode functions, and the worked binary example. + * - 7.3.3 + - 27-29 + - **L1A to L1B.** + * - 7.3.4 + - 29-32 + - **L1B to L1C**, including all six interpolation methods and the CIC + filter code template. + * - 7.3.5 + - 32-35 + - **L1B/C to L1D.** The longest procedure - 14 steps. + * - 7.3.6 + - 35-37 + - **L1B/C to L2.** + * - 7.4 + - 37-42 + - **I-ALiRT**, including Table 7-1 (packet structure), Table 7-2 (STATUS + field over four packets), Table 7-3 (decommutated structure) and the + clock-angle formulae. + * - 8 + - 42-44 + - Appendix 1 - the compression algorithm rationale, zig-zag encoding, and + Fibonacci encoding with Zeckendorf's theorem. + +What is only in the external document +-------------------------------------- + +The short list of things you cannot get from this repository: + +* **Figure 1**, the pipeline workflow diagram. :ref:`mag-data-products` has an + ASCII equivalent, but the original shows the reference-frame regions as shaded + bands, which is a useful mental model. +* **Table 7-2**, the exact I-ALiRT STATUS bit layout. The code's + ``Packet0``-``Packet3`` classes are the operative version and their docstrings + record the bit positions, but the document is the source of truth if they ever + disagree. +* **Narrative rationale** - why the Leinweber method, why a CIC filter, what the + Solar Orbiter heritage code did. The equations and procedures are transcribed + here; the reasoning is summarised in a sentence at most. +* **The change log and change record**, useful for understanding why something + looks odd (for example, kappa becoming a matrix in Issue 5 Revision 1, and + clock angles being added to I-ALiRT in Issue 5 Revision 2). diff --git a/docs/source/algorithm-code-documentation/swapi.rst b/docs/source/algorithm-code-documentation/swapi.rst index 6e91613b5b..0866d98fd6 100644 --- a/docs/source/algorithm-code-documentation/swapi.rst +++ b/docs/source/algorithm-code-documentation/swapi.rst @@ -24,4 +24,4 @@ The modules below contain various utility classes and functions for processing S :template: autosummary.rst :recursive: - swapi_utils + swapi_utils \ No newline at end of file diff --git a/docs/source/algorithm-code-documentation/swapi/ancillary.rst b/docs/source/algorithm-code-documentation/swapi/ancillary.rst new file mode 100644 index 0000000000..dc358fd684 --- /dev/null +++ b/docs/source/algorithm-code-documentation/swapi/ancillary.rst @@ -0,0 +1,469 @@ +.. _swapi-ancillary: + +Ancillary Files, Calibration and External Dependencies +====================================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**Document:** sections 8, 9, 10.3 and 11. + +SWAPI needs remarkably little ancillary data to reach L2 - two tables, both +derived from the same delivered workbook. Everything else on this page is +either L3's problem or operations' problem, and is documented so you can tell +which is which. + +What L2 actually reads +---------------------- + +.. list-table:: + :header-rows: 1 + :widths: 26 20 54 + + * - Ancillary + - SDC descriptor + - Read by + * - ESA unit conversion table + - ``esa-unit-conversion`` + - ``swapi_l2.solve_full_sweep_energy`` (and the I-ALiRT code) + * - LUT notes table + - ``lut-notes`` + - ``swapi_l2.solve_full_sweep_energy`` + +Both L2 tables are loaded through ``swapi_utils.read_swapi_lut_table(path)``, +which is a thin ``pd.read_csv`` plus one important cleanup: + +.. code-block:: python + + df["Energy"] = (df["Energy"].astype(str) + .str.replace(",", "", regex=False) + .replace("Solve", -1) + .astype(np.float64)) + +Two things that will bite you: the ``Energy`` column arrives as +**comma-thousands-separated strings** (``"1,163"``), and the sentinel +``"Solve"`` becomes **-1**. Everything downstream detects solve steps as +``Energy < 0``. Only the ``Energy`` column is cleaned - ``Voltage`` keeps its +literal ``"Solve"`` strings. + +.. warning:: + + The CSVs are read with a UTF-8 BOM, so the first column name is + ``"timestamp"``, not ``"timestamp"``. Pandas handles it; a hand-rolled + ``csv.DictReader`` will not. + +.. _swapi-l3-descriptors: + +What the SDC holds but this repository never reads +--------------------------------------------------- + +SWAPI delivers a further **thirteen** ancillary files to the SDC. All of them +are inputs to the SWAPI team's L3 container (:ref:`swapi-l3-scope`), and none +of them appear anywhere in this codebase. Their descriptors are recorded here +because the algorithm document refers to most of them only by the symbol they +carry, and that mapping is otherwise written down nowhere. + +.. list-table:: + :header-rows: 1 + :widths: 34 20 46 + + * - SDC descriptor + - Symbol + - Purpose + * - ``central-effective-area`` + - :math:`\mathcal{A}_0^s(V)` + - Central effective area vs ESA voltage. + :ref:`swapi-instrument-response`. + * - ``passband-fit-coefficients`` + - :math:`P_r(v/v_0, \theta, V)` + - Energy-angle passband, one set for the sunglasses region and one for the + open aperture. :ref:`swapi-instrument-response`. + * - ``azimuthal-transmission`` + - :math:`T(\phi)` + - Azimuthal transmission. :ref:`swapi-instrument-response`. + * - ``instrument-response-lut`` + - n/a + - The three functions above combined into the tabulated response that the + L3 forward model evaluates directly. + * - ``efficiency-lut`` + - :math:`\varepsilon_H`, :math:`\varepsilon_{He}` + - Time-varying detection efficiency. :ref:`swapi-efficiency-lut`. + * - ``energy-gf-sw-lut`` + - :math:`G(E/q)` + - Energy-dependent geometric factor, solar wind. + * - ``energy-gf-pui-lut`` + - :math:`G(E/q)` + - Energy-dependent geometric factor, pickup ions. + * - ``density-of-neutral-helium-lut`` + - :math:`n_{\mathrm{He}}` + - Hot-model interstellar neutral helium density, the source population for + the L3 pickup helium fit. + * - ``helium-inflow-vector`` + - n/a + - Interstellar neutral helium inflow vector. + * - ``hydrogen-inflow-vector`` + - n/a + - Interstellar neutral hydrogen inflow vector. + * - ``proton-density-temperature-lut`` + - n/a + - **Unidentified.** + * - ``alpha-density-temperature-lut`` + - n/a + - **Unidentified.** + * - ``clock-angle-and-flow-deflection-lut`` + - n/a + - **Unidentified.** + +.. warning:: + + The last three have no counterpart in any section of the algorithm document + summarised on these pages, and nothing in the L2-to-L3 contract explains + them. Ask the SWAPI team what they contain rather than inferring it from the + descriptor names. + +.. _swapi-esa-unit-conversion-adp: + +ESA Unit Conversion ADP +----------------------- + +**[DOC]** Section 10.3. Delivered by the SWAPI Instrument Team "as needed". + +**Naming.** The ADP file naming convention is the IMAP standard: + +.. code-block:: text + + imap_____. + +so the ESA Unit Conversion ADP is + +.. code-block:: text + + imap_swapi_esa-unit-conversion___.xlsx + +The initial ground-test delivery was +``imap_swapi_esa-unit-conversion_20250211_v00.xlsx``. + +**[CODE]** The SDC ingests it as CSV. The vendored test copies are +``imap_processing/tests/swapi/lut/imap_swapi_esa-unit-conversion_20250626_v001.csv`` +and ``.../imap_swapi_lut-notes_20250626_v006.csv``. + +**Columns** (main sheet): + +.. list-table:: + :header-rows: 1 + :widths: 24 76 + + * - Column + - Meaning + * - ``timestamp`` + - Validity start of this table version, ``M/D/YYYY H:MM``. Parsed with + ``format="%m/%d/%Y %H:%M"``. + * - ``ESA Step #`` + - 0-71. + * - ``K factor`` + - eV/V/e. 1.88 in the 2025-02-11 rows, 1.93 in the 2025-05-19 rows. + **Read but never used.** See :ref:`swapi-k-factor`. + * - ``Voltage`` + - ESA voltage in V, or the literal ``Solve``. **Not used.** + * - ``Energy`` + - eV/q, or ``Solve``. **This is what L2 uses.** + * - ``Sweep #`` + - Which sweep table this row applies to. Matched against the L1 + ``sweep_table`` variable. The test file contains sweeps 0, 1 and 2. + * - ``ESA Index Number`` + - For fixed steps, the row on the DAC ladder. For solve steps, the + **offset** from the final solve step (``-16 ... +16`` in the nominal + table). This is the column that drives the fine-step walk. + * - ``LUT version number`` + - Which ``LUT_Notes_vx`` sheet the offsets refer to. + +.. note:: + + The table is keyed on **(timestamp, Sweep #)**, and L2 selects the latest + version at or before the sweep's start time. A single CSV therefore contains + several stacked table versions - the 288-row test file holds two versions of + sweep 0 plus one each of sweeps 1 and 2. When adding a new version, append + rows; do not replace the file's history, or reprocessing older data will + silently pick the wrong energies (or fall back to the earliest version). + +LUT notes table +--------------- + +**[DOC]** The ``LUT_Notes_vx`` tab of the ADP workbook, one tab per LUT +version. **[CODE]** ingested as ``imap_swapi_lut-notes__v.csv`` - +note the file version tracks the **LUT** version (``v006`` in the test file), +not the delivery. + +**Columns:** + +.. list-table:: + :header-rows: 1 + :widths: 26 74 + + * - Column + - Meaning + * - ``ESA Index Number`` + - Row index on the ladder, 0-based, ascending. + * - ``ESA Voltage`` + - ESA voltage in V, descending down the table (10240 V at row 0). + * - ``Energy`` + - eV/q. **The column L2 reads.** + * - ``Lower Energy``, ``Upper Energy`` + - Passband edges. Not used by any code here. + * - ``ESA Range`` + - HV range selector. + * - ``ESA DAC (Dec)``, ``ESA DAC (Hex)`` + - The commanded DAC value. **``ESA DAC (Hex)`` is what + ``esa_lvl5`` is matched against**, formatted as 4 uppercase hex digits. + +The test file has 1024 data rows. Adjacent rows can share a DAC value (rows 0 +and 1 are both ``1FFE``), which is why "first match wins" in +``solve_full_sweep_energy`` is a real behavioural choice. + +Calibration data +---------------- + +**[DOC]** Section 9. Three ground sources plus two on-orbit ones. None of this +is read by L1 or L2; it feeds the L3 response function and the operations +voltage updates. + +Ground calibration +^^^^^^^^^^^^^^^^^^ + +.. list-table:: + :header-rows: 1 + :widths: 26 74 + + * - Source + - What it produces + * - **End-to-end (ETE) model** + - A full simulation of instrument plus environment, driven by real 1 au + solar wind (ACE and Wind, via the OMNI database), used to generate the + LUTs and to validate the processing. Five modules: OMNI observations, + SW+PUI distribution function model, coordinate transformation matrices, + a SIMION-based electro-optical model, and a telemetry formatter. **This + is also how SWAPI processing code was verified** - see below. + * - **SIMION module** + - Ions flown through the instrument geometry at a range of energies, entry + locations and angles, per ESA voltage. Integrated over a uniform beam to + give the response function, then tabulated: ion energies, entry angles, + and relative intensities per ESA voltage. 72 response files per ESA step. + * - **Princeton lab** + - Instrument response over 0.4-18 keV/q required (0.1-20 keV/q tested), a + broad range of angles, each optical element (through the attenuation + grid as well as the open aperture), and H and He. Energy-angle + passbands from azimuthal positioning; entrance-aperture response from + elevation-angle rolls. Also CEM gain curves, absolute detection + efficiency, carbon-foil active area, transmission and scattering, UV + suppression, rate dependence and backgrounds. Details in Rankin et al. + (2025). + * - **CoDICE cross-calibration** + - Both instruments in the same chamber at the same time, rotating in and + out of the beam, with an absolute beam monitor. Primarily 2 keV/q and + 16 keV/q protons and helium, at several elevation angles and at fixed + azimuths of 0 degrees (through the sunglasses) and 90 degrees (through + the open aperture). Rankin et al. (2025), Livi et al. (2025). + +On-orbit calibration +^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Two activities: + +1. **Gain curve tests**, every ~3 to 6 months (multiple times during + commissioning), to find the best PCEM and SCEM voltages as the CEMs age. + The optimal setting is where the count rate "flattens out" - above some + bias, rates become independent of further voltage increase. CEM gain decays + with total output charge over its lifetime, so the voltage must be raised + to compensate. +2. **In-flight comparison with CoDICE**, once both instruments' data are + validated and processed, using overlapping species and energy ranges. + +The absolute efficiency is computed from the three counters: + +.. math:: + + \varepsilon = \frac{\mathrm{COIN}^2}{\mathrm{PRM} \times \mathrm{SEC}} + +which is independent of individual detector performance and aging +(Funsten et al. 2005). + +.. _swapi-efficiency-lut: + +The efficiency LUT +^^^^^^^^^^^^^^^^^^ + +**[DOC]** Section 9.5.1. The efficiency calibration table has **two columns**, +:math:`\varepsilon_H` for hydrogen and :math:`\varepsilon_{He}` for helium, +versus time. It scales the central effective area: + +.. math:: + + \mathcal{A}_0^{H^+}(V) = \mathcal{A}_{0,\mathrm{lab}}^{H^+}(V) + \frac{\varepsilon_H(t)}{\varepsilon_H(t_{\mathrm{lab}})} + +.. math:: + + \mathcal{A}_0^{He^{2+}}(V) = \mathcal{A}_0^{He^{+}}(V) + = \mathcal{A}_{0,\mathrm{lab}}^{H^+}(V) + \frac{\varepsilon_{He}(t)}{\varepsilon_H(t_{\mathrm{lab}})} + +where :math:`\varepsilon_H(t_{\mathrm{lab}})` is the **first proton entry in +the table on or after 2025-11-01** and :math:`\varepsilon_H(t)` is the most +recent entry preceding :math:`t`. Only relative values matter, so the table is +agnostic to whether it is ever rescaled to an absolute efficiency. + +Initial values: hydrogen column **1**, helium column **1.05**. The 1.05 comes +from the high-energy limit observed in the lab for He\ :sup:`+` versus +H\ :sup:`+`; above a few keV per charge the ratio was consistently 1.05. Since +both solar wind alphas and pickup ions tend to be above that threshold, the +increase of the ratio at low energies has not been accounted for. +He\ :sup:`+` and He\ :sup:`2+` are assumed to have the same efficiency. + +**[CODE]** No efficiency table is read anywhere in this repository. It is an +L3 input, delivered to the SDC under the ``efficiency-lut`` descriptor. + +.. _swapi-instrument-response: + +Instrument response CSVs +^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Three functions, stored as CSV files and loaded by the production +(L3) code from ancillary files: + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Function + - Notes + * - :math:`\mathcal{A}_0^s(V)` + - Central effective area vs ESA voltage. ~0.75 cm\ :sup:`2` at 100 V + falling to ~0.29 cm\ :sup:`2` near 3 kV, rising slightly above that. + Carries an uncertainty of **up to 20%**, which is a systematic floor on + every density measurement. + * - :math:`P_r(v/v_0, \theta, V)` + - Region-specific energy-angle passband, one set for SG and one for OA. + Interpolated on speed ratio and elevation. + * - :math:`T(\phi)` + - Azimuthal transmission, tabulated from 0 to 180 degrees at 0.1 degree + spacing. Missing entries treated as zero; the interpolator uses + :math:`|\phi|` after wrapping azimuth into + :math:`[-180^\circ, 180^\circ)`. The flat portions are hard-coded for + performance: :math:`T = 10^{-3}` for :math:`|\phi| \le 9^\circ` and + :math:`T = 1` for :math:`31^\circ \le |\phi| \le 115^\circ`. + +Normalizations of :math:`\mathcal{A}_0^s` and :math:`P_r` are aligned at +:math:`\theta = 0` and :math:`k^{*} = 1.89` eV/V/e. + +**[CODE]** None of these files are read here. They are delivered to the SDC as +``central-effective-area``, ``passband-fit-coefficients`` and +``azimuthal-transmission`` - see :ref:`swapi-l3-descriptors`. + +The gain test LUT +----------------- + +**[DOC]** Section 11, "Maintenance of SWAPI ground data processing". This is +the *only* regular maintenance item the document names: + + The only regular maintenance of the SWAPI ground data processing is updating + the PCEM and SCEM voltages periodically. + +The gain test is SWAPI's only calibration activity during routine operations - +multiple times during commissioning, then every ~3 months. Its output is a +**1x2 LUT**: + +.. list-table:: + :header-rows: 1 + :widths: 50 50 + + * - PCEM setting (V) + - SCEM level (V) + * - (value) + - (value) + +LUT improvements are made asynchronously with SWAPI data production, and +**"the ground processing will also need to incorporate the resulting efficiency +changes acquired after gain curve tests."** + +**[CODE]** Nothing reads this LUT. Its effect reaches the pipeline only +indirectly - via the efficiency table at L3, and via ``SWP_HK.PCEM_LVL`` / +``SWP_HK.SCEM_LVL`` in housekeeping. Since L2 does not apply efficiency, no L2 +reprocessing is triggered by a gain test. + +External dependencies +--------------------- + +**[DOC]** Section 8. Three, and their status here is worth being precise about. + +Spacecraft thruster data +^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** SWAPI data will be flagged in the L1 data product **and beyond** if +obtained during spacecraft thruster activity, for data quality reasons. + +* SWAPI is **off** for at least 25 minutes before, during, and 1 hour after + delta-V and stationkeeping maneuvers - so no data exists to flag for those. +* SWAPI is **fully operational** for the daily repointing maneuvers - so those + are exactly the ones that need flagging. +* Thruster firing history comes from the MOC. + +**[CODE]** Not implemented. ``swapi_l2.py`` carries +``# TODO: add thruster firing flag`` and there is no bit for it in +``SWAPIFlags``. Section 10.2 of the document places the flag at L1-L2 (as +``SWP_SC_THRUSTER``, 0 or 1), so L2 is the right place for it. + +MAG Level 2 data +^^^^^^^^^^^^^^^^ + +**[DOC]** The magnetic field direction is needed to constrain the solar wind +**alpha** flow vector, because the alpha-proton differential drift is modelled +as lying along :math:`\hat{B}`. Where MAG L2 is unavailable, MAG **L1D** is +used and the product is reprocessed once L2 exists; **only the L2-processed +data is released publicly**, and the interim product carries a +``PRELIMINARY_MAG`` flag. + +**[CODE]** L3a's concern. No SWAPI code in this repository reads MAG. + +Timing, attitude, ephemeris (SPICE) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** The L3 solar wind flow deflection angle at ~1 min resolution needs +**IMAP/SWAPI spin phase for each ESA energy bin of the individual sweep** and +the spacecraft pointing in a standard coordinate system. IMAP SPICE kernels are +used. At L3 this becomes 72 SWAPI-to-RTN rotation matrices per sweep; if the +matrices are unavailable the chunk is skipped and assigned fill values, and if +the spacecraft velocity is unavailable the fit still runs but the Sun-frame +outputs cannot be computed. + +**[CODE]** In this repository SPICE is used only for **time conversion**: the +L1 and L2 CLI branches require time kernels as a dependency, and +``swapi_l1`` uses ``met_to_utc`` / ``ttj2000ns_to_met`` to write +``sci_start_time``. No geometry, no spin phase, no rotation matrices. + +Instrument status summary +^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Figure 4 shows an "instrument status summary" feeding L3 alongside +MAG and SPICE. The document does not define its format. The nearest thing that +exists is the L1A/L1B housekeeping products and the ``swp_l1a_flags`` bitfield. + +Testing and code delivery +------------------------- + +**[DOC]** Sections 12 and 13, quoted because they define the working +relationship: + +* **Testing.** "SWAPI processing code was tested by the SDC working with the + SWAPI data team. The SWAPI team ran the ETE model and provided inputs and + outputs to the SDC to allow the SDC to verify the implementation of the SWAPI + ground processing code." So the intended validation path for L1/L2 is + ETE-model input/output pairs from the instrument team, not synthetic data + invented here. +* **Code delivery.** "SWAPI delivers the SWAPI L3 processing code using a + Docker container via AWS." This is the document's own statement that L3 is + not the SDC's to write. + +**[CODE]** The tests that exist validate decommutation against SWAPI CSV +exports of a single pre-launch idle packet, plus unit tests of the grouping, +decompression and energy-solve logic. There are **no ETE-model +input/output validation pairs** in the repository. See +:ref:`swapi-implementation-status`. diff --git a/docs/source/algorithm-code-documentation/swapi/data-products.rst b/docs/source/algorithm-code-documentation/swapi/data-products.rst new file mode 100644 index 0000000000..2a38778d6e --- /dev/null +++ b/docs/source/algorithm-code-documentation/swapi/data-products.rst @@ -0,0 +1,390 @@ +.. _swapi-data-products: + +Data Products and What Feeds What +================================= + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This is the "what goes into what" map. It is the page to read before touching +``cli.py`` or adding a product. + +Product inventory +----------------- + +**[CODE]** Everything this repository produces for SWAPI. The +``logical_source`` strings come from +``imap_processing/cdf/config/imap_swapi_global_cdf_attrs.yaml`` and are what +``write_cdf()`` uses to name the output file. + +.. list-table:: + :header-rows: 1 + :widths: 30 10 12 48 + + * - ``logical_source`` + - Level + - Cadence + - Contents + * - ``imap_swapi_l1_sci`` + - L1 + - 1 sweep (12 s) + - PCEM/SCEM/COIN **counts** and uncertainties as 72-element arrays, plus + quality flags and sweep metadata. + * - ``imap_swapi_l1a_hk`` + - L1A + - 1 s (raw) + - Housekeeping, **raw** DN values. All 102 ``SWP_HK`` fields. + * - ``imap_swapi_l1b_hk`` + - L1B + - 1 s (derived) + - Housekeeping, **derived/engineering-unit** values (same fields, + ``use_derived_value=True``). + * - ``imap_swapi_l2_sci`` + - L2 + - 1 sweep (12 s) + - PCEM/SCEM/COIN **count rates** and uncertainties, plus the solved + ``esa_energy`` (72 values per sweep). + +Plus one non-CDF product: + +.. list-table:: + :header-rows: 1 + :widths: 30 10 12 48 + + * - Product + - Level + - Cadence + - Contents + * - SWAPI I-ALiRT record + - -- + - 12 s + - ``list[dict]`` of pseudo proton speed, density and temperature destined + for the I-ALiRT database, **not** a CDF. See :ref:`swapi-ialirt`. + +Not produced here +----------------- + +.. list-table:: + :header-rows: 1 + :widths: 26 74 + + * - Product + - Status + * - L3a solar wind proton (``1m-sw-p``) + - **SWAPI team's code.** Density, speed, temperature at 1 min. + * - L3a solar wind alpha (``1m-sw-a``) + - **SWAPI team's code.** Needs MAG L2. + * - L3a pickup helium (``10m-pui-he``) + - **SWAPI team's code.** Cooling index, cutoff speed, ionization rate, + background rate, density, temperature at 10 min. + * - L3b combined differential flux (``10m-combined``) + - **SWAPI team's code.** :math:`J(E/q)` at 10 min. + * - Quick-look plots + - **[DOC]** Section 15 specifies colour-coded spectrograms of count rate + vs energy/charge vs time (with MAG L1D overlaid on the daily plot). + **No SWAPI quicklook code exists in this repository.** + +See :ref:`swapi-l3-scope` for what those products need from L2. + +The processing chain +-------------------- + +**[DOC]** Algorithm document figure 4, annotated with what is and is not here. + +.. code-block:: text + + L0 CCSDS packets (SWP_SCI, SWP_HK) + | + +------------------+------------------+ + | | + Decommutation table Level 1A + (the XTCE) Housekeeping <-- HERE + | | + v v + Level 1 Level 1B + PCEM/SCEM/COIN counts per Housekeeping (derived) <-- HERE + ESA voltage step and time <-- HERE + | + ESA Unit Conversion ADP -->| + | + v + Level 2 I-ALiRT + PCEM/SCEM/COIN count rates <-- HERE (separate APID) <-- HERE + per ESA energy step and time + | + | <-- B-field from MAG L2 (L1D fallback) + | <-- timing, attitude, ephemeris (SPICE) + | <-- instrument status summary + | <-- geometric factor, efficiency, instrument response + | + +-------+-------+ + | | + v v + Level 3a Level 3b <-- NOT HERE. SWAPI team's Docker container. + SW + PUI combined + plasma fits differential + intensity + +Dependency wiring +----------------- + +**[CODE]** ``imap_processing/cli.py``, ``class Swapi``. The CLI is the only +entry point; ``do_processing`` dispatches on ``self.data_level`` and +``self.descriptor``. + +.. list-table:: + :header-rows: 1 + :widths: 14 12 34 40 + + * - Level + - Descriptor + - Dependencies (count is checked) + - Call + * - ``l1`` / ``l1a`` + - ``sci`` + - **3**: SWAPI L0 ``raw``, SWAPI L1 ``hk`` CDF, time kernels + - ``swapi_l1(dependencies, descriptor="sci")`` + * - ``l1`` / ``l1a`` + - ``hk`` + - **2**: SWAPI L0 ``raw``, time kernels + - ``swapi_l1(dependencies, descriptor="hk")`` + * - ``l2`` + - (any) + - **3**: SWAPI L1 ``sci`` CDF, ``esa-unit-conversion`` ancillary, + ``lut-notes`` ancillary + - ``swapi_l2(l1_dataset, esa_table_df, lut_notes_df)`` + +Two things worth knowing about this wiring: + +* **The L1 science job depends on the L1 housekeeping job.** The HK CDF is + loaded to populate the science quality flags, using nearest-epoch matching. + You cannot produce ``l1_sci`` without first producing ``l1a_hk``/``l1b_hk`` + for the same day. +* **Both ``l1`` and ``l1a`` are accepted** as the data level for the same + branch. ``l1a`` exists because the housekeeping products are named + ``l1a_hk`` / ``l1b_hk`` while the science product is named ``l1_sci``. +* **The HK descriptor branch returns two datasets** (L1A raw and L1B derived) + from a single invocation. +* **Anything other than l1/l1a/l2 raises** ``NotImplementedError``. + +L1 output is passed through ``filter_day_boundary_data(ds, self.start_date)`` +(``imap_processing/utils.py``) so that packets spilling over the UTC day +boundary are trimmed. L2 is **not** filtered - it inherits L1's boundaries. + +Filenames and descriptors +------------------------- + +**[DOC]** The algorithm document specifies: + +.. code-block:: text + + imap_swapi___-_. + +.. list-table:: + :header-rows: 1 + :widths: 20 16 64 + + * - Field + - Required? + - SWAPI options per the document + * - ``dataLevel`` + - required + - ``l1``, ``l2``, ``l3a``, ``l3b`` + * - ``descriptor`` + - optional + - ``12s`` (L1 or L2 data, default), ``1m-sw-p`` (L3a), ``1m-sw-a`` (L3a), + ``10m-pui-he`` (L3a), ``10m-combined`` (L3b) + * - ``extension`` + - required + - ``cdf`` + +L0 files arrive from the SDC as +``imap_l0_sci__YYYYMMDD_vNN.pkts``. + +.. warning:: + + **[CODE]** The descriptors actually used are ``sci``, ``hk`` and the + implicit ``l1a_hk``/``l1b_hk`` split - **not** ``12s``. The document's + ``12s`` descriptor appears nowhere in the code or the CDF configs. This is + a real deviation and is tracked in :ref:`swapi-implementation-status`. Do + not "fix" it by renaming products; the SDC file catalog and every downstream + consumer are keyed on the current strings. + +Ancillary inputs +---------------- + +Summarized here; details in :ref:`swapi-ancillary`. + +.. list-table:: + :header-rows: 1 + :widths: 24 12 20 44 + + * - Ancillary + - Used at + - SDC descriptor + - What for + * - Decommutation table + - L1 + - (the XTCE itself) + - Packet field definitions. + * - ESA unit conversion ADP + - L2 + - ``esa-unit-conversion`` + - Fixed energies for ESA steps 0-62 and the fine-step offset indices. + * - LUT notes table + - L2 + - ``lut-notes`` + - The ESA DAC-to-energy ladder used to solve the fine-step energies. + * - Efficiency LUT + - L3 + - -- + - :math:`\varepsilon_H`, :math:`\varepsilon_{He}` vs time, from gain tests. + * - Instrument response CSVs + - L3 + - -- + - Central effective area, energy-angle passbands, azimuthal transmission. + * - Gain test 1x2 LUT + - operations + - -- + - New PCEM/SCEM voltage settings. Not read by any pipeline code. + * - MAG L2 (L1D fallback) + - L3a alpha + - -- + - Constrains the alpha-proton differential flow along **B**. + * - SPICE kernels + - L1, L3 + - (standard) + - Time conversion at L1; spin phase and rotation matrices at L3. + * - Thruster firing history + - L1/L2 + - -- + - **[DOC]** Should set a thruster flag. **[CODE]** Not implemented. + +CDF variable inventory +---------------------- + +**[CODE]** What is actually in the files. + +L1 science (``imap_swapi_l1_sci``) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Coordinates: ``epoch`` (one per sweep), ``esa_step`` (0-71), +``esa_step_label``. + +.. list-table:: + :header-rows: 1 + :widths: 44 14 42 + + * - Variable + - Shape + - Notes + * - ``swp_pcem_counts`` + - (sweep, 72) + - Decompressed counts. Index 0 forced to ``NaN``. + * - ``swp_scem_counts`` + - (sweep, 72) + - as above + * - ``swp_coin_counts`` + - (sweep, 72) + - as above + * - ``swp_pcem_counts_stat_uncert_plus`` / ``_minus`` + - (sweep, 72) + - :math:`\sqrt{N}`; plus and minus are identical + * - ``swp_scem_counts_stat_uncert_plus`` / ``_minus`` + - (sweep, 72) + - as above + * - ``swp_coin_counts_stat_uncert_plus`` / ``_minus`` + - (sweep, 72) + - as above + * - ``swp_l1a_flags`` + - (sweep, 72) + - ``uint16`` bitfield, ``SWAPIFlags`` + * - ``sweep_table`` + - (sweep,) + - ``SWP_SCI.SWEEP_TABLE``, taken from packet 0 + * - ``plan_id`` + - (sweep,) + - ``SWP_SCI.PLAN_ID``, taken from packet 0 + * - ``sci_start_time`` + - (sweep,) + - UTC string of the **first** packet's epoch, ms precision. Added at + SWAPI's request for L3. + * - ``esa_lvl5`` + - (sweep,) + - ``SWP_SCI.ESA_LVL5`` from packet 11 (``SEQ_NUMBER == 11``). The key to + the L2 fine-step solve. + * - ``lut_choice`` + - (sweep,) + - ``SWP_HK.LUT_CHOICE``, nearest HK packet + * - ``fpga_type`` + - (sweep,) + - ``SWP_HK.FPGA_TYPE``, nearest HK packet + * - ``fpga_rev`` + - (sweep,) + - ``SWP_HK.FPGA_REV``, nearest HK packet + +.. note:: + + The flag variable is named ``swp_l1a_flags`` even though the product is + ``imap_swapi_l1_sci`` (not ``l1a``). It is carried through to L2 under the + same name. Cosmetic, but it trips people up when grepping. + +L2 science (``imap_swapi_l2_sci``) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +L2 copies a subset of L1 - ``epoch``, ``esa_lvl5``, ``esa_step``, +``esa_step_label``, ``fpga_rev``, ``fpga_type``, ``lut_choice``, ``plan_id``, +``sci_start_time``, ``sweep_table``, ``swp_l1a_flags`` - and adds: + +.. list-table:: + :header-rows: 1 + :widths: 44 14 42 + + * - Variable + - Shape + - Notes + * - ``esa_energy`` + - (sweep, 72) + - eV/q. The solved energy of every step of every sweep. + ``VALIDMIN``/``VALIDMAX`` = 0 / 21000. + * - ``swp_pcem_rate`` + - (sweep, 72) + - counts / 0.145 s. Negative values (L1 fill) become ``NaN``. + * - ``swp_scem_rate`` + - (sweep, 72) + - as above + * - ``swp_coin_rate`` + - (sweep, 72) + - as above + * - ``swp_pcem_rate_stat_uncert_plus`` / ``_minus`` + - (sweep, 72) + - L1 count uncertainty / 0.145 s + * - ``swp_scem_rate_stat_uncert_plus`` / ``_minus`` + - (sweep, 72) + - as above + * - ``swp_coin_rate_stat_uncert_plus`` / ``_minus`` + - (sweep, 72) + - as above + +**The L1 counts themselves are not copied to L2.** Only rates. + +Every rate variable, every rate uncertainty and ``swp_l1a_flags`` get +``DEPEND_1 = "esa_energy"`` at L2, replacing the L1 ``DEPEND_1 = "esa_step"``. +That is what makes the L2 file plottable against physical energy. + +Epoch convention +---------------- + +**[CODE]** The L1/L2 ``epoch`` for a sweep is **the creation time of the 7th +packet** (``SEQ_NUMBER == 6``), i.e. the centre of the 12-second sweep, chosen +to line up with mission conventions. The *start* of the sweep is available +separately as the ``sci_start_time`` UTC string. + +Do not assume ``epoch`` is the sweep start. **[DOC]** The L3 measurement-time +formula is written against the sweep start: + +.. math:: + + t_i = t_{\mathrm{start}} + i \cdot \frac{12}{72}\ \mathrm{s} + = t_{\mathrm{start}} + (i+1) \cdot 0.16 - \frac{0.145}{2}\ \mathrm{s} + +for 0-indexed ESA step :math:`i` (recalling that step 0 is skipped). diff --git a/docs/source/algorithm-code-documentation/swapi/ialirt.rst b/docs/source/algorithm-code-documentation/swapi/ialirt.rst new file mode 100644 index 0000000000..61cfdcc0b8 --- /dev/null +++ b/docs/source/algorithm-code-documentation/swapi/ialirt.rst @@ -0,0 +1,351 @@ +.. _swapi-ialirt: + +I-ALiRT - Real-Time Space Weather Stream +======================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**Code:** ``imap_processing/ialirt/l0/process_swapi.py`` (368 lines, the whole +algorithm), ``imap_processing/ialirt/constants.py`` +(``class IalirtSwapiConstants``), +``imap_processing/ialirt/packet_definitions/ialirt_swapi.xml`` (APID 1187). + +**Document:** section 14. + +.. note:: + + The SWAPI I-ALiRT code does **not** live under ``imap_processing/swapi/``. + It lives with the other instruments' I-ALiRT parsers and imports two things + from the SWAPI modules: ``process_sweep_data`` (the 6-per-packet to + 72-per-sweep reorder) and ``SWAPI_LIVETIME``. Its output is a **list of + dicts destined for the I-ALiRT database, not a CDF**, and there is no + production caller in this repository - ``process_swapi_ialirt`` is invoked + by the SDC's real-time service. + +What it is +---------- + +**[DOC]** The I-ALiRT cadence requirement for instruments is 15 seconds. A +single SWAPI sweep is 12 seconds, so SWAPI delivers **one record per sweep at a +12-second cadence**, which comfortably beats the requirement. + +.. list-table:: + :header-rows: 1 + :widths: 26 74 + + * - Property + - Value + * - Data products + - Pseudo speed, density and temperature of H\ :sup:`+` solar wind + * - Time resolution + - 12 s + * - Components + - :math:`u_p`, :math:`n_p`, :math:`T_p` (isotropic) + * - Inputs required + - Coincidence count rates and count rate uncertainties as a function of + E/q for **62 coarse energy bins (1 sweep)** + * - Ancillary files + - Effective area :math:`A_{\mathrm{eff}}`, FWHM of energy width + :math:`\Delta E/E` + +The word **pseudo** is doing real work here. These are not the L3a moments. +They come from an analytical model with no angular response function, fit to a +handful of bins around the peak, on a single sweep. They are meant for space +weather forecasting, not science. + +The packet +---------- + +**[CODE]** ``ialirt_swapi.xml``, APID **1187**, fields: + +.. code-block:: text + + SWAPI_SHCOARSE packet time + SWAPI_ACQ acquisition time (used for the record's midpoint) + SWAPI_FLAG + SWAPI_RESERVED + SWAPI_SEQ_NUMBER 0-11, as in SWP_SCI + SWAPI_VERSION used as the sweep table id for the LUT lookup + SWAPI_COIN_CNT0..5 six coincidence counts, one per 1/6 s + SWAPI_SPARE + +**Coincidence counts only** - no PCEM, no SCEM, no per-sample range-status +bits. This is the subset the document describes. + +The analytical count-rate model +------------------------------- + +**[DOC]** Originally developed for SWAP (Elliott et al. 2016). It uses the +instrument effective area :math:`A_{\mathrm{eff}}`, the azimuthal width of the +field of view :math:`\Delta\phi`, and the energy passband :math:`E_e`, and +**does not require any angular response function**. The count rate for each ESA +energy passband is + +.. math:: + + C(E_e) = \left( n A_{\mathrm{eff}} + \left(\frac{\beta}{\pi}\right)^{3/2} + e^{-\beta\left(v_e^2 + u^2 - 2 v_e u\right)} \right) + \sqrt{\frac{\pi}{\beta u v_e}}\; + \mathrm{erf}\!\left(\sqrt{\beta u v_e}\,\frac{\Delta\phi}{2}\right) + \left( v_e^4 \frac{\Delta v}{v_e} + \sin^{-1}\!\left(v_{\mathrm{th}}/v_e\right) \right) + +where :math:`n` is the solar wind density, :math:`u` the bulk flow speed, +:math:`v_e = \sqrt{2 E_e/m}` the centre speed of the passband, +:math:`\Delta v` the speed width of the passband, +:math:`v_{\mathrm{th}} = \sqrt{2 k_B T/m}` the thermal velocity, and +:math:`\beta = 1/v_{\mathrm{th}}^2`. + +The speed width comes from the FWHM of the energy width: + +.. math:: + + \frac{\Delta v}{v} = \frac{1}{2}\left(\frac{\Delta E}{E}\right) + +**[CODE]** ``count_rate(energy_pass, speed, density, temp)`` implements this +term for term, with unit conversions (energy eV to J, speed km/s to m/s, +density cm\ :sup:`-3` to m\ :sup:`-3`). + +Constants +--------- + +**[CODE]** ``IalirtSwapiConstants``: + +.. list-table:: + :header-rows: 1 + :widths: 26 26 48 + + * - Constant + - Value + - Source + * - ``eff_area`` + - :math:`1.633 \times 10^{-4}` cm\ :sup:`2` + - **[DOC]** matches + * - ``az_fov`` + - 30 degrees (as radians) + - **[DOC]** matches + * - ``fwhm_width`` + - 0.085 + - **[DOC]** matches ("SWAPI's energy resolution for solar wind protons") + * - ``speed_ew`` + - :math:`0.5 \times 0.085` + - **[DOC]** matches + * - ``boltz``, ``at_mass``, ``prot_mass``, ``e_charge`` + - SI + - -- + * - ``speed_coeff`` + - :math:`\sqrt{2 e/m_p}/10^3` + - Used for the initial speed guess from the peak energy + * - ``temporary_density_factor`` + - :math:`e^1` + - **[DOC]** the flight-data density correction, see below + +The flight-data corrections +--------------------------- + +**[DOC]** Section 14, "Processing update for flight data". Two corrections were +added after first light: + +1. **Spin-phase smearing.** "The solar wind proton parameters appear to show + significant change within 5 sweeps due to S/C spin phase variation. A simple + correction using 5 sweeps averaged (2 sweeps before, 2 sweeps after, and 1 + current) values is provided." +2. **Density offset.** "The count rates in the space appear to be much higher + compared to those shown in Figure 24, which made the pseudo density much + higher compared to a realistic value. This systematic offset is accounted for + by correcting the pseudo density by scaling it to :math:`1/e` times the + 5-sweep-averaged fitted density." + +**[CODE]** Both are implemented, the second exactly and the first with a +difference: + +* The density correction is applied *inside the model* rather than to the + output: ``density = density * exp(1)`` in ``count_rate`` inflates the model + count by :math:`e`, which makes the fitted density come out :math:`e^{-1}` + times smaller. Algebraically equivalent, and the comment says so - "this will + increase the model count by a factor of :math:`e^1`, changing the output + density by a factor of :math:`e^{-1}` ... to be replaced once SWAPI's L3 + processing pipeline is finalized." +* The 5-sweep average is a **trailing** window (``[-5:]``: the current sweep + plus the 4 before it), not the document's **centred** window (2 before, the + current, 2 after). It is also a **geometric** mean, which the document does + not specify. See :ref:`swapi-implementation-status`. + +The algorithm as implemented +---------------------------- + +**[CODE]** ``process_swapi_ialirt(unpacked_data, calibration_lut_table)``: + +.. code-block:: text + + 1. sort by epoch; compute MET from sc_sclk_sec / sc_sclk_sub_sec + 2. find_groups(..., (0, 11), "swapi_seq_number", "met") group into sweeps + 3. per group: + a. require seq numbers to be exactly [0..11], else skip the group + b. process_sweep_data(subset, "swapi_coin_cnt") -> 72 values + c. counts = counts * 16 + 8 decompression + d. truncate to the first 63 steps + e. rate = counts / SWAPI_LIVETIME + error = sqrt(counts) / SWAPI_LIVETIME + f. look up the 63 energies from the esa-unit-conversion table, using + swapi_version as the sweep id and the LATEST timestamp + g. optimize_pseudo_parameters(rate, error, energies) + h. once 5 consecutive sweeps (12.0 +/- 0.05 s apart) are in hand, + geometric-mean the last 5 and emit one record + +Grouping differs from the science pipeline: ``find_groups`` plus an explicit +``np.array_equal(seq_values, np.arange(12))`` check, rather than the +timestamp-difference sliding window used at L1. Incomplete or duplicated groups +are collected and logged, not raised. + +Decompression +^^^^^^^^^^^^^ + +**[CODE]** + +.. code-block:: python + + # I-ALiRT packets have counts compressed by a factor of 16. + # Add 8 to avoid having counts truncated to 0 and to avoid + # counts being systematically too low + raw_coin_count = raw_coin_count * 16 + 8 + +Unconditional - there are no range-status bits in the I-ALiRT packet, so every +count is assumed compressed. The ``+ 8`` (half of 16) is a mid-bin correction +for the truncation the on-board divide introduces. **Neither the unconditional +multiply nor the ``+8`` appears in the document.** + +The fit +^^^^^^^ + +**[DOC]** The pseudo speed, density and temperature are found by minimizing the +sum of **inverse-variance weighted** squares of the difference between modelled +and observed count rates, using a non-linear least squares algorithm (e.g. +Levenberg-Marquardt). "The energy range considered for the analytical model fit +are **three energy bins on the left and two on the right** of +:math:`E_{\mathrm{peak}}`", where :math:`E_{\mathrm{peak}}` is the energy of +the ESA bin with the highest count rate. + +**[CODE]** ``optimize_pseudo_parameters``: + +.. code-block:: python + + max_index = np.argmax(count_rates) + five_point_range = range(max_index - 2, max_index + 2 + 1) + xdata = energy_passbands.take(five_point_range, mode="clip") + ... + curve_fit(f=count_rate, xdata=xdata, ydata=ydata, sigma=sigma, p0=initial_param_guess) + +``scipy.optimize.curve_fit`` with ``sigma`` is inverse-variance weighted +least squares (Levenberg-Marquardt by default for unbounded problems), so the +minimization matches. **The window does not**: the code takes 2 bins left, the +peak, and 2 bins right - five points, symmetric - where the document specifies +3 left and 2 right. ``mode="clip"`` keeps the window in range near the array +edges by repeating the edge bin. + +Initial guess: + +.. math:: + + u^{(0)} = \sqrt{E_{\mathrm{peak}}} \cdot \sqrt{2e/m_p}\,/\,10^3 + \qquad + n^{(0)} = 5 \left(\frac{400}{u^{(0)}}\right)^2 + \qquad + T^{(0)} = 60000 \left(\frac{u^{(0)}}{400}\right)^2 + +i.e. the speed corresponding to the peak bin, and density/temperature scaled +from nominal 400 km/s values. + +Failure handling +^^^^^^^^^^^^^^^^ + +**[CODE]** A fit is rejected if any of: + +* ``curve_fit`` raises ``RuntimeError`` (no convergence); +* the returned covariance matrix is not finite (scipy may be echoing back the + initial guess); +* :math:`R^2 < 0.7` on the fitted window. + +On rejection, **the speed is still reported** - from the initial guess, i.e. +the peak-bin speed - and density and temperature are set to ``NaN``: + +.. code-block:: python + + if sol is None: + sol = initial_param_guess.copy() + sol[1:] = np.nan + +This mirrors the L3a behaviour the document describes for the science product +("whenever the fit fails and fill values are reported, ``PROTON_SW_SPEED`` is +still populated based on the peak of the sweep-averaged coincidence rates"), +though the :math:`R^2 \ge 0.7` threshold is the code's own - the document +specifies :math:`R^2 \ge 0.9` for the L3a fits and nothing for I-ALiRT. + +Averaging +^^^^^^^^^ + +**[CODE]** ``geometric_mean(...)``. Sweeps with ``NaN`` in any of the three +parameters are excluded; if none of the 5 are valid, the record reports all +three as ``NaN`` with the arithmetic mean MET. Otherwise each parameter is +averaged in log space: + +.. math:: + + \bar{x} = \exp\left(\frac{1}{N}\sum_i \ln x_i\right) + +and the record's ``swapi_epoch`` is the arithmetic mean of the valid sweeps' +midpoint METs, converted with ``met_to_ttj2000ns``. + +The 5-sweep window only emits when the sweeps are contiguous: + +.. code-block:: python + + if len(swapi_met_list) >= 5 and np.all( + np.isclose(np.diff(swapi_met_list[-5:]), 12.0, atol=0.05) + ): + +so a data gap suppresses records until 5 clean sweeps have accumulated again. + +Output record +------------- + +**[CODE]** One dict per emitted window, values as ``Decimal`` to 3 decimal +places (or ``None`` when not finite), merged with the standard I-ALiRT +instrument header: + +.. code-block:: python + + { + ... instrument header items ... + "instrument": "swapi", + "swapi_epoch": , + "swapi_pseudo_proton_speed": Decimal | None, + "swapi_pseudo_proton_density": Decimal | None, + "swapi_pseudo_proton_temperature": Decimal | None, + } + +Reference numbers +----------------- + +**[DOC]** Figure 24 is the validation case: a simulated coarse sweep for +:math:`u = 550` km/s, :math:`n = 5.27` cm\ :sup:`-3`, +:math:`T = 1 \times 10^5` K, generated with the **full** instrument response +function, then fit with the analytical model. Recovered: +:math:`n_p = 4.67 \pm 0.17` cm\ :sup:`-3`, +:math:`u_p = 545.3 \pm 1.24` km/s, +:math:`T_p = (1.18 \pm 0.08) \times 10^5` K. + +That is the accuracy to expect: speed close, density and temperature "within +acceptable ranges". **[DOC]** "Future improvements may be made to correct the +values of pseudo speed, density, and temperature based on the pre-computed LUT +for a range of input parameters" - i.e. a bias-correction table, not yet +delivered. + +Tests +----- + +**[CODE]** ``imap_processing/tests/ialirt/unit/test_process_swapi.py``, which +exercises ``count_rate``, ``optimize_pseudo_parameters``, +``geometric_mean`` and the full ``process_swapi_ialirt`` including the returned +dictionary keys. diff --git a/docs/source/algorithm-code-documentation/swapi/implementation-status.rst b/docs/source/algorithm-code-documentation/swapi/implementation-status.rst new file mode 100644 index 0000000000..7bb2a19f9d --- /dev/null +++ b/docs/source/algorithm-code-documentation/swapi/implementation-status.rst @@ -0,0 +1,471 @@ +.. _swapi-implementation-status: + +Implementation Status and Known Gaps +==================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This page is the honest accounting of where the code stands against the +algorithm document. **Read it before proposing or estimating work.** + +Accurate as of the most recent survey of ``imap_processing/swapi`` and +``imap_processing/ialirt/l0/process_swapi.py`` against algorithm document +version 07 (2026-06-03). If you change something material, update this page in +the same commit. + +Summary +------- + +.. list-table:: + :header-rows: 1 + :widths: 16 20 64 + + * - Level + - State + - Notes + * - L0 / L1 science + - **Mature, with a real gap** + - Grouping, reordering, decompression and flags are complete and tested. + **Three of the five document rejection criteria are missing**, including + checksum verification. Uncertainty is Poisson-only by design (the + compression term has not been delivered). + * - L1A / L1B housekeeping + - **Complete** + - There is no algorithm - raw and derived decommutation of the same + packet. All 102 fields. + * - L2 + - **Complete for what it claims** + - Rates and the energy solve both work and the solve is more general than + the document. Missing the thruster flag and any energy uncertainty. + * - I-ALiRT + - **Working, several undocumented choices** + - Produces records; the fit window, the averaging window, the + :math:`R^2` threshold and the ``+8`` decompression offset all differ + from or extend the document. + * - L3a / L3b + - **Out of scope, correctly** + - The SWAPI team's Docker container. See :ref:`swapi-l3-scope`. + * - Quick look + - **Not started** + - Section 15 specifies spectrograms. No SWAPI quicklook code exists here. + +Hard failures in the code +------------------------- + +Explicit exceptions, so you know what a bad input looks like: + +.. list-table:: + :header-rows: 1 + :widths: 40 60 + + * - Location + - Condition + * - ``cli.py`` ``Swapi.do_processing`` + - ``NotImplementedError`` for any data level other than l1/l1a/l2. + * - ``cli.py`` ``Swapi.do_processing`` + - ``ValueError`` if the dependency count is not exactly 3 (l1 sci), 2 + (l1 hk) or 3 (l2). + * - ``swapi_l1.swapi_l1`` + - ``ValueError`` if not exactly one L0 ``raw`` file, or (for ``sci``) not + exactly one L1 ``hk`` file. + * - ``swapi_l1.decompress_count`` + - ``ValueError`` if any count is negative (the field must be unsigned). + * - ``swapi_l2.solve_full_sweep_energy`` + - ``ValueError`` if no ESA table entry exists for a sweep table id even at + the earliest timestamp. + * - ``swapi_l2.solve_full_sweep_energy`` + - ``ValueError`` if an ``esa_lvl5`` hex value is not in the LUT notes + ``ESA DAC (Hex)`` column. + * - ``process_swapi.process_swapi_ialirt`` + - ``ValueError`` if the ESA unit conversion table has no rows for the + packet's ``swapi_version`` sweep id. + +Missing L1 rejection criteria +----------------------------- + +**This is the most substantive gap in the SWAPI pipeline.** Document section +10.1.2 lists five conditions under which L0 data must be marked ``NaN``. + +.. list-table:: + :header-rows: 1 + :widths: 30 14 56 + + * - Criterion + - State + - Detail + * - ``SWP_SCI.MODE`` is not ``HVSCI`` + - **Done** + - ``filter_good_data``, all 12 packets must be HVSCI. + * - ``SWP_HK.CHKSUM`` is wrong + - **Missing** + - The field is parsed by the XTCE (``SWP_HK.CHKSUM`` and + ``SWP_SCI.CHKSUM``) and never checked. The document specifies the + algorithm precisely - "an 8-bit xor of all packet bytes (including the + CCSDS header), except for the checksum byte" - and notes that it is + computed on board *and* on the ground. Implementing it needs raw packet + bytes, which ``packet_file_to_datasets`` does not return, so this is not + a one-line fix. + * - Saturation above 4.0 MHz + - **Missing** + - "If count rates exceed 4.0 MHz using ``SWP_SCI.PCEM_CNT0`` through + ``PCEM_CNT5`` or ``SCEM_CNT0`` through ``SCEM_CNT5`` the sweep may later + be discarded." At 0.145 s live time, 4.0 MHz is ~580,000 counts - i.e. + compression region 2 or 3. Cheap to add once decompressed. + * - ``SWP_HK.PCEM_RATE_ST == 1`` during the sweep + - **Missing** + - The field exists in the XTCE and in the HK products. It is neither a + rejection criterion nor a bit in ``SWAPIFlags``. Note the near-miss: + ``PCEM_CNT_ST`` **is** flagged (bit 7), and it means something + different - "tripped but handled by FSW" versus "still exceeded despite + FSW measures". + * - ``SWP_HK.SCEM_RATE_ST == 1`` during the sweep + - **Missing** + - as above, and ``SCEM_CNT_ST`` is bit 8. + +Consequence: a sweep taken while the detector was saturated, or one whose +packets failed their checksum, currently flows through L1 and L2 unmarked. L3 +assumes L2 is already clean. + +Missing flags +------------- + +.. list-table:: + :header-rows: 1 + :widths: 26 12 62 + + * - Flag + - State + - Detail + * - ``SWP_SC_THRUSTER`` + - **Missing** + - **[DOC]** Section 10.2 adds this to ``SWP_L1_FLAGS`` at L1-L2 to + indicate data taken during spacecraft thruster activity, "namely, daily + repointing maneuvers". ``swapi_l2.py`` has + ``# TODO: add thruster firing flag``; ``SWAPIFlags`` has no bit for it. + Needs the MOC thruster firing history as an ancillary input, which the + SDC does not currently deliver to this pipeline. + * - "other flags" + - **Missing** + - ``swapi_l2.py`` ``# TODO: add other flags``. Unspecified. + * - ``PRELIMINARY_MAG`` + - N/A + - L3a alpha only. Correctly absent. + * - ``FIT_ERROR`` / ``BAD_FIT`` + - N/A + - L3a only. Correctly absent. + +Uncertainty gaps +---------------- + +.. list-table:: + :header-rows: 1 + :widths: 24 14 62 + + * - Item + - State + - Detail + * - Poisson term + - **Done** + - :math:`\sqrt{N}` at L1, divided by :math:`t_{\mathrm{live}}` at L2. + * - Compression term + - **Missing** + - **[DOC]** "The compression of counts also contributes to the + uncertainty. This uncertainty is estimated in an empirical way ... + generate some count rate numbers that span the range of higher count + rates, compress and un-compress these count rates and compare the + difference between the two cases. These differences are then used to + construct an empirical expression to estimate the error." The empirical + expression has never been delivered; the code carries an explicit + ``TODO``. Note that :math:`\sqrt{N}` on decompressed counts + **understates** the true uncertainty for every compressed sample - which + is exactly the high-rate solar wind core. + * - Asymmetric uncertainties + - **Deferred, per the document** + - ``_plus`` and ``_minus`` are identical arrays at both levels. The + document explicitly allows this ("will be the same, except if we modify + them to be asymmetric uncertainties in the future"), so the duplicated + variables are placeholders for a decision not yet made, not a bug. + * - Energy uncertainty :math:`\Delta(E/q)` + - **Missing** + - Required by the L3b flux uncertainty (equation 14). The passband edges + are in the LUT notes table's unused ``Lower Energy`` / ``Upper Energy`` + columns. + +Deviations from the document +---------------------------- + +These are design decisions or judgement calls, not bugs, but they will surprise +anyone reading the document first. + +Product descriptors do not match the document +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Document:** the L1 and L2 descriptor is ``12s``. + +**Code:** the descriptors are ``sci`` and ``hk``; logical sources are +``imap_swapi_l1_sci``, ``imap_swapi_l1a_hk``, ``imap_swapi_l1b_hk``, +``imap_swapi_l2_sci``. ``12s`` appears nowhere. + +**Consequence:** cosmetic for processing, but the document's filename table is +not a reliable guide to what is on disk. Do **not** rename products to match - +the SDC file catalog and every downstream consumer key on the current strings. + +The quality gate uses grouping criteria as rejection criteria +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Document:** constant ``PLAN_ID`` and ``SWEEP_TABLE`` across the 12 packets is +how you *group* packets into a sweep (10.1.1 step 3.b.ii). The rejection +criteria are the five in 10.1.2. + +**Code:** ``filter_good_data`` groups by timestamp continuity and then +*rejects* sweeps whose ``plan_id`` or ``sweep_table`` varies. Same outcome for +well-formed data; different outcome at a plan change boundary, where the +document's reading would start a new sweep and the code's discards one. + +Sweep grouping tests time, not sequence +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Document:** "the entire sweep is marked by ``SEQ_NUMBER`` 0-11; all packets +must be present to process a sweep." + +**Code:** ``find_sweep_starts`` requires ``seq_number == 0`` at the start and +then eleven consecutive 1-second timestamp gaps. It does not verify that the +intervening sequence numbers are 1..11. + +**Note:** the I-ALiRT code does the stricter check +(``np.array_equal(seq_values, np.arange(12))``). The two grouping +implementations are not equivalent, which is worth knowing if the two products +ever disagree about which sweeps exist. + +The overflow sentinel is a finite number +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Document:** on overflow, "``actual_value = ``". + +**Code:** ``np.iinfo(np.int32).max`` = 2147483647, cast to ``float32``. + +**Consequence:** it survives into L2 as a count rate of ~1.5e10 Hz. It is not +``NaN``, not a CDF ``FILLVAL``, and not flagged. Any downstream statistic that +does not filter on ``VALIDMAX`` will be destroyed by a single overflow sample. +The compression flag bit *is* set for those samples, so they are detectable - +but only if you know to look. + +Fine-step index flooring is silent +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Document:** the fine-step energies are at rows ``r_p16 - 4k``. It does not +say what to do if that goes negative. + +**Code:** negative ladder indices are clamped to 0, per SWAPI instruction +("flooring"). Nothing records that it happened - no flag, no log line - so +several fine steps can silently share one energy. + +L2 blanks negative rates but not their uncertainties +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Code:** ``l2_dataset[var].where(l2_dataset[var] >= 0, np.nan)`` is applied to +the three rate variables only. A step can therefore have a ``NaN`` rate and a +finite uncertainty. + +The ``k`` factor is never used +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Document:** :math:`k_{L2} = 1.93` eV/V/e is "the ``k``-factor estimated +pre-launch from lab measurements and used to convert from ESA energy in the L2 +CDF files to the actual ESA voltage of the instrument", and it differs from the +SIMION value :math:`k^{*} = 1.89` for reasons still under investigation. + +**Code:** the ADP's ``K factor`` and ``Voltage`` columns are read into the +DataFrame and ignored; L2 takes ``Energy`` directly. This is correct arithmetic +but the L2 product does not record which ``k`` its energies were built with, +and the ADP's own value changed between deliveries (1.88 in the 2025-02-11 +rows, 1.93 in the 2025-05-19 rows). + +I-ALiRT: fit window is 2+1+2, not 3+1+2 +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Document:** "three energy bins on the left and two on the right of +:math:`E_{\mathrm{peak}}`" - six points. + +**Code:** ``range(max_index - 2, max_index + 2 + 1)`` - five points, symmetric. + +**Consequence:** one fewer constraint, and asymmetric relative to the document, +which matters because the low-energy side of the proton peak is where the +shoulder lives. Worth asking the SWAPI team which is intended. + +I-ALiRT: averaging window is trailing and geometric +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Document:** "5 sweeps averaged (2 sweeps before, 2 sweeps after, and 1 +current)" - a **centred** window. The kind of average is not specified. + +**Code:** ``swapi_met_list[-5:]`` - the current sweep and the 4 before it, a +**trailing** window - and a **geometric** mean in log space, with ``NaN`` +sweeps excluded. + +**Consequence:** the reported ``swapi_epoch`` is the mean MET of the window, so +the timestamp is honest, but the record is offset ~24 s later than a centred +average would place it. For a real-time stream, trailing is arguably the right +engineering choice (a centred window cannot be computed until 2 sweeps in the +future have arrived, costing 24 s of latency) - but it is a deviation and it is +undocumented in the code. + +I-ALiRT: 63 energy steps, not 62 +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Document:** the I-ALiRT input is "coincidence count rates ... for 62 coarse +energy bins (1 sweep)". + +**Code:** ``NUM_IALIRT_ENERGY_STEPS = 63`` and the arrays are truncated to +``[:, :63]``. + +**Consequence:** step 0 - the ESA ramp-up step - is included. The science +pipeline explicitly ``NaN``\ s it; the I-ALiRT path does not. Since the fit +window is centred on the peak, which is far from step 0, this is unlikely to +change a result, but ``np.argmax`` over an array that includes a ramp-up +sample is not obviously safe. + +I-ALiRT: undocumented decompression offset +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Code:** ``raw_coin_count = raw_coin_count * 16 + 8``, unconditionally, with +the rationale in a comment ("add 8 to avoid having counts truncated to 0 and to +avoid counts being systematically too low"). The document says nothing about +I-ALiRT count compression at all. The mid-bin correction is defensible; it is +just not written down anywhere but the comment. + +I-ALiRT: :math:`R^2` threshold +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Code:** :math:`R^2 \ge 0.7` to accept a fit. The document specifies +:math:`R^2 \ge 0.9` for the **L3a** fits and gives no threshold for I-ALiRT. +The 0.7 is the code's own choice. + +I-ALiRT: density factor placement +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Document:** correct the pseudo density by scaling it to :math:`1/e` times the +5-sweep-averaged fitted density. + +**Code:** multiplies the *model* density by :math:`e` inside ``count_rate``, +which produces the same fitted result. The constant is named +``temporary_density_factor`` and the comment says it is "to be replaced once +SWAPI's L3 processing pipeline is finalized". Algebraically fine; flagged here +because it is applied before averaging rather than after, and because it is an +empirical fudge with a shelf life. + +Not written at all +------------------ + +.. list-table:: + :header-rows: 1 + :widths: 26 74 + + * - Item + - Notes + * - Quick-look products + - **[DOC]** Section 15: colour-coded spectrograms of count rate vs + energy/charge vs time. Figure 25 shows the daily plot - COIN rate vs + E/q (top), spectrogram (middle), MAG L1D :math:`B_R, B_T, B_N, |B|` + (bottom), with the He\ :sup:`+` PUI cutoff energy marked. No SWAPI code + in this repository produces it. + * - ``SWP_LGSCI`` processing + - "Large" science - all 6 ESA levels per sample instead of one, used for + testing. Not in the XTCE, no APID enum, not processed. + * - ``SWP_AUT`` processing + - APID 1192 is in ``SWAPIAPID`` but there is no XTCE container and no + code path. Autonomy power off/cycle flags. + * - ``SWP_MG`` / ``SWP_MD`` processing + - Event messages and memory dump. Not in the XTCE, not processed. Probably + correctly - they are engineering-only. + * - ``orbnum`` in the L1 product + - **[DOC]** Table 2 lists it ("orbit number in which observation began, + generated by MOC and stored at SDC"). Absent from the CDF. + * - ETE-model validation data + - **[DOC]** Section 12 defines the intended verification method: the SWAPI + team runs the end-to-end model and supplies input/output pairs so the + SDC can verify the implementation. No such pairs are in the repository. + The only SWAPI-supplied validation data is a decommutation check on one + pre-launch idle packet. + * - Compression-uncertainty empirical expression + - See above. Blocked on the SWAPI team. + +Test coverage +------------- + +**[CODE]** ``imap_processing/tests/swapi/`` and +``imap_processing/tests/ialirt/unit/test_process_swapi.py``. + +.. list-table:: + :header-rows: 1 + :widths: 34 22 44 + + * - Area + - Coverage + - Notes + * - Decommutation + - **Validated** + - Field-by-field against SWAPI CSV exports for the first ``SWP_SCI`` and + ``SWP_HK`` packet, plus packet counts. + * - Sweep grouping + - **Unit tested** + - ``find_sweep_starts``, ``get_indices_of_full_sweep``, + ``filter_good_data``. + * - Decompression + - **Unit tested** + - ``decompress_count``. + * - Reordering + - **Unit tested** + - via ``test_swapi_algorithm`` / ``test_process_swapi_science``. + * - L1 CDF write + - **Tested** + - Round trip through ``write_cdf()``. + * - L2 rates + - **Tested** + - Rate and uncertainty arrays checked against + ``counts / SWAPI_LIVETIME``. + * - L2 energy solve + - **Tested** + - ``test_solve_full_sweep_energy`` pins the 63 fixed energies and the 9 + fine energies for ``esa_lvl5 = 4663``, sweep table 0. + * - L2 CDF attributes + - **Heavily tested** + - ISTP attributes, ``DEPEND_1`` rewiring, ``VALIDMIN``/``VALIDMAX``, + ``DELTA_PLUS_VAR``/``DELTA_MINUS_VAR``. + * - I-ALiRT + - **Unit tested** + - ``count_rate``, ``optimize_pseudo_parameters``, ``geometric_mean``, + and the full ``process_swapi_ialirt`` output keys. + * - Overflow sentinel handling + - **Untested end to end** + - ``decompress_count`` is tested; nothing checks what the sentinel does to + L2 rates. + * - Index flooring + - **Untested** + - The clamp in ``solve_full_sweep_energy`` has no test. + * - Flight-representative science + - **Not tested** + - The only L0 test data is pre-launch idle data from 2024-09-24. + +Where to start, if you are picking up work +------------------------------------------ + +Rough order of value, highest first: + +1. **The three missing L1 rejection criteria** (saturation, ``PCEM_RATE_ST``, + ``SCEM_RATE_ST``). Cheap, and they are the difference between "L2 is clean" + being true and being assumed. Saturation needs a threshold check on + decompressed counts; the two ``RATE_ST`` conditions need two more + ``SWAPIFlags`` bits and two more entries in ``hk_flags_name``. +2. **Do something visible about the overflow sentinel** - at minimum a test + showing what it produces at L2, ideally a fill value or a flag. +3. **The thruster flag.** Blocked on an ancillary input, so the real work is + agreeing with the SDC on how thruster history reaches this pipeline. +4. **Checksum verification.** Real work, because it needs raw packet bytes that + ``packet_file_to_datasets`` does not surface. +5. **Reconcile the I-ALiRT fit window** with the document, or record the + deviation as intentional. +6. **Ask the SWAPI team for the compression-uncertainty expression** and for + ETE input/output validation pairs. Both are blocking, neither is ours to + invent. diff --git a/docs/source/algorithm-code-documentation/swapi/index.rst b/docs/source/algorithm-code-documentation/swapi/index.rst new file mode 100644 index 0000000000..46863cb4c1 --- /dev/null +++ b/docs/source/algorithm-code-documentation/swapi/index.rst @@ -0,0 +1,203 @@ +:orphan: + +.. _swapi-index: + +SWAPI +===== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +.. currentmodule:: imap_processing.swapi + +This is the SWAPI (Solar Wind and Pickup Ion) instrument module, which contains +the code for processing data from the SWAPI instrument. + +Purpose of these pages +---------------------- + +These pages are a **condensed, self-contained working reference** for the SWAPI +processing algorithms, written so that a developer (human or AI agent) can get +productive without reading the full algorithm document. + +They are a summary of the source document below plus what the code in +``imap_processing/swapi`` actually does. Where the two disagree, that is called +out explicitly in :ref:`swapi-implementation-status`. + +.. _swapi-source-documents: + +Source documents +---------------- + +**None of these are redistributed in this repository.** This is an open-source +repository and the mission documents are not ours to publish. Request them from +the SWAPI instrument team at Princeton or the SDC document store. + +.. list-table:: + :header-rows: 1 + :widths: 22 78 + + * - Short name + - Reference + * - **Algorithm document** + - 05899-Algorithms_AN, *IMAP SWAPI Instrument Algorithms Document*, + Version 07, 2026-06-03. Prepared by Bishwas Shrestha (SWAPI Algorithms + Lead) and Margaret Shaw-Lecerf; approved by Eric Zirnstein (Data + Analysis Lead), Jamie Rankin (Instrument Lead) and Scott Weidner + (Project Manager). Princeton University, Department of Astrophysical + Sciences, Space Physics Group. 80 pages. The primary source for these + pages. + * - **ESA Unit Conversion ADP** + - ``imap_swapi_esa-unit-conversion___.xlsx``, + delivered by the SWAPI team. The workbook's main sheet and its + ``LUT_Notes_vx`` sheet are both required to derive L2 energies. The SDC + ingests them as two CSV ancillary files (``esa-unit-conversion`` and + ``lut-notes``) - see :ref:`swapi-ancillary`. + * - **Packet ICD** + - Field-level packet definitions. Superseded in practice by + ``imap_processing/swapi/packet_definitions/swapi_packet_definition.xml``, + which is what the code parses. + * - **SDC ICDs** + - LASP 167441 (POC-to-Instrument-Team) and LASP 167442 + (SDC-to-Instrument-Team). Referenced by the algorithm document for + filenames and L0 delivery; not needed by any code here. + * - **Instrument paper** + - Rankin et al. 2025, SSRv, 221, 108. The authority for the calibration + numbers (geometric factor, passbands, ``k`` factor) the algorithm + document quotes. + +.. tip:: + + If you hold a copy of the algorithm document, put it in ``docs/reference/``. + That directory is gitignored, so it will never be committed, and the section + index in :ref:`swapi-reference-tables` is written against that location. + +.. important:: + + Two conventions used throughout these pages: + + * **[DOC]** marks a statement taken from the algorithm document. It describes + the intended behavior, which may not be what the code does yet. + * **[CODE]** marks a statement verified against ``imap_processing/swapi`` + (or ``imap_processing/ialirt`` for the real-time product). + + When those conflict, the code is what runs and the document is what the + instrument team expects. Both matter; do not silently "fix" one to match the + other without asking. + +.. warning:: + + **This repository stops at L2.** SWAPI's L3a (solar wind proton, alpha and + pickup-helium plasma parameters) and L3b (combined differential flux) are + produced by the SWAPI team's own code, delivered as a separate Docker + container. Section 13 of the algorithm document states this explicitly. + Roughly two thirds of the algorithm document (sections 7.1.1, 9.5, 10.4, + 10.5) describes work that does **not** belong here. See + :ref:`swapi-l3-scope` before starting anything that looks like a fit. + +Which page to read +------------------ + +Read only what you need. Each page is designed to be loaded on its own. + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Page + - Read it when you need to know... + * - :ref:`swapi-overview` + - What SWAPI physically is, how a measurement happens, and the vocabulary + (sweep, ESA step, coarse/fine, sweep plan, PCEM/SCEM/COIN, sunglasses, + open aperture, live time). **Start here if you are new.** + * - :ref:`swapi-data-products` + - The full product inventory, exact ``logical_source`` strings, what feeds + what, and how the CLI is wired. **The "what goes into what" map.** + * - :ref:`swapi-l1` + - Packet decommutation, sweep grouping, count decompression, quality + flags, count uncertainty, and the L1 CDF contents. + * - :ref:`swapi-l2` + - Counts to rates, the live-time constant, and the ESA step-to-energy + solve (the only real algorithm at L2). + * - :ref:`swapi-ancillary` + - The ESA unit conversion ADP, the LUT notes table, the efficiency and + gain-test LUTs, the instrument response CSVs, and the external + dependencies (MAG, SPICE, thruster history). + * - :ref:`swapi-ialirt` + - The 12-second real-time space-weather product and its analytical + pseudo-moment fit. + * - :ref:`swapi-l3-scope` + - What L3a/L3b are, why they are not in this repository, and exactly what + L2 has to hand them. **Read before writing any fitting code.** + * - :ref:`swapi-implementation-status` + - What is implemented, what is stubbed, where the code deviates from the + document, and what is not written at all. **Read before proposing + work.** + * - :ref:`swapi-reference-tables` + - Where the big tables live (XTCE, ancillary CSVs, PDF page ranges). + Deliberately *not* reproduced inline. + +.. toctree:: + :maxdepth: 1 + + overview + data-products + l1 + l2 + ancillary + ialirt + l3-scope + implementation-status + reference-tables + +Ten-second orientation +---------------------- + +* SWAPI is a **top-hat electrostatic analyzer with a coincidence detector**, + derived from New Horizons' SWAP. It measures solar wind protons + (H\ :sup:`+`) and alphas (He\ :sup:`2+`) plus interstellar pickup ions + (H\ :sup:`+` and He\ :sup:`+`, helium dominated). +* Solar wind enters through a **0.1% transmissive grid** ("the sunglasses") so + the bright core beam does not saturate the detector. Pickup ions enter + through the **open aperture** unattenuated. Both populations land on the same + detector; which aperture a count came from is an *angular* distinction, not a + separate channel. +* Two channel electron multipliers - **PCEM** (primary, sees the ion) and + **SCEM** (secondary, sees carbon-foil secondary electrons). Hits within + 100 ns are also counted as **COIN** (coincidence). Every science product is + three numbers per energy step: PCEM, SCEM, COIN counts. +* The **sweep is the fundamental unit**: 12 seconds, 72 ESA steps, telemetered + as 12 one-second packets of 6 steps each. One L1/L2 record is one sweep. +* The 72 steps are **1 ramp-up + 62 coarse (0.1-20 keV/q) + 9 fine** steps. The + fine steps move with the solar wind peak and are defined by the active + **sweep plan**; their energies must be *solved for* at L2, they are not in a + fixed table. +* Processing chain in this repository: + ``CCSDS packets -> L1 (counts/sweep, CDF) -> L2 (count rates + energies)``. + L3a/L3b are the SWAPI team's, not ours. +* There is also a **12-second I-ALiRT** product (coincidence counts only, + pseudo speed/density/temperature) built from a different APID and living + under ``imap_processing/ialirt/``. + +Where the code lives +-------------------- + +.. code-block:: text + + imap_processing/swapi/ + constants.py NUM_PACKETS_PER_SWEEP=12, NUM_ENERGY_STEPS=72 + swapi_utils.py SWAPIAPID, SWAPIMODE, read_swapi_lut_table() + packet_definitions/ + swapi_packet_definition.xml SWP_HK (102 fields) + SWP_SCI (46 fields) + l1/swapi_l1.py packets -> L1 science + L1A/L1B housekeeping + l2/swapi_l2.py counts -> rates, ESA step -> energy + + imap_processing/ialirt/ + l0/process_swapi.py the whole I-ALiRT algorithm + constants.py (IalirtSwapiConstants) A_eff, dE/E, azimuth FOV, density factor + packet_definitions/ialirt_swapi.xml SWAPI I-ALiRT packet (APID 1187) + + imap_processing/cdf/config/imap_swapi_global_cdf_attrs.yaml + imap_processing/cdf/config/imap_swapi_variable_attrs.yaml + imap_processing/quality_flags.py (class SWAPIFlags) + imap_processing/cli.py (class Swapi) dependency wiring per level + imap_processing/tests/swapi/ tests, L0 test data, validation CSVs, LUTs diff --git a/docs/source/algorithm-code-documentation/swapi/l1.rst b/docs/source/algorithm-code-documentation/swapi/l1.rst new file mode 100644 index 0000000000..3f315e2f88 --- /dev/null +++ b/docs/source/algorithm-code-documentation/swapi/l1.rst @@ -0,0 +1,462 @@ +.. _swapi-l1: + +L0 to L1 - Decommutation, Grouping and Counts +============================================= + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**Code:** ``imap_processing/swapi/l1/swapi_l1.py`` (845 lines, the whole of L1). + +**Document:** section 10.1. + +L1 has one job: turn a day of 1-second CCSDS packets into an array of complete +12-second sweeps, with the counts decompressed, the bad sweeps removed, and the +quality flags set. No physical units are applied. **[DOC]** "Fully +decommutated but uncalibrated raw data at full resolution." + +The pipeline +------------ + +**[CODE]** ``process_swapi_science(sci_dataset, hk_dataset)``: + +.. code-block:: text + + 1. get_indices_of_full_sweep() find complete 12-packet sweeps + 2. filter_good_data() drop sweeps failing the quality gate + 3. process_sweep_data() reorder 6-per-packet -> 72-per-sweep (x6) + 4. decompress_count() undo the divide-by-16 compression (x3) + 5. force step 0 to NaN the ESA ramp-up step + 6. build quality flag bitfield from compression flags + nearest HK + 7. assemble the xr.Dataset epoch = packet 7's time + 8. uncertainty = sqrt(counts) + +Housekeeping takes a much shorter path: decommutate twice +(``use_derived_value=False`` then ``True``), attach attributes, return both. + +Step 1 - finding complete sweeps +-------------------------------- + +**[DOC]** Twelve consecutive seconds of science data correspond to a single +sweep, which is the unit of L1 and all downstream products. A sweep begins at +``SWP_SCI.SEQ_NUMBER == 0`` and ends at ``SEQ_NUMBER == 11``; **all 12 packets +must be present** to process a sweep. Only complete sweeps (72 steps) are +processed. + +**[CODE]** ``find_sweep_starts(packets)`` implements this as a sliding-window +test. An index qualifies as a sweep start when: + +* ``seq_number == 0`` at that index, **and** +* the ``shcoarse`` difference to the next packet is exactly 1 second, for each + of the following 11 gaps. + +.. code-block:: python + + diff = packets["shcoarse"].data[1:] - packets["shcoarse"].data[:-1] + ione = diff == 1 + valid = (packets["seq_number"] == 0)[:-11] & ione[:-10] & ione[1:-9] & ... & ione[10:] + +``get_indices_of_full_sweep()`` then expands each start index by +``+ np.arange(12)`` to give the flat list of packet indices belonging to +complete sweeps. + +.. note:: + + The check is on **packet timestamps being 1 second apart**, not on + ``seq_number`` running 0..11. A sequence of 12 one-second-spaced packets + starting at ``seq_number == 0`` is accepted even if the intervening sequence + numbers were wrong. In practice they are not, but the guard is on time, not + on sequence. (Contrast with the I-ALiRT code, which explicitly asserts + ``np.array_equal(seq_values, np.arange(12))``.) + +.. note:: + + ``find_sweep_starts`` returns early with an empty array if fewer than 12 + packets are present, so a short file produces an empty product rather than + an error. + +Step 2 - the quality gate +------------------------- + +**[DOC]** Section 10.1.2, "Marking data not qualified for L0-L1 processing". +L0 data will be marked as ``NaN`` at L0-L2 and potentially rejected for L3 for +any of: + +.. list-table:: + :header-rows: 1 + :widths: 34 40 26 + + * - Condition + - Meaning + - Implemented? + * - ``SWP_HK.CHKSUM`` is wrong + - 8-bit XOR of all packet bytes (including the CCSDS header, excluding the + checksum byte) does not match + - **No** + * - ``SWP_SCI.MODE`` is not ``HVSCI`` + - Instrument not in high-voltage science mode + - **Yes** + * - Saturation: any of ``PCEM_CNT0..5`` or ``SCEM_CNT0..5`` implies a count + rate above **4.0 MHz** + - Detector saturated; the sweep "may later be discarded" + - **No** + * - ``SWP_HK.PCEM_RATE_ST == 1`` at any point during the sweep + - PCEM count-rate threshold exceeded and *still* exceeded despite FSW + action + - **No** + * - ``SWP_HK.SCEM_RATE_ST == 1`` at any point during the sweep + - as above for SCEM + - **No** + +**[CODE]** ``filter_good_data(full_sweep_sci)`` implements a *different* gate. +A sweep is kept when, across its 12 packets: + +* ``sweep_table`` is constant, **and** +* ``plan_id`` is constant, **and** +* ``mode == SWAPIMODE.HVSCI`` for every packet. + +The first two are the document's *grouping* criteria (10.1.1 step 3.b.ii) +promoted to rejection criteria; the last is the only one of the five +document rejection criteria that made it in. Checksum verification, the 4 MHz +saturation test, and the two ``RATE_ST`` tests are **not implemented**. See +:ref:`swapi-implementation-status`. + +.. warning:: + + There is a naming trap in ``filter_good_data``. The local variable + ``bad_data_indices`` holds the boolean mask of **good** sweeps; the code + then takes ``np.where(bad_data_indices == 0)`` to find the bad ones. The + logic is correct, the name is inverted. + +**[DOC]** For context, the flight software also transitions the instrument to +SAFE automatically if PCEM/SCEM current, voltage or temperature sensors are +out of limit on two consecutive samples, or if PCEM/SCEM count rate is out of +limit (>4.0 MHz) on six or more consecutive samples. Ground processing never +has to do anything about this beyond noticing the mode change. + +Step 3 - reordering 6-per-packet into 72-per-sweep +-------------------------------------------------- + +**[CODE]** ``process_sweep_data(full_sweep_sci, cem_prefix)``. This is the +fiddliest part of L1 and it is worth understanding before changing anything. + +Each packet carries 6 samples as **6 separate named fields** +(``PCEM_CNT0`` ... ``PCEM_CNT5``), one per 1/6-second interval. After +decommutation those become 6 flat arrays indexed by packet. The sweep's 72 +values need packet-major, then sample-minor ordering: + +.. code-block:: text + + Wanted: step 0 1 2 3 4 5 6 7 8 9 10 11 12 ... + from pkt0 pkt1 pkt2 + field CNT0..CNT5 CNT0..CNT5 CNT0..CNT5 + +The implementation is three reshapes: + +.. code-block:: python + + current = np.concatenate([ds[f"{prefix}{i}"] for i in range(6)], axis=0) + current = current.reshape(6, -1, NUM_PACKETS_PER_SWEEP) # (cem_sample, sweep, packet) + all_cem = np.stack(current, axis=-1) # (sweep, packet, cem_sample) + all_cem = all_cem.reshape(-1, NUM_ENERGY_STEPS) # (sweep, 72) + +Called six times per invocation - three for the counts +(``pcem_cnt``, ``scem_cnt``, ``coin_cnt``) and three for the compression +indicators (``pcem_rng_st``, ``scem_rng_st``, ``coin_rng_st``). + +.. tip:: + + ``process_sweep_data`` is also imported and reused by the I-ALiRT code + (``imap_processing/ialirt/l0/process_swapi.py``) with the prefix + ``swapi_coin_cnt``. If you change its signature or ordering, you change the + real-time product too. + +Step 4 - count decompression +---------------------------- + +**[DOC]** Counts are compressed on board by a **divide-by-16**. Each sample +carries a range-status bit (``XXX_RNG_ST0..5``) saying whether the value was +compressed. There are three compression regions: + +.. list-table:: + :header-rows: 1 + :widths: 14 34 20 32 + + * - Region + - True count range + - ``RNG_ST`` + - Telemetered value + * - 1 + - :math:`0 \le n \le 65{,}535` + - 0 + - the count itself + * - 2 + - :math:`65{,}536 \le n \le 1{,}048{,}575` + - 1 + - :math:`n / 16` + * - 3 + - :math:`n \ge 1{,}048{,}576` + - 1 + - ``0xFFFF`` (overflow) + +The document's pseudocode: + +.. code-block:: text + + if XXX_RNG_ST0 == 0: # Uncompressed + actual_value = XXX_CNT0 + elif XXX_RNG_ST0 == 1 and XXX_CNT0 == 0xFFFF: # Overflow + actual_value = + elif XXX_RNG_ST0 == 1 and XXX_CNT0 != 0xFFFF: + actual_value = XXX_CNT0 * 16 + +**[CODE]** ``decompress_count(count_data, compression_flag)`` implements this +exactly, with two concrete choices the document leaves open: + +* The **overflow sentinel** is ``np.iinfo(np.int32).max`` (2147483647), + "per SWAPI's suggestion to use a big value". +* The function raises ``ValueError`` if any input count is negative, because + the field must be unsigned. +* The return type is ``float32``, so the overflow sentinel survives as + ``2.1474836e+09`` and the array can carry ``NaN``. + +.. warning:: + + The overflow sentinel is an enormous *finite* number, not a fill value. It + propagates straight into L2: ``2147483647 / 0.145`` is a count rate of + ``1.48e10`` Hz. Nothing in the current L1 or L2 code flags or masks it. If + you are debugging absurd rates, check for the sentinel first. + +Step 5 - the ramp-up step +------------------------- + +**[CODE]** Immediately after decompression: + +.. code-block:: python + + swp_pcem_counts[:, 0] = np.nan + swp_scem_counts[:, 0] = np.nan + swp_coin_counts[:, 0] = np.nan + +**[DOC]** ESA step 0 is the "1 step to ramp up to full ESA voltage" and carries +no usable science. The code comment records that this was SWAPI's instruction +and that ``NaN`` (rather than dropping the step) was chosen because it helps +with plotting. + +Note that the compression flags and quality flags for step 0 are **not** +blanked - only the counts. + +Step 6 - quality flags +---------------------- + +**[CODE]** ``swp_l1a_flags``, one ``uint16`` per (sweep, ESA step), assembled +from two sources. The bit assignments live in +``imap_processing/quality_flags.py``, ``class SWAPIFlags``: + +.. list-table:: + :header-rows: 1 + :widths: 10 30 18 42 + + * - Bit + - Flag + - Source + - Meaning + * - 0 + - ``INF`` + - common + - Value is infinite + * - 1 + - ``NEG`` + - common + - Value is negative + * - 2 + - ``SWP_PCEM_COMP`` + - science + - PCEM count was compressed at this step + * - 3 + - ``SWP_SCEM_COMP`` + - science + - SCEM count was compressed + * - 4 + - ``SWP_COIN_COMP`` + - science + - COIN count was compressed + * - 5 + - ``OVR_T_ST`` + - HK + - Upper temperature limit of a thermistor exceeded + * - 6 + - ``UND_T_ST`` + - HK + - Lower temperature limit of a thermistor exceeded + * - 7 + - ``PCEM_CNT_ST`` + - HK + - PCEM count rate tripped but handled by FSW + * - 8 + - ``SCEM_CNT_ST`` + - HK + - SCEM count rate tripped but handled by FSW + * - 9 + - ``PCEM_V_ST`` + - HK + - PCEM voltage tolerance exceeded + * - 10 + - ``PCEM_I_ST`` + - HK + - PCEM current threshold exceeded despite FSW measures + * - 11 + - ``PCEM_INT_ST`` + - HK + - PCEM current interrupt tripped but handled by FSW + * - 12 + - ``SCEM_V_ST`` + - HK + - SCEM voltage tolerance exceeded + * - 13 + - ``SCEM_I_ST`` + - HK + - SCEM current threshold exceeded despite FSW measures + * - 14 + - ``SCEM_INT_ST`` + - HK + - SCEM current interrupt tripped but handled by FSW + +Compression bits are per-step and come straight from the reordered +``*_rng_st`` arrays. The ten HK bits are per-**packet** and are broadcast: + +.. code-block:: python + + hk_dataset = hk_dataset.drop_duplicates("epoch") + good_sweep_hk_data = hk_dataset.sel({"epoch": good_sweep_times}, method="nearest") + ... + current_flag = np.repeat(good_sweep_hk_data[flag_name.lower()].data, 6).reshape(-1, 72) + +**[CODE]** The reasoning, recorded in the source: HK and SCI cadences are not +both 1 second in flight - in HVSCI, HK nominally comes every 60 s and SCI every +12 s - but *both are sampled at 1 Hz on board*, so ground processing uses the +**nearest-timestamp HK packet** for each SCI packet, per the SWAPI team. Since +one SCI packet holds 6 measurements from the same packet, its HK flag is +repeated 6 times. + +**[DOC]** Table 3 also lists ``SWP_PCEM_RNG_ST#COMP`` (singular, "pull from +``SWP_SCI.PCEM_RNG_ST0``") where the code sets a single ``SWP_PCEM_COMP`` bit +from any of the six. The code's behaviour is the sensible reading. + +Two document flags that are **not** in ``SWAPIFlags``: ``PCEM_RATE_ST`` and +``SCEM_RATE_ST``. In the document these are rejection criteria (section +10.1.2), not flags - and they are neither rejected nor flagged here. + +Step 8 - count uncertainty +-------------------------- + +**[DOC]** Uncertainty is quantified for the PCEM, SCEM and COIN counts. The +Poisson contribution is + +.. math:: + + \Delta n = \sqrt{n} + +The upper (``_ERR_PLUS``) and lower (``_ERR_MINUS``) uncertainties are the same +"except if we modify them to be asymmetric uncertainties in the future." + +**[DOC]** *Compression of the counts also contributes to the uncertainty.* That +contribution is to be estimated empirically: generate count rates spanning the +high-rate range, compress and un-compress them, compare, and fit an empirical +expression to the differences. + +**[CODE]** Only the Poisson term is implemented, and the source carries an +explicit ``TODO`` saying the formula "will change in the future - replace it +with the actual formula once SWAPI provides it." Plus and minus are identical +arrays. + +.. code-block:: python + + dataset["swp_pcem_counts_stat_uncert_plus"] = np.sqrt(swp_pcem_counts) + dataset["swp_pcem_counts_stat_uncert_minus"] = np.sqrt(swp_pcem_counts) + +Note that :math:`\sqrt{N}` is applied to the **decompressed** counts, which is +correct for region 1 but understates the uncertainty in regions 2 and 3 - which +is precisely the gap the compression term is meant to fill. + +Housekeeping processing +----------------------- + +**[CODE]** The ``hk`` descriptor branch of ``swapi_l1()``: + +.. code-block:: python + + l1a_hk_data = packet_file_to_datasets(f, xtce, use_derived_value=False)[SWP_HK] + l1b_hk_data = packet_file_to_datasets(f, xtce, use_derived_value=True)[SWP_HK] + +Two datasets from the same packets - L1A is raw DN, L1B is the derived +engineering value from the XTCE calibrators and enumerations. There is no +algorithm; the difference is entirely ``use_derived_value``. + +One wrinkle worth knowing: some derived HK values are **strings** (e.g. +``SWP_HK.PCEM_SAFE`` is raw 0/1 but derives to ``'OK'``/``'ERR'``), so the L1B +branch selects a different attribute set (``l1b_hk_string_attrs``) for string +variables to stay ISTP compliant. + +L1 CDF contents vs the document +------------------------------- + +**[DOC]** Table 2 lists the L1 product fields. Many of them are *filename and +provenance* fields that the SDC supplies rather than the algorithm: + +.. list-table:: + :header-rows: 1 + :widths: 30 20 50 + + * - Document field + - Where it lives now + - Note + * - ``Instrument``, ``dataLevel``, ``versionInfo``, ``filetype``, + ``dataProductDescriptor`` + - CDF global attributes / filename + - Handled by ``ImapCdfAttributes`` and ``write_cdf()`` + * - ``startTime`` + - ``sci_start_time`` + - UTC string, ms precision. The document's format is + ``yyyymmddthhmmss``; the code writes ISO with milliseconds (``A23``). + * - ``orbnum`` + - **absent** + - Orbit number, "generated by MOC and stored at SDC". Not in the CDF. + * - ``mode`` + - **absent as a variable** + - Only implicitly present: non-HVSCI sweeps are dropped. + * - ``predict/reconstruct`` + - filename + - ``p``/``r`` per ephemeris kind; SDC-level concern. + * - ``SWP_HK.LUT_CHOICE``, ``FPGA_TYPE``, ``FPGA_REV`` + - ``lut_choice``, ``fpga_type``, ``fpga_rev`` + - Present. + * - ``SWP_SCI.SWEEP_TABLE``, ``SWP_SCI.PLAN_ID`` + - ``sweep_table``, ``plan_id`` + - Present. + * - ``SWP_L1_FLAGS`` + - ``swp_l1a_flags`` + - Present, renamed. + * - ``ESA_LVL5`` + - ``esa_lvl5`` + - Present. The document calls it a 16-bit int; the code carries the raw + integer and converts to a 4-digit hex string at L2. + * - ``SWP_{PCEM,SCEM,COIN}_COUNTS`` (+ ``_ERR_PLUS``/``_ERR_MINUS``) + - ``swp_*_counts``, ``swp_*_counts_stat_uncert_{plus,minus}`` + - Present, renamed to the mission-wide ``stat_uncert`` convention. + +Tests +----- + +**[CODE]** ``imap_processing/tests/swapi/``: + +* ``test_swapi_decom.py`` - validates decommutation of the first ``SWP_SCI`` + and ``SWP_HK`` packet field-by-field against SWAPI-supplied CSV exports + (``idle_export_raw.SWP_SCI_20240924_080204.csv`` and the HK equivalent), and + checks the packet counts. +* ``test_swapi_l1.py`` - unit tests for ``filter_good_data``, + ``decompress_count``, ``find_sweep_starts``, ``get_indices_of_full_sweep``, + the reordering, and an end-to-end ``write_cdf()`` round trip. + +The L0 test data (``imap_swapi_l0_raw_20240924_v001.pkts``) is **pre-launch +idle data from 2024-09-24**, so the tests exercise the plumbing rather than +flight-representative science. diff --git a/docs/source/algorithm-code-documentation/swapi/l2.rst b/docs/source/algorithm-code-documentation/swapi/l2.rst new file mode 100644 index 0000000000..1e0112a6a3 --- /dev/null +++ b/docs/source/algorithm-code-documentation/swapi/l2.rst @@ -0,0 +1,293 @@ +.. _swapi-l2: + +L1 to L2 - Count Rates and ESA Energies +======================================= + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**Code:** ``imap_processing/swapi/l2/swapi_l2.py`` (312 lines). + +**Document:** sections 10.2 and 10.3. + +L2 is **the last level this repository produces for SWAPI**. It does two +things: divide counts by the live time to get rates, and work out what energy +each of the 72 ESA steps actually was. The second is the only non-trivial +algorithm at L2, and it is the reason L2 needs ancillary files at all. + +**[DOC]** "No additional rejection, culling, or flagging of suspect/bad data is +performed during L1-L2 processing." + +Counts to rates +--------------- + +**[DOC]** L1-L2 processing divides counts by time to produce count rates and +updates the uncertainty: + +.. math:: + + \mathrm{SWP\_PCEM\_RATE} = \frac{\mathrm{SWP\_PCEM\_COUNTS}}{t_{\mathrm{live}}} + \qquad + \mathrm{SWP\_PCEM\_RATE\_ERR\_PLUS} = + \frac{\mathrm{SWP\_PCEM\_COUNTS\_ERR\_PLUS}}{t_{\mathrm{live}}} + +and the same for SCEM and COIN. + +.. _swapi-livetime: + +The live time constant +^^^^^^^^^^^^^^^^^^^^^^ + +.. math:: + + t_{\mathrm{live}} = 0.145\ \mathrm{s} + +**[DOC]** The exposure time per energy bin is one complete sweep (12 s, coarse ++ fine) divided by the total energy steps (72), giving 0.167 s - but the actual +time spent acquiring counts is **less** than that, because some of each step is +spent settling the HVPS. The live time is 0.145 s. + +**[CODE]** ``swapi_l2.SWAPI_LIVETIME = 0.145``. It is a module-level constant, +not an ancillary input, and it is **also imported by the I-ALiRT code**. + +.. warning:: + + ``SWAPI_LIVETIME`` is hard-coded. **[DOC]** The L2 ancillary list in section + 10.6 says the L2 product requires an "accumulation time per ESA step (TBC)" + ancillary file. If the SWAPI team ever delivers a per-step or time-varying + live time, this constant becomes wrong for both L2 and I-ALiRT + simultaneously. + +**[DOC]** The Poisson uncertainty in the rate is + +.. math:: + + \Delta C = \frac{\sqrt{N}}{t_{\mathrm{bin}}} + +with :math:`N` the counts per energy bin. Since L1 already stored +:math:`\sqrt{N}`, dividing the L1 uncertainty by :math:`t_{\mathrm{live}}` is +the same thing. Upper and lower rate uncertainties are the same, "except if we +modify them to be asymmetric uncertainties in the future." + +**[CODE]** Implemented verbatim, with one addition the document does not +mention: + +.. code-block:: python + + for var in ["swp_pcem_rate", "swp_scem_rate", "swp_coin_rate"]: + l2_dataset[var] = l2_dataset[var].where(l2_dataset[var] >= 0, np.nan) + +Negative rates (which can only arise from L1 fill values) become ``NaN``. The +uncertainties are **not** given the same treatment, so a step can end up with +a ``NaN`` rate and a finite uncertainty. + +.. _swapi-esa-energy-solve: + +The ESA step-to-energy solve +---------------------------- + +**[DOC]** Section 10.3. Ground processing from L1 to L2 includes the conversion +of ESA step number (0 to 71) to energy (keV). The SWAPI Instrument Team +provides ESA Unit Conversion ADP updates as needed. + +**Code:** ``solve_full_sweep_energy(esa_lvl5_data, sweep_table, esa_table_df, +lut_notes_df, data_time)``. + +Why it is not a simple lookup +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +The 62 coarse steps sit at fixed voltages, so their energies are tabulated. The +**9 fine steps move with the solar wind peak**, so their voltages - and +therefore energies - differ sweep to sweep. The instrument tells you where they +were only indirectly: ``SWP_SCI.ESA_LVL5`` from the last packet of the sweep +gives the ESA DAC setting of the **71st** step, and the other 8 fine steps are +at fixed *index offsets* from it on a monotonic DAC-to-energy ladder. + +So the algorithm is: look up the fixed energies, then walk backwards down the +ladder from the last step. + +The document's procedure +^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** To convert the ESA step number to an energy for a single 72-step +sweep: + +* For ESA step numbers **0-62**, simply use the corresponding ``Energy`` in the + ADP. +* For ESA step numbers **63-71** (where ``Voltage`` and ``Energy`` both read + ``Solve``): + + a. Find ``v_p16`` = ``SWP_SCI.ESA_LVL5`` at ``SWP_SCI.SEQ_NUMBER == 11`` in + this sweep. + b. On the ``LUT_Notes_vx`` tab of the ADP (matching the ``LUT version + number`` column), find ``r_p16``, the row number of that voltage in + column B (ESA Voltage). + c. Energy for step 71 is column C (Energy) on row ``r_p16``. + d. Energy for step 70 is column C on row ``r_p16 - 4``. + e. ...continuing in steps of 4: step 69 at ``r_p16 - 8``, 68 at + ``r_p16 - 12``, 67 at ``r_p16 - 16``, 66 at ``r_p16 - 20``, 65 at + ``r_p16 - 24``, 64 at ``r_p16 - 28``, 63 at ``r_p16 - 32``. + +The ``-4`` spacing is not magic: it is exactly the ``ESA Index Number`` column +of the ADP, which reads ``-16, -12, -8, -4, 0, 4, 8, 12, 16`` for steps 63-71. + +The implementation +^^^^^^^^^^^^^^^^^^ + +**[CODE]** ``solve_full_sweep_energy`` loops over sweeps. Per sweep: + +**1. Select the right ADP rows.** Filter the ESA table to rows whose +``timestamp <= sci_start_time`` **and** whose ``Sweep #`` equals this sweep's +``sweep_table`` value, sort by ``(timestamp, ESA Step #)`` and take the last 72 +rows - i.e. the most recent applicable table version for this sweep table. + +.. code-block:: python + + subset = esa_table_df[(esa_table_df["timestamp"] <= time) + & (esa_table_df["Sweep #"] == sweep_id)] + subset = subset.sort_values(["timestamp", "ESA Step #"]).iloc[-NUM_ENERGY_STEPS:] + +If no row qualifies - the data predates every table entry - the code falls back +to the **earliest** timestamp for that sweep table, and only raises +``ValueError`` if even that is empty. This fallback is why the L2 test can run +against 2024-09-24 data using a table that starts 2025-02-11. + +**2. Fill the fixed energies.** ``read_swapi_lut_table`` maps the string +``"Solve"`` to ``-1`` when reading the CSV, so solve steps are simply the +negative entries: + +.. code-block:: python + + solve_steps = sweep_esa_energies < 0 + energy_data[i_sweep, ~solve_steps] = sweep_esa_energies[~solve_steps] + if not np.any(solve_steps): + continue + +.. important:: + + The code does **not** assume the solve steps are the last 9. The comment is + explicit: "Solve steps are the fine sweep steps. This can be variable + numbers and is not always the final 9 steps." Anything the ADP marks + ``Solve`` is solved, wherever it sits. This is more general than the + document's fixed 63-71 recipe, and it is the right call given that the fine + allocation is sweep-plan dependent. + +**3. Locate the sweep on the DAC ladder.** ``esa_lvl5`` is formatted as a +4-digit uppercase hex string and matched against the LUT notes table's +``ESA DAC (Hex)`` column: + +.. code-block:: python + + esa_lvl5_hex = np.vectorize(lambda x: format(x, "04X"))(l1_dataset["esa_lvl5"].values) + ... + matching_indices = np.nonzero(lut_notes_df["ESA DAC (Hex)"].values == esa_lvl5_val)[0] + last_energy_step_index = matching_indices[0] + +The **first** match is used. (The LUT notes table has repeated DAC values - +e.g. index 0 and 1 are both ``1FFE`` - so "first match" is a real choice, not a +formality.) A missing DAC value raises ``ValueError``. + +**4. Walk the ladder.** The ``ESA Index Number`` offsets are re-referenced to +the final solve step, then added to the anchor row: + +.. code-block:: python + + fine_offsets = subset["ESA Index Number"].values[solve_steps] + fine_offsets -= fine_offsets[-1] # -> [-32, -28, ..., -4, 0] + fine_lut_indices = last_energy_step_index + fine_offsets + +This reproduces the document's ``r_p16 - 4k`` pattern for the nominal table +while generalizing to any offset pattern the ADP declares. + +**5. Floor negative indices.** **[CODE]**, not in the document: + +.. code-block:: python + + fine_lut_indices[fine_lut_indices < 0] = 0 + +The source records the rationale as a SWAPI instruction: SWAPI calls this +"flooring" the index. If the 71st step's DAC lands near the top of the ladder +(row < 32), back-tracking would run off the end. Example from the comment: + +.. code-block:: text + + 71st index = 31 + nine fine energy indices = [31, 27, 23, 19, 15, 11, 7, 3, -1] + flooring = [31, 27, 23, 19, 15, 11, 7, 3, 0] + +.. warning:: + + Flooring silently duplicates energies rather than producing a fill value. + When it triggers, several fine steps get the same energy and there is + **nothing in the product to say it happened** - no flag, no log message. If + fine-step energies look degenerate, this is why. + +**6. Write it out.** ``esa_energy`` is a (sweep, 72) float array, and every +rate, rate uncertainty and the flag array get ``DEPEND_1 = "esa_energy"``. + +.. code-block:: python + + depend_on_esa_energy = ["swp_l1a_flags", "swp_pcem_rate", "swp_scem_rate", + "swp_coin_rate", "swp_pcem_rate_stat_uncert_plus", ...] + for variable in depend_on_esa_energy: + l2_dataset[variable].attrs["DEPEND_1"] = "esa_energy" + +Worked example +^^^^^^^^^^^^^^ + +From the vendored test table (2025-02-11, sweep table 0, LUT version 4): + +.. code-block:: text + + ESA Step # K factor Voltage Energy Sweep # ESA Index Number + 0 1.88 619 1,163 0 544 + 1 1.88 619 1,163 0 544 + ... + 60 1.88 69 129 0 960 + 61 1.88 64 120 0 976 + 62 1.88 57 107 0 992 + 63 1.88 Solve Solve 0 -16 + 64 1.88 Solve Solve 0 -12 + ... + 70 1.88 Solve Solve 0 12 + 71 1.88 Solve Solve 0 16 + +Steps 0-34 all read 1163 eV: the ESA is still at the top of the ramp. Energy +then falls monotonically to 107 eV at step 62. The 9 ``Solve`` rows carry +offsets ``-16 ... +16`` in steps of 4, which after re-referencing become +``-32 ... 0``. + +**[CODE]** ``test_swapi_l2.py::test_solve_full_sweep_energy`` pins this down: +for ``esa_lvl5 = 4663`` (hex ``1237``) and sweep table 0, the last 9 energies +come out as ``[4290, 4199, 4109, 4020, 3934, 3850, 3767, 3687, 3608]`` eV. + +Deviations from the document at L2 +---------------------------------- + +Full list in :ref:`swapi-implementation-status`; the L2-specific ones: + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Item + - Status + * - ``SWP_SC_THRUSTER`` flag + - **[DOC]** "an additional flag is added in ``SWP_L1_FLAGS`` to indicate + data taken during SC thruster activity, namely, daily repointing + maneuvers." **[CODE]** ``# TODO: add thruster firing flag``. Not + implemented, and there is no bit reserved for it in ``SWAPIFlags``. + * - Other flags + - **[CODE]** ``# TODO: add other flags``. + * - The ``k`` factor + - The ADP's ``K factor`` and ``Voltage`` columns are read into the + DataFrame and never used. L2 records energy but not the ``k`` used to + derive it, which matters because :math:`k_{L2} = 1.93` is what L3 needs + to invert energy back to ESA voltage - see :ref:`swapi-k-factor`. + * - ADP file format + - **[DOC]** The ADP is an ``.xlsx`` workbook with a main sheet and a + ``LUT_Notes_vx`` sheet. **[CODE]** Two separate CSVs are ingested. The + xlsx-to-CSV conversion happens outside this repository. + * - Energies at L2, not voltages + - The document's title for section 10.3 says "ESA Step Number ... to + Energy (keV)". The ADP energies and the code both work in **eV/q** + (``VALIDMAX`` = 21000, ``LABLAXIS`` = "Energy (eV/q)"). diff --git a/docs/source/algorithm-code-documentation/swapi/l3-scope.rst b/docs/source/algorithm-code-documentation/swapi/l3-scope.rst new file mode 100644 index 0000000000..80fbcbe6af --- /dev/null +++ b/docs/source/algorithm-code-documentation/swapi/l3-scope.rst @@ -0,0 +1,216 @@ +.. _swapi-l3-scope: + +L3 Scope - What Is Not in This Repository +========================================= + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +**Document:** sections 7.1.1, 9.5, 10.4, 10.5, 13. + +.. important:: + + **This repository takes SWAPI to L2 and stops.** L3a and L3b are produced by + the SWAPI team's own code, delivered as a separate Docker container run on + AWS. Section 13 of the algorithm document says so in one line: "SWAPI + delivers the SWAPI L3 processing code using a Docker container via AWS." + + Roughly two thirds of the algorithm document describes L3. If you are + reading the document and find yourself in a section full of Maxwellian + integrals, Jacobians or Vasyliunas-Siscoe distributions, you are outside + this repository's scope. + +This page exists for two reasons: so that nobody starts implementing L3 here by +accident, and so that the **contract** between L2 and L3 is written down. That +contract is the only part of section 10.4 and 10.5 that constrains what we do. + +The contract - what L3 needs from L2 +------------------------------------ + +**[DOC]** Section 10.6, "Inputs Required" and "Ancillary files" for each L3 +product. Consolidated: + +.. list-table:: + :header-rows: 1 + :widths: 24 12 18 46 + + * - L3 product + - Cadence + - L2 input + - Other inputs + * - Solar wind proton velocity, density, temperature + - 1 min + - Count rates and count rate uncertainties as a function of E/q for + **62 coarse energy bins (5 sweeps)** + - SPICE spin phase, SPICE frame kernel, efficiency LUT, instrument + response LUT + * - Solar wind alpha velocity, density, temperature + - 1 min + - as above, **coarse steps only** + - as above, plus the **magnetic field vector** (MAG L2, L1D fallback) + * - Pickup He\ :sup:`+` fit parameters, density, temperature + - 10 min + - Count rates and uncertainties as a function of E/q for **62 coarse + energy bins (50 sweeps)** + - as above, plus interstellar neutral He density LUT and interstellar + neutral H and He flow vector LUT + * - Combined differential flux :math:`J(E/q)` + - 10 min + - as above (50 sweeps) + - :math:`\Delta E/E`, energy-dependent geometric factor + :math:`G(E/q)`, efficiency :math:`\varepsilon` + +Read off the consequences for L2: + +1. **L3 consumes count rates keyed on E/q, not on step index.** That is why L2 + sets ``DEPEND_1 = "esa_energy"`` on every rate variable. If the energy solve + is wrong, every L3 product is wrong and there is nothing downstream that + can catch it. +2. **Rate uncertainties are a required input, not a nicety.** The proton and + alpha fits use them for the covariance estimate and the PUI fit uses them + in the Poisson likelihood. +3. **Chunking is L3's job, but the boundaries are ours.** L3 groups + non-overlapping 5-sweep and 50-sweep chunks. It needs contiguous sweeps with + trustworthy times, which is what ``sci_start_time`` and the epoch convention + provide. +4. **L3 needs the sweep start time, per ESA step.** Hence + ``sci_start_time`` - a UTC string of the first packet's epoch, added to L1 + "for L3 purposes per SWAPI requests" - alongside the centre-time ``epoch``. +5. **L3 needs to invert energy back to ESA voltage** using + :math:`k_{L2} = 1.93` eV/V/e, because the response function is tabulated + against voltage. See :ref:`swapi-k-factor`. L2 does not record the ``k`` it + used. +6. **Coarse-only for some products.** The alpha fit uses coarse steps only; the + proton fit uses coarse and fine. L2 must therefore keep the fine steps + distinguishable, which is why ``plan_id`` and ``sweep_table`` are carried + through and why the solve marks fine steps explicitly. +7. **L2 must not pre-average.** Every L3 product does its own chunking from + 12-second sweeps. +8. **L2 must not apply efficiency, deadtime or geometric factor.** Those live + in the L3 forward model (:math:`\varepsilon`, :math:`\tau = 183.7` ns, + :math:`G(E/q)`). + +What L3a does, in one paragraph each +------------------------------------ + +Enough to recognize the algorithms, not enough to implement them. If you need +the equations, they are in document sections 10.4.1 (proton, pages 43-54), +10.4.2 (alpha, pages 55-57) and 10.4.3 (pickup helium, pages 58-67). + +Solar wind protons (H\ :sup:`+`) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** A **forward model fit**, not a moment integration. The proton VDF is +modelled as a drifting Maxwellian with parameters +:math:`\boldsymbol{x} = (\ln n, \ln T, v_R, v_T, v_N)` (logs so density and +temperature stay positive), and the model coincidence rate is the VDF +integrated against the instrument response over both azimuthal regions +(sunglasses and open aperture) by nested Gauss-Legendre quadrature with +:math:`(N_\theta, N_\phi, N_v) = (21, 21, 15)` points, with dynamic integration +limits from the passband and the VDF width. A deadtime correction +:math:`\mathcal{D} = 1/(1 + \tau C^{\mathrm{model}})` with +:math:`\tau = 183.7` ns maps the model rate to the observed rate. +Minimization is **unweighted** least squares - deliberately, because Poisson +inverse-variance weighting would over-weight the low-count wings where pickup +ions, alphas and the proton shoulder contribute, an effect that the sunglasses +exaggerate by attenuating the cold core. Uncertainties use the +heteroscedasticity-consistent **HC3** sandwich estimator rather than the +Jacobian covariance. Because the MSE has a spurious local minimum with the bulk +velocity flipped about the spin axis, the fit is re-run up to six times from +the flipped solution and the better minimum kept. Flags: ``FIT_ERROR`` if the +optimizer fails, ``BAD_FIT`` if :math:`R^2 < 0.9` or +:math:`T > 5 \times 10^5` K - and even then the speed is reported from the +peak of the sweep-averaged rates. + +Solar wind alphas (He\ :sup:`2+`) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Protons are fit first and **held fixed**, then three alpha parameters +are fit: :math:`(\log n^{He^{2+}}, \log T^{He^{2+}}, \Delta v^{He^{2+}})`, +where the alpha bulk velocity is constrained to +:math:`\boldsymbol{v}^{He^{2+}} = \boldsymbol{v}^{p} + \Delta v^{He^{2+}}\hat{B}` +- the drift lies **along the magnetic field**, which is where the MAG +dependency comes from. The alpha peak is found from the proton-subtracted +residual :math:`R_i = C_i - 2 C_i^p` (the factor of two so that few-percent +errors in the proton model are not mistaken for the alpha peak), searched +between 1.5 and 4 times the proton peak E/q; five points around it are fit with +``scipy.optimize.least_squares(method='lm')``. Coarse steps only. + +Pickup helium (He\ :sup:`+`) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** This is the part inherited from SWAP. The generalized +**Vasyliunas & Siscoe** model in the Chen et al. (2014) form is fit to 50 +combined sweeps over the range +:math:`1.25 E_{b,H^+} < E_e < 1.2 E_{b,He^+}`: + +.. math:: + + f(r,w,\psi) = \frac{\alpha}{4\pi}\frac{\beta_E r_E^2}{r u_{\mathrm{sw}} v_b^3} + w^{\alpha-3}\, n_{\mathrm{He}}(r w^\alpha, \psi)\, \Theta(1-w) + +with free parameters cooling index :math:`\alpha`, ionization rate +:math:`\beta_E`, cutoff speed :math:`v_b` and a constant background rate +:math:`C_{\mathrm{bg}}`. The interstellar neutral helium density +:math:`n_{\mathrm{He}}` is precomputed from the **hot model** (Thomas 1978) and +supplied as a LUT. Optimization is Nelder-Mead on a bounds-transformed +parameter vector, maximizing the **Poisson likelihood**; density and +temperature then come from numerical integration of the best-fit distribution, +with uncertainties propagated by re-integrating at +:math:`j \pm \sigma_j`. Bounds and initial values are the document's Table 5 +(:math:`\alpha \in [1,5]` starting 1.5; +:math:`\beta_E \in [0.6\times10^{-9}, 8\times10^{-7}]` s\ :sup:`-1` starting +:math:`10^{-7}`; :math:`v_b \in [0.8, 1.2] v_{\mathrm{sw}}` starting +:math:`v_{\mathrm{sw}}`; :math:`C_{\mathrm{bg}} \in [0, 10]` Hz starting +0.1 Hz). :math:`C_{\mathrm{bg}} > 1` Hz is reported as fill (suprathermal +contamination) while keeping the other parameters. + +What L3b does +------------- + +**[DOC]** Section 10.5. The one L3 equation worth having here, because it is +the only place efficiency and geometric factor enter and it is what a naive +reader might otherwise try to put at L2: + +.. math:: + + J\!\left(\frac{E}{q}\right) = + \frac{C(E/q)}{\frac{E}{q} \cdot G(E/q) \cdot \varepsilon} + +with :math:`J` the differential flux in #/[cm\ :sup:`2` s sr eV/q], +:math:`C` the count rate, :math:`G` the geometric factor and +:math:`\varepsilon` the time-varying efficiency (updated after each on-orbit +gain test and provided in a LUT). The calculation **assumes the same efficiency +for hydrogen and helium** - the H efficiency is used for both, efficiency +treated as mass-independent. Uncertainty: + +.. math:: + + \Delta J = J \sqrt{\left(\frac{\Delta C}{C}\right)^2 + + \left(\frac{\Delta (E/q)}{E/q}\right)^2} + +Note the second term: **L3b needs an energy uncertainty** +:math:`\Delta(E/q)`. L2 does not produce one. The energy passband edges are in +the LUT notes table (``Lower Energy`` / ``Upper Energy`` columns, which no code +here reads), so the information exists but is not propagated. + +If you are asked to add L3 here +------------------------------- + +Push back, and point at section 13. If the answer is still yes, these are the +things that would have to change on our side of the boundary, in rough order: + +1. L2 would need to publish an energy uncertainty (from the LUT notes + passband edges). +2. L2 or L1 would need the thruster flag, since L3 rejects thruster-contaminated + data. +3. The remaining L1 rejection criteria (checksum, saturation, ``RATE_ST``) + would need implementing, because L3's flags assume L2 data is already clean. +4. The response-function ancillary files (central effective area, passbands, + azimuthal transmission), the efficiency table and the interstellar-neutral + LUTs would all need ingest paths here. The files themselves are delivered to + the SDC and their descriptors are listed in :ref:`swapi-l3-descriptors`, but + nothing in this repository reads any of them. +5. SPICE geometry would need wiring in: per-ESA-step SWAPI-to-RTN rotation + matrices and spacecraft velocity in the solar inertial frame. +6. New logical sources and descriptors (``1m-sw-p``, ``1m-sw-a``, + ``10m-pui-he``, ``10m-combined``) and a new ``PROCESSING_LEVELS`` entry. diff --git a/docs/source/algorithm-code-documentation/swapi/overview.rst b/docs/source/algorithm-code-documentation/swapi/overview.rst new file mode 100644 index 0000000000..68c6a879d7 --- /dev/null +++ b/docs/source/algorithm-code-documentation/swapi/overview.rst @@ -0,0 +1,486 @@ +.. _swapi-overview: + +Instrument and Mission Concepts +=============================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +Everything on this page is background needed to read the algorithm pages. It is +mostly **[DOC]** (algorithm document sections 4, 5, 6 and 7). + +What SWAPI measures +------------------- + +SWAPI (Solar Wind and Pickup Ion) is one of ten instruments on IMAP, built by +the Space Physics Group at Princeton University. It sits at the Sun-Earth L1 +Lagrange point and measures four ion populations: + +.. list-table:: + :header-rows: 1 + :widths: 24 20 56 + + * - Population + - Symbol + - Notes + * - Solar wind protons + - H\ :sup:`+` + - The bright core beam. Enters through the attenuating grid. + * - Solar wind alphas + - He\ :sup:`2+` + - Sits at roughly twice the proton E/q. Also attenuated. + * - Interstellar pickup helium + - He\ :sup:`+` + - The primary pickup-ion science target. Helium dominated. + * - Interstellar pickup hydrogen + - H\ :sup:`+` + - Present but much weaker at 1 au than in the outer heliosphere. + +The instrument is a modified version of New Horizons' **SWAP** (Solar Wind +Around Pluto). The heritage matters: the pickup-ion fitting code is inherited +from SWAP, and the I-ALiRT analytical model is the SWAP model of +Elliott et al. (2016). SWAP's own ground processing also stopped at L2. + +.. note:: + + **[DOC]** Heritage NH-SWAP code is *not* reused for L0-L2 processing. The + processing *steps* were adapted, drawing on the SWAP team's decades of + operations, but the code here is new. Only the SWAP **PUI** code is reused + and modified, and that happens at L3 - outside this repository. + +How a single measurement happens +-------------------------------- + +This chain explains nearly every variable name in the data products. + +1. An ion enters through one of two paths: + + * the **sun-facing aperture grid** ("the sunglasses"), which is **0.1% + transmissive**. This attenuates the solar wind so the detector is not + saturated, without attenuating anything from the other directions. + * the **open aperture**, unattenuated, which is where pickup ions come from. + +2. It passes through the toroidal **electrostatic analyzer (ESA)**, which + selects a narrow band of **energy per charge (E/q)**. The ESA also blocks UV + light and neutrals. +3. It crosses a **field-free flight path** and is post-accelerated into the + detector section. +4. It passes through an **ultrathin carbon foil**, liberating secondary + electrons. +5. The ion itself lands on the **primary CEM (PCEM)**; the secondary electrons + are steered onto the **secondary CEM (SCEM)**. +6. Pulses from both are amplified and accumulated in counters. If a PCEM and an + SCEM event fall within a **100 ns window**, a **coincidence (COIN)** is also + registered. + +Three counters, one measurement +------------------------------- + +Every SWAPI science measurement is the triple (PCEM, SCEM, COIN) for one ESA +step. + +.. list-table:: + :header-rows: 1 + :widths: 14 18 68 + + * - Counter + - Also called + - Meaning + * - ``PCEM`` + - PRM, primary + - Ions that passed through the carbon foil and hit the primary CEM. + * - ``SCEM`` + - SEC, secondary + - Secondary electrons liberated from the carbon foil. + * - ``COIN`` + - coincidence + - Both within 100 ns. **This is the science channel** - it is what the + L3 fits and the I-ALiRT product use, because it has by far the lowest + background. + +**[DOC]** The three-counter arrangement lets absolute detection efficiency be +computed on orbit, independent of how the individual detectors age +(Funsten et al. 2005): + +.. math:: + + \varepsilon = \frac{\mathrm{COIN}^2}{\mathrm{PRM} \times \mathrm{SEC}} + +This is why all three are carried all the way through L2 even though only COIN +is fit downstream. + +High voltage +------------ + +**[DOC]** Three separate high-voltage power supplies (ESA, PCEM, SCEM), all +controlled by an 8051 microcontroller. + +* The **ESA** steps its voltage through the sweep, up to **10.2 kV**. +* The **CEM** voltages are held at a steady value, up to **4 kV**. They are + raised over the mission to compensate for CEM gain decay - that is what the + periodic gain test is for (see :ref:`swapi-ancillary`). + +Sweeps, steps, and the 12-second cadence +---------------------------------------- + +This is the single most important set of definitions in SWAPI processing. + +.. list-table:: + :header-rows: 1 + :widths: 28 72 + + * - Concept + - Definition + * - **Spin** + - One spacecraft rotation about the Sun-pointing axis. **4 rpm, so ~15 s.** + SWAPI does not use spin as an aggregation unit, but L3 needs spin phase + per ESA step. + * - **Sweep** + - **12 seconds, 72 ESA steps.** The fundamental unit of L1 and L2: one + record in the L1/L2 CDF is one sweep. ``NUM_ENERGY_STEPS = 72``. + **[CODE]** + * - **ESA step** + - One energy setting. **6 steps per second, so 0.167 s per step.** Only + **0.145 s** of that is actually counting - see live time below. + * - **Packet** + - One ``SWP_SCI`` packet holds **1 second = 6 ESA steps**, and is a fixed + **54 bytes / 432 bits**. **12 packets make one sweep**; + ``NUM_PACKETS_PER_SWEEP = 12``. **[CODE]** + * - **Sequence number** + - ``SWP_SCI.SEQ_NUMBER``, 0-11, identifies which 6-step group of the + sweep a packet holds. ``0`` marks the start of a sweep, ``11`` the end. + All 12 must be present to process the sweep. + * - **Coarse sweep** + - The **62 fixed E/q steps** covering **0.1-20 keV/q**, high energy to low + energy. These have fixed, tabulated energies. + * - **Fine sweep** + - **9 variable steps**, placed according to the active *sweep plan*. Their + energies are not fixed and must be solved for at L2. + * - **Ramp-up step** + - **1 step** at the start of the sweep, used to transition the ESA from + the low fine-sweep voltage back to full voltage. It carries no usable + science. **[CODE]** ``swapi_l1`` sets index 0 of all three count arrays + to ``NaN``. + * - **Live time** + - ``SWAPI_LIVETIME = 0.145`` s. The actual accumulation time per energy + bin, less than the 0.167 s step duration because the HVPS needs settling + time. **[CODE]** ``swapi_l2.SWAPI_LIVETIME``. + * - **Five-sweep chunk** + - **60 s = 5 sweeps = 4 spacecraft spins.** 5 is the smallest number of + 12-second sweeps that closes on the ~15 s spin period, so a five-sweep + chunk samples spin phase evenly. This is the L3a proton/alpha cadence + and the I-ALiRT averaging window. + * - **Fifty-sweep chunk** + - **10 minutes.** The L3a pickup-helium and L3b cadence. + +So: ``72 steps = 1 ramp-up + 62 coarse + 9 fine``, and +``12 s / 72 steps = 0.167 s/step``, of which ``0.145 s`` counts. + +.. warning:: + + **Step ordering trap.** The coarse sweep runs from the **highest** energy to + the **lowest** (ESA step 0 is ~1.2 keV at the top of the ramp, step 62 is + ~107 eV in the 2025-02-11 table). Energy is *decreasing* with step index + through the coarse sweep, then the 9 fine steps jump back up around the + solar wind peak. Never assume step index is monotonic in energy across the + whole 72. + +Sweep plans +----------- + +**[DOC]** The 9 fine steps are programmable. There are **16 sweep plans**; the +active plan is telemetered per packet as ``SWP_SCI.PLAN_ID``, and the table +within the plan as ``SWP_SCI.SWEEP_TABLE``. + +.. list-table:: + :header-rows: 1 + :widths: 16 84 + + * - Plan + - Fine-step allocation + * - **Plan 4** + - **The nominal operations plan since 2026-02-01.** 3 fine steps dedicated + to low-energy background at ESA voltages **35 V, 20 V and 5 V**, plus + **6 fine steps** distributed above and below the peak E/q bin. + * - **Plan 5** + - Used for electron-hoovering background tests. All 9 fine steps at ESA + voltages **40, 35, 30, 25, 20, 15, 10, 4 and 0 V**. + +.. important:: + + The background fine steps exist because the L1 background at L1 is not + constant - it varies with solar UV output (Bzowski et al. 2013; Sokol et al. + 2013) and with galactic/anomalous cosmic ray intensity (Leske et al. 2013). + **[DOC]** L3 excludes these background bins from the solar wind fit and uses + them to constrain the background level instead. At L1 and L2 they are + ordinary energy steps and get no special treatment. + +Both ``plan_id`` and ``sweep_table`` are carried into the L1 and L2 CDFs, and +``sweep_table`` is the key used to pick the right rows out of the ESA unit +conversion table at L2. **[CODE]** + +Telemetry types +--------------- + +**[DOC]** There are seven SWAPI telemetry types. Note that SWAPI produces no +summary or histogram data (SWAP did), but does produce large science, I-ALiRT +and autonomy data. + +.. list-table:: + :header-rows: 1 + :widths: 16 12 14 58 + + * - Telemetry + - APID + - Kind + - Description + * - ``SWP_HK`` + - 1184 + - Engineering + - Housekeeping: instrument status, counters, ADC values. In HVSCI the HK + packet nominally comes every **60 s**, but is sampled at 1 Hz. 102 + fields in the XTCE. **Processed.** + * - ``SWP_SCI`` + - 1188 + - Science + - PCEM, SCEM and COIN counts for each of 6 samples in a 1-second period. + Reports the ESA level for a single sample (``ESA_LVL5``); the other + levels come from the lookup table. 46 fields. **Processed.** + * - ``SWP_IAL`` + - 1187 + - Science + - I-ALiRT: a subset of the science packet carrying **coincidence counts + only**. **Processed, under** ``imap_processing/ialirt/``. + * - ``SWP_LGSCI`` + - -- + - Science + - "Large" science - same as SCI but includes all 6 ESA levels rather than + one. Used for testing. **Not processed; not in the XTCE.** + * - ``SWP_AUT`` + - 1192 + - Engineering + - Autonomy: minimum required per ICD (power off/cycle flags). **Enum + exists in** ``SWAPIAPID`` **but not processed and not in the XTCE.** + * - ``SWP_MG`` + - -- + - Engineering + - Asynchronous event message packets. **Not processed.** + * - ``SWP_MD`` + - -- + - Engineering + - Memory dump, fixed 128-byte payload. **Not processed.** + +**[DOC]** CCSDS packets have variable (even) length up to 4096 bytes including +headers, per the GI ICD. All ``SWP_SCI`` packets are fixed size. **All packet +types carry a CHKSUM parameter**, computed on board and intended to be +recomputed on the ground as a data check. + +Instrument modes +---------------- + +``SWP_SCI.MODE`` / ``SWP_HK.MODE``. **[CODE]** ``swapi_utils.SWAPIMODE``: + +.. list-table:: + :header-rows: 1 + :widths: 12 18 70 + + * - Value + - Name + - Meaning + * - 0 + - ``LVENG`` + - Low voltage, engineering + * - 1 + - ``LVSCI`` + - Low voltage, science + * - 2 + - ``HVENG`` + - High voltage, engineering + * - 3 + - ``HVSCI`` + - **High voltage, science - the only mode that produces valid science.** + Every packet of a sweep must be in HVSCI or the sweep is dropped. + +Apertures, regions, and the response function +--------------------------------------------- + +**[DOC]** The instrument response is decomposed by *azimuthal region*, because +the sunglasses only cover part of the aperture. The azimuth angle +:math:`\phi` is measured in instrument coordinates: + +.. list-table:: + :header-rows: 1 + :widths: 24 22 54 + + * - Region + - Azimuth range + - Transmission + * - Sunglasses (**SG**) + - :math:`|\phi| \le 20^\circ` + - :math:`T = 10^{-3}` across the flat central part + (:math:`|\phi| \le 9^\circ`) + * - Open aperture (**OA**) + - :math:`20^\circ \le |\phi| \le 150^\circ` + - :math:`T = 1` across the flat part + (:math:`31^\circ \le |\phi| \le 115^\circ`) + +The full effective-area function factorizes as + +.. math:: + + \mathcal{A}^s(v,\theta,\phi,V) = + \mathcal{A}^s_0(V)\; + P_{\mathrm{region}(\phi)}\!\left(\frac{v}{v_0^s},\theta,V\right)\; + T(\phi) + +where :math:`\mathcal{A}^s_0` is the central effective area, +:math:`P_r` is the region-specific energy-angle passband, :math:`T(\phi)` is +the azimuthal transmission tabulated from 0 to 180 degrees at 0.1 degree +spacing, and the central speed is +:math:`v_0^s = \sqrt{2 k^{*} q^s |V| / m^s}`. + +**These three functions are ancillary CSVs used only by L3.** They are listed +here because they are the reason L2 must report ESA *energy* in a way that can +be converted back to ESA *voltage* - see the ``k`` factor below. + +.. _swapi-k-factor: + +The k factor +------------ + +**[DOC]** Two values are in play and they are not the same: + +.. list-table:: + :header-rows: 1 + :widths: 16 22 62 + + * - Symbol + - Value + - Use + * - :math:`k^{*}` + - 1.89 eV/V/e + - The peak :math:`(E/q)/|V|` at :math:`\theta = 0`, from high-resolution + SIMION simulations. Used to normalize the L3 response functions. + * - :math:`k_{L2}` + - 1.93 eV/V/e + - Estimated pre-launch from lab measurements (Rankin et al. 2025). **This + is the factor used to convert the ESA energy in the L2 CDF files back to + the actual ESA voltage of the instrument.** + +The discrepancy is believed to come from inaccuracies in beam energy and +orientation in the lab measurements. It is still under investigation; as of the +initial IMAP data release, L3 uses the SIMION :math:`k^{*}`. + +**[CODE]** The pipeline never applies a ``k`` factor. The ESA unit conversion +table carries a ``K factor`` column (1.88 in the 2025-02-11 rows, 1.93 in the +2025-05-19 rows) but ``swapi_l2`` reads only the ``Energy`` column and ignores +``K factor`` and ``Voltage`` entirely. This is correct as far as it goes - +energies in the table are already energies - but it means the L2 product does +not record which ``k`` was used to build them. + +Energy resolution and effective area +------------------------------------ + +**[DOC]** Numbers quoted by the algorithm document for the solar wind: + +.. list-table:: + :header-rows: 1 + :widths: 26 22 52 + + * - Quantity + - Value + - Where used + * - :math:`\Delta E / E` (FWHM) + - 0.085 + - I-ALiRT passband width. **[CODE]** + ``IalirtSwapiConstants.fwhm_width`` + * - Speed width :math:`\Delta v / v` + - :math:`\tfrac{1}{2}\,\Delta E/E` = 0.0425 + - I-ALiRT. **[CODE]** ``IalirtSwapiConstants.speed_ew`` + * - Effective area :math:`A_{\mathrm{eff}}` + - :math:`1.633 \times 10^{-4}`\ cm\ :sup:`2` + - I-ALiRT analytical model. **[CODE]** + ``IalirtSwapiConstants.eff_area`` (converted to m\ :sup:`2`) + * - Azimuthal FOV :math:`\Delta\phi` + - 30 degrees + - I-ALiRT. **[CODE]** ``IalirtSwapiConstants.az_fov`` + * - Detector deadtime :math:`\tau` + - 183.7 ns + - **L3 only.** 5% correction at ~2.7e5 Hz, routine for high-flux solar + wind. Not applied at L1 or L2. + +Definitions of terms +-------------------- + +**[DOC]** Algorithm document section 5, reproduced because these units appear +verbatim in the CDF attributes. + +.. list-table:: + :header-rows: 1 + :widths: 20 56 24 + + * - Term + - Definition + - Unit + * - Bulk velocity + - Measure of the peak of the solar wind distribution + - km/s + * - Count + - Number of particles recorded on the instrument + - # + * - Count rate + - Counts per unit time + - #/s + * - Density + - Number of particles per unit volume + - #/cm\ :sup:`3` + * - Efficiency + - Correction factor applied in L1-L2 processing to correct for particle + detection which may change over time + - dimensionless + * - Flux + - Particle count rate per unit area per unit solid angle per unit + energy/charge + - #/[cm\ :sup:`2` s sr eV/q] + * - Geometric factor + - Energy width per energy multiplied by effective area, + :math:`A_{\mathrm{eff}} \cdot \Delta E/E` + - cm\ :sup:`2` sr eV/eV + * - Temperature + - Measure of the broadness of the distribution + - K + +.. note:: + + The document's own definition of *efficiency* says it is "applied in L1-L2 + processing". **[CODE]** It is not. Efficiency is applied where the flux is + formed, which is L3b (equation 13), and the L2 product is a bare count rate. + Treat the definition table as a glossary, not as a specification of where + the correction lives. + +Reference frames +---------------- + +**[DOC]** Frames named by the algorithm document. None of them are used by any +code in this repository - L1 and L2 are frame-free (counts and rates per ESA +step). They are listed so that the L3 equations can be read. + +* **Instrument frame** - speed :math:`v`, elevation :math:`\theta`, azimuth + :math:`\phi`. The frame the response function is tabulated in. +* **Spacecraft RTN frame** - :math:`(v_R, v_T, v_N)`. The frame the L3a bulk + velocity is fit in. Per-ESA-step SWAPI-to-RTN rotation matrices come from + SPICE. +* **Solar inertial frame** - the Sun rest frame. L3a also reports velocity here, + obtained by adding the spacecraft velocity from SPICE. +* **GSE** - used for the end-to-end model and for comparison with WIND/SWE + (:math:`v_R \approx -V_{x,\mathrm{GSE}}`, + :math:`v_T \approx -V_{y,\mathrm{GSE}}`, + :math:`v_N \approx V_{z,\mathrm{GSE}}`). +* **Ecliptic J2000** - the frame the interstellar neutral inflow directions are + quoted in (He from 255.7, 5.1 degrees at 25.4 km/s; H from 252.2, 9.0 degrees + at 22 km/s). + +Data volume +----------- + +**[DOC]** The SWAPI team anticipates **4.786560 MB/day** of raw data and +**22 MB/day** of processed data across L1, L2 and L3. diff --git a/docs/source/algorithm-code-documentation/swapi/reference-tables.rst b/docs/source/algorithm-code-documentation/swapi/reference-tables.rst new file mode 100644 index 0000000000..cf805c177b --- /dev/null +++ b/docs/source/algorithm-code-documentation/swapi/reference-tables.rst @@ -0,0 +1,418 @@ +.. _swapi-reference-tables: + +Reference Tables - Where to Look Them Up +======================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +The algorithm document's large tables are **deliberately not reproduced** in +these pages: they are long, they go stale, and in almost every case a +machine-readable version already exists in the repository that the code +actually reads. + +The document itself is **not in this repository** - see +:ref:`swapi-source-documents`. This page tells you where to look instead, and +gives a page index for the parts that only exist in the PDF. + +Rule of thumb +------------- + +.. list-table:: + :header-rows: 1 + :widths: 40 60 + + * - If you need... + - Go to + * - A packet field's name, bit offset, width, type or enumeration + - ``imap_processing/swapi/packet_definitions/swapi_packet_definition.xml`` + * - An I-ALiRT packet field + - ``imap_processing/ialirt/packet_definitions/ialirt_swapi.xml`` + * - A fixed ESA step energy, or a fine-step offset + - the ``esa-unit-conversion`` ancillary CSV + * - The ESA DAC-to-energy ladder + - the ``lut-notes`` ancillary CSV + * - A quality flag's bit position + - ``imap_processing/quality_flags.py``, ``class SWAPIFlags`` + * - A CDF variable's units, fill value, valid range or description + - ``imap_processing/cdf/config/imap_swapi_variable_attrs.yaml`` + * - A ``logical_source`` string or global attribute + - ``imap_processing/cdf/config/imap_swapi_global_cdf_attrs.yaml`` + * - A pipeline constant (12 packets, 72 steps, 0.145 s) + - ``imap_processing/swapi/constants.py``, ``swapi_l2.SWAPI_LIVETIME`` + * - An I-ALiRT physical constant + - ``imap_processing/ialirt/constants.py``, + ``class IalirtSwapiConstants`` + * - An equation from L1, L2 or I-ALiRT + - :ref:`swapi-l1` / :ref:`swapi-l2` / :ref:`swapi-ialirt` - the + load-bearing ones are transcribed + * - An L3 equation + - :ref:`swapi-l3-scope` for the shape of it, then the document + * - Anything else + - the algorithm document, using the page index below + +Machine-readable tables in the repository +----------------------------------------- + +Packet definitions +^^^^^^^^^^^^^^^^^^ + +``imap_processing/swapi/packet_definitions/swapi_packet_definition.xml`` is the +authoritative field definition for the two processed APIDs. It supersedes the +document's telemetry description and, unlike it, cannot be out of date with +respect to processing - it is what the code parses. + +.. list-table:: + :header-rows: 1 + :widths: 20 14 66 + + * - Container + - Fields + - Notes + * - ``SWP_HK`` + - 102 + - APID 1184. Includes ``PKT_APID``. Carries the enumerated status bits, + the ADC monitors, and ``CHKSUM``. + * - ``SWP_SCI`` + - 46 + - APID 1188. ``SHCOARSE``, ``SEQ_NUMBER``, ``SWEEP_TABLE``, ``PLAN_ID``, + ``MODE``, ``ESA_LVL5``, 18 ``*_RNG_ST0..5`` bits, 18 ``*_CNT0..5`` + counts, ``CHKSUM``. + +Query it directly rather than trusting a transcription: + +.. code-block:: bash + + # every SWP_HK field name + grep -o 'parameterRef="SWP_HK[^"]*"' \ + imap_processing/swapi/packet_definitions/swapi_packet_definition.xml + + # the enumeration for a status field + grep -A 12 'name="SWP_HK.MODE"' \ + imap_processing/swapi/packet_definitions/swapi_packet_definition.xml + +The I-ALiRT packet is separate: +``imap_processing/ialirt/packet_definitions/ialirt_swapi.xml``, APID **1187**, +13 fields, coincidence counts only. + +APIDs +^^^^^ + +.. list-table:: + :header-rows: 1 + :widths: 22 14 18 46 + + * - Telemetry + - APID + - In ``SWAPIAPID``? + - In an XTCE? + * - ``SWP_HK`` + - 1184 + - yes + - yes (``swapi_packet_definition.xml``) + * - ``SWP_IAL`` + - 1187 + - **no** + - yes (``ialirt_swapi.xml``) + * - ``SWP_SCI`` + - 1188 + - yes + - yes (``swapi_packet_definition.xml``) + * - ``SWP_AUT`` + - 1192 + - yes + - **no** + * - ``SWP_LGSCI`` + - unknown + - no + - no + * - ``SWP_MG`` + - unknown + - no + - no + * - ``SWP_MD`` + - unknown + - no + - no + +Ancillary CSVs +^^^^^^^^^^^^^^ + +Vendored **test copies** live in ``imap_processing/tests/swapi/lut/``. These are +test fixtures, not the operational ancillary files - the real ones come from the +SDC's ancillary store by descriptor. + +.. list-table:: + :header-rows: 1 + :widths: 30 12 58 + + * - File + - Rows + - Contents + * - ``imap_swapi_esa-unit-conversion_20250626_v001.csv`` + - 288 + - Two stacked versions of sweep 0 plus one each of sweeps 1 and 2 + (72 steps each). Columns as in + :ref:`swapi-esa-unit-conversion-adp`. + * - ``imap_swapi_lut-notes_20250626_v006.csv`` + - 1024 + - The DAC-to-energy ladder. ``ESA Index Number``, ``ESA Voltage``, + ``Energy``, ``Lower Energy``, ``Upper Energy``, ``ESA Range``, + ``ESA DAC (Dec)``, ``ESA DAC (Hex)``. + +.. code-block:: bash + + # what sweep tables and table versions does the ESA table cover? + cut -d, -f1,6,8 imap_processing/tests/swapi/lut/imap_swapi_esa-unit-conversion_20250626_v001.csv \ + | sort -u + +Validation and test data +^^^^^^^^^^^^^^^^^^^^^^^^ + +``imap_processing/tests/swapi/``: + +.. list-table:: + :header-rows: 1 + :widths: 40 60 + + * - Path + - Contents + * - ``l0_data/imap_swapi_l0_raw_20240924_v001.pkts`` + - Raw CCSDS packets, **pre-launch idle data, 2024-09-24**. + * - ``l0_validation_data/idle_export_raw.SWP_SCI_20240924_080204.csv`` + - SWAPI-supplied decommutation truth for the first science packet. + * - ``l0_validation_data/idle_export_raw.SWP_HK_20240924_080204.csv`` + - Same for the first housekeeping packet. + +Constants +^^^^^^^^^ + +.. list-table:: + :header-rows: 1 + :widths: 34 16 50 + + * - Constant + - Value + - Where + * - ``NUM_PACKETS_PER_SWEEP`` + - 12 + - ``swapi/constants.py`` + * - ``NUM_ENERGY_STEPS`` + - 72 + - ``swapi/constants.py`` + * - ``SWAPI_LIVETIME`` + - 0.145 s + - ``swapi/l2/swapi_l2.py`` + * - ``NUM_IALIRT_ENERGY_STEPS`` + - 63 + - ``ialirt/l0/process_swapi.py`` + * - ``eff_area`` + - 1.633e-4 cm\ :sup:`2` + - ``ialirt/constants.py`` + * - ``az_fov`` + - 30 deg + - ``ialirt/constants.py`` + * - ``fwhm_width`` + - 0.085 + - ``ialirt/constants.py`` + * - ``temporary_density_factor`` + - :math:`e^1` + - ``ialirt/constants.py`` + +.. note:: + + ``swapi/constants.py`` is four lines long. Numbers that arguably belong + there - the live time, the I-ALiRT step count, the overflow sentinel, the + ``4.0`` MHz saturation threshold if it is ever implemented - are scattered + across the modules that use them. If you add a tunable, consider whether it + belongs in ``constants.py``. + +Document page index +------------------- + +Page numbers are the **PDF page**, which is the printed document page + 1 +(the PDF has one unnumbered cover). Use these when you need something these +pages deliberately do not reproduce. + +.. list-table:: + :header-rows: 1 + :widths: 12 34 12 42 + + * - Section + - Title + - PDF pages + - In scope here? + * - 1 + - Scope + - 6-7 + - background + * - 2 + - Applicable documents + - 8 + - -- + * - 3 + - Abbreviations + - 9 + - -- + * - 4 + - Instrument description + - 10-12 + - :ref:`swapi-overview` + * - 5 + - Definitions of terms + - 14 + - :ref:`swapi-overview` + * - 6.1 + - Data volume, filenames + - 15 + - :ref:`swapi-data-products` + * - 6.2 + - Data product definitions (L0-L3) + - 15-16 + - :ref:`swapi-data-products` + * - 6.3 + - L0 data content and format (the 7 telemetry types) + - 16-17 + - :ref:`swapi-overview` + * - 6.4 + - L1/L2/L3 grouping, sweep structure, sweep plans + - 17 + - :ref:`swapi-overview` + * - 6.5 + - Processing flow (figure 4) + - 18 + - :ref:`swapi-data-products` + * - 7 + - Heritage instrument (NH SWAP) + - 19-23 + - background; 7.1.1 is L3 + * - 8 + - External dependencies + - 24 + - :ref:`swapi-ancillary` + * - 9.1-9.4 + - Calibration: ETE model, SIMION, lab, CoDICE, on-orbit + - 25-30 + - :ref:`swapi-ancillary` + * - 9.5 + - Instrument response function, efficiency, passbands + - 30-33 + - **L3.** :ref:`swapi-ancillary` summarizes. + * - 10.1 + - **L0 to L1 processing** + - 35-40 + - :ref:`swapi-l1` + * - 10.1.3 + - L1 CDF contents (table 2) and flag array (table 3) + - 37-39 + - :ref:`swapi-l1` + * - 10.2 + - **L1 to L2 processing** + - 40-41 + - :ref:`swapi-l2` + * - 10.3 + - **ESA Unit Conversion ADP** (table 3, and the solve procedure) + - 41-43 + - :ref:`swapi-l2`, :ref:`swapi-ancillary` + * - 10.4.1 + - L3a solar wind protons + - 44-56 + - **L3.** :ref:`swapi-l3-scope` + * - 10.4.2 + - L3a solar wind alphas + - 56-59 + - **L3.** :ref:`swapi-l3-scope` + * - 10.4.3 + - L3a pickup helium + - 59-69 + - **L3.** :ref:`swapi-l3-scope` + * - 10.5 + - L3b combined differential flux + - 69 + - **L3.** :ref:`swapi-l3-scope` + * - 10.6 + - **Summary of SWAPI data products** (inputs and ancillaries per product) + - 69-72 + - :ref:`swapi-data-products`, :ref:`swapi-l3-scope` + * - 11 + - Maintenance: the gain test 1x2 LUT (table 6) + - 73 + - :ref:`swapi-ancillary` + * - 12 + - Recommended testing approaches + - 74 + - :ref:`swapi-ancillary` + * - 13 + - **Code delivery** (L3 is a SWAPI Docker container) + - 75 + - :ref:`swapi-l3-scope` + * - 14 + - **I-ALiRT products** + - 76-78 + - :ref:`swapi-ialirt` + * - 15 + - Quick look products + - 79-80 + - not implemented + * - 16 + - References + - 81 + - -- + +Figures and tables worth knowing about +-------------------------------------- + +.. list-table:: + :header-rows: 1 + :widths: 16 14 70 + + * - Item + - PDF page + - What it shows + * - Figure 2 + - 10 + - Instrument block diagram: three HVPS, three counters, 8051. + * - Figure 3 + - 12 + - Electro-optics cross-section - the clearest single picture of the + sunglasses / open-aperture / ESA / carbon foil / PCEM / SCEM path. + * - Figure 4 + - 18 + - **The data pipeline flow diagram.** Worth having open when reading + :ref:`swapi-data-products`. + * - Table 1 + - 34 + - **NH/SWAP vs IMAP/SWAPI operations comparison.** The concise statement + of the 12 s / 72 step / 0.167 s cadence and the sweep-plan regime. + * - Table 2 + - 37-38 + - L1 data product contents. + * - Table 3 (flags) + - 39 + - L1 flag array contents. + * - Table 3 (ADP) + - 41-42 + - Start and end of the ESA Unit Conversion ADP. Note the document reuses + the number "Table 3" for both the flag array and the ADP. + * - Figure 12 + - 31 + - Central effective area and azimuthal transmission curves. + * - Figure 13 + - 33 + - Interpolated energy-angle passbands with integration limits. + * - Table 4 + - 52 + - SWAPI proton fitted parameters vs WIND/SWE - the flight-data sanity + check for L3a. + * - Table 5 + - 66 + - He\ :sup:`+` PUI fitting parameter bounds and initial values. + * - Table 6 + - 73 + - The 1x2 gain test LUT format. + * - Figure 24 + - 78 + - **The I-ALiRT validation case** (550 km/s, 5.27 cm\ :sup:`-3`, + 1e5 K in; 545.3 km/s, 4.67 cm\ :sup:`-3`, 1.18e5 K out). + * - Figure 25 + - 80 + - The daily quick-look plot layout. diff --git a/docs/source/algorithm-code-documentation/swe.rst b/docs/source/algorithm-code-documentation/swe.rst index 291a43bb28..6b5a264359 100644 --- a/docs/source/algorithm-code-documentation/swe.rst +++ b/docs/source/algorithm-code-documentation/swe.rst @@ -36,4 +36,4 @@ The modules below contains various utility classes and functions used for proces :template: autosummary.rst :recursive: - utils.swe_utils + utils.swe_utils \ No newline at end of file diff --git a/docs/source/algorithm-code-documentation/swe/ancillary.rst b/docs/source/algorithm-code-documentation/swe/ancillary.rst new file mode 100644 index 0000000000..898d530139 --- /dev/null +++ b/docs/source/algorithm-code-documentation/swe/ancillary.rst @@ -0,0 +1,350 @@ +.. _swe-ancillary: + +Ancillary Files, Calibration and External Dependencies +====================================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +What the pipeline actually reads +-------------------------------- + +.. list-table:: + :header-rows: 1 + :widths: 20 12 16 52 + + * - Descriptor + - Level + - Format + - Consumed by + * - ``esa-lut`` + - L1B + - CSV + - ``get_checker_board_pattern()``, ``get_esa_energy_pattern()`` + * - ``l1b-in-flight-cal`` + - L1B, I-ALiRT + - CSV + - ``read_in_flight_cal_data()`` + * - ``eu-conversion`` + - L1B + - CSV + - ``convert_raw_to_eu()`` + * - SPICE kernels + - L1A, L1B, L2 + - SPICE + - ``met_to_ttj2000ns``, ``get_instrument_spin_phase`` + +Representative copies of all three CSVs live in +``imap_processing/tests/swe/lut/``. They are test fixtures, not the operational +files, but the **column names and shapes are the contract** and the code will +break on anything else. + +.. note:: + + SWE also delivers a fourth ancillary file under the ``config`` descriptor. + It holds the tunable constants of the **L3** algorithms - geometric + fractions, pitch angle, gyrophase and energy bin definitions, + ``in_vs_out_energy_index``, ``core_halo_breakpoint_initial_guess`` and about + a dozen similar values. **Nothing in this repository reads it**, because SWE + L3 is produced elsewhere (:ref:`swe-l3-scope`). It is mentioned here only so + that finding it in a dependency list or an archive manifest does not look + like a missing L1B or L2 input. + +.. _swe-esa-lut: + +ESA lookup table (``esa-lut``) +------------------------------ + +Fixture: ``imap_swe_esa-lut_20250301_v000.csv``, 384 data rows. + +This single file encodes the entire onboard ESA stepping scheme, and it is what +makes the checkerboard data-driven rather than hard-coded. + +.. list-table:: + :header-rows: 1 + :widths: 16 84 + + * - Column + - Meaning + * - ``table_idx`` + - Which onboard table, **0-7**. This matches the document's statement that + "SWE flight software does include the ability to select from 8 onboard + look up tables". 48 rows each. + * - ``esa_step`` + - Step index within the full cycle. Only the **first 12 steps of each + quarter cycle** are listed - 0-11, 180-191, 360-371, 540-551 - because + the 12-step pattern repeats through the remaining 168 steps of each + quarter cycle. + * - ``esa_v`` + - ESA plate voltage in volts for that step. + * - ``v_index`` + - Which of the 24 energy rows, **1-24** (the code subtracts 1). + * - ``ialirt`` + - 1 if this step is one of the eight downlinked in real time. + +What the tables contain in the fixture: + +.. list-table:: + :header-rows: 1 + :widths: 14 20 66 + + * - ``table_idx`` + - ``ialirt`` rows + - Contents + * - 0 + - 16 + - **The nominal science table.** All 24 voltages, 0.56 V to 1108.66 V, in + the interleaved even/odd arrangement. 16 flagged rows = 8 energies, + each appearing twice in the 48-row block. + * - 1 + - 0 + - **A calibration (gain sweep) table.** All 48 rows hold the single + voltage 5.64 V at ``v_index`` 8 - exactly the fixed-ESA configuration + the weekly gain sweep needs. + * - 2-7 + - 0 + - Additional tables. Not exercised by any test. + +.. important:: + + **[CODE]** The relationship "``esa_table_num == 0`` means science, anything + else means calibration" is what L1B filters on, and it is only true because + of how the operational LUT is populated. Two places disagree about how many + tables are legitimate: + + * ``swe_l1b_science()`` keeps only ``esa_table_num == 0``. + * ``get_esa_dataframe()`` raises ``ValueError`` for anything outside + ``[0, 1]``. + * ``get_checker_board_pattern()`` and ``get_esa_energy_pattern()`` default + to ``esa_table_num=0`` and are **always called with the default** - the + packet's actual ``esa_table_num`` is never passed through. + + If SWE ever commands a different science table, L1B will silently drop every + packet. See :ref:`swe-implementation-status`. + +In-flight calibration (``l1b-in-flight-cal``) +--------------------------------------------- + +Fixture: ``imap_swe_l1b-in-flight-cal_20240510_20260716_v000.csv``. + +.. code-block:: text + + met_time,cem1,cem2,cem3,cem4,cem5,cem6,cem7 + 453050308,1,1,1,1,1,1,1 + 553051294,1,1,1,1,1,1,1 + 1782864000,2,2,2,2,2,2,2 + +One row per weekly gain sweep: a MET timestamp followed by seven multiplicative +factors, one per CEM, applied to counts to correct for gain degradation. + +**[DOC]** section 3.3.4 specifies: + +* Filename ``imap_swe_l1b-in-flight-cal_YYYYMMDD_vXXX.csv`` where ``YYYYMMDD`` + is the date from which the file should first be used. Version nominally 001. +* All factors are **1.0 for times before commissioning**. +* When a new calibration point is added, the **filename date is set to the date + of the previous calibration** - because linear interpolation only becomes + possible back to that point once the new point exists. This is what drives + reprocessing. +* The file carries a **trailing row with a far-future timestamp** repeating the + most recent factors, so quicklook processing can run before the next + calibration. (The fixture's third row, MET 1782864000, is that padding row.) +* **[DOC]** Section 3.4.3: factors come from comparing counts at the operating + CEM level (step 2 of the gain run) to counts at the next higher level (step + 3), assuming the higher level is correct. + +**[CODE]** ``read_in_flight_cal_data()`` accepts a list of files, concatenates, +drops rows with no MET, sorts, and de-duplicates on MET keeping the last. The +interpolation and the ``LAST_CAL_INTERVAL`` flag are described in +:ref:`swe-l1`. + +.. note:: + + **[DOC]** The same factors are used for I-ALiRT, but I-ALiRT "will + necessarily use the most recent calibration factors available, as + interpolation between data points is of course not possible for real time + analysis". **[CODE]** ``process_swe.py`` does exactly that: a + ``searchsorted(..., side="right") - 1`` to pick the last row at or before + the group midpoint, with no interpolation. + +EU conversion table (``eu-conversion``) +--------------------------------------- + +Fixture: ``imap_swe_eu-conversion_20240510_v000.csv``, 63 rows. Standard IMAP +format consumed by ``convert_raw_to_eu()``: ``packetName``, ``mnemonic``, +``convertAs`` (all ``UNSEGMENTED_POLY``), and coefficients ``c0``-``c7``. + +**[DOC]** Appendix A gives the algebraic expressions. Science packet: + +.. list-table:: + :header-rows: 1 + :widths: 34 66 + + * - Mnemonic + - Conversion + * - ``SPIN_PHASE`` + - ``0.005493 * x`` + * - ``SPIN_PERIOD`` + - ``0.00032 * x`` + * - ``THRESHOLD_DAC`` + - ``0.001221 * x`` + +App housekeeping packet (selection - the full set is in the CSV): + +.. list-table:: + :header-rows: 1 + :widths: 34 66 + + * - Mnemonic + - Conversion + * - ``HVPS_CEM_DAC`` + - ``1.025641 * x`` + * - ``HVPS_VBULK`` + - ``0.51282 * x`` + * - ``HVPS_VCEM`` + - ``1.34616 * x`` + * - ``HVPS_VESA`` + - ``0.38462 * x`` + * - ``HVPS_VESA_LOW_RANGE`` + - ``0.0078526 * x`` + * - ``HVPS_ICEM`` + - ``0.064103 * x`` **[DOC]** / ``0.000064103 * x`` **[CODE]** + * - ``FEE_TEMP``, ``SENSOR_TEMP``, ``HVPS_TEMP``, ``CDH_*_TEMP`` + - 6th- or 5th-order polynomials (see Appendix A / the CSV) + * - ``LVPS_*_BOARD_TEMP`` + - ``-273.2 + 0.1444619083 * x`` + * - ``LVPS_*_VMON`` / ``IMON``, ``CDH_*_VMON`` + - Linear, some with negative slopes for the negative rails + * - ``HVPS_ESA_DAC`` + - **Two conversions.** Low range ``0.007852613 * x``; high range + ``0.384617788 * x``. + +.. warning:: + + **[DOC]** "Please note for ``HVPS_ESA_DAC``, there will be two different + conversion based on whether we are in high range or low range of the ESA + voltage." + + **[CODE]** ``convert_raw_to_eu()`` selects on ``mnemonic`` alone; there is + no range-dependent branch and no ``HVPS_ESA_DAC`` row in the fixture CSV. + The dual-range handling is not implemented. See + :ref:`swe-implementation-status`. + + Separately, ``HVPS_ICEM`` in the fixture differs from the document by a + factor of 1000 (amps versus milliamps, most likely). Worth a question to + the SWE team rather than a unilateral edit. + +Calibration constants that are **not** ancillary files +------------------------------------------------------ + +Two quantities the algorithm document says "will be stored in a calibration +data file" are instead hard-coded in +``imap_processing/swe/utils/swe_constants.py``, both under a shared +``# TODO: add these to instrument status summary``: + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Constant + - Status + * - ``ENERGY_CONVERSION_FACTOR = 4.75`` + - The analyzer constant ``k = E/V``. **[DOC]** "determined from ground + calibration. This value will be stored in a calibration data file, where + it can be read by the processing code." + * - ``GEOMETRIC_FACTORS`` + - ``[424.4, 564.5, 763.8, 916.9, 792.0, 667.7, 425.2] * 1e-6`` + cm^2 sr eV/eV. **[DOC]** "nominal SWE geometric factors ... as of + October 2025, and may be refined by further analysis"; "geometric + factors can also be stored in a calibration data file to be read by the + processing routine." + +Changing either one changes every L2 number, so promoting them to an ancillary +file is a real (and probably inevitable) piece of work. Note that a geometric +factor update would **not** require reprocessing L1B - the time-varying part of +the detector response is handled entirely by the in-flight calibration at L1B. + +Constants worth knowing +----------------------- + +.. list-table:: + :header-rows: 1 + :widths: 34 20 46 + + * - Name + - Value + - Meaning + * - ``N_ESA_STEPS`` + - 24 + - Distinct ESA voltages / energies per full cycle. + * - ``N_ANGLE_SECTORS`` / ``N_ANGLE_BINS`` + - 30 + - Spin sectors (L1B) and 12-degree angle bins (L2). Same number, different + meanings. + * - ``N_CEMS`` + - 7 + - Detectors. + * - ``N_QUARTER_CYCLES`` + - 4 + - Packets per full cycle. + * - ``N_QUARTER_CYCLE_STEPS`` + - 180 + - Measurements per packet. + * - ``ENERGY_CONVERSION_FACTOR`` + - 4.75 + - Analyzer constant k. + * - ``VELOCITY_CONVERSION_FACTOR`` + - 1.237e31 + - ``v^4 / E^2`` for electrons, cm/s and eV. + * - ``FLUX_CONVERSION_FACTOR`` + - 6.187e30 + - ``j / (fv * E)``. Exactly half of the above. + * - ``ELECTRON_MASS`` + - 9.10938356e-31 kg + - Defined but **not referenced anywhere** - the two conversion factors + above already have it baked in. + * - ``CEM_DETECTORS_ANGLE`` + - -63 ... +63 + - Polar angle per CEM. + * - deadtime + - 360e-9 s + - Local to ``deadtime_correction()``. **[DOC]** shows 1.5e-6 as a + placeholder. + * - BDE threshold + - 1.75 + - Default in ``determine_streaming()``. **[DOC]** initial value from + Genesis, to be refined in flight. + * - BDE minimum steps + - 3 of 8 + - Default in ``compute_bidirectional()``. **[DOC]** initial value. + +External dependencies +--------------------- + +SPICE / spin +^^^^^^^^^^^^ + +The only hard external dependency below L3. + +* ``imap_processing.spice.time.met_to_ttj2000ns`` - L1B epoch, I-ALiRT epoch. +* ``imap_processing.spice.spin.get_instrument_spin_phase`` and ``get_spin_angle`` + - L2 spin angles. The CLI declares "spin data" as the second L2 dependency. + +MAG and SWAPI +^^^^^^^^^^^^^ + +**[DOC]** L3 needs the MAG field vector (pitch angle, and the field-aligned +temperature rotation) and SWAPI ion velocity, density and temperature (solar +wind frame transformation, spacecraft potential, and the assumption that the +bulk electron velocity equals the proton velocity). + +**Neither is a dependency of anything in this repository.** See +:ref:`swe-l3-scope`. + +Ultra deflector state +^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** An unusual cross-instrument coupling: the L3 break-point finding +algorithm "has been optimized for times when the Ultra deflector voltage is at +the nominal 3500 V level. The algorithm may fail at times when the Ultra +deflectors are turned off, and a flag is added to the L3 data to indicate these +times." L3's problem, but worth knowing the coupling exists. diff --git a/docs/source/algorithm-code-documentation/swe/data-products.rst b/docs/source/algorithm-code-documentation/swe/data-products.rst new file mode 100644 index 0000000000..c36ad12c13 --- /dev/null +++ b/docs/source/algorithm-code-documentation/swe/data-products.rst @@ -0,0 +1,350 @@ +.. _swe-data-products: + +Data Products and What Feeds What +================================= + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This is the map. If you are trying to work out which file a variable comes from +or which CLI invocation produces it, start here. + +Product inventory +----------------- + +**[CODE]** Every ``Logical_source`` below is defined in +``imap_processing/cdf/config/imap_swe_global_cdf_attrs.yaml``. That file is the +authority; if you add a product, add it there first. + +.. list-table:: + :header-rows: 1 + :widths: 26 10 64 + + * - ``logical_source`` + - Level + - Contents + * - ``imap_swe_l1a_sci`` + - L1A + - Decompressed 16-bit CEM counts, one record per ``SWE_SCIENCE`` packet + (one quarter cycle), shaped ``(epoch, 180 spin_sector, 7 cem_id)``. + Carries the raw 8-bit counts alongside, plus every science packet + metadata field in raw (unconverted) units. + * - ``imap_swe_l1a_hk`` + - L1A + - ``SWE_APP_HK`` decommutated, **raw units**. No algorithm. + * - ``imap_swe_l1a_cem-raw`` + - L1A + - ``SWE_CEM_RAW`` decommutated. Engineering-mode 1-second CEM counts, + latched and live, uncompressed. No algorithm. + * - ``imap_swe_l1b_sci`` + - L1B + - Deadtime-corrected, gain-calibrated **count rates** on the full-cycle + checkerboard grid, shaped ``(epoch, 24 esa_step, 30 spin_sector, + 7 cem_id)``. Plus per-measurement acquisition times, ESA energies, + counting uncertainty and a quality flag. + * - ``imap_swe_l1b_hk`` + - L1B + - ``SWE_APP_HK`` decommutated with **derived (engineering) units** applied + by the XTCE. Note this is produced from the **L0 file**, not from + ``imap_swe_l1a_hk``. + * - ``imap_swe_l2_sci`` + - L2 + - Phase space density and number flux, both in the original + ``(esa_step, spin_sector)`` organization **and** binned into 30 fixed + 12-degree spin angle bins. Plus statistical uncertainties, spin angles, + acquisition times and the quality flag. + +Not produced here +----------------- + +.. list-table:: + :header-rows: 1 + :widths: 26 74 + + * - Product + - Why not + * - **SWE L3** + - Separate repository, run closer to the science team. Spacecraft + potential, core/halo break, pitch angle and gyrophase distributions, + and moments (density, velocity, temperature, heat flux). See + :ref:`swe-l3-scope`. + * - **I-ALiRT SWE** + - Produced here, but **not by** ``imap_processing/swe`` and **not as a + SWE CDF**. It lives in ``imap_processing/ialirt/l0/process_swe.py`` and + is merged with the other instruments' real-time records into a single + I-ALiRT dataset. See :ref:`swe-ialirt`. + * - **In-flight calibration analysis** + - **[DOC]** Section 3.3.4: the weekly gain sweep data are examined **on + the ground by a SWE science team member** to decide whether CEM bias + needs raising. The SDC's job is only to keep those data out of L1B and + above. The resulting factors arrive back as an ancillary file. + * - **Quicklook** + - Nothing SWE-specific exists in this repository. + +The processing chain +-------------------- + +.. code-block:: text + + L0 CCSDS (.pkts) + | + | swe_l1a() -- packet_file_to_datasets, use_derived_value=False + | dispatch on APID + +-- APID 1344 SWE_SCIENCE --> swe_science() --> imap_swe_l1a_sci + | (8-bit -> 16-bit decompression) + +-- APID 1330 SWE_APP_HK ------------------> imap_swe_l1a_hk + +-- APID 1334 SWE_CEM_RAW ------------------> imap_swe_l1a_cem-raw + | + | (all three then pass through filter_day_boundary_data) + v + imap_swe_l1a_sci + eu-conversion + esa-lut + l1b-in-flight-cal + | + | swe_l1b_science() + | 1. convert_raw_to_eu on science metadata + | 2. drop calibration-mode data (esa_table_num != 0) + | 3. keep only complete 0,1,2,3 quarter-cycle runs + | 4. checkerboard reorganization -> (n, 24, 30, 7) + | 5. acquisition time per measurement + | 6. deadtime correction + | 7. in-flight gain calibration (+ LAST_CAL_INTERVAL flag) + | 8. counts -> rate + | 9. sqrt(counts) uncertainty, ESA energies + v + imap_swe_l1b_sci + | + | swe_l2() + | 1. phase space density from count rate + | 2. number flux from phase space density + | 3. spin phase from SPICE -> spin angle + | 4. bin into 30 fixed 12-degree spin angle bins + v + imap_swe_l2_sci ----> (separate repository) ----> SWE L3 + + L0 CCSDS (.pkts) -- swe_l1b(), use_derived_value=True --> imap_swe_l1b_hk + +Dependency wiring +----------------- + +**[CODE]** ``imap_processing/cli.py``, ``class Swe``. The CLI checks the +*number* of dependencies before doing anything, so an extra or missing +ancillary file fails loudly rather than silently changing behavior. + +.. list-table:: + :header-rows: 1 + :widths: 12 12 34 42 + + * - Level + - Descriptor + - Dependencies (exact count enforced) + - Notes + * - ``l1a`` + - - + - **2**: SWE L0 file, time kernels + - ``swe_l1a(path)`` takes a plain path, not the collection. Returns a list + of up to three datasets. + * - ``l1b`` + - ``sci`` + - **5**: L1A science, ``l1b-in-flight-cal``, ``esa-lut``, + ``eu-conversion``, time kernels + - Exactly one science file; multiple is rejected. + * - ``l1b`` + - ``hk`` + - **2**: SWE L0 file, time kernels + - Reparses L0 with ``use_derived_value=True``. Does **not** read L1A HK. + * - ``l2`` + - - + - **2**: L1B science, spin data + - Exactly one science file. Spin data is consumed indirectly through + SPICE in ``get_instrument_spin_phase``. + +Any other data level raises ``NotImplementedError``. + +Ancillary inputs +---------------- + +Covered in full in :ref:`swe-ancillary`. Summary of descriptors: + +.. list-table:: + :header-rows: 1 + :widths: 24 14 62 + + * - Descriptor + - Used at + - Purpose + * - ``esa-lut`` + - L1B + - The 8 onboard ESA stepping tables. Supplies both the checkerboard index + map and the per-cell ESA voltage. + * - ``l1b-in-flight-cal`` + - L1B, I-ALiRT + - Per-CEM gain factors versus MET, one row per weekly gain sweep. + * - ``eu-conversion`` + - L1B + - Polynomial raw-to-engineering conversions for science packet metadata. + * - SPICE kernels + - L1A, L1B, L2 + - Time conversion everywhere; spin phase at L2. + +CDF variable inventory +---------------------- + +L1A science (``imap_swe_l1a_sci``) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Coordinates: ``epoch``, ``spin_sector`` (0-179), ``cem_id`` (0-6), plus label +variables. + +.. list-table:: + :header-rows: 1 + :widths: 26 24 50 + + * - Variable + - Shape + - Meaning + * - ``science_data`` + - (epoch, 180, 7) + - **Decompressed** 16-bit counts. + * - ``raw_science_data`` + - (epoch, 180, 7) + - The 8-bit values as telemetered. Dropped at L1B. + * - science packet metadata + - (epoch,) + - ``shcoarse``, ``acq_start_coarse``, ``acq_start_fine``, + ``acq_duration``, ``settle_duration``, ``spin_phase``, ``spin_period``, + ``quarter_cycle``, ``esa_table_num``, ``esa_acq_cfg``, ``threshold_dac``, + ``stim_enabled``, ``stim_cfg_reg``, ``cem_nominal_only``, + ``spin_period_validity``, ``spin_phase_validity``, + ``spin_period_source``, ``repoint_warning``, ``high_count``, ``cksum``. + All in **raw** units at L1A. + +**[CODE]** The CCSDS header fields (``version``, ``type``, ``sec_hdr_flg``, +``pkt_apid``, ``seq_flgs``, ``src_seq_ctr``, ``pkt_len``) are dropped in +``swe_science()`` before the merge, with a ``TODO`` noting they should not be +returned by ``packet_file_to_datasets`` in the first place. The APID is +preserved as the global attribute ``packet_apid``, which L1B reads back to pick +the EU conversion table. + +L1B science (``imap_swe_l1b_sci``) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Coordinates: ``epoch`` (one per full cycle), ``esa_step`` (0-23), +``spin_sector`` (0-29), ``cem_id`` (0-6), ``cycle`` (0-3), plus labels. + +.. list-table:: + :header-rows: 1 + :widths: 26 28 46 + + * - Variable + - Shape + - Meaning + * - ``science_data`` + - (epoch, 24, 30, 7) + - **Count rate**, counts/s. Deadtime corrected and gain calibrated. + * - ``counts_stat_uncert`` + - (epoch, 24, 30, 7) + - ``sqrt`` of the decompressed counts. **In counts, not counts/s** - see + :ref:`swe-implementation-status`. + * - ``acquisition_time`` + - (epoch, 24, 30) + - MET seconds at the **center** of each measurement's accumulation + window. + * - ``acq_duration`` + - (epoch, 24, 30) + - Microseconds, per measurement. + * - ``esa_energy`` + - (epoch, 24, 30) + - Electron energy in eV for each cell: ESA voltage from the LUT times the + analyzer constant. + * - ``data_quality`` + - (epoch,) + - ``SweL1bFlags`` bitfield. Currently only bit 2, ``LAST_CAL_INTERVAL``. + * - packet metadata + - (epoch, 4) + - Every L1A metadata field, reshaped so the four quarter cycles of a full + cycle sit along the ``cycle`` dimension. Engineering units where the EU + table defines a conversion. + +``epoch`` is ``met_to_ttj2000ns`` of the acquisition start time of the **third** +quarter cycle (index 2) of each full cycle, i.e. approximately the center of +the ~1 minute cycle. + +L2 science (``imap_swe_l2_sci``) +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Coordinates: ``epoch``, ``esa_step`` (0-23), ``energy`` (24 values in eV), +``spin_sector`` (0-29), ``inst_az`` (30 bin centers, 6...354 deg), ``cem_id`` +(0-6), ``inst_el`` (the 7 CEM polar angles), plus labels. + +.. list-table:: + :header-rows: 1 + :widths: 32 30 38 + + * - Variable + - Dimensions + - Meaning + * - ``phase_space_density`` + - epoch, energy, inst_az, inst_el + - Binned into 12-degree spin angle bins. **The primary L2 product.** + Units s^3 / cm^6. + * - ``flux`` + - epoch, energy, inst_az, inst_el + - Same, as differential number flux. Units 1 / (eV cm^2 s ster). + * - ``phase_space_density_spin_sector`` + - epoch, esa_step, spin_sector, cem_id + - **Unbinned**, on the checkerboard grid. Retained explicitly for L3, + which needs per-measurement resolution to compute pitch angles. + * - ``flux_spin_sector`` + - epoch, esa_step, spin_sector, cem_id + - Same. + * - ``inst_az_spin_sector`` + - epoch, energy, inst_az (see note) + - The **actual** SWE spin angle in degrees of each measurement, before + binning. Needed by L3. + * - ``psd_stat_uncert``, ``flux_stat_uncert`` + - epoch, esa_step, spin_sector, cem_id (see note) + - Statistical uncertainties. + * - ``acquisition_time``, ``acq_duration``, ``data_quality`` + - carried through from L1B + - Carried "for L3 purposes" per comments in ``swe_l2.py``. + +.. warning:: + + **[CODE]** Two of the L2 variables carry dimension names that do not match + the data in them. ``inst_az_spin_sector`` holds unbinned + ``(epoch, esa_step, spin_sector)`` data but is declared + ``(epoch, energy, inst_az)``; ``psd_stat_uncert`` and ``flux_stat_uncert`` + hold **binned** data but are declared ``(esa_step, spin_sector)``. The sizes + happen to match (24 energies vs 24 ESA steps, 30 sectors vs 30 bins), so + nothing raises. See :ref:`swe-implementation-status`. + +Epoch convention +---------------- + +.. list-table:: + :header-rows: 1 + :widths: 24 76 + + * - Level + - ``epoch`` + * - L1A + - Straight from ``packet_file_to_datasets``: one per packet, derived from + the packet's ``SHCOARSE``. + * - L1B science + - One per full cycle. ``met_to_ttj2000ns`` of + ``ACQ_START_COARSE + ACQ_START_FINE/1e6`` of the **third** quarter cycle + packet. + * - L2 + - Inherited unchanged from L1B. + * - I-ALiRT + - ``met_to_ttj2000ns`` of the midpoint of each 30-second half cycle, + stored as ``swe_epoch``. + +Filenames and descriptors +------------------------- + +Standard IMAP convention: ``imap_swe____v.cdf``, +derived by ``write_cdf()`` from ``Logical_source`` and ``Data_version``. +Descriptors in use: ``sci``, ``hk``, ``cem-raw``. + +**[CODE]** ``filter_day_boundary_data(ds, self.start_date)`` is applied to every +L1A dataset, so a file dated ``YYYYMMDD`` contains only packets from that UTC +day even when the L0 file spans a boundary. It is **not** applied at L1B or L2. diff --git a/docs/source/algorithm-code-documentation/swe/ialirt.rst b/docs/source/algorithm-code-documentation/swe/ialirt.rst new file mode 100644 index 0000000000..fe8e90ea64 --- /dev/null +++ b/docs/source/algorithm-code-documentation/swe/ialirt.rst @@ -0,0 +1,336 @@ +.. _swe-ialirt: + +I-ALiRT - Real-Time Bidirectional Electrons +=========================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This page covers algorithm document sections 3.3.5, 3.4.2.2 and 3.4.5. + +.. important:: + + **None of this lives in** ``imap_processing/swe``. The entire SWE I-ALiRT + algorithm is ``imap_processing/ialirt/l0/process_swe.py``, which imports + three things from the SWE package: + ``decompressed_counts``, ``deadtime_correction`` and + ``read_in_flight_cal_data``. If you change any of those three, you have + changed the real-time product. + +What it is +---------- + +**[DOC]** SWE is part of the IMAP Active Link for Real-Time (i-ALiRT) system. +The downlinked subset is **electron counts from all 7 CEMs at all spin angles +for 8 of the 24 energy channels**, chosen to cover the suprathermal range +**100 - 1000 eV** where the distribution is governed by field topology. + +Two products come out: + +#. **BDE** - the bidirectional electron flag. One value per time step: **1 = + counterstreaming**, **0 = nominal unidirectional flow**. Counterstreaming is + typically the signature of a coronal mass ejection, where the field is + connected to the Sun at both ends. **[DOC]** cautions that it can also come + from connection to Earth's bow shock, so it is "not a definitive CME + signature but must be considered in the context of other measurements". The + algorithm derives from the onboard search flown on Genesis/GEM + [Neugebauer et al., 2003]. +#. **Normalized counts** - counts at each of the 8 ESA steps summed over all + azimuths and all CEMs, giving a time series that shows suprathermal + variability and the energy distribution. + +Which 8 energies +---------------- + +**[DOC]** "for the nominal stepping table ... the SWE i-ALiRT packet includes +ESA steps **12-19**" (1-based). + +**[CODE]** Zero-based rows 11-18 of ``ESA_VOLTAGE_ROW_INDEX_DICT``: + +.. list-table:: + :header-rows: 1 + :widths: 20 20 20 40 + + * - row (0-based) + - ESA V + - E (eV) + - ``ialirt.utils.constants.swe_energy`` + * - 11 + - 21.13 + - 100.4 + - 100.4 + * - 12 + - 29.39 + - 139.6 + - 140.0 + * - 13 + - 40.88 + - 194.2 + - 194.0 + * - 14 + - 56.87 + - 270.1 + - 270.0 + * - 15 + - 79.10 + - 375.7 + - 376.0 + * - 16 + - 110.03 + - 522.6 + - 523.0 + * - 17 + - 153.05 + - 726.9 + - 727.0 + * - 18 + - 212.89 + - 1011.2 + - 1011.0 + +The same selection is encoded a **third** time as the ``ialirt`` column of the +ESA LUT CSV, where exactly 16 of table 0's 48 rows are flagged (8 energies +appearing twice each). Three independent encodings of one fact; see +:ref:`swe-implementation-status`. + +The packet +---------- + +**[DOC]** Section 3.4.2.2. Unlike ``SWE_SCIENCE``, the I-ALiRT packet gives +**each 8-bit value its own named field**. Starting at byte 18, bit 0 there are +**28 fields**: 4 ESA steps × 7 CEMs, laid out CEM-major (CEM1's four steps, +then CEM2's four, and so on). + +Per nominal 1-second packet: 7 CEMs × 4 energies, with **2 energies in one +spin-angle bin and 2 in the next**. Field names follow +``SWE_IALIRT.ELECTRON_COUNTS_SPIN_I_POL__E_J``. + +The counts use the **same 8-bit compression** as the science packet - see +:ref:`swe-decompression`. + +**[CODE]** ``imap_processing/ialirt/packet_definitions/ialirt_swe.xml`` names +them ``SWE_CEM_E`` (n = 1-7, m = 1-4), plus ``SWE_SHCOARSE``, +``SWE_ACQ_SEC``, ``SWE_ACQ_SUB``, ``SWE_SEQ``, ``SWE_NOM_FLAG``, +``SWE_OPS_FLAG``. + +Cadence +------- + +**[DOC]** "2 quarter-cycles are required for i-ALiRT measurements at all 8 +energy steps. Since one criterion for defining the SWE i-ALiRT bidirectional +electron parameter will be the identification of counterstreaming electrons +over a range of energies, this means that the appropriate time cadence for SWE +i-ALiRT data will be **30 seconds**." + +**[CODE]** ``process_swe()`` accumulates a full minute (``swe_seq`` 0-59, one +packet per second), then splits it into two halves - ``swe_seq`` 0-29 and +30-59 - and emits **two records per minute**. A group with any missing or +duplicate sequence number is skipped and logged. + +The algorithm as implemented +---------------------------- + +.. code-block:: text + + 60 accumulated 1-second packets + | + | drop groups where swe_nom_flag == 0 + | require swe_seq to be exactly 0..59 + | + +-- first half (swe_seq 0-29) --+ + +-- second half (swe_seq 30-59) --+ + | + 1. prepare_raw_counts -> (8 energies, 7 CEMs, 30 phi bins) + 2. decompress_counts -> 16-bit + 3. deadtime_correction -> acq_duration hard-coded to 80 ms + 4. normalize_counts -> * cal_factor / geometric_factor + 5a. sum over CEMs -> azimuthal_check_counterstreaming + 5b. sum over azimuth -> polar_check_counterstreaming + 6. BDE = max(azimuthal, polar) + 7. sum over CEMs and azimuth -> swe_normalized_counts (8 values) + +Step 1 - building the (8, 7, 30) array +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[CODE]** ``prepare_raw_counts()``. Two lookups do all the work: + +.. code-block:: python + + ENERGY_BINS = np.array([ + [1, 5, 7, 3], # swe_seq 0-14 (Q1) + [2, 6, 4, 0], # swe_seq 15-29 (Q2) + [3, 7, 5, 1], # swe_seq 30-44 (Q3) + [0, 4, 6, 2], # swe_seq 45-59 (Q4) + ]) + + phi = [(12 + 24*seq) % 360, # energy fields e1, e2 + (24 + 24*seq) % 360] # energy fields e3, e4 + bin = ((phi - 12) // 12) % 30 + +The parity is the thing to notice: ``(12 + 24*seq)`` always maps to an **even** +bin index and ``(24 + 24*seq)`` to an **odd** one. Combined with +``ENERGY_BINS``, each of the 8 energies is written into **only even or only odd +bins**, 15 of the 30, within a half cycle. The other 15 stay zero. + +This is not a defect - it reflects the instrument, where a given energy is +sampled in only the odd or only the even spin-angle bins of a quarter cycle - +and the peak-offset arithmetic below preserves parity, so the zeros are never +read. + +Steps 2-4 - counts to normalized counts +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** "For i-ALiRT calculations, we do not need to convert counts to phase +space distribution or intensity, but do need to **normalize the counts based on +the geometric factors** for each CEM detector", and to apply the in-flight +calibration, using **the most recent factors** rather than an interpolation. + +.. code-block:: text + + norm_counts[i][j][k] = ccounts[i][j][k] * cal_factor[j] / g[j] + +**[CODE]** + +* ``deadtime_correction(counts, 80 * 10**3)`` - ``ACQ_DURATION`` is not in the + I-ALiRT packet, so the nominal 80 ms is hard-coded in microseconds. +* Calibration factor: ``searchsorted(cal_met, group_mid, side="right") - 1``, + i.e. the last row at or before the midpoint of the half cycle. No + interpolation, as specified. +* ``normalize_counts(counts, interp_cal)`` - + ``counts * (interp_cal / GEOMETRIC_FACTORS)[:, np.newaxis]``, then clamps + negatives to 0 (the equivalent of the heritage ``ccounts < 0`` guard, which + is **absent** from the science L2 path). + +.. note:: + + **[DOC]** The I-ALiRT C fragment assigns a **different** geometric factor + array (``435.0e-6, 599.0e-6, 808.0e-6, 781.0e-6, 876.0e-6, 548.0e-6, + 432.0e-6``) from the one in its own comment header and from the science + ``fspace()`` fragment. **[CODE]** uses the single shared + ``swe_constants.GEOMETRIC_FACTORS``, i.e. the SWE nominal set. The document + is internally inconsistent here; the code's choice (one set of geometric + factors for the instrument) is the defensible one, but it is worth + confirming with SWE. + +Step 5a - the azimuthal search +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Sum normalized counts over all CEMs, leaving counts as a function of +energy and azimuth. At each energy: + +* Find the azimuth ``Apeak`` with maximum counts ``Cpeak``. **This direction is + assumed to be the magnetic field direction.** +* ``C180`` = counts 180 degrees away, i.e. 15 bins - but since only every other + bin is measured, use the **average of the two neighbours**: bins ``N+14`` and + ``N+16``. +* ``C90`` = counts 90 degrees away, the average of ``N+6``, ``N+8``, ``N+22`` + and ``N+24``. All arithmetic mod 30. +* ``Cmin = min(C180, C90)``. +* **Bidirectional at this step if both ``Cpeak/Cmin`` and ``C180/Cmin`` exceed + the threshold**, initially **1.75** from Genesis experience. + +The physical reasoning: for unidirectional flow, both 90 and 180 degrees away +are low. For bidirectional flow, 180 degrees away is *also* high, and only the +90-degree directions are low. + +**[CODE]** ``find_min_counts()`` computes **three** offset averages rather than +two: + +.. code-block:: python + + counts_90 = average_counts(peak_bin, summed, ( 6, 8)) + counts_180 = average_counts(peak_bin, summed, (14, 16)) + counts_neg_90 = average_counts(peak_bin, summed, (-6, -8)) + cmin = np.min(np.hstack([counts_90, counts_180, counts_neg_90]), axis=1) + +Note ``-6 mod 30 == 24`` and ``-8 mod 30 == 22``, so the code's +``counts_neg_90`` is the document's other half of ``A90``. The deviation is +that the code takes the **minimum of the two 90-degree sides separately** +rather than their average. That makes ``Cmin`` smaller or equal, so the ratios +are larger or equal, so the code is **slightly more willing to declare +bidirectional flow** than the document. All offsets are even, which preserves +the bin parity noted above. + +``determine_streaming(cpeak, counts_180, cmin, threshold=1.75)`` then applies +the two-ratio test elementwise over the 8 energies. + +Step 5b - the polar search +^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** "Depending on the direction of the interplanetary magnetic field, it +is possible that counterstreaming electrons could be observed in the polar +angle direction rather than the azimuthal angle direction. Thus if +counterstreaming is not observed in the above analysis, a second search will be +done in the other dimension." + +Sum over azimuth instead of over CEMs, leaving counts as a function of energy +and CEM. Then ``Cmin`` = the mean of CEMs 3, 4 and 5 (the middle three), and +bidirectional is declared if both ``C_CEM1/Cmin`` and ``C_CEM7/Cmin`` exceed +1.75 - i.e. both **end** detectors are enhanced relative to the middle. + +**[CODE]** ``polar_check_counterstreaming()``: +``summed[:, 2:5].mean(axis=1)`` for ``Cmin`` (zero-based indices 2, 3, 4 = +CEMs 3, 4, 5), then ``determine_streaming(summed[:, 0], summed[:, 6], cmin)``. + +**[CODE]** The polar search runs **unconditionally**, not only when the +azimuthal search fails, and the results are combined with ``max()``. That is +logically equivalent to the document's "if not observed, do the second search", +and simpler to vectorize. + +Step 6 - the BDE flag +^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** "The bidirectional electron parameter BDE will then be set to 1 if +bidirectional electrons are identified in **at least a minimum number of ESA +steps** ... with an initial value of **3 of the 8** ESA steps." + +**[CODE]** ``compute_bidirectional(streaming_first, streaming_second, +min_esa_steps=3)`` sums each half's per-energy flags and compares to 3. +``bde_first_half = max(bde_first_search[0], bde_second_search[0])`` combines the +azimuthal and polar searches. + +Step 7 - normalized counts product +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** "counts at each ESA step will be summed over both azimuthal and polar +angles ... this data product will have a 30 second cadence." + +**[CODE]** ``normalized_first_half.sum(axis=(1, 2))`` -> 8 values, cast to +``int`` for the output record. + +Output record +------------- + +**[CODE]** Each half cycle appends a dict: + +.. code-block:: python + + { + "instrument": "swe", + "swe_epoch": int(met_to_ttj2000ns(group_time_mid)), + "swe_normalized_counts": [int(v) for v in summed], # 8 values + "swe_counterstreaming_electrons": bde, # 0 or 1 + } + +merged with ``_populate_instrument_header_items(met)``. +``ialirt/utils/create_xarray.py`` writes these into the shared I-ALiRT dataset +against dimensions ``("swe_epoch", "swe_electron_energy")`` and +``("swe_epoch",)``, with dtypes ``int64`` and ``uint8`` +(``ialirt/utils/constants.py``). + +.. note:: + + **[DOC]** section 3.4.5 warns: "if a SWE quarter-cycle is sufficiently + longer than a spacecraft spin, it may be necessary to leave out the last + azimuthal step of each quarter cycle in the analysis". **[CODE]** does not + do this, and there is no flag or diagnostic that would tell you whether it + has become necessary. That decision was explicitly deferred to flight + experience. + +Tests +----- + +``imap_processing/tests/ialirt/unit/test_process_swe.py`` covers +``prepare_raw_counts``, ``decompress_counts``, ``normalize_counts``, +``find_bin_offsets``, ``average_counts``, ``find_min_counts``, +``determine_streaming``, ``compute_bidirectional`` and the end-to-end +``process_swe``. diff --git a/docs/source/algorithm-code-documentation/swe/implementation-status.rst b/docs/source/algorithm-code-documentation/swe/implementation-status.rst new file mode 100644 index 0000000000..71f7a7869e --- /dev/null +++ b/docs/source/algorithm-code-documentation/swe/implementation-status.rst @@ -0,0 +1,520 @@ +.. _swe-implementation-status: + +Implementation Status and Known Gaps +==================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This page is the honest accounting of where the code stands against the +algorithm document. **Read it before proposing or estimating work.** + +Accurate as of a survey of ``imap_processing/swe`` and +``imap_processing/ialirt/l0/process_swe.py`` against algorithm document +CN102D-D0001, Issue Draft, 15 June 2026. If you change something material, +update this page in the same commit. + +Summary +------- + +.. list-table:: + :header-rows: 1 + :widths: 16 20 64 + + * - Level + - State + - Notes + * - L0 / L1A + - **Complete** + - Decommutation and decompression are done and validated against GSEOS + exports. No rejection criteria exist in the document, so none are + missing. + * - L1A / L1B housekeeping + - **Complete** + - No algorithm - the same packet decommutated raw and derived. The + ``HVPS_ESA_DAC`` dual-range conversion is the one document requirement + not met. + * - L1B science + - **Mature** + - The checkerboard, timing, deadtime and gain calibration all work and are + tested. The active ESA table number is hard-wired to 0 and the + uncertainty is in the wrong units. + * - L2 + - **Complete for what it claims, with real gaps** + - Phase space density, flux, spin angle binning all work. **The + end-detector correction from the heritage code is missing**, the frame + is instrument rather than despun spacecraft, and the uncertainty path + has a double-binning bug. + * - I-ALiRT + - **Working, a few undocumented choices** + - Produces records; the ``Cmin`` definition, the unconditional polar + search and the hard-coded 80 ms all differ from or extend the document. + * - L3 + - **Out of scope, correctly** + - Separate repository. See :ref:`swe-l3-scope`. + * - Quicklook + - **Not started** + - No SWE quicklook code exists here. + +Ranked list of things to fix +---------------------------- + +Highest value first, in the judgement of whoever last surveyed this. Each has a +detailed entry below. + +#. :ref:`swe-gap-end-detector` - missing factor of 2 on CEMs 1 and 7 at L2. +#. :ref:`swe-gap-uncert-units` - uncertainty is in counts, data in counts/s. +#. :ref:`swe-gap-double-bin` - ``flux_stat_uncert`` binned twice. +#. :ref:`swe-gap-esa-table` - the active ESA table number is never read. +#. :ref:`swe-gap-frame` - instrument frame vs despun spacecraft at L2. +#. :ref:`swe-gap-none-dataset` - ``swe_l1b`` appends ``None`` when no full cycle + is found. +#. :ref:`swe-gap-dims` - mislabelled dimensions on three L2 variables. +#. :ref:`swe-gap-esa-dac` - ``HVPS_ESA_DAC`` dual-range conversion missing. +#. :ref:`swe-gap-calfile` - ``k`` and the geometric factors are hard-coded. +#. :ref:`swe-gap-dead-code` - three dead functions, one of which cannot run. + +Hard failures in the code +------------------------- + +Explicit exceptions, so you know what a bad input looks like: + +.. list-table:: + :header-rows: 1 + :widths: 40 60 + + * - Location + - Condition + * - ``cli.py`` ``Swe.do_processing`` + - ``NotImplementedError`` for any data level other than l1a/l1b/l2. + * - ``cli.py`` ``Swe.do_processing`` + - ``ValueError`` if the dependency count is not exactly 2 (l1a), 5 + (l1b sci), 2 (l1b hk) or 2 (l2). + * - ``cli.py`` ``Swe.do_processing`` + - ``ValueError`` if more than one science file is supplied for l1b or l2. + * - ``swe_l1b.calculate_calibration_factor`` + - ``ValueError`` if any acquisition time falls outside the in-flight + calibration time range. **Deliberate** - SWE does not want + extrapolation, matching the heritage ``electron_cal()`` which exits. + * - ``swe_l1b.get_esa_dataframe`` + - ``ValueError`` for an ESA table number outside ``[0, 1]``. Unreachable - + the function is never called. + * - ``swe_l2.find_angle_bin_indices`` + - ``ValueError`` if any spin angle is outside ``[0, 360)``. + +Detailed entries +---------------- + +.. _swe-gap-end-detector: + +1. The end-detector correction is missing +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**Severity: high. A factor of 2 on two of the seven detectors.** + +**[DOC]** The heritage ``fspace()`` routine reproduced in section 3.4.4 does not +end after the phase space density loop. It has a second loop: + +.. code-block:: c + + for (i=0; i95% of 4π sr. +* One **ESA step** (an ESA voltage setting) lasts nominally 83.333 ms = + 3.333 ms settle + 80 ms accumulate, and covers ~2 degrees of spin. Each step + yields 7 numbers, one per CEM. +* A **quarter cycle** is one telemetry packet: 15 seconds, 180 ESA steps + (12 per second), 1260 compressed bytes. Four quarter cycles make a **full + cycle** (~1 minute), which is the L1B/L2 record. A full cycle covers + **24 energies × 30 spin angles × 7 CEMs**. +* Within a quarter cycle only 6 of the 24 energies are sampled per spin-angle + bin, and odd- and even-numbered spin bins get *different* sets of 6. Sorting + the 4 × 180 measurements back into a (24, 30) grid is the **checkerboard** + reorganization and is the single most confusing piece of SWE code. +* Counts are telemetered **8-bit compressed** (SWEPAM scheme) and must be + expanded to 16 bits through a 16-entry base/step_size table. +* Processing chain in this repository: + ``CCSDS packets -> L1A (decompressed counts, per packet) -> L1B (checkerboard, + deadtime, gain cal, count rates) -> L2 (phase space density, flux, spin + angle bins)``. L3 is not ours. +* There is also a **30-second I-ALiRT** product (8 of the 24 energies, + normalized counts and a bidirectional-electron flag) living under + ``imap_processing/ialirt/``. + +Where the code lives +-------------------- + +.. code-block:: text + + imap_processing/swe/ + utils/swe_constants.py N_ESA_STEPS=24, N_ANGLE_SECTORS=30, N_CEMS=7, + N_QUARTER_CYCLES=4, N_QUARTER_CYCLE_STEPS=180, + GEOMETRIC_FACTORS, ENERGY_CONVERSION_FACTOR=4.75, + CEM_DETECTORS_ANGLE, ESA_VOLTAGE_ROW_INDEX_DICT + utils/swe_utils.py SWEAPID, acquisition-time helpers + packet_definitions/ + swe_packet_definition.xml SWE_APP_HK + SWE_CEM_RAW + SWE_SCIENCE + l1a/swe_l1a.py APID dispatch for science / HK / CEM raw + l1a/swe_science.py count decompression, L1A science dataset + l1b/swe_l1b.py checkerboard, deadtime, in-flight cal, rates + l2/swe_l2.py phase space density, flux, spin angle binning + + imap_processing/ialirt/ + l0/process_swe.py the whole SWE I-ALiRT algorithm (BDE) + utils/constants.py swe_energy (the 8 I-ALiRT energies in eV) + packet_definitions/ialirt_swe.xml SWE I-ALiRT packet fields + + imap_processing/cdf/config/imap_swe_global_cdf_attrs.yaml + imap_processing/cdf/config/imap_swe_l1a_variable_attrs.yaml + imap_processing/cdf/config/imap_swe_l1b_variable_attrs.yaml + imap_processing/cdf/config/imap_swe_l2_variable_attrs.yaml + imap_processing/quality_flags.py (class SweL1bFlags) + imap_processing/cli.py (class Swe) dependency wiring per level + imap_processing/tests/swe/ tests, L0 test data, validation CSVs, LUTs \ No newline at end of file diff --git a/docs/source/algorithm-code-documentation/swe/l1.rst b/docs/source/algorithm-code-documentation/swe/l1.rst new file mode 100644 index 0000000000..e3cf323780 --- /dev/null +++ b/docs/source/algorithm-code-documentation/swe/l1.rst @@ -0,0 +1,573 @@ +.. _swe-l1: + +L0 to L1B - Decompression, Reorganization and Rates +=================================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This page covers algorithm document sections 3.4.2 (L0 to L1A) and 3.4.3 +(L1A to L1B). + +The pipeline +------------ + +.. code-block:: text + + L0 packets + | + | ---------------- L1A: swe_l1a.py / swe_science.py ---------------- + | 1. decommutate by APID (use_derived_value=False) + | 2. unpack the 1260-byte SCIENCE_DATA field to (180, 7) uint8 + | 3. expand 8-bit -> 16-bit counts through the decompression table + v + imap_swe_l1a_sci (per quarter cycle) + | + | ---------------- L1B: swe_l1b.py --------------------------------- + | 4. raw -> engineering units for science metadata + | 5. drop calibration-mode quarter cycles + | 6. keep only runs of quarter_cycle == 0,1,2,3 + | 7. checkerboard: (n*4, 180, 7) -> (n, 24, 30, 7) + | 8. per-measurement acquisition time + | 9. deadtime correction + | 10. in-flight gain calibration + | 11. counts -> count rate + | 12. sqrt(counts) uncertainty; ESA energies from the LUT + v + imap_swe_l1b_sci (per full cycle) + +Step 1 - decommutation +---------------------- + +**[CODE]** ``swe_l1a()`` calls ``packet_file_to_datasets`` with +``use_derived_value=False`` and the SWE XTCE, then dispatches on APID: + +.. list-table:: + :header-rows: 1 + :widths: 14 24 62 + + * - APID + - Packet + - Handling + * - 1344 + - ``SWE_SCIENCE`` + - Full science processing (``swe_science()``). + * - 1330 + - ``SWE_APP_HK`` + - Attributes attached, passed through raw. + * - 1334 + - ``SWE_CEM_RAW`` + - Attributes attached, passed through raw. + +Unknown APIDs are logged, not raised. **[DOC]** Section 3.4.2.5 says autonomy, +static housekeeping, event message and memory dump packets "should also be +decommutated at this level"; none of them are in the SWE XTCE or the APID enum. + +Step 2 - the SCIENCE_DATA field +------------------------------- + +**[DOC]** ``SWE_SCIENCE.SCIENCE_DATA`` is a **1260-byte** fixed field starting +at byte 32, bit 0, and is the last field before the checksum. Interpret it as a +3D array: + +.. code-block:: text + + 7 (CEM_COUNTS) x 12 (STEPS_EACH_SECOND) x 15 (SECONDS) + +Layout, byte by byte: + +.. code-block:: text + + bytes 0- 6 CEM 1..7, ESA step 1 of second 1 (spin angle bin 1) + bytes 7-13 CEM 1..7, ESA step 2 of second 1 + ... + bytes 77-83 CEM 1..7, ESA step 12 of second 1 (spin angle bin 2) + bytes 84-90 CEM 1..7, ESA step 1 of second 2 (spin angle bin 3) + ... + bytes 1253-1259 CEM 1..7, ESA step 12 of second 15 (spin angle bin 30) + +Note the pairing: 12 ESA steps per second span **two** 12-degree spin-angle +bins (6 steps each). + +**[CODE]** ``swe_science()`` does this in one vectorized pass: + +.. code-block:: python + + raw_science_array = np.array([ + np.frombuffer(binary_string, dtype=np.uint8).reshape(180, N_CEMS) + for binary_string in l0_dataset["science_data"].values + ]) + +giving ``(n_packets, 180, 7)``. The 180 axis is named ``spin_sector`` in the +L1A CDF - at L1A that name means "step index within the quarter cycle", **not** +a 12-degree angle bin. + +.. _swe-decompression: + +Step 3 - count decompression +---------------------------- + +**[DOC]** Flight software compresses each 16-bit counter to 8 bits using the +scheme inherited from ACE/SWEPAM. This is lossy and irreversible. + +To decompress: + +1. Split the 8-bit value: upper 4 bits are ``index``, lower 4 bits are + ``multi``. + + .. code-block:: text + + Bits: | 7 6 5 4 | 3 2 1 0 | + index multi + +2. Look up ``base[index]`` and ``step_size[index]``. +3. ``N = base[index] + multi * step_size[index] + (step_size[index] - 1) / 2``, + keeping only the **integer part** of the last term. + +The last term places the result in the middle of the compressed bin rather than +at its lower edge. + +.. list-table:: + :header-rows: 1 + :widths: 16 30 30 24 + + * - index + - base + - step_size + - covers + * - 0 + - 0 + - 1 + - 0-15 + * - 1 + - 16 + - 1 + - 16-31 + * - 2 + - 32 + - 2 + - 32-63 + * - 3 + - 64 + - 4 + - 64-127 + * - 4 + - 128 + - 8 + - 128-255 + * - 5 + - 256 + - 16 + - 256-511 + * - 6 + - 512 + - 16 + - 512-767 + * - 7 + - 768 + - 16 + - 768-1023 + * - 8 + - 1024 + - 32 + - 1024-1535 + * - 9 + - 1536 + - 32 + - 1536-2047 + * - 10 + - 2048 + - 64 + - 2048-3071 + * - 11 + - 3072 + - 128 + - 3072-5119 + * - 12 + - 5120 + - 256 + - 5120-9215 + * - 13 + - 9216 + - 512 + - 9216-17407 + * - 14 + - 17408 + - 1024 + - 17408-33791 + * - 15 + - 33792 + - 2048 + - 33792-66559 + +Worked example from the document: the byte ``230`` = ``0xE6`` gives +``index = 14``, ``multi = 6``, so +``17408 + 6*1024 + (1024-1)//2 = 17408 + 6144 + 511 = 24063``. The true value +was somewhere in 23552-24575. + +**[CODE]** ``swe_science.decompressed_counts(cem_count)`` implements exactly +this with the table as a dict. ``swe_science()`` precomputes all 256 results +once and then decompresses the whole array by fancy indexing: + +.. code-block:: python + + decompression_table = np.array([decompressed_counts(i) for i in range(256)]) + science_array = decompression_table[raw_science_array] + +**[DOC]** The document says decompression "could be done here, or could be part +of L1a to L1b processing". **[CODE]** SWE does it at L1A, and keeps the raw +8-bit values in ``raw_science_data`` so the choice is reversible. That variable +is deleted at the top of L1B. + +The same function is imported by the I-ALiRT code, which uses the identical +scheme (**[DOC]** section 3.4.2.2). + +Step 4 - engineering units +-------------------------- + +**[CODE]** ``swe_l1b_science()`` reads the APID back out of the L1A global +attribute ``packet_apid``, looks up the matching ``SWEAPID`` name, and calls +``convert_raw_to_eu()`` with the ``eu-conversion`` ancillary file. For science +packets that converts ``SPIN_PHASE``, ``SPIN_PERIOD`` and ``THRESHOLD_DAC`` +(**[DOC]** Appendix A). The full table is in :ref:`swe-ancillary`. + +Step 5 - dropping calibration data +---------------------------------- + +**[DOC]** Section 3.3.4: in-flight calibration runs use the **same telemetry +format** as science mode, so L0 and L1 processing is identical, but the data +"should not be included in the Level 2 or higher science products". + +**[CODE]** L1B drops them, not L2: + +.. code-block:: python + + science_data = l1a_data_copy["esa_table_num"].data == 0 + l1a_data_copy = l1a_data_copy.isel({"epoch": science_data}) + +The comment explains the reasoning: calibration-mode data "looks same as +science data but it only measures one energy or specific energy steps during +the whole duration. Right now, only index 0 in LUT collects science data." + +This is confirmed by the ESA LUT itself: table 0 cycles through all 24 +voltages, while table 1 holds a single voltage (5.64 V) for all 48 rows - +exactly what a gain sweep needs. See :ref:`swe-ancillary`. + +Step 6 - finding complete full cycles +------------------------------------- + +A full cycle needs four consecutive packets with ``QUARTER_CYCLE`` = 0, 1, 2, 3. +Partial cycles at the ends of a file, or gaps, must be discarded. + +**[CODE]** ``find_cycle_starts()`` (credited to Brandon Stone) does this with a +sliding window on the first difference, avoiding any Python loop: + +.. code-block:: python + + diff = cycles[1:] - cycles[:-1] + ione = diff == 1 + valid = (cycles == 0)[:-3] & ione[:-2] & ione[1:-1] & ione[2:] + first_quarter_indices = np.where(valid)[0] + +i.e. "this element is 0, and the next three differences are all +1". +``get_indices_of_full_cycles()` then broadcasts each start index to +``[i, i+1, i+2, i+3]`` and flattens. + +If no full cycle is found, ``swe_l1b_science()`` logs and returns ``None``. + +.. _swe-l1-checkerboard: + +Step 7 - the checkerboard reorganization +---------------------------------------- + +This is the heart of SWE L1B, and it has no direct counterpart in the heritage +codes - **[DOC]** section 3.4.1 says explicitly that "the SWEPAM code used to +combine data from different energies and angles to build up a full measurement +will need to be modified for use with SWE data". + +**Goal.** Turn four packets' worth of measurements - +``(4, 180, 7)`` = 720 measurements - into a ``(24 esa_step, 30 spin_sector, 7)`` +grid, exactly one measurement per cell. + +**Inputs.** The ESA LUT ancillary file, filtered to the active table index +(``esa_table_num``, default 0). Its columns: + +.. list-table:: + :header-rows: 1 + :widths: 16 84 + + * - Column + - Meaning + * - ``table_idx`` + - Which of the 8 onboard tables (0-7). 48 rows each. + * - ``esa_step`` + - Step index within the full cycle. Takes values 0-11, 180-191, 360-371, + 540-551 - i.e. the **first 12 steps of each quarter cycle**, since the + pattern then repeats every 12 steps through the 180. + * - ``esa_v`` + - The ESA plate voltage for that step. + * - ``v_index`` + - Which of the 24 energy rows (1-24) that voltage belongs to. + * - ``ialirt`` + - 1 if this step is one of the 8 downlinked in real time. + +**[CODE]** ``get_checker_board_pattern()`` builds the index map: + +1. Take the 48 rows of the active table, reshape ``v_index - 1`` and + ``esa_step`` to ``(4 quarter cycles, 12 steps)``. +2. Within each quarter cycle, split the 12 steps into ``(2, 6)``: the first 6 + belong to **even** spin-sector columns, the second 6 to **odd** columns. +3. Write into a ``(24, 2)`` scratch array: for quarter cycle *i* whose first + step index is ``start_esa_step`` (0, 180, 360 or 540), the even column gets + ``start_esa_step + 0..5`` at the six ``v_index`` rows for that half, and the + odd column gets ``start_esa_step + 6..11`` at its six rows. +4. ``np.tile`` that ``(24, 2)`` block 15 times to ``(24, 30)``. +5. Add the column offsets ``[0, 0, 12, 12, 24, 24, ...]`` so that each + successive pair of columns advances by 12 steps - 12 steps is exactly one + second, i.e. two spin-angle bins. + +The result is a ``(24, 30)`` array of indices into the flattened 720-element +full-cycle array. + +``get_esa_energy_pattern()`` runs the identical construction but writes +``esa_v`` instead of a step index, giving the (24, 30) grid of ESA **voltages** +that L1B converts to energies. + +``populated_data_in_checkerboard_pattern()`` applies the map. It flattens the +pattern in Fortran order (``order="F"``, column-major) and uses it to index six +variables: + +.. list-table:: + :header-rows: 1 + :widths: 30 70 + + * - Variable + - Handling + * - ``science_data`` + - Reshaped ``(n_cycles, 720, 7)``, indexed, reshaped ``(n, 24, 30, 7)``. + * - ``acq_start_coarse``, ``acq_start_fine``, ``acq_duration``, + ``settle_duration`` + - One value per packet, so repeated 180 times to length 720 before + indexing. Result ``(n, 24, 30)``. + * - ``esa_step_number`` + - **Synthesized, not telemetered.** ``tile(arange(180), 4)`` per cycle, + put through the same permutation. Result: the 0-179 position of each + measurement within its quarter cycle, which is what the acquisition + time formula needs. + +.. important:: + + The checkerboard is a **pure permutation**. Nothing is averaged, summed or + dropped, and every one of the 720 measurements lands in exactly one cell. + That is what lets L1B keep per-measurement acquisition times, and lets L2 + emit both a binned and an unbinned product. + +Step 8 - acquisition time per measurement +----------------------------------------- + +**[DOC]** Section 3.4.4. ``ACQ_START_COARSE`` and ``ACQ_START_FINE`` give the +start of the **first data acquisition period** of the quarter cycle, *after* +the first settle time. Subsequent steps follow at the step period: + +.. code-block:: text + + sci_step_acquisition_time_sec = + ACQ_START_COARSE + ACQ_START_FINE/1e6 + + step * (ACQ_DURATION + SETTLE_DURATION)/1e6 + + sci_step_acquisition_middle_time_sec = + sci_step_acquisition_time_sec + (ACQ_DURATION/1e6)/2 + +with ``step`` running 0-179 within the quarter cycle. Both durations are in +**microseconds** in the L1 files. + +**[CODE]** Split across two helpers in ``swe_utils.py``: + +* ``combine_acquisition_time(coarse, fine)`` -> ``coarse + fine/1e6`` +* ``calculate_data_acquisition_time(start, step, acq_duration, settle_duration)`` + -> the center-time formula above. + +The result becomes the L1B ``acquisition_time`` variable, ``(epoch, 24, 30)``, +in MET seconds. It is the **center** time, which matters: L2 uses it to look up +the spin angle, and getting the center rather than the edge is what makes the +spin angle correct to half a bin. + +Step 9 - deadtime correction +---------------------------- + +**[DOC]** Non-paralyzable detector model, ``n = m / (1 - m*t)``. The heritage C: + +.. code-block:: c + + correct = 1.0 - ((1.5e-6) * c[i][j][k] / tsampl); + /* arbitrary x10 cutoff */ + correct = (correct < 0.1) ? 0.1 : correct; + ccounts[i][j][k] = c[i][j][k] / correct; + +The 0.1 floor caps the correction at a factor of 10, so a saturated detector +produces a large but finite number rather than a negative or infinite one. + +The document's fragment uses **1.5e-6 s**, but states in the same paragraph +that "the dead time for SWE will be based on the SWE electronics, with the +actual value defined during instrument calibration", and the C comment reads +``dead time = XXXXX sec``. + +**[CODE]** ``deadtime_correction(counts, acq_duration)`` uses **360 ns**: + +.. code-block:: python + + deadtime = 360e-9 + correct = 1.0 - (deadtime * (counts / (acq_duration[..., np.newaxis] * 1e-6))) + correct = np.maximum(0.1, correct) + corrected_count = counts.astype(np.float64) / correct + +This is a deliberate substitution of the measured SWE value for the document's +placeholder, and the 0.1 floor is preserved. ``acq_duration`` is converted from +microseconds to seconds inline. The same function is reused by the I-ALiRT +code. + +Step 10 - in-flight gain calibration +------------------------------------ + +**[DOC]** Sections 3.3.4 and 3.4.3. Weekly, SWE runs a macro that fixes the ESA +voltage and steps the CEM bias to ``Vnom - 200``, ``Vnom - 100``, ``Vnom``, +``Vnom + 100``, ``Vnom + 200`` V. On the ground, someone compares the counts at +the operating level (step 2 of the run) to the counts at the next higher level +(step 3), assumes the higher level is correct, and uses the ratio as a scale +factor. That yields **seven multiplicative factors**, one per CEM, per +calibration run. + +The factors arrive as +``imap_swe_l1b-in-flight-cal_YYYYMMDD_vXXX.csv``, one row per calibration point: +a MET timestamp followed by seven factors. All factors are 1.0 for times before +commissioning. + +The **filename date convention matters**: when a new calibration measurement is +added, the file's date is set to the date of the **previous** calibration, +because only then does linear interpolation become possible back to that +previous point. Files also carry a trailing row with a far-future timestamp +repeating the most recent factors, so quicklook processing can proceed before +the next calibration exists - and so that reprocessing is correctly triggered +once it does. + +Interpolation is **linear in time between the two surrounding calibration +points**, per measurement. The heritage ``electron_cal()`` exits with an error +rather than extrapolating past the last point. + +**[CODE]** Three functions: + +* ``read_in_flight_cal_data(files)`` - concatenates the CSVs, drops rows with + no MET, sorts by MET, drops duplicate METs keeping the last. +* ``calculate_calibration_factor(acquisition_times, cal_times, cal_data)`` - + **raises ``ValueError`` if any acquisition time falls outside the calibration + time range**, matching the heritage refusal to extrapolate. Then + ``np.searchsorted`` to find the bracketing pair and a manual linear + interpolation, vectorized over the whole ``(n, 24, 30)`` time array, giving + ``(n, 24, 30, 7)``. +* ``apply_in_flight_calibration(...)`` - multiplies, and sets a quality flag. + +**[CODE, beyond the document]** ``SweL1bFlags.LAST_CAL_INTERVAL`` (bit 2) is +set on any epoch where at least one acquisition time falls after the +second-to-last calibration entry. That is precisely the region covered by the +far-future padding row, so the flag marks "these factors are effectively the +last measured ones held constant, not a true interpolation". This is a sensible +addition that the document does not ask for. + +Step 11 - counts to rate +------------------------ + +**[DOC]** "counts can be converted from counts to count rate, simply by +dividing the corrected counts by the accumulation time for each measurement". +The accumulation time is ``ACQ_DURATION`` from the science packet +(nominally 80 ms), settable in flight, expected to change only in response to +spin rate changes. + +**[CODE]** ``convert_counts_to_rate(data, acq_duration)`` - divide by +``acq_duration * 1e-6``, broadcasting a new trailing axis if needed. This runs +**after** the gain calibration, so the output ``science_data`` is +deadtime-corrected, gain-calibrated counts per second. + +Step 12 - uncertainty and energies +---------------------------------- + +**[CODE]** ``counts_stat_uncert = np.sqrt(populated_data["science_data"])`` - +Poisson uncertainty on the **decompressed, uncorrected** counts, with a +``TODO`` noting that SWE may want deadtime correction included. This is not in +the algorithm document at all; the document never discusses uncertainty. + +Two consequences to be aware of, both covered in +:ref:`swe-implementation-status`: + +* The uncertainty is in **counts** while ``science_data`` is in **counts/s**, + and L2 applies the same phase-space-density conversion to both. +* The compression quantization (which for large counts dominates Poisson - a + count of 24063 is known only to within ±512) is not included. + +**[CODE]** ESA energies are built from the LUT and the analyzer constant: + +.. code-block:: python + + esa_energies = get_esa_energy_pattern(esa_lut_files[0]) # (24, 30) volts + esa_energies = np.repeat(esa_energies[np.newaxis, :, :], n_cycles // 4, axis=0) + esa_energies = esa_energies * swe_constants.ENERGY_CONVERSION_FACTOR + +and stored as ``esa_energy`` in eV. See :ref:`swe-l2` for what the conversion +factor is and :ref:`swe-ancillary` for why it is hard-coded. + +Housekeeping and CEM raw +------------------------ + +**[DOC]** Section 3.3.2 lists what is in ``SWE_APP_HK``: operation mode, CDH +monitors, LV monitors, HV monitors, HV settings (enable/disable, limit plug +status, range), temperature sensors, FEE thresholds, stim pulse status. +Processing is "decommutation of the CCSDS packets, expansion of each +housekeeping parameter to the appropriate length in bytes, and conversion from +raw units to engineering units." + +**[CODE]** There is no HK algorithm. The same packet is decommutated twice from +the same L0 file: + +* ``imap_swe_l1a_hk`` - ``swe_l1a()``, ``use_derived_value=False``, raw counts. +* ``imap_swe_l1b_hk`` - ``swe_l1b()``, ``use_derived_value=True``, XTCE-derived + engineering units. + +Note the L1B HK product reads the **L0 file**, not the L1A HK CDF. String-valued +HK variables get a separate attribute set (``l1b_hk_string_attrs``). + +**[DOC]** Section 3.4.2.4: ``SWE_CEM_RAW`` starts at byte 14, bit 0 with seven +consecutive 4-byte fields, CEM1 through CEM7, uncompressed. **[CODE]** The XTCE +actually defines **fourteen** count fields - ``CEM1..7_COUNTS_LATCHED`` and +``CEM1..7_COUNTS_LIVE`` - plus ``SHCOARSE``, ``THRESHOLD_DAC``, +``STIM_CFG_REG`` and ``CKSUM``. The XTCE is the authority; Appendix B warns that +the packet definitions were still moving. + +Tests +----- + +``imap_processing/tests/swe/``: + +.. list-table:: + :header-rows: 1 + :widths: 36 64 + + * - File + - Covers + * - ``test_swe_l1a_science.py`` + - Packet count, ``decompressed_counts`` against the document's worked + example, raw and derived field values against + ``idle_export_*.SWE_SCIENCE_*.csv``, data ordering, and full + decompression against the raw/EU validation pair. + * - ``test_swe_l1a_hk.py``, ``test_swe_l1a_cem_raw.py`` + - Decommutation against ``idle_export_eu`` CSVs. + * - ``test_swe_l1a.py`` + - CDF creation for all three L1A products. + * - ``test_swe_l1b.py`` + - Full cycle index finding, calibration interpolation, the + ``LAST_CAL_INTERVAL`` flag, the checkerboard pattern against + ``checker-board-indices.csv``, the end-to-end L1B, and count rates. + * - ``test_swe_l2.py`` + - Phase space density, flux, angle bin lookup, binning, and end-to-end L2 + at both 15 s and 14.6 s spin periods. + +Test L0 data are real GSEOS exports from 2024-05-10 (``l0_data/``), with +matching validation CSVs in ``l0_validation_data/`` and ancillary files in +``lut/``. diff --git a/docs/source/algorithm-code-documentation/swe/l2.rst b/docs/source/algorithm-code-documentation/swe/l2.rst new file mode 100644 index 0000000000..cb403bb3b2 --- /dev/null +++ b/docs/source/algorithm-code-documentation/swe/l2.rst @@ -0,0 +1,388 @@ +.. _swe-l2: + +L1B to L2 - Phase Space Density, Flux and Spin Angle +==================================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +This page covers algorithm document section 3.4.4. + +**[DOC]** "SWE Level 1b to Level 2 processing consists of converting the counts +values to physical phase space distribution or intensity units, and organizing +the data by energy step, polar angle, and spin phase." + +Four things happen. Two are trivial lookups, two are real. + +.. code-block:: text + + count rate (n, 24, 30, 7) + | + | 1. energy from ESA voltage (done at L1B, using the LUT + k) + | 2. polar angle from CEM index (a constant array) + | 3. phase space density, then flux + | 4. spin phase from SPICE -> 12-degree spin angle bins + v + imap_swe_l2_sci + +Step 1 - particle energy +------------------------ + +**[DOC]** "The particle energy measured at each step is determined from the +voltage applied to the electrostatic analyzer (ESA) plates. The ratio of +measured energy to applied voltage is referred to as the **analyzer constant** +``k = E/V``, and is determined from ground calibration. This value will be +stored in a calibration data file, where it can be read by the processing +code." + +**[CODE]** ``swe_constants.ENERGY_CONVERSION_FACTOR = 4.75``, hard-coded, with +a ``TODO: add these to instrument status summary`` above it. The multiplication +happens at **L1B** (``esa_energy``), not L2, so L2 receives energies already in +eV. L2 also builds its own ``energy`` coordinate independently from +``ESA_VOLTAGE_ROW_INDEX_DICT`` times the same constant. + +With k = 4.75 the 24 ESA voltages 0.56 V - 1108.66 V map to +**2.66 eV - 5266 eV**, consistent with the document's stated 1 eV - 5 keV +range. + +.. list-table:: + :header-rows: 1 + :widths: 12 22 22 12 22 22 + + * - row + - ESA V + - E (eV) + - row + - ESA V + - E (eV) + * - 0 + - 0.56 + - 2.66 + - 12 + - 29.39 + - 139.6 + * - 1 + - 0.78 + - 3.71 + - 13 + - 40.88 + - 194.2 + * - 2 + - 1.08 + - 5.13 + - 14 + - 56.87 + - 270.1 + * - 3 + - 1.51 + - 7.17 + - 15 + - 79.10 + - 375.7 + * - 4 + - 2.10 + - 9.98 + - 16 + - 110.03 + - 522.6 + * - 5 + - 2.92 + - 13.87 + - 17 + - 153.05 + - 726.9 + * - 6 + - 4.06 + - 19.29 + - 18 + - 212.89 + - 1011.2 + * - 7 + - 5.64 + - 26.79 + - 19 + - 296.14 + - 1406.7 + * - 8 + - 7.85 + - 37.29 + - 20 + - 411.93 + - 1956.7 + * - 9 + - 10.92 + - 51.87 + - 21 + - 572.99 + - 2721.7 + * - 10 + - 15.19 + - 72.15 + - 22 + - 797.03 + - 3785.9 + * - 11 + - 21.13 + - 100.4 + - 23 + - 1108.66 + - 5266.1 + +Rows 11-18 (100.4 - 1011.2 eV) are the eight energies downlinked in real time; +see :ref:`swe-ialirt`. + +Step 2 - polar angle +-------------------- + +**[DOC]** "As each CEM detector is oriented at a given angle relative to the +spacecraft spin axis, the polar angle is the same for all measurements by a +given detector. The central detector points outward perpendicular to the +spacecraft spin axis. The other detectors point at nominally +/-21, +/-42, and ++/-63 degrees relative to this detector." + +**[CODE]** ``swe_constants.CEM_DETECTORS_ANGLE = [-63, -42, -21, 0, 21, 42, 63]``, +exposed as the L2 ``inst_el`` coordinate. There is no computation. + +.. _swe-l2-psd: + +Step 3 - phase space density +---------------------------- + +**[DOC]** Heritage ``fspace()``: + +.. code-block:: text + + fv(v, theta, phi) = 2 * C(E, theta, phi) / (G * v^4 * tau) + + C = corrected counts + E = electron energy, eV + v = electron speed from energy, cm/s + tau = sampling time, s + G = geometric factor, cm^2 sr + fv = phase space density, s^3 / cm^6 + +with ``v^4`` evaluated as ``1.237e31 * E^2`` for E in eV. The heritage C guards +``ccounts < 0`` by writing ``fv = 0``. + +The **geometric factors** are constant in time (time variation is handled by the +L1B gain calibration). **[DOC]** "The geometric factors below are the nominal +SWE geometric factors for the 7 CEM detectors (1-7) as of October 2025, and may +be refined by further analysis": + +.. code-block:: text + + g = {424.4e-6, 564.5e-6, 763.8e-6, 916.9e-6, 792.0e-6, 667.7e-6, 425.2e-6} + +The document notes the factors "are hard-coded into the routine; geometric +factors can also be stored in a calibration data file to be read by the +processing routine." + +.. note:: + + The C fragment in the document is **stale**: the array it actually assigns + (``gg[0] = 255.6e-6`` ...) is the **ACE/SWEPAM** set, clearly labelled + ``/* Ace geometric factors */``, while the SWE values sit above it in a + comment. Use the SWE set. + +**[CODE]** ``swe_constants.GEOMETRIC_FACTORS`` holds the SWE October 2025 +values, hard-coded, under the same ``TODO: add these to instrument status +summary``. ``swe_l2.calculate_phase_space_density()``: + +.. code-block:: python + + phase_space_density = (2 * data) / ( + GEOMETRIC_FACTORS[np.newaxis, np.newaxis, np.newaxis, :] + * VELOCITY_CONVERSION_FACTOR # 1.237e31 + * particle_energy_data[:, :, :, np.newaxis] ** 2 + ) + +``data`` is the L1B ``science_data``, which is **already** ``C/tau``, so the +``tau`` in the document's formula is absorbed and does not appear. + +The ``1.237e31`` factor, derived in the function's docstring: + +.. code-block:: text + + v = sqrt(2E/m) with E = eV * 1.60219e-19 J/eV, m = 9.10938356e-31 kg + = sqrt(3.5176e16 * eV) cm/s + v^4 = 1.237e31 * eV^2 + +**[CODE]** The heritage ``ccounts < 0`` guard is **not** implemented; negative +inputs would propagate. In practice counts are unsigned and the deadtime floor +keeps the correction positive, so this has not bitten anyone. + +.. warning:: + + **The heritage ``fspace()`` ends with a second loop that is not implemented + here**, halving the phase space density of the two **end** detectors (CEM 1 + and CEM 7, indices 0 and 6) at every azimuth - and, for even azimuth + indices, taking the value from the neighbouring odd index first. This is a + real factor-of-2 correction on the outermost detectors. See + :ref:`swe-implementation-status`. + +Step 4 - number flux +-------------------- + +**[DOC]** "In the heritage code, electron distributions are always saved in +units of phase space distribution rather than intensity", but the conversion is +given: + +.. code-block:: text + + j = C / (G * E * tau) # differential number flux + j(v, theta, phi) = fv * v^4 / (2 * E) + = 6.187e30 * fv * E # E in eV + +Units of ``j``: 1 / (cm^2 s eV ster). + +**[CODE]** ``swe_l2.calculate_flux()``: + +.. code-block:: python + + flux = FLUX_CONVERSION_FACTOR * esa_energy[:, :, :, np.newaxis] * phase_space_density + +with ``FLUX_CONVERSION_FACTOR = 6.187e30``. The docstring records that Ruth +Skoug confirmed both this factor and the 1.237e31 above. + +.. _swe-l2-spin: + +Step 5 - spin phase and angle binning +------------------------------------- + +**[DOC]** "In Level 1 SWE data products, the data are organized by the spin +sector of each measurement... As part of Level 2 processing, the spin sectors +are converted to spin angles of the spacecraft by combining SWE timing +information with knowledge of the spacecraft spin from spacecraft ephemeris +data." + +Because SWE is not synced to the spin, **the same spin sector index views a +different physical angle in each quarter cycle.** Combining four quarter cycles +therefore requires binning by actual angle: + +**[DOC]** "For Level 2, these bins will be defined to be **12 degrees wide, in +despun spacecraft coordinates**. Binning will be determined by the **central +spin phase angle** of each spin bin." + +**[CODE]** ``swe_l2()``: + +.. code-block:: python + + inst_spin_phase = get_instrument_spin_phase( + query_met_times=l1b_dataset["acquisition_time"].data.ravel(), + instrument=SpiceFrame.IMAP_SWE, + ) + inst_spin_angle = get_spin_angle(inst_spin_phase, degrees=True).reshape( + -1, N_ESA_STEPS, N_ANGLE_SECTORS + ) + +Note that the query times are the **center** acquisition times computed at L1B, +which is what makes "the central spin phase angle of each spin bin" work out. + +The angles are saved as ``inst_az_spin_sector`` before binning, because L3 +needs the per-measurement angle, not the bin. + +**[CODE]** ``find_angle_bin_indices()`` does the binning. Bin centers are +``6, 18, 30, ... 354``; edges are ``np.arange(0, 360, 12)``. The lookup is + +.. code-block:: python + + spin_angle_bins_indices = np.searchsorted(edges, inst_spin_angle, side="right") - 1 + +so an angle in ``[0, 12)`` lands in bin 0 (center 6), ``[12, 24)`` in bin 1, and +so on. Angles outside ``[0, 360)`` raise ``ValueError``. + +**[CODE]** ``put_data_into_angle_bins()`` accumulates with ``np.add.at`` and +divides by a parallel count array, i.e. it takes the **mean** of whatever +measurements fall in a bin. Empty bins are set to ``NaN`` (via dividing by +``NaN``) rather than 0, deliberately, "because zero physical counts could be +valid data". + +.. note:: + + In the nominal case each ``(energy, bin)`` cell receives exactly one + measurement, so the mean is a no-op. It stops being a no-op when the spin + period drifts relative to the quarter cycle - which is exactly why the test + suite runs L2 at both a 15 s and a 14.6 s spin period. + +**[CODE]** ``put_uncertainty_into_angle_bins()`` is the same accumulation in +quadrature: it sums ``data**2`` into the bins and returns ``sqrt``. The +docstring records the SWE instruction verbatim, and notes that SWE intends to +replace this with a formula based on spin data in future. + +Reference frame +^^^^^^^^^^^^^^^ + +**[DOC]** asks for **despun spacecraft (DSC)** coordinates, twice - once in +3.4.4 for the spin phase and once in 3.4.6, where L3 needs SWE and MAG in a +common frame. + +**[CODE]** produces the **instrument** spin angle +(``SpiceFrame.IMAP_SWE``, variables named ``inst_az`` / ``inst_el``). The two +differ by a fixed instrument-to-spacecraft mounting offset, which +``get_instrument_spin_phase`` applies as +``(spacecraft_spin_phase + instrument_spin_offset) % 1``. Flagged in +:ref:`swe-implementation-status`; confirm with the SWE team which frame the L2 +deliverable is expected to be in before changing anything. + +What L2 emits +------------- + +**[DOC]** "The primary Level 2 data product is a matrix every full SWE cycle +(nominally 1 minute) of electron phase space distribution with size +**(7 CEMs x 24 energy steps x 30 spin phase bins)**, and the same matrix with +units of electron intensity." + +"In addition, to facilitate Level 2 to Level 3 processing, the SWE Level 2 data +set will also include phase space distribution and intensity matrices for the +**original measurements** (7 CEM detectors x 180 energy/spin phase bins per +quarter-cycle), together with the **time stamps and spin phase angles** for each +individual measurement. These data will allow calculation of the electron pitch +angle for each individual measurement, leading to the most accurate pitch angle +distributions." + +**[CODE]** Both are present: + +.. list-table:: + :header-rows: 1 + :widths: 36 28 36 + + * - Variable + - Shape + - Document counterpart + * - ``phase_space_density`` + - (n, 24, 30, 7) + - The primary binned product. + * - ``flux`` + - (n, 24, 30, 7) + - The intensity version of the same. + * - ``phase_space_density_spin_sector`` + - (n, 24, 30, 7) + - The "original measurements" matrix. + * - ``flux_spin_sector`` + - (n, 24, 30, 7) + - Ditto. + * - ``inst_az_spin_sector`` + - (n, 24, 30) + - The per-measurement spin phase angles. + * - ``acquisition_time`` + - (n, 24, 30) + - The per-measurement time stamps. + +The document's "180 energy/spin phase bins per quarter-cycle" and the code's +``(24, 30)`` per full cycle are the **same 720 numbers** - the checkerboard is a +permutation, so preserving it full-cycle preserves every original measurement. +The code's layout is strictly more convenient for L3 because energy and spin +sector are already separate axes. + +Deviations from the document at L2 +---------------------------------- + +Summarized here, detailed in :ref:`swe-implementation-status`: + +#. The **end-detector halving** in heritage ``fspace()`` is not implemented. +#. Spin angles are in the **instrument** frame, not despun spacecraft. +#. The **analyzer constant** and **geometric factors** are hard-coded rather + than read from a calibration file, as the document says they may be. +#. ``flux_stat_uncert`` is **binned twice**, and both uncertainty variables + carry ``esa_step``/``spin_sector`` dimension names while holding binned + data. +#. The heritage ``ccounts < 0`` guard is absent. diff --git a/docs/source/algorithm-code-documentation/swe/l3-scope.rst b/docs/source/algorithm-code-documentation/swe/l3-scope.rst new file mode 100644 index 0000000000..10ec6df293 --- /dev/null +++ b/docs/source/algorithm-code-documentation/swe/l3-scope.rst @@ -0,0 +1,198 @@ +.. _swe-l3-scope: + +L3 Scope - What Is Not in This Repository +========================================= + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +.. warning:: + + **SWE L3 is not produced here.** ``imap-processing`` takes SWE to L2. A + separate repository, run closer to the science team, takes L2 to L3. + + If a task sounds like fitting Maxwellians, finding a spacecraft potential, + computing a pitch angle, or integrating a moment, **it does not belong in + this repository.** Stop and check before writing code. + +Section 3.4.6 of the algorithm document is four of its 38 pages and is by far +the most algorithmically dense part of the whole document. It is summarized +here for one reason only: **so you can tell what L2 owes L3**, and recognise an +L3 task when one arrives misfiled. + +The contract - what L3 needs from L2 +------------------------------------ + +**[DOC]** "Level 3 processing starts from the SWE Level 2 data files. In +particular, Level 3 processing will use Level 2 variables which contain the +electron phase space distributions as a function of energy, polar angle (CEM) +and spin angle (SWE azimuthal angle bin) **in despun spacecraft coordinates for +each individual nominal 80 millisecond measurement** (7 CEM detectors x 24 ESA +voltage x 30 spin angle bins per SWE full cycle). The Level 2 files also include +the energy corresponding to each ESA voltage step, and the polar angle +corresponding to each CEM detector." + +The rationale is explicit: "For pitch angle calculations, starting from this +full data set will allow calculation of pitch angle using the magnetic field +vector measurement from MAG **during each nominal 80 millisecond SWE data +acquisition period**." Pitch angle is computed per measurement, not per bin - +so L2 must not throw away per-measurement resolution. + +.. list-table:: + :header-rows: 1 + :widths: 34 18 48 + + * - What L3 needs + - L2 variable + - State + * - Per-measurement phase space density + - ``phase_space_density_spin_sector`` + - Present, ``(epoch, 24, 30, 7)``. + * - Per-measurement intensity + - ``flux_spin_sector`` + - Present. + * - Energy per ESA step + - ``energy`` coordinate; ``esa_energy`` at L1B + - Present. + * - Polar angle per CEM + - ``inst_el`` coordinate + - Present. + * - Per-measurement spin angle + - ``inst_az_spin_sector`` + - Present, **but in the instrument frame, not despun spacecraft**. + * - Per-measurement time stamp + - ``acquisition_time`` + - Present, center of the accumulation window, MET seconds. + * - Accumulation duration + - ``acq_duration`` + - Present, microseconds. + * - Data quality + - ``data_quality`` + - Present, one ``SweL1bFlags`` byte per full cycle. + +.. important:: + + The single open question on the L2-to-L3 contract is the **reference + frame**. The document asks for despun spacecraft coordinates at L2 so that + SWE and MAG can be combined without further rotation. The code produces + instrument-frame angles. The difference is a fixed mounting offset that + ``get_instrument_spin_phase`` already knows about, so this is a small change + - but it is a change to a deliverable's definition and needs SWE team + agreement, not a unilateral patch. See :ref:`swe-implementation-status`. + +What L3 does +------------ + +Three stages. All of them need external inputs that this repository never +loads. + +Stage 1 - spacecraft potential and the core/halo break +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** The spacecraft charges, which shifts the measured electron energies +and makes the low-energy end unreliable. The first L3 step is to find the +potential by **fitting two Maxwellians twice** to the 1D (angle-averaged) phase +space distribution versus energy: + +* **below 30 eV** -> the spacecraft potential; +* **30 - 400 eV** -> the break between the thermal (**core**) and suprathermal + (**halo**) populations. + +Steps listed: compute fit weights; build the ``log(fv)`` vs ``E`` spectrum; +perform the fits on distributions **averaged over 7 full SWE cycles (~7 minute +running average)**; if no potential break is seen (potential below the minimum +SWE energy), **set the potential to 2.5 V**; refine using the bins adjacent to +the fitted break; if either fit fails, **fall back to a smoothed-spline maximum +curvature** method. Then correct the energies using the potential. + +**[DOC]** notes the heritage ACE/SWEPAM code instead did a nonlinear +least-squares fit of **three** Maxwellians to find both break points, and that +the SWE break-point finder is tuned for Ultra deflectors at 3500 V, may fail +when they are off (a flag marks those times), and may fail when the potential +drops below SWE's ~3 eV floor. + +Stage 2 - pitch angle and gyrophase +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +**[DOC]** Requires the spacecraft potential, the **MAG** field vector, and the +**SWAPI** solar wind velocity, all in despun spacecraft coordinates. Steps: + +* Correct energies for spacecraft potential. +* Compute velocity vector and energy in the **solar wind frame** per + measurement. +* Compute pitch angle and gyrophase per measurement. +* Bin: **20 pitch angle bins of 9 degrees**, at the 24 nominal energies; and - + not in the heritage code - **24 gyrophase bins of 15 degrees**. +* Fit the distribution versus energy in each pitch angle bin (and each + pitch/gyrophase combination if gyrophase binning is used) and evaluate the + fit at the nominal energies. +* Integrate the pitch angle distribution to get the 1D energy spectrum, and + integrate the two halves (0-90 and 90-180 degrees) separately to define "in" + and "out" from the Sun, where "out" is the direction of maximum phase space + density. + +**[DOC]** flags a genuine statistics problem with gyrophase binning: a full SWE +measurement gives only ``7 x 30 = 210`` measurements per energy step, against +``20 x 24 = 240`` pitch/gyrophase bins. Accumulating 5-10 full cycles may be +needed. Deferred to a future data release. + +Stage 3 - moments +^^^^^^^^^^^^^^^^^ + +**[DOC]** Density, velocity, temperature and heat flux, computed **two ways** +(bi-Maxwellian fit and direct integration) and for **three populations** (core, +halo, total). + +The integrals are considered more accurate because they assume no spectral +form, but SWE cannot measure down to zero energy, so **the integral densities +are corrected using the fitted moments to fill in the low-energy end**. + +Every variant ends with: rotate V to RTN; compute the temperature tensor +eigenvalues and eigenvectors and pick the primary; rotate T to RTN; rotate T to +field-aligned coordinates using **B** from MAG to get ``Tpar`` and ``Tperp``. + +**[DOC]** makes an important physical caveat: because spacecraft charging makes +low-energy electrons hard to measure, SWE will **ultimately assume the bulk +electron velocity equals the proton velocity and the electron density equals +the ion density, both from SWAPI**, and use the SWE moments analysis only for +**electron temperature and heat flux**. + +External inputs L3 needs that this repository never loads +--------------------------------------------------------- + +.. list-table:: + :header-rows: 1 + :widths: 20 80 + + * - Source + - Used for + * - **MAG L2** + - Field vector per 80 ms measurement, in despun spacecraft coordinates. + Pitch angle, gyrophase, and the field-aligned temperature rotation. + * - **SWAPI L3** + - Ion velocity (solar wind frame transformation), density and temperature + (spacecraft potential context, and the substitution for electron bulk + moments). + * - **Ultra HK** + - Deflector voltage state, to flag times when the break-point finder is + unreliable. + * - **SWE** ``config`` + - An SDC ancillary file of L3 tuning constants: geometric fractions, the + pitch angle / gyrophase / energy bin definitions, the in-versus-out + energy index, the core/halo breakpoint initial guess and similar. + Delivered to the SDC under the ``config`` descriptor, but read only by + the L3 repository. + +If you are asked to add L3 here +------------------------------- + +Push back, and route it to the L3 repository. If the request survives that, +the things to settle first are: + +#. **Frame.** L3 needs despun spacecraft coordinates. Resolve the L2 frame + question before anything else. +#. **Cross-instrument dependencies.** This repository's design principle is + that each batch job is self-contained. L3 needs MAG and SWAPI products as + inputs - that is a real change to the dependency model, not just new code. +#. **The ~7 minute running average.** L2 records are 1 minute. Any L3 step that + averages 7 cycles needs data from neighbouring files, which crosses the + day-boundary filtering done at L1A. diff --git a/docs/source/algorithm-code-documentation/swe/overview.rst b/docs/source/algorithm-code-documentation/swe/overview.rst new file mode 100644 index 0000000000..906247e909 --- /dev/null +++ b/docs/source/algorithm-code-documentation/swe/overview.rst @@ -0,0 +1,308 @@ +.. _swe-overview: + +Instrument and Measurement Concepts +=================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +Everything on this page is background. No code depends on it directly, but +almost every design decision in :ref:`swe-l1` and :ref:`swe-l2` only makes +sense once you have it. + +What SWE measures +----------------- + +**[DOC]** SWE measures the **3D distribution of solar wind thermal and +suprathermal electrons from 1 eV to 5 keV**. It is a heritage design, closely +following NASA's Ulysses/SWOOPS, ACE/SWEPAM and Genesis/GEM solar wind electron +instruments. + +Electrons are the diagnostic of magnetic field topology in the solar wind. +Their pitch angle distribution tells you whether the local field line is +connected to the Sun at one end (unidirectional strahl), both ends +(counterstreaming - typically a coronal mass ejection), or disconnected. That +is why the two headline SWE science products are a **pitch angle distribution** +and a **bidirectional-electron flag**, and why SWE needs MAG data to be useful +at L3. + +The sensor +---------- + +**[DOC]** The Sensor Head (SH) is a **spherical-section electrostatic analyzer +(ESA)** followed by **seven channel electron multiplier (CEM) detectors**. + +* Electrons enter through an aperture oriented **normal to the spacecraft spin + axis**. +* A **positive high voltage on the inner ESA plate** admits only electrons in a + narrow band of energy and azimuthal angle. Stepping that voltage sweeps the + energy range. +* Electrons arriving at different **polar** angles land on different CEMs, + giving **21-degree polar resolution** across a fan-shaped field of view. +* As the spacecraft spins, the fan sweeps out **>95% of 4π steradians**, + missing only small conical holes centered on the spin axis (parallel and + antiparallel). + +The consequence for the code: **polar angle is a property of the detector +index, not something you compute.** The central detector looks radially +outward, perpendicular to the spin axis, and the others are at nominally +±21, ±42 and ±63 degrees from it. + +**[CODE]** ``swe_constants.CEM_DETECTORS_ANGLE = [-63, -42, -21, 0, 21, 42, 63]``, +which becomes the L2 ``inst_el`` coordinate. Note the ordering: index 0 is the +-63 degree detector and index 6 is +63. The algorithm document numbers the CEMs +1-7; the code indexes them 0-6 as ``cem_id``. + +Electronics worth knowing about +------------------------------- + +**[DOC]** SWE uses the IMAP Common Electronics (ICE) EBOX. Two things in it +matter for data processing: + +* **HVPS.** Two independent supplies: the ESA supply (+1200 V, **dual range**) + and the CEM bias supply (up to +4200 V, **commandable**, nominally +2800 V at + start of mission). The ESA supply is stepped every ~83.333 ms. + + The dual-range ESA supply is why ``HVPS_ESA_DAC`` has **two different raw-to- + engineering conversions** depending on whether the instrument is in low or + high range (see :ref:`swe-ancillary`). + + The commandable CEM bias is why the **in-flight gain calibration** exists: + as the CEMs age their gain drops, the weekly calibration sequence measures + how far off nominal they are, and the bias is occasionally stepped up. + +* **CDH.** Collects science data into memory, controls the HVPS, generates + telemetry packets, and runs the ESA stepping tables from MRAM. + +Spacecraft time +--------------- + +**[DOC]** SWE's FPGA has a **32-bit MET coarse counter** (whole seconds of +Mission Elapsed Time) and a **20-bit MET fine counter** (microseconds within +the second). The spacecraft delivers a time-and-status packet every second +carrying the SCLK of the next 1PPS; SWE loads it on the 1PPS edge and zeroes +the fine counter. If the packet does not arrive, the counters free-run. + +Both counters tag science telemetry as ``ACQ_START_COARSE`` (seconds) and +``ACQ_START_FINE`` (microseconds). The 1PPS also paces FSW operations such as +telemetry generation. + +Operating modes +--------------- + +**[DOC]** Six FSW modes, two in the boot FSW and four in the App FSW: + +.. list-table:: + :header-rows: 1 + :widths: 16 84 + + * - Mode + - What it is + * - Boot + - Boot FSW. Always transitions to LVENG next. + * - LVENG + - Low voltage engineering. LV monitors and engineering HK; memory + upload/dump/diagnostics allowed. **Safing from any mode lands here.** + * - LVSCI + - Low voltage science. End-to-end science data acquisition driven by a + **stim pulse** into the FEE, rather than by real electrons. + * - HVENG + - High voltage engineering. High voltages set to fixed levels by ground + command. Used heavily in ground test and during commissioning ramp-ups. + * - HVSCI + - **High voltage science. The only mode that produces science data.** ESA + voltages are stepped by FSW; CEM counters accumulate at each step. + Transitions out of HVSCI only happen after a complete 15-second quarter + cycle. + +Engineering modes (LVENG/HVENG) produce ``SWE_CEM_RAW`` packets: raw CEM counts +every 1 second, no compression, used for ground test and commissioning. + +.. note:: + + **[CODE]** Nothing in ``imap_processing/swe`` checks the operating mode. The + L1B filter that separates science from calibration data keys on + ``esa_table_num``, not on mode. See :ref:`swe-implementation-status`. + +The measurement cycle +--------------------- + +This is the part to get right. Everything about the SWE data layout follows +from it. + +**[DOC]** The ESA level is updated nominally every **83.333 ms**, which +corresponds to **2 degrees of spin** at a nominal 15-second spin period. The +83.333 ms is composed of: + +.. code-block:: text + + SETTLE_DURATION (nominally 3.333 ms) high voltage settling, no counting + ACQ_DURATION (nominally 80.000 ms) counters accumulate + ------------------------------------- + step period (nominally 83.333 ms) + +Both are telemetered per packet **in microseconds** and are commandable in +flight. + +**Spin-angle bin.** Six consecutive ESA steps take 0.5 seconds and cover +12 degrees of spin. Those six measurements are treated as belonging to one +**spin-angle bin**. A 15-second quarter cycle therefore contains 30 spin-angle +bins of 12 degrees each. + +**Quarter cycle.** One quarter cycle is: + +.. code-block:: text + + 15 seconds x 12 ESA steps per second = 180 measurements + 180 measurements x 7 CEMs = 1260 counter values + 1260 counter values, 8-bit compressed = 1260 bytes = the SCIENCE_DATA field + +Nominally one quarter cycle is one spacecraft spin, but **SWE is not synced to +the spin**. Measurements are time-based, paced off the 1PPS. The quarter cycle +length is configurable and will be set **slightly longer than a spin period** +so that every energy step gets full angular coverage. IMAP's spin rate is +planned to be 3.9-4.1 RPM, i.e. a period of **14.6-15.4 seconds**. + +.. note:: + + The algorithm document writes this rate as "3.9 - 4.1 Hz". That is a slip - + 4 Hz would be a 0.25 s period, not the 14.6-15.4 s stated in the same + sentence. Read it as RPM. + +**Full cycle.** A SWE measurement covers **24 energies × 30 spin angles**. Only +a subset of those energy-angle bins is measured in each quarter cycle: + +* In each quarter cycle, **6 of the 24 ESA levels** are measured at each spin + angle. +* **Odd-numbered spin-angle bins get one set of 6 levels; even-numbered bins + get a different set of 6.** +* Over four quarter cycles, all 24 ESA levels are measured at all 30 spin + angles. + +So the **full cycle** - four quarter cycles, nominally one minute - is the +smallest unit that is a complete measurement, and it is the L1B and L2 record. + +**[DOC]** The 720 ESA steps used in a full cycle are precomputed on board +before acquisition begins, from a 24-element ESA table stored in CDH MRAM. +**Eight such tables** can be stored; one is selected before acquisition, and +new tables can be uploaded. The processing here assumes the nominal scheme. + +.. _swe-checkerboard-concept: + +The checkerboard +---------------- + +The odd/even alternation above is why the SWE code talks about a +**"checkerboard pattern"**. Picture the (24 energies × 30 spin angles) grid you +want to end up with. In a single quarter cycle you fill only a scattered +subset of its cells; the pattern of filled cells alternates between adjacent +columns, and four quarter cycles' patterns interlock to fill the grid exactly +once. + +**[CODE]** ``swe_l1b.get_checker_board_pattern()`` builds a (24, 30) integer +array whose value at ``[energy_row, spin_column]`` is the index into the flat +720-element quarter-cycle-concatenated measurement array that belongs there. +``swe_l1b.populated_data_in_checkerboard_pattern()`` then applies it with fancy +indexing. The mapping is read from the **ESA LUT ancillary file**, not +hard-coded - see :ref:`swe-ancillary`. + +The important property: **the checkerboard is a lossless permutation.** All 720 +measurements of a full cycle land in the (24, 30) grid, exactly one per cell, +and nothing is averaged or dropped. That is why L1B can carry the original +per-measurement acquisition times alongside the counts. + +Heritage, and what it implies +----------------------------- + +**[DOC]** Section 3.4.1 is explicit that SWE processing is based on the +ACE/SWEPAM C codes (themselves descended from Ulysses/SWOOPS Fortran), because +SWEPAM is the only one of the three heritage instruments still operating. + +Two differences from heritage that matter: + +1. **Energy steps went from 20 to 24**, and the stepping scheme changed. +2. **Heritage instruments stepped contiguous energies per spin** (lowest *n* + energies in spin 1, next *n* in spin 2, ...). **SWE covers the full energy + range in every spin** at reduced energy and angle resolution. + +Consequence: the SWEPAM code that *combines* measurements into a full +distribution had to be rewritten for SWE (that is the checkerboard), but the +code that *processes* a complete distribution - deadtime, gain calibration, +phase space density, moments - is taken essentially verbatim from heritage. The +algorithm document reproduces those heritage C fragments directly, which is why +they appear in these pages as C rather than as equations. + +Vocabulary +---------- + +.. list-table:: + :header-rows: 1 + :widths: 26 74 + + * - Term + - Meaning + * - **CEM** + - Channel electron multiplier. Seven of them, at fixed polar angles. + ``cem_id`` 0-6 in code; CEM 1-7 in the document. + * - **ESA step** + - One setting of the ESA plate voltage, held for + ``SETTLE_DURATION + ACQ_DURATION``. 180 per quarter cycle, 720 per full + cycle, drawn from 24 distinct voltages. + * - **ESA level / ESA voltage** + - One of the 24 distinct voltages in the active ESA table. Maps to a + particle energy through the analyzer constant *k*. + * - **Quarter cycle** + - One ``SWE_SCIENCE`` packet. 15 s, 180 ESA steps, 1260 compressed bytes. + ``QUARTER_CYCLE`` in telemetry counts 0-3. + * - **Full cycle** + - Four consecutive quarter cycles with ``QUARTER_CYCLE`` 0,1,2,3. ~1 + minute. The L1B and L2 record. + * - **Spin sector** + - In L1A, the index 0-179 of a measurement within a quarter cycle. In L1B + and L2, the index 0-29 of a column of the checkerboard grid. **The same + word is used for both; check the dimension size.** + * - **Spin angle bin** + - One of 30 fixed 12-degree-wide bins in *physical* spin angle, centered + at 6, 18, ... 354 degrees. Produced at L2 by looking up each + measurement's actual spin angle from SPICE. ``inst_az`` in the L2 CDF. + * - **Checkerboard** + - The (24, 30) reorganization of a full cycle's 720 measurements. See + :ref:`swe-checkerboard-concept`. + * - **Deadtime** + - Detector recovery time after a count, during which arrivals are missed. + Corrected with a non-paralyzable model. + * - **In-flight calibration / gain sweep** + - Weekly sequence that steps CEM bias around nominal to measure gain + degradation. Produces the multiplicative per-CEM factors applied at + L1B. + * - **Stim pulse** + - Electronic pulser into the FEE preamps, used to exercise the chain + without real electrons. ``STIM_ENABLED`` / ``STIM_CFG_REG`` in + telemetry. + * - **BDE** + - Bidirectional electrons. The I-ALiRT flag: 1 = counterstreaming, + 0 = nominal unidirectional flow. + +Reference frames +---------------- + +.. list-table:: + :header-rows: 1 + :widths: 26 74 + + * - Frame + - Where it appears + * - **Instrument (IMAP_SWE)** + - **[CODE]** What L2 actually produces. ``inst_az`` is the SWE spin angle + from ``get_instrument_spin_phase(..., SpiceFrame.IMAP_SWE)``; + ``inst_el`` is the fixed CEM polar angle. + * - **Despun spacecraft (DSC)** + - **[DOC]** What section 3.4.4 asks for at L2 - "to facilitate comparison + with other IMAP instruments, required for Level 3 processing, the spin + phase angles will be calculated in despun spacecraft coordinates" - and + what L3 uses to combine SWE with MAG. See + :ref:`swe-implementation-status` for the discrepancy. + * - **RTN** + - **[DOC]** L3 only. All moments are rotated to RTN. + * - **Field-aligned** + - **[DOC]** L3 only. Temperatures are additionally rotated to + parallel/perpendicular to **B** from MAG. diff --git a/docs/source/algorithm-code-documentation/swe/reference-tables.rst b/docs/source/algorithm-code-documentation/swe/reference-tables.rst new file mode 100644 index 0000000000..174517883e --- /dev/null +++ b/docs/source/algorithm-code-documentation/swe/reference-tables.rst @@ -0,0 +1,295 @@ +.. _swe-reference-tables: + +Reference Tables - Where to Look Them Up +======================================== + +.. include:: /algorithm-code-documentation/_ai_generated_notice.inc + +Rule of thumb +------------- + +Big tables are **not** reproduced in these pages. They go stale, and every one +of them already exists in a machine-readable form that the code actually reads. +This page tells you which file, and which page of the algorithm document if you +have a copy. + +The one exception is the **count decompression table**, reproduced in full in +:ref:`swe-decompression`, because it is small, frozen in flight software, and +you cannot debug a count without it. + +Authority order, highest first: + +#. **The code and its ancillary files.** What runs. +#. **The XTCE** (``swe_packet_definition.xml``). What is parsed. +#. **The algorithm document.** What the instrument team intends. + +Appendix B of the algorithm document says so itself: "Note: packet definitions +continue to change as the instrument development progresses." + +Machine-readable tables in the repository +----------------------------------------- + +Packet definitions +^^^^^^^^^^^^^^^^^^ + +.. list-table:: + :header-rows: 1 + :widths: 46 54 + + * - File + - Contains + * - ``imap_processing/swe/packet_definitions/swe_packet_definition.xml`` + - ``SWE_APP_HK`` (APID 1330), ``SWE_CEM_RAW`` (APID 1334), + ``SWE_SCIENCE`` (APID 1344). Field names, bit widths, and derived-value + conversions. + * - ``imap_processing/ialirt/packet_definitions/ialirt_swe.xml`` + - The SWE I-ALiRT fields: ``SWE_CEM<1-7>_E<1-4>``, ``SWE_SHCOARSE``, + ``SWE_ACQ_SEC``, ``SWE_ACQ_SUB``, ``SWE_SEQ``, ``SWE_NOM_FLAG``, + ``SWE_OPS_FLAG``. + +APIDs +^^^^^ + +.. list-table:: + :header-rows: 1 + :widths: 14 26 60 + + * - APID + - ``SWEAPID`` member + - Packet + * - 1330 + - ``SWE_APP_HK`` + - Application housekeeping. Feeds ``imap_swe_l1a_hk`` and + ``imap_swe_l1b_hk``. + * - 1334 + - ``SWE_CEM_RAW`` + - Engineering-mode 1-second CEM counts. Feeds + ``imap_swe_l1a_cem-raw``. + * - 1344 + - ``SWE_SCIENCE`` + - One quarter cycle. Feeds everything else. + +The SWE I-ALiRT packet is not in ``SWEAPID``; it is handled by the I-ALiRT +packet machinery. + +``SWE_SCIENCE`` field list +^^^^^^^^^^^^^^^^^^^^^^^^^^ + +From the XTCE, in order. This is the full set; every one of them reaches L1B as +an ``(epoch, cycle)`` metadata variable. + +.. code-block:: text + + SHCOARSE ACQ_START_COARSE ACQ_START_FINE + CEM_NOMINAL_ONLY SPIN_PERIOD_VALIDITY SPIN_PHASE_VALIDITY + SPIN_PERIOD_SOURCE SETTLE_DURATION ACQ_DURATION + SPIN_PHASE SPIN_PERIOD REPOINT_WARNING + HIGH_COUNT STIM_ENABLED QUARTER_CYCLE + ESA_TABLE_NUM ESA_ACQ_CFG THRESHOLD_DAC + STIM_CFG_REG SCIENCE_DATA CKSUM + +Ancillary CSVs +^^^^^^^^^^^^^^ + +Operational copies come from the SDC ancillary store; representative fixtures +live in ``imap_processing/tests/swe/lut/``. Described in full in +:ref:`swe-ancillary`. + +.. list-table:: + :header-rows: 1 + :widths: 46 54 + + * - Fixture + - Columns + * - ``imap_swe_esa-lut_20250301_v000.csv`` + - ``table_idx, esa_step, esa_v, v_index, ialirt``. 8 tables x 48 rows. + * - ``imap_swe_l1b-in-flight-cal_20240510_20260716_v000.csv`` + - ``met_time, cem1 ... cem7``. + * - ``imap_swe_eu-conversion_20240510_v000.csv`` + - ``index, packetName, mnemonic, convertAs, segNumber, lowValue, + highValue, c0 ... c7``. 63 rows. + * - ``checker-board-indices.csv`` + - The expected ``(24, 30)`` checkerboard index map, used only as a test + oracle for ``get_checker_board_pattern()``. + +Validation and test data +^^^^^^^^^^^^^^^^^^^^^^^^ + +``imap_processing/tests/swe/``: + +.. list-table:: + :header-rows: 1 + :widths: 50 50 + + * - File + - What it is + * - ``l0_data/2024051010_SWE_SCIENCE_packet.bin`` + - Real ``SWE_SCIENCE`` packets, 2024-05-10. + * - ``l0_data/2024051010_SWE_HK_packet.bin`` + - Real ``SWE_APP_HK`` packets. + * - ``l0_data/2024051011_SWE_CEM_RAW_packet.bin`` + - Real ``SWE_CEM_RAW`` packets. + * - ``l0_validation_data/idle_export_raw.SWE_SCIENCE_*.csv`` + - GSEOS export of the science packets in **raw** units. + * - ``l0_validation_data/idle_export_eu.SWE_SCIENCE_*.csv`` + - Same in **engineering** units. The raw/EU pair is what validates + decompression. + * - ``l0_validation_data/idle_export_eu.SWE_APP_HK_*.csv`` + - HK validation. + * - ``l0_validation_data/idle_export_eu.SWE_CEM_RAW_*.csv`` + - CEM raw validation. + +Constants in code +^^^^^^^^^^^^^^^^^ + +``imap_processing/swe/utils/swe_constants.py`` holds every SWE magic number +except the deadtime (local to ``deadtime_correction``) and the I-ALiRT +thresholds (default arguments in ``process_swe.py``). The full list with +values is in :ref:`swe-ancillary`. + +CDF attribute configs +^^^^^^^^^^^^^^^^^^^^^ + +``imap_processing/cdf/config/``: ``imap_swe_global_cdf_attrs.yaml`` (the +``Logical_source`` authority) plus ``imap_swe_l1a_variable_attrs.yaml``, +``imap_swe_l1b_variable_attrs.yaml`` and ``imap_swe_l2_variable_attrs.yaml``. + +Document page index +------------------- + +Against CN102D-D0001, Issue Draft, 15 June 2026 (38 pages). Put your copy in +``docs/reference/IMAP_SWE_Algorithms_v8.pdf`` - that directory is gitignored. + +.. list-table:: + :header-rows: 1 + :widths: 10 14 30 46 + + * - Pages + - Section + - Title + - Worth reading for + * - 4 + - 1 + - Introduction + - Scope. Two paragraphs. + * - 5-8 + - 2 + - SWE Instrument Description + - Sensor head, ESA + 7 CEMs, HVPS ranges, CDH. Figures 1-3. + * - 8-9 + - 3.1 + - Operating Modes + - The six FSW modes and the MET coarse/fine counters. Figure 4. + * - 9-11 + - 3.2 + - Nominal Science Operations + - **The measurement cycle.** Figure 5 is the single most useful page in + the document - it is the picture of the checkerboard. + * - 11-15 + - 3.3 + - Data Products + - Science, HK, engineering, in-flight calibration and I-ALiRT product + definitions. Figure 6 is the pipeline diagram. + * - 13 + - 3.3.4 + - In-Flight Calibration Data + - The calibration file format, the filename date convention, and the + far-future padding row. Read this before touching calibration code. + * - 16 + - 3.4.1 + - Heritage data processing + - Why the code looks the way it does, and the 8-table caveat. + * - 16-18 + - 3.4.2 + - L0 to L1A + - The ``SCIENCE_DATA`` byte layout and the **decompression table** with a + worked example. Also the I-ALiRT, HK and CEM raw packet notes. + * - 19-21 + - 3.4.3 + - L1A to L1B + - Deadtime C fragment, counts-to-rate, and the ``electron_cal()`` + interpolation C fragment. + * - 21-25 + - 3.4.4 + - L1B to L2 + - Analyzer constant, polar angles, the ``fspace()`` C fragment + (**including the end-detector loop**), the flux conversion, the + acquisition time formula, and the spin angle binning specification. + * - 25-29 + - 3.4.5 + - I-ALiRT Processing + - The BDE algorithm in full, with the worked bin-offset examples. + Figure 7 overlays the I-ALiRT subset on Figure 5. + * - 29-32 + - 3.4.6 + - L2 to L3 + - **Not our work.** Spacecraft potential, pitch angle, moments. See + :ref:`swe-l3-scope`. + * - 33-34 + - Appendix A + - Telemetry packet conversions + - Every raw-to-engineering polynomial, including the ``HVPS_ESA_DAC`` + dual-range note. + * - 35-38 + - Appendix B + - Telemetry Packet Definitions + - Sample byte/bit layouts for ``SWE_SCIENCE`` and ``SWE_IALIRT``. + **Superseded by the XTCE**; the appendix says so. + +Figures worth knowing about +--------------------------- + +.. list-table:: + :header-rows: 1 + :widths: 14 20 66 + + * - Figure + - Page + - What it shows + * - 1 + - 5 + - Front and back views of the instrument: sensor head cylinder, FEE box, + EBOX. + * - 2 + - 6 + - Sensor cross section: the spherical ESA plates and the seven gold CEM + cones. + * - 3 + - 7 + - Electronics block diagram, colour-coded by assembly. + * - 4 + - 9 + - Mode transition diagram. (The text extraction of this figure is + garbled; you need the PDF.) + * - 5 + - 11 + - **Science data acquisition timing and flow.** The energy/spin-angle + grid with the quarter-cycle colouring. If you only look at one figure, + this is it. + * - 6 + - 15 + - The SWE data processing pipeline, L0 through L3, as defined in the + SDMP. + * - 7 + - 25 + - Figure 5 again with a yellow box around the 8 I-ALiRT ESA steps. + +Where else to look +------------------ + +.. list-table:: + :header-rows: 1 + :widths: 34 66 + + * - Question + - Answer + * - "What does this packet field mean?" + - The XTCE, then Appendix B. + * - "What is this CDF variable?" + - ``imap_processing/cdf/config/imap_swe_*_variable_attrs.yaml``. + * - "Where does this number come from?" + - ``swe_constants.py``, then :ref:`swe-ancillary`. + * - "Is this implemented?" + - :ref:`swe-implementation-status`. + * - "Should this be implemented here at all?" + - :ref:`swe-l3-scope`. diff --git a/docs/source/index.rst b/docs/source/index.rst index 2ae3dbb730..4b1007a590 100644 --- a/docs/source/index.rst +++ b/docs/source/index.rst @@ -21,6 +21,7 @@ The explicit code interfaces and structure are described in the :ref:`algorithm- .. toctree:: :maxdepth: 1 + Mission Overview Onboarding & Collaboration IMAP Data Access Tool CDF Metadata Resources diff --git a/docs/source/mission-overview.rst b/docs/source/mission-overview.rst new file mode 100644 index 0000000000..fd7c3e16d1 --- /dev/null +++ b/docs/source/mission-overview.rst @@ -0,0 +1,407 @@ +.. _mission-overview: + +Mission Overview +================ + +This page is the mission-level context. It answers three questions: what IMAP is trying to find out, why +that requires these particular ten instruments, and why several of them appear +to measure the same thing. + +It is deliberately short. Each instrument's own ``overview`` page covers how +that instrument works; this page only covers **why it is on the spacecraft.** + +What IMAP is +------------ + +The Interstellar Mapping and Acceleration Probe is a NASA +Heliophysics Solar Terrestrial Probe built by a team of 25 partner institutions. +It carries **ten instruments** on a simple Sun-pointed spinner orbiting the +**Sun-Earth L1 Lagrange point**, and launched on **24 September 2025**. Its design is a follow up to the IBEX mission. + +* Spins at **4 RPM**, like IBEX. +* Unlike IBEX, the spin axis is **repointed roughly 1° each day** to track the + nominal solar wind aberration direction (~4° off the Sun in the ecliptic). + Tracking the average solar wind direction, plus being far from terrestrial + backgrounds, is why IMAP's ENA measurements are substantially cleaner than + IBEX's. +* Three instruments have two sensor heads each, and one sits on its own + dedicated pivot platform. +* IMAP was the top new mission priority in the 2013 Heliophysics Decadal + Survey, formed by merging two separate white-paper proposals - one for + expanded ENA imaging after IBEX, one for in-situ particle acceleration + measurements. + +That merger matters, and it is the reason this payload can look like two missions +bolted together. The Decadal Survey group judged the two proposals +*"not just complementary, but synergistic, as some of the particles accelerated +in the inner heliosphere are ultimately 'recycled' through charge exchange in +the outer heliosphere and return to L1 as ENAs."* + + +The science objectives +---------------------- + +IMAP is framed around **two coupled topics**: + +1. the **acceleration of charged particles**, and +2. the **interaction of the solar wind with the local interstellar medium** + (LISM, or VLISM for the *very* local interstellar medium). + +These are coupled because particles accelerated in the inner heliosphere +propagate outward and then *mediate* that interaction. + +Formally, the mission is organised around **four Science Objectives** from the +NASA Announcement of Opportunity, listed from the LISM inward. Instrument pages +refer to these as O1-O4: + +.. list-table:: + :header-rows: 1 + :widths: 6 94 + + * - + - Objective + * - **O1** + - Improve understanding of the **composition and properties of the LISM**. + * - **O2** + - Advance understanding of the **temporal and spatial evolution of the + boundary region** in which the solar wind and the interstellar medium + interact. + * - **O3** + - Identify and advance understanding of processes related to the + **interactions of the magnetic field of the Sun and the LISM**. + * - **O4** + - Identify and advance understanding of **particle injection and + acceleration** near the Sun, in the heliosphere and heliosheath. + +A third, operational goal explains a large amount of code in this repository. +**I-ALiRT** (IMAP Active Link for Real-Time) continuously +telemeters real-time space weather data from **SWAPI, CoDICE, HIT, SWE and +MAG**, which the SOC analyses and posts with a **latency of under 5 minutes**. + +Imaging a boundary you cannot visit +----------------------------------- + +The heliosphere is the bubble the solar wind inflates in the VLISM. +Its boundary region - the termination shock, the heliosheath beyond it, and the +heliopause separating heliospheric plasma from the VLISM - extends from hundreds +to roughly **1000 au** in the upwind direction. You cannot survey that by flying +to it. + +Instead IMAP images it with **energetic neutral atoms**. An ion in the +heliosheath charge-exchanges with a neutral atom and becomes neutral; having no +charge, it is no longer steered by magnetic fields, so its arrival direction at +1 au still points back to where it was born. Three instruments image this +across overlapping energy ranges: **IMAP-Lo, IMAP-Hi and IMAP-Ultra**. + +The difficulty is that an ENA map depends on both the ion population out there +*and* the neutral density it charge-exchanged with - and the latter depends on +the solar wind's history. **This is why the in-situ instruments exist.** They +are not a parallel experiment; they supply the boundary conditions that make the +maps interpretable. + +That dependency is concrete, not rhetorical. GLOWS light curves +yield heliolatitude profiles of the ISN hydrogen ionization rate; those decompose +into photoionization and charge-exchange rates, and the charge-exchange rates +into **profiles of 3-D solar wind speed and density**. Those profiles are then +*"used to calculate survival probabilities of ENAs observed by IMAP."* In this +repository that appears as GLOWS L3 producing ENA survival probabilities for Lo, +Hi and Ultra. (:ref:`glows`) + +Why pickup ions +--------------- + +The solar wind *"picks up locally ionized interstellar neutrals +drifting into the heliosphere, creating the PUI population."* Interstellar +neutrals cross into the heliosphere unimpeded because they carry no charge; once +ionized, the solar wind sweeps them up and carries them outward. They are +recognised in an E/q spectrum by a characteristic **PUI cutoff**, and SWAPI is +specified to observe the He+ distribution *"from low energies to beyond the PUI +cutoff."* + +Pickup ions serve both goals at once. + +* **They are a sample of the LISM delivered to 1 au.** Combining SWAPI, CoDICE, + IMAP-Lo and GLOWS *"allows determination of the LISM flow properties with + unprecedented accuracy"* (O1). +* **They are the seed population for acceleration.** *"Suprathermal ions, + including PUIs, serve as seed populations for acceleration at interplanetary + shocks and within the magnetosphere"* (O4). +* **They mediate the interaction itself.** The solar wind flows outward + *"incorporating an increasing fraction of PUIs and PUI pressure all the way + out to the termination shock"* (O2). + +IMAP also distinguishes **interstellar** PUIs from **inner source** ones, and +asks explicitly *"what is the composition of dust at 1 au that provides a seed +population for the inner source PUIs?"* - which is part of why a dust instrument +shares the spacecraft. + +.. _mission-overview-instruments: + +The ten instruments +------------------- + +Ranges, resolutions and cadences are from Table 4 of the mission +paper; the objective mapping is from each instrument's own section where the +paper states it explicitly. + +.. list-table:: + :header-rows: 1 + :widths: 13 22 65 + + * - Instrument + - Range + - Observable and contribution + * - **SWAPI** + - 0.1-20 keV/q + - Solar wind H+ and He++ **and the lighter PUIs (H+ and He+)**, as 1-D + VDFs. Solar wind bulk parameters at ~12 s; PUI He+ at ~10 min, enabling + the PUI gravitational focusing cone. **O1, O4.** + * - **CoDICE-Lo** + - 0.5-80 keV/q + - Solar wind and suprathermal ions **with composition and charge states**, + 3-D VDFs: solar wind He-Fe and **interstellar pickup He, O and Ne**. + m/Δm ≥ 2. + * - **CoDICE-Hi** + - 0.03-5 MeV/nuc + - Suprathermal and energetic ion mass composition and arrival direction. + m/Δm ≥ 4. See :ref:`mission-overview-discrepancies`. + * - **HIT** + - Ions 2-50 MeV/nuc + (species dependent); + electrons 0.5-1 MeV + - Energetic H-Ni composition, spectra, angular distributions and arrival + times, linked to CoDICE suprathermals. Resolves 3He, a tracer of SEP + acceleration. 8 of 10 apertures view the full sky; 2 are modified for + electrons. **O4.** + * - **SWE** + - 1-5000 eV + - Solar wind electron 3-D VDFs. Pitch-angle distributions diagnose + magnetic topology. **[REPO]** (:ref:`swe-overview`) + * - **MAG** + - ±512 nT / ±60,000 nT + (auto-ranging) + - Vector interplanetary magnetic field, 2 Hz (64 Hz for ~8 hours/day). + Needed by any pitch-angle calculation anywhere in the pipeline. + * - **IMAP-Lo** + - 5-1000 eV + - ISN **and** ENA flux and composition. On a **pivot platform**, which + breaks the measurement degeneracy in ISN flow parameters. Tracks ISN H, + He, O, Ne and D over >180° of ecliptic longitude. **O1-O3.** + * - **IMAP-Hi** + - 0.41-15.6 keV FWHM + - Hydrogen ENA flux maps in **nine contiguous energy passbands**. Two + identical single-pixel cameras, 4.1° FWHM conical FOV: **Hi-90** + perpendicular to the spin axis, **Hi-45** at 45° anti-sunward. Hi-90 + makes a full sky map every 6 months. **O2-O3.** + * - **IMAP-Ultra** + - ENA 3-300 keV; + ions 3-5000 keV + - The highest-energy ENAs, 2° angular resolution for H above 30 keV. Two + slit-optics imagers at 45° and 90° covering **~3π sr per spin**; full sky + map every 3 months. Near-copy of JUICE/JENI, with ~35x the collecting + power of Cassini/INCA. **O2-O4.** + * - **IDEX** + - 2x10⁻¹³ - 5x10⁻¹¹ g; + 1-286 amu + - Interstellar and interplanetary dust composition by impact-ionization + TOF mass spectrometry, m/Δm > 120 at 56 amu. **Links the interstellar + gas-phase composition from IMAP-Lo and the PUI measurements from CoDICE + and SWAPI to the composition of dust grains.** **O1.** + * - **GLOWS** + - 120.5 ± 4.3 nm + - Hydrogen Lyman-α helioglow light curves along Sun-centred rings, one per + pointing. Yields heliolatitude profiles of 3-D solar wind speed and + density, and thence ENA survival probabilities. **O1, O2.** + +.. _mission-overview-ladder: + +The energy ladder and deliberate overlaps +------------------------------------------ + +The three ENA cameras *"have overlapping energy ranges that roughly +match in-situ ion measurements above."* That matching is the design, not a +coincidence: + +.. code-block:: text + + in-situ ions SWAPI 0.1 - 20 keV/q + CoDICE-Lo 0.5 - 80 keV/q + CoDICE-Hi 0.03 - 5 MeV/nuc + HIT 2 - 50 MeV/nuc + + ENA imaging IMAP-Lo 5 - 1000 eV + IMAP-Hi 0.41 - 15.6 keV + IMAP-Ultra 3 - 300 keV + +The overlaps were engineered and then verified on the ground. Three +pairs - **SWAPI & CoDICE, IMAP-Lo & IMAP-Hi, and IMAP-Hi & IMAP-Ultra** - were +cross-calibrated *"as separate pairs in the same vacuum chamber at the same +time"*, rotated into steady ion and neutral beams across their overlapping +ranges. Instruments with overlapping ranges continue to cross-calibrate in +flight, and IMAP-Lo and IMAP-Hi are additionally cross-calibrated against +IBEX-Lo and IBEX-Hi for as long as IBEX survives. + +This is what the ladder buys: following one population from thermal solar wind, +through the pickup shell, into the suprathermal tail and out to energetic +particle energies, as a single cross-calibrated spectrum. A gap loses the trail. + +It also explains a structural oddity in this repository. **CoDICE-Hi's neighbour +in the spectrum is HIT, not CoDICE-Lo** - the paper says HIT links its +measurements to *"suprathermal measurements from CoDICE"*. Lo hands off to Hi, +Hi hands off to HIT. CoDICE is one box spanning a seam that falls in its middle; +see :ref:`codice-overview` for how deep that split runs in the code. + +Why two instruments measure pickup ions +--------------------------------------- + +SWAPI and CoDICE both observe pickup ions and are the cross-calibrated pair in +that energy range - but they measure different properties, and the paper's own +wording separates them cleanly. SWAPI measures *"the lighter species +of PUIs (H+ and He+)"*; CoDICE-Lo measures *"interstellar pickup He, O, and Ne +ions"* with composition and charge state. + +.. list-table:: + :header-rows: 1 + :widths: 22 39 39 + + * - + - SWAPI + - CoDICE-Lo + * - Measures + - **E/q only** - no mass, no charge state + - E/q **plus** TOF and residual energy, giving **M, q and M/q** per + particle **[REPO]** (:ref:`codice-overview`) + * - Species ID + - Inferred from the *shape* of the E/q spectrum; L3 is model fitting + **[REPO]** (:ref:`swapi-l3-scope`) + - Measured directly, per event + * - Range + - 0.1-20 keV/q + - 0.5-80 keV/q + * - PUI species + - H+ and He+ - the light ones + - He, O and Ne - the heavy ones + * - Cadence + - ~12 s solar wind, ~10 min PUI He+ + - ≤ 1 hour + * - Strength + - Energy resolution and cadence on the dominant species + - Breadth of species and charge states + +So SWAPI supplies the precise *shape* of the distribution for the two species +that dominate - fast enough to resolve the PUI gravitational focusing cone - and +CoDICE supplies the *inventory* of which heavier elements and charge states are +present. + +CoDICE has a second, unrelated solar wind job worth knowing about: charge-state +ratios such as O7+/O6+ and C6+/C5+ freeze in close to the Sun and are unchanged +by transport, so they reach L1 as a record of coronal conditions. The mission paper +notes these charge-state ratios as one of the novel I-ALiRT measurements +improving on ACE. In this repository they are the L3a ratio products and the +I-ALiRT pseudo-density ratios.(:ref:`codice-l3-scope`, +:ref:`codice-ialirt`) + +.. _mission-overview-discrepancies: + +Where the numbers disagree +-------------------------- + +.. warning:: + + The mission paper and the instrument algorithm documents do not always agree + on energy ranges, and **the mission paper is not internally consistent + either.** Do not "fix" one to match the other. + + .. list-table:: + :header-rows: 1 + :widths: 16 28 28 28 + + * - Instrument + - Mission paper Table 4 + - Mission paper prose + - Instrument pages here + * - **CoDICE-Hi** + - 0.05-2 MeV/nuc + - ~0.03-5 MeV/nuc (§4.2 opening); *"~0.03 and >2 MeV/nuc"* a paragraph + later + - ~0.03-5 MeV/nuc (:ref:`codice-overview`) + * - **HIT** + - Ions 2-70 MeV/nuc + - 2-50 MeV/nuc, *"species dependent"* + - ~2-40 MeV/nuc (:ref:`hit-overview`) + * - **IMAP-Lo** + - 5-1000 eV + - ENA maps *"down to 100 eV and below and up to 1 keV"* + - ENAs 40 eV - 1 keV + + These are mostly the same instrument described at different confidence + levels and with different qualifiers - ``species dependent`` does a lot of + work in the HIT row, and the IMAP-Lo rows differ because Table 4 covers ISN + and ENA together while the other two quote the ENA range. **For anything + that affects code, the instrument algorithm document wins**, per the + convention on each instrument index page. This page quotes Table 4 because + it is the only self-consistent mission-wide set. + +.. _mission-overview-sources: + +Sources +------- + +**Primary source for this page:** + + D.J. McComas et al., *Interstellar Mapping And Acceleration Probe: The NASA + IMAP Mission*, Space Science Reviews (2025) **221**:100, 82 pp. + `doi:10.1007/s11214-025-01224-z + `_ + +This paper is **open access** +(CC BY-NC-ND 4.0), so it can be linked and quoted freely. It is also the citable +reference for the mission-level **CMAD** (Calibration and Measurement Algorithms +Document), supplied as a supplemental file to the paper. + +.. tip:: + + If you hold a copy, put it in ``docs/reference/``. That directory is + gitignored, so it will never be committed. + +The paper is the lead article in a **17-paper IMAP collection** in Space Science +Reviews, which includes a dedicated paper for each of the ten instruments. When +an instrument page here lacks the background you need, the relevant paper is: + +.. list-table:: + :header-rows: 1 + :widths: 20 40 40 + + * - SWAPI + - Rankin et al. 2025 + - CoDICE - Livi et al. 2025 + * - HIT + - Christian et al. 2025 + - SWE - Skoug et al. 2025 + * - MAG + - Horbury et al. 2025 + - IMAP-Lo - Schwadron et al. 2025 + * - IMAP-Hi + - Funsten et al. 2025 + - IMAP-Ultra - Gkioulidou et al. 2025 + * - IDEX + - Horányi et al. 2025 + - GLOWS - Bzowski et al. 2025 + * - I-ALiRT + - Lee et al. 2025 + - Observatory - Hegarty et al. 2025 + +The statements on this page come from the instrument overview pages +cited inline, each of which names its own algorithm document; see the +``Source documents`` section of any instrument index, for example +:ref:`codice-source-documents`. + +Where to go next +---------------- + +* For how an instrument works: its ``overview`` page, linked from + :ref:`algorithm-code-documentation`. +* For what it produces and what the files are called: its ``data-products`` + page. +* For what is actually built versus merely specified: its + ``implementation-status`` page. diff --git a/pyproject.toml b/pyproject.toml index 9d3145f302..5ca12e49c6 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -118,7 +118,7 @@ lint.ignore = [ convention = "numpy" [tool.codespell] -ignore-words-list = "livetime,nd" +ignore-words-list = "livetime,livetimes,nd,anc" [tool.poetry-dynamic-versioning] enable = true