diff --git a/.issueflows/03-solved-issues/issue164_original.md b/.issueflows/03-solved-issues/issue164_original.md new file mode 100644 index 00000000..6d4db864 --- /dev/null +++ b/.issueflows/03-solved-issues/issue164_original.md @@ -0,0 +1,21 @@ +# Issue #164: Allow for c.update() + +- GitHub: https://github.com/jepegit/cellpy/issues/164 +- Labels: enhancement, v2, cellpy2-stage4, cellpy2-stage5 +- Epic: #783 (Epic L, live/incremental), Stage 2. Depends on: #780. + +## Original description + +After loading a file (e.g. a cellpyfile), it should be possible to update it by + +```python +c = cellpy.get("cellpyfile.h5) +# the cellpy file contains the paths to the original raw files +c.update() +# only the new data will be loaded and processed +``` + +To achieve this, cellpy will need to have an easy way to find the last loaded +data and load from that. And update summaries from starting from the first new +step (or the last old if it is not complete) etc. This means that a smart way +of lookup and merging if several raw files are used is needed. diff --git a/.issueflows/03-solved-issues/issue164_plan.md b/.issueflows/03-solved-issues/issue164_plan.md new file mode 100644 index 00000000..091fc216 --- /dev/null +++ b/.issueflows/03-solved-issues/issue164_plan.md @@ -0,0 +1,43 @@ +# Plan: #164 `CellpyCell.update()` + +Confirmed approach (autonomous run under #783 stage 2; user asked to +"process the issues"). Builds on #778 (`update_core_data`), #779 (protocol), +#780 (`load_since` on four loaders). + +## Approach + +1. `CellpyCell.update(force=False, **loader_kwargs) -> bool` in + `cellpy/readers/cellreader.py`, before `merge`. +2. Change detection: fresh `FileID` size/mtime vs stored `raw_data_files`; + db sources always "changed". +3. Loader recovery from `data._provenance["source_type"]` via + `set_instrument` (cellpy-file loads carry the default tester). +4. Marker derived from `data.raw` (`_marker_from_raw`: both `row_count` and + `last_source_datapoint_num`, rewound to the last cycle start). No new + cellpy-file field. +5. Incremental path (single fid + native schema + harmonized raw + loader + matches `SupportsIncrementalLoad`): `load_since` → stamp `test_id`, align + dtypes → `_update_from_raw_rows` (`core.update_core_data` + + `_add_summary_extras` + `_refresh_scaled_summary_columns`). +6. Fallback full reload (`ValueError`/`LoaderError`, multi-file, + non-incremental loader): `from_raw` on all sources, restore + `meta_common` / `cycle_mode` / `cell_name`, `make_step_table`, + `make_summary(find_ir=...)`. +7. `_refresh_fid`: size/mtimes, `last_data_point`, `raw_data_files_length`. +8. `tests/incremental_support.incremental_update` delegates to the new engine. + +## Tests (`tests/test_cell_update.py`, essential) + +no-op, growth == full load, two growths (marker), FileID refresh, cellpy-file +round trip, single-cycle fallback keeps meta, force, no source raises, +non-incremental loader routes to full reload. + +## Docs + +`incremental-load-protocol.md` section for #164, `docs/agents/index.md`, +root `AGENTS.md` quick facts, HISTORY bullet, test-registry rows. + +## Out of scope + +Poll loop (#781), batch live refresh (#782), multi-file incremental merge +(falls back to full reload). diff --git a/.issueflows/03-solved-issues/issue164_status.md b/.issueflows/03-solved-issues/issue164_status.md new file mode 100644 index 00000000..3c373fe0 --- /dev/null +++ b/.issueflows/03-solved-issues/issue164_status.md @@ -0,0 +1,34 @@ +# Status: #164 `CellpyCell.update()` + +- [x] Done + +## Done + +- `CellpyCell.update()` + helpers (`_raw_sources_changed`, + `_ensure_loader_for_update`, `_marker_from_raw`, `_update_incremental`, + `_update_from_raw_rows`, `_summary_has_ir`, `_refresh_fid`, + `_update_full_reload`) and module helpers `_frame_to_pandas`, + `_align_dtypes` in `cellpy/readers/cellreader.py`; `_load_marker` attribute + initialised in `__init__`. +- `tests/incremental_support.incremental_update` now delegates to + `CellpyCell._update_from_raw_rows` (the #778 oracle covers the shipped engine). +- `tests/test_cell_update.py`: 9 essential tests. +- Docs: `incremental-load-protocol.md` (#164 section), `docs/agents/index.md`, + root `AGENTS.md`, HISTORY `[Unreleased]`, test-registry. + +## Verification + +- `uv run pytest tests/test_cell_update.py tests/test_incremental_update.py tests/test_load_since.py` green. +- `uv run pytest -m essential` — see PR. + +## Notes + +- Branch `164-cell-update` stacked on `780-load-since` (PR #1101); PR base + is `780-load-since` until #1101 merges. +- Multi-file cells and non-incremental loaders take the full-reload path + (documented; the smart multi-file merge from the original issue is not + needed for the live use case, one file per running test). + +## Remaining + +- None for this issue. Stage 3: #781 (poll loop), #782 (batch live refresh). diff --git a/.issueflows/04-designs-and-guides/incremental-load-protocol.md b/.issueflows/04-designs-and-guides/incremental-load-protocol.md index a5b38801..d1825864 100644 --- a/.issueflows/04-designs-and-guides/incremental-load-protocol.md +++ b/.issueflows/04-designs-and-guides/incremental-load-protocol.md @@ -61,6 +61,55 @@ expose `load_since`). cycle-local rebase may differ from a full load. Only loader-made markers carry the equality guarantee. +## `CellpyCell.update()` (#164) + +The public consumer of the protocol, in `cellpy/readers/cellreader.py` +(section "incremental refresh"). Returns `bool` (frames changed). + +- **Change detection** compares a fresh `FileID(fid.full_name)` size and + mtime with the stored `raw_data_files` entry (same stats + `check_file_ids` uses). Databases (`is_db`) and unreadable stats always + count as changed. `force=True` skips the check. +- **Loader recovery.** A cell loaded from a cellpy-file carries the config + default tester; `data._provenance["source_type"]` (persisted) names the + loader that read the raw. `update()` calls `set_instrument` from it (and + forwards `**loader_kwargs`, e.g. `model=`, since the model is not + persisted). Decision: no new cellpy-file field. +- **Marker without state.** The marker is not persisted either. When the + cell has no in-memory `_load_marker`, `_marker_from_raw` derives one from + `data.raw` with **both** seek fields filled (`row_count` = index of the + first row of the last cycle; `last_source_datapoint_num` = the datapoint + before it), so text and arbin loaders each find their field. Rewinding to + the cycle start mirrors the loaders' own policy above. Alternative + rejected: storing the marker in the cellpy-file (extra schema, and the + derived one is exact for loader-made markers anyway). +- **Incremental path** only when: one raw file, `native_schema`, + `config.reader.use_harmonized_raw`, and + `isinstance(loader_class, SupportsIncrementalLoad)`. The chunk gets + `test_id = active_test_id` and its dtypes cast to the existing raw's + (`_align_dtypes`; harmonize can yield Int32 where raw has Int64), then + goes through `_update_from_raw_rows` → `core.update_core_data` with the + by-value inputs cellpy owns (`nom_cap_abs`, current factor, raw limits), + `_add_summary_extras`, and `_refresh_scaled_summary_columns`. + `find_ir` follows whether the current summary has `ir_charge`. +- **Fallback = full reload** on `ValueError` / `LoaderError` from the + incremental path (typically core refusing a chunk whose start is at or + before the first kept row, i.e. a single-cycle head), for multi-file + cells, and for non-incremental loaders. `from_raw` on all recorded + sources, then `meta_common` (deep copy), `cycle_mode`, and `cell_name` + are restored before `make_step_table()` / `make_summary(find_ir=...)`. +- **FileID refresh** after either path: size / mtimes, `last_data_point` + (max datapoint), `raw_data_files_length[-1]`. A second `update()` on the + same file is then a no-op. +- `tests/incremental_support.incremental_update` (the #778 oracle) now + delegates to `CellpyCell._update_from_raw_rows`, so the equality tests + cover the shipped engine. + +## Link + +Design §3 in `cellpy-design-and-development/active/cellpy2-live-incremental-design.md`. +Tests: `tests/test_load_since.py` (#780), `tests/test_cell_update.py` (#164). +Consumers: `live.py` poll loop (#781), batch live refresh (#782). ## Link Design §3 in `cellpy-design-and-development/active/cellpy2-live-incremental-design.md`. diff --git a/.issueflows/04-designs-and-guides/test-registry.md b/.issueflows/04-designs-and-guides/test-registry.md index c8e5d666..9f5fdcd9 100644 --- a/.issueflows/04-designs-and-guides/test-registry.md +++ b/.issueflows/04-designs-and-guides/test-registry.md @@ -200,6 +200,15 @@ current issue**. `/iflow-doctor` may audit the whole suite against this table. | tests/test_load_since.py::test_arbin_res_load_since_seeks_on_datapoint_and_rewinds_to_cycle_start | yes | yes | arbin_res.load_since | #780 | skip without mdbtools | | tests/test_load_since.py::test_arbin_sql_load_since_filters_on_datapoint | yes | yes | arbin_sql.load_since | #780 | mocked `_query_sql` | | tests/test_load_since.py::test_arbin_sql_query_gets_a_datapoint_clause | yes | yes | arbin_sql._query_sql | #780 | SQL clause on the fully qualified table | +| tests/test_cell_update.py::test_update_on_unchanged_source_is_a_noop | yes | yes | CellpyCell.update / _raw_sources_changed | #164 | size+mtime unchanged → False | +| tests/test_cell_update.py::test_update_after_growth_equals_full_load | yes | yes | CellpyCell.update (incremental) | #164 | head + grown tail == full cellpy.get | +| tests/test_cell_update.py::test_update_twice_tracks_the_marker | yes | yes | CellpyCell._marker_from_raw / _load_marker | #164 | two growths; marker rewinds | +| tests/test_cell_update.py::test_update_refreshes_file_id | yes | yes | CellpyCell._refresh_fid | #164 | size / last_data_point / lengths; second update no-op | +| tests/test_cell_update.py::test_update_after_cellpy_file_round_trip | yes | yes | CellpyCell._ensure_loader_for_update | #164 | loader from provenance; mass kept | +| tests/test_cell_update.py::test_update_falls_back_to_full_reload_and_keeps_meta | yes | yes | CellpyCell._update_full_reload | #164 | single-cycle head → core rejects → reload | +| tests/test_cell_update.py::test_update_force_reloads_an_unchanged_source | yes | yes | CellpyCell.update(force=True) | #164 | | +| tests/test_cell_update.py::test_update_without_raw_source_raises | yes | yes | CellpyCell.update | #164 | NoDataFound | +| tests/test_cell_update.py::test_update_uses_full_reload_for_non_incremental_loader | yes | yes | CellpyCell.update (protocol gate) | #164 | loader without load_since → full reload | | tests/test_dbreader.py::test_missing_column_warns_once | yes | yes | readers.dbreader.Reader._pick_info | #1008 | warn-once per missing header | | tests/test_dbreader.py::test_nom_cap_specifics_column_reaches_pages | yes | yes | batch._dbengine._create_pages_dict | #1008 | db value → pages | | tests/test_dbreader.py::test_simple_db_engine_skip_file_search_excel_reader | yes | yes | batch._dbengine.simple_db_engine / find_files | #1017 | skip_file_search frames one row per cell | diff --git a/AGENTS.md b/AGENTS.md index 596ecb77..722bb490 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -374,6 +374,9 @@ Quick facts: `kind=raw` lists raw and sets `needs_metadata`). - Metadata peek (no frames): `cellpy.read_meta(path)` → dict with `cell` / `tests`. - Ingestion form fields: `cellpy.instrument_meta_schema(instrument)` → `fields` / `units`. +- Live/running test: `c.update()` re-reads only the new raw rows (incremental + loaders) or reloads fully, returns `True` when frames changed; `False` if + the raw file did not change on disk. Also works after `cellpy.get(".cellpy")`. - Frames: `c.data.raw` / `.steps` / `.summary`; columns via `c.schema.*`. After a raw load, each cycle's raw capacity starts at 0. A forgotten tester reset that 1.x plotted as doubled capacity is rebased on load for every diff --git a/HISTORY.md b/HISTORY.md index f4b0b801..05f5dc20 100644 --- a/HISTORY.md +++ b/HISTORY.md @@ -11,6 +11,13 @@ next marker, which rewinds to the start of the last cycle so cycle-local capacity normalisation stays correct. Other loaders stay full-read. (#780) +* `CellpyCell.update()`: refresh a cell from a raw file that grew. Detects + changes from file size/mtime, reads only the new rows for incremental + loaders (arbin_res, arbin_sql, neware_txt, maccor_txt) and appends them + through cellpy-core, otherwise reloads fully while keeping mass, area, + nominal capacity, cycle mode and cell name. Works on cells loaded from a + cellpy-file. Returns `True` when the frames changed. (#164) + ## [2.1.5.post6] - 2026-09-25 * `summary_collector(...).plot()` keeps a lone charge or discharge series diff --git a/cellpy/readers/cellreader.py b/cellpy/readers/cellreader.py index 603f1e52..8a06c4fe 100644 --- a/cellpy/readers/cellreader.py +++ b/cellpy/readers/cellreader.py @@ -37,6 +37,7 @@ from cellpy.exceptions import ( DeprecatedFeature, + LoaderError, MixedCycleModesError, NoDataFound, ) @@ -123,6 +124,29 @@ } +def _frame_to_pandas(frame): + """pandas view of a core (polars) frame; pandas passes through.""" + return frame.to_pandas() if hasattr(frame, "to_pandas") else frame + + +def _align_dtypes(chunk, existing): + """Cast the shared columns of a polars ``chunk`` to ``existing``'s dtypes. + + A freshly harmonized chunk can carry narrower integer types (Int32 for a + literal ``test_id``, for example) than the raw frame it is appended to; + core's vertical concat requires an exact dtype match. + """ + import polars as pl + + target = existing if isinstance(existing, pl.DataFrame) else pl.from_pandas(existing) + casts = [ + pl.col(name).cast(dtype) + for name, dtype in target.schema.items() + if name in chunk.columns and chunk.schema[name] != dtype + ] + return chunk.with_columns(casts) if casts else chunk + + def normalize_summary_meta_fields(fields=None): """Normalize meta field names for ``SUMMARY_META_DEPENDENCIES`` / ``refresh_after``. @@ -338,6 +362,9 @@ def __init__( self.tester = tester self.loader = None # this will be set in the function set_instrument + #: Incremental-load position for ``update()`` (#164); derived from + #: the raw frame when None, so it is never persisted. + self._load_marker = None self.debug = debug logging.debug("created CellpyCell instance") @@ -1870,6 +1897,235 @@ def _convert2fid_list(tbl): # -------------------- cellpy file handling end ---------------------- + # -------------------- incremental refresh (#164) -------------------- + + def update(self, force=False, **loader_kwargs): + """Refresh this cell from its raw source(s) if they have grown. + + The headline live/incremental feature (Epic L, cellpy 2.2). Works on a + cell loaded from raw and on a cell loaded from a cellpy-file (the + raw-file ids stored there name the source). + + Flow: + + 1. Change detection on the recorded raw files (size and mtime, as in + ``check_file_ids``). Unchanged and not ``force`` → no-op. + 2. Single source whose loader implements ``SupportsIncrementalLoad`` + (arbin_res, arbin_sql, neware_txt, maccor_txt): read only the rows + since the load marker and append them through core + ``update_core_data`` (overlap trimmed, affected steps rebuilt, + summary refreshed). The marker is derived from the loaded raw when + this cell has none yet, so a cellpy-file round trip needs no extra + state. + 3. Otherwise, or when the incremental path rejects the chunk (for + example the head held a single cycle): full reload of every raw + file, then step table and summary. Cell metadata (mass, nominal + capacity, area, cycle mode, name) is kept. + + Args: + force: refresh even when the file stats did not change. + **loader_kwargs: forwarded to ``set_instrument`` when the loader has + to be (re)created from the stored provenance, e.g. + ``model="UIO"`` for a neware export. + + Returns: + bool: ``True`` when the frames changed, ``False`` for a no-op. + + Raises: + NoDataFound: if the cell has no recorded raw source. + + Examples: + ```python + c = cellpy.get("running_test.csv", instrument="neware_txt") + ... # the tester keeps writing + if c.update(): + print(c.data.summary.tail(1)) + ``` + """ + data = self.data + fids = [f for f in (data.raw_data_files or []) if f is not None] + if not fids: + raise NoDataFound("cannot update: no raw source recorded on this cell") + + if not force and not self._raw_sources_changed(fids): + logging.info("update: raw source(s) unchanged") + return False + + self._ensure_loader_for_update(**loader_kwargs) + + from cellpy.readers.instruments.contract import SupportsIncrementalLoad + + loader = self.loader_class + can_go_incremental = ( + len(fids) == 1 + and self.native_schema + and getattr(config.reader, "use_harmonized_raw", True) + and isinstance(loader, SupportsIncrementalLoad) + ) + if can_go_incremental: + try: + return self._update_incremental(fids[0], loader) + except (ValueError, LoaderError) as exc: + logging.info(f"update: incremental path declined ({exc}); reloading") + return self._update_full_reload(fids, **loader_kwargs) + + def _raw_sources_changed(self, fids) -> bool: + """True if any recorded raw file differs from disk in size or mtime. + + Sources without file stats (databases, missing files) count as changed + so the caller still tries to refresh them. + """ + for fid in fids: + if getattr(fid, "is_db", False) or not fid.full_name: + return True + current = ds.FileID(fid.full_name) + if current.name is None: + return True + if fid.size is None or fid.last_modified is None: + return True + if int(current.size) != int(fid.size): + return True + if float(current.last_modified) != float(fid.last_modified): + return True + return False + + def _ensure_loader_for_update(self, **loader_kwargs): + """Recreate the loader from stored provenance when the tester changed. + + A cell loaded from a cellpy-file carries the default instrument, not + the one that read the raw file; ``_provenance['source_type']`` + remembers it. + """ + provenance = getattr(self.data, "_provenance", None) or {} + source_type = provenance.get("source_type") + if loader_kwargs or (source_type and source_type != self.tester): + instrument = source_type or self.tester + logging.debug(f"update: setting instrument {instrument} ({loader_kwargs})") + self.set_instrument(instrument=instrument, **loader_kwargs) + self.tester = instrument + + def _marker_from_raw(self): + """Derive a `LoadMarker` from the loaded raw when none is stored. + + Both seek fields are filled so any implementing loader can read the + one it uses: ``row_count`` = row index of the first row of the last + cycle (text loaders); ``last_source_datapoint_num`` = the datapoint + just before that row (arbin). Rewinding to the cycle start mirrors the + loaders' own marker policy (see ``instruments/incremental.py``). + """ + import polars as pl + + from cellpy.readers.instruments.contract import LoadMarker + from cellpy.readers.instruments.incremental import last_cycle_start + + raw = self.data.raw + frame = raw if isinstance(raw, pl.DataFrame) else pl.from_pandas(raw) + if frame.height == 0: + return None + cycle_column = self.schema.raw.cycle_num + start_row = last_cycle_start(frame, cycle_column) + datapoint_column = self.schema.raw.datapoint_num + datapoint = int(frame.get_column(datapoint_column)[start_row]) + return LoadMarker(last_source_datapoint_num=datapoint - 1, row_count=start_row) + + def _update_incremental(self, fid, loader) -> bool: + import polars as pl + + marker = getattr(self, "_load_marker", None) or self._marker_from_raw() + if marker is None: + raise ValueError("no raw rows to derive a load marker from") + source = fid.full_name if not getattr(fid, "is_db", False) else fid.name + chunk = loader.load_since(source, marker) + self._load_marker = chunk.marker + if chunk.new_raw is None or chunk.new_raw.height == 0: + self._refresh_fid(fid) + return False + + test_id_column = self.schema.raw.test_id + new_raw = chunk.new_raw.with_columns(pl.lit(int(self.data.active_test_id)).alias(test_id_column)) + new_raw = _align_dtypes(new_raw, self.data.raw) + self._update_from_raw_rows(new_raw, find_ir=self._summary_has_ir()) + self._refresh_fid(fid) + logging.info(f"update: appended {chunk.new_raw.height} raw rows (incremental)") + return True + + def _update_from_raw_rows(self, new_raw, find_ir=True): + """Append ``new_raw`` (native schema) through core ``update_core_data``. + + Mirrors the cellpy-side orchestration in ``make_step_table`` / + ``make_summary`` (by-value nominal capacity, current factor, raw + limits) and re-applies the summary extras and the scaled (mass/area) + columns, so the result compares with a full ``cellpy.get``. Raises + ``ValueError`` when core rejects the chunk (full reload territory). + """ + from cellpy.readers.native_core import _add_summary_extras + + data = self.data + factor = core_units.calculate_current_conversion_factor( + data.raw_units["current"], to_units=self.cellpy_units + ) + nom_cap_abs = self._resolve_nom_cap_abs(data) + out = self.core.update_core_data( + data, + new_raw, + nom_cap_abs=nom_cap_abs, + current_conversion_factor=factor, + find_ir=find_ir, + raw_limits=self.raw_limits, + ) + # ``update_core_data`` returns a bare cellpycore ``Data``; copy the + # frames back so cellpy's metadata-bearing ``Data`` stays the owner. + data.raw = _frame_to_pandas(out.raw) + data.steps = _frame_to_pandas(out.steps) + data.summary = _add_summary_extras(_frame_to_pandas(out.summary), self.core.schema) + self._refresh_scaled_summary_columns() + return self + + def _summary_has_ir(self) -> bool: + summary = getattr(self.data, "summary", None) + if summary is None or getattr(summary, "empty", True): + return True + return self.schema.summary.ir_charge in summary.columns + + def _refresh_fid(self, fid): + """Re-stat the raw source and record the new tail on its `FileID`.""" + if not getattr(fid, "is_db", False) and fid.full_name: + current = ds.FileID(fid.full_name) + if current.name is not None: + fid.size = current.size + fid.last_modified = current.last_modified + fid.last_accessed = current.last_accessed + fid.last_info_changed = current.last_info_changed + raw = self.data.raw + datapoint_column = self.schema.raw.datapoint_num + if datapoint_column in raw.columns and len(raw): + fid.last_data_point = int(raw[datapoint_column].max()) + if self.data.raw_data_files_length: + self.data.raw_data_files_length[-1] = len(raw) + + def _update_full_reload(self, fids, **loader_kwargs) -> bool: + sources = [f.name if getattr(f, "is_db", False) else f.full_name for f in fids] + old = self.data + keep_meta = copy.deepcopy(old.meta_common) + keep_cycle_mode = self.cycle_mode + keep_name = self._cell_name + find_ir = self._summary_has_ir() + is_a_file = self.tester not in DB_READER_INSTRUMENTS + + self.from_raw(file_names=sources, is_a_file=is_a_file) + self.data.meta_common = keep_meta + if keep_cycle_mode is not None: + self.cycle_mode = keep_cycle_mode + if keep_name is not None: + self._cell_name = keep_name + self.make_step_table() + self.make_summary(find_ir=find_ir) + self._load_marker = None + logging.info("update: full reload") + return True + + # -------------------- incremental refresh end ----------------------- + def merge(self, cells, mode="campaign", renumber_cycles=True, **kwargs): """Merge other cells/datasets into this one. diff --git a/docs/agents/index.md b/docs/agents/index.md index e59187f4..ffffd19e 100644 --- a/docs/agents/index.md +++ b/docs/agents/index.md @@ -169,6 +169,15 @@ Useful methods on `CellpyCell` (non-exhaustive): meta-dependent columns (cheaper than a full `make_summary()`). See `cellpy.readers.cellreader.SUMMARY_META_DEPENDENCIES` for the map GUIs can use for messaging. +- `update()` — refresh a cell from a raw file that is still being written + (a running test). Returns `True` if the frames changed, `False` when the + file's size/mtime are unchanged (`force=True` overrides). With + `arbin_res` / `arbin_sql` / `neware_txt` / `maccor_txt` only the new rows + are read and appended (the last cycle is re-read whole); other loaders, + or a head too short to append to, fall back to a full reload that keeps + mass / area / nominal capacity / cycle mode. Works on a cell loaded from a + `.cellpy` file too (the raw path is stored in it); pass loader kwargs + such as `model="UIO"` when the instrument needs them. - `save` / `to_csv` / Excel helpers — persist for the user's workflow Deeper shape docs: [Data structure](../fundamentals/data_structure.md). diff --git a/tests/incremental_support.py b/tests/incremental_support.py index 52e4736e..c4c0733c 100644 --- a/tests/incremental_support.py +++ b/tests/incremental_support.py @@ -4,11 +4,9 @@ ``update_core_data`` must produce the same ``raw`` / ``steps`` / ``summary`` as a single full load. -``incremental_update`` is a test-side prototype of what L3 (#164) will expose as -``CellpyCell.update()``: it drives ``update_core_data`` with the cellpy-owned -by-value inputs (nominal capacity, current factor, instrument raw limits) and then -re-applies the cellpy-side summary extras and the scaled (mass/area) columns. -Replace this helper with the public API once L3 lands. +``incremental_update`` feeds a tail frame through the same code path L3 (#164) +uses inside ``CellpyCell.update()`` (``CellpyCell._update_from_raw_rows``), so +these tests stay the oracle for the shipped implementation. """ from __future__ import annotations @@ -17,9 +15,6 @@ import pandas as pd import pandas.testing as pdt -from cellpycore import units as core_units - -from cellpy.readers.native_core import _add_summary_extras REPO_ROOT = Path(__file__).resolve().parents[1] NEWARE_UIO = REPO_ROOT / "testdata" / "data" / "neware_uio.csv" @@ -39,33 +34,9 @@ def tail_rows(raw: pd.DataFrame, datapoint_col: str, since: int, overlap: int = return raw[raw[datapoint_col] > since - overlap] -def _to_pandas(frame): - return frame.to_pandas() if hasattr(frame, "to_pandas") else frame - - def incremental_update(cell, new_raw: pd.DataFrame, find_ir: bool = True): - """Append ``new_raw`` to ``cell`` in place via ``update_core_data``. - - Mirrors the cellpy-side orchestration in ``make_step_table`` / - ``make_summary`` so the result is comparable with a full ``cellpy.get``. - """ - factor = core_units.calculate_current_conversion_factor(cell.data.raw_units["current"], to_units=cell.cellpy_units) - nom_cap_abs = cell._resolve_nom_cap_abs(cell.data) - out = cell.core.update_core_data( - cell.data, - new_raw, - nom_cap_abs=nom_cap_abs, - current_conversion_factor=factor, - find_ir=find_ir, - raw_limits=cell.raw_limits, - ) - # ``update_core_data`` returns a bare cellpycore ``Data``; copy the frames back - # so cellpy's metadata-bearing ``Data`` stays the owner. - cell.data.raw = _to_pandas(out.raw) - cell.data.steps = _to_pandas(out.steps) - cell.data.summary = _add_summary_extras(_to_pandas(out.summary), cell.core.schema) - cell._refresh_scaled_summary_columns() - return cell + """Append ``new_raw`` to ``cell`` in place through ``CellpyCell.update()``'s engine.""" + return cell._update_from_raw_rows(new_raw, find_ir=find_ir) def _normalize(frame: pd.DataFrame, sort_by) -> pd.DataFrame: diff --git a/tests/test_cell_update.py b/tests/test_cell_update.py new file mode 100644 index 00000000..ecb8972d --- /dev/null +++ b/tests/test_cell_update.py @@ -0,0 +1,145 @@ +"""``CellpyCell.update()`` (#164): refresh a cell from a raw source that grew. + +The oracle is a full ``cellpy.get`` of the complete file: a cell loaded from a +truncated copy, then ``update()``-ed after the copy grew, must carry the same +raw / steps / summary frames. +""" + +from __future__ import annotations + +import shutil + +import pytest + +import cellpy +from cellpy.exceptions import NoDataFound +from cellpy.readers.cellreader import CellpyCell +from tests.incremental_support import ( + NEWARE_KWARGS, + NEWARE_UIO, + assert_cell_frames_equal, + truncate_text_file, +) + +pytestmark = pytest.mark.essential + +MID_CYCLE_3 = 6000 # rows; well inside the third of four cycles +MID_CYCLE_4 = 8800 +SINGLE_CYCLE = 1000 # rows; still inside the first cycle + + +@pytest.fixture(scope="module") +def full_cell(): + return cellpy.get(NEWARE_UIO, testing=True, **NEWARE_KWARGS) + + +@pytest.fixture +def live_file(tmp_path): + return truncate_text_file(NEWARE_UIO, tmp_path / "live.csv", MID_CYCLE_3) + + +def _grow(path, n_rows=None): + if n_rows is None: + shutil.copyfile(NEWARE_UIO, path) + else: + truncate_text_file(NEWARE_UIO, path, n_rows) + + +def test_update_on_unchanged_source_is_a_noop(live_file): + c = cellpy.get(live_file, testing=True, **NEWARE_KWARGS) + n_raw = len(c.data.raw) + assert c.update() is False + assert len(c.data.raw) == n_raw + + +def test_update_after_growth_equals_full_load(live_file, full_cell): + c = cellpy.get(live_file, testing=True, **NEWARE_KWARGS) + _grow(live_file) + assert c.update() is True + assert_cell_frames_equal(c, full_cell) + + +def test_update_twice_tracks_the_marker(live_file, full_cell): + c = cellpy.get(live_file, testing=True, **NEWARE_KWARGS) + _grow(live_file, MID_CYCLE_4) + assert c.update() is True + assert len(c.data.raw) == MID_CYCLE_4 + marker = c._load_marker + assert marker is not None and marker.row_count is not None + assert marker.row_count < MID_CYCLE_4 # rewound to the last cycle start + _grow(live_file) + assert c.update() is True + assert_cell_frames_equal(c, full_cell) + + +def test_update_refreshes_file_id(live_file): + c = cellpy.get(live_file, testing=True, **NEWARE_KWARGS) + fid = c.data.raw_data_files[0] + old_size, old_last = fid.size, fid.last_data_point + _grow(live_file) + c.update() + assert fid.size > old_size + assert fid.last_data_point > old_last + assert fid.last_data_point == int(c.data.raw[c.schema.raw.datapoint_num].max()) + assert c.data.raw_data_files_length[-1] == len(c.data.raw) + assert c.update() is False # stats now match the file again + + +def test_update_after_cellpy_file_round_trip(live_file, tmp_path): + c = cellpy.get(live_file, testing=True, mass=1.3, **NEWARE_KWARGS) + cellpy_file = tmp_path / "live.cellpy" + c.save(cellpy_file) + reloaded = cellpy.get(cellpy_file, testing=True) + assert reloaded.tester != "neware_txt" # the file does not carry the loader + _grow(live_file) + assert reloaded.update() is True + assert reloaded.tester == "neware_txt" + assert reloaded.mass == pytest.approx(1.3) + expected = cellpy.get(NEWARE_UIO, testing=True, mass=1.3, **NEWARE_KWARGS) + assert_cell_frames_equal(reloaded, expected) + + +def test_update_falls_back_to_full_reload_and_keeps_meta(tmp_path): + live = truncate_text_file(NEWARE_UIO, tmp_path / "live.csv", SINGLE_CYCLE) + c = cellpy.get(live, testing=True, mass=2.5, **NEWARE_KWARGS) + c.cell_name = "keep-me" + _grow(live) + # A single-cycle head leaves nothing before the rewind point, so core + # rejects the chunk and update() reloads the whole file instead. + assert c.update() is True + assert c.mass == pytest.approx(2.5) + assert c.cell_name == "keep-me" + expected = cellpy.get(NEWARE_UIO, testing=True, mass=2.5, **NEWARE_KWARGS) + assert_cell_frames_equal(c, expected) + + +def test_update_force_reloads_an_unchanged_source(live_file): + c = cellpy.get(live_file, testing=True, **NEWARE_KWARGS) + before = c.data.raw.copy() + assert c.update(force=True) is True + assert len(c.data.raw) == len(before) + + +def test_update_without_raw_source_raises(): + c = CellpyCell(initialize=True) + with pytest.raises(NoDataFound): + c.update() + + +def test_update_uses_full_reload_for_non_incremental_loader(live_file, full_cell, monkeypatch): + c = cellpy.get(live_file, testing=True, **NEWARE_KWARGS) + calls = [] + + def _no_incremental(*args, **kwargs): + calls.append(1) + raise AssertionError("incremental path must not run") + + monkeypatch.setattr(c, "_update_incremental", _no_incremental) + # Pretend the loader lacks load_since (a fresh subclass sidesteps the ABC + # isinstance cache): the protocol check must route to full reload. + not_incremental = type("NotIncremental", (type(c.loader_class),), {"load_since": None}) + c.loader_class.__class__ = not_incremental + _grow(live_file) + assert c.update() is True + assert calls == [] + assert_cell_frames_equal(c, full_cell)