From 61b17cc4f30ff01efe978190521521f6931a7804 Mon Sep 17 00:00:00 2001 From: jepegit Date: Sat, 26 Sep 2026 00:00:54 +0200 Subject: [PATCH 1/3] chore: record drive #783 stop before picking #779 Keep the stage-2 halt notes locally. This replaces the leftover #1074 auto status in the same file. Co-authored-by: Cursor --- .issueflows/01-current-issues/auto_status.md | 41 +++++++------------ .issueflows/01-current-issues/drive_status.md | 16 ++++++++ .../cycle_status_2026-09-25.md | 14 +++++++ 3 files changed, 44 insertions(+), 27 deletions(-) create mode 100644 .issueflows/01-current-issues/drive_status.md create mode 100644 .issueflows/03-solved-issues/cycle_status_2026-09-25.md diff --git a/.issueflows/01-current-issues/auto_status.md b/.issueflows/01-current-issues/auto_status.md index 1f392342..b18889d0 100644 --- a/.issueflows/01-current-issues/auto_status.md +++ b/.issueflows/01-current-issues/auto_status.md @@ -1,29 +1,16 @@ -# Auto #1074 +# Auto #783 -- epic: 1074 -- stage: 1 -- stage_title: Find by project and number range -- loop_count: 0 -- budget: 2 -- last_outcome: epoch_gated -- overnight_authorization: yes -- started: 2026-09-22T10:16:00+02:00 -- updated: 2026-09-22T10:30:00+02:00 +epic: 783 +stage: 2 +stage_title: Cell update from a marker +loop_count: 0 +budget: 2 +started: 2026-09-25T23:45:00+02:00 +overnight: authorized by drive confirm -## Cycle - -- #1075 merged via PR #1078 -- #1076 not started — not yolo-fit (`onfail:stop`) - -## Adversarial - -- Stage goal not met: library cannot yet list by project/range. -- #1075 spec honesty: clear. Parser + essential tests match the must-match / must-miss examples. No reopen. -- Remaining gap owned by existing #1076. No new issue created. -- Did not re-queue #1076 through cycle (would weaken yolo). - -## Gate - -- Stage 1 `done`: false -- Open blocker: #1076 -- Stage 2 not started +last_outcome: stopped +stop_reason: > + Stage 2 queue is yolo: no. Cycle halted on #779 before implementation + (public protocol, not a small change). #780 and #164 were not reached. + Did not re-queue: another loop would hit the same safeguard. + Did not open duplicate issues. Stage 3 was not started (epoch_gated). diff --git a/.issueflows/01-current-issues/drive_status.md b/.issueflows/01-current-issues/drive_status.md new file mode 100644 index 00000000..15439da3 --- /dev/null +++ b/.issueflows/01-current-issues/drive_status.md @@ -0,0 +1,16 @@ +# Drive #783 + +anchor: 783 +started: 2026-09-25T23:25:00+02:00 + +- [x] draft +- [x] publish — no new issues; #778 #779 #780 #164 #781 #782 already exist. Plan PR https://github.com/jepegit/cellpy/pull/1099 +- [ ] auto — stopped on stage 2. #779 is yolo: no. Stage 3 not started. +- [ ] final_review +- [ ] cleanup +- [ ] status + +findings: + +last_outcome: pending +auto_stop: stage 2 halted on #779 (yolo: no). #780 and #164 not reached. #781 and #782 not started. diff --git a/.issueflows/03-solved-issues/cycle_status_2026-09-25.md b/.issueflows/03-solved-issues/cycle_status_2026-09-25.md new file mode 100644 index 00000000..45bfa8ae --- /dev/null +++ b/.issueflows/03-solved-issues/cycle_status_2026-09-25.md @@ -0,0 +1,14 @@ +# Cycle epic 783 stage 2 + +queue: epic 783 +onfail: stop +started: 2026-09-25T23:45:00+02:00 + +- [x] Done — halted + +- [ ] #779 — SupportsIncrementalLoad protocol and marker types — stopped: yolo: no (public protocol; not a small change). No branch, no PR. +- [ ] #780 — load_since for cheap-partial loaders — not reached +- [ ] #164 — CellpyCell.update loads only new data — not reached + +skipped_closed: none in this stage (#778 is stage 1 and already closed) +blocked: none outside the queue From 91af9f2ffb47aac7265593262ee4150cd96e43b4 Mon Sep 17 00:00:00 2001 From: jepegit Date: Sat, 26 Sep 2026 00:31:57 +0200 Subject: [PATCH 2/3] Add load_since() to the four cheap-partial loaders. arbin_res, arbin_sql, neware_txt, and maccor_txt now match the optional SupportsIncrementalLoad protocol. Each returns the harmonized rows read since a LoadMarker and the next marker. The marker rewinds to the first row of the last cycle read, because harmonize's reset-granularity normalisation rebases each cycle against its first row; a mid-cycle chunk would be rebased against the wrong row. Text loaders seek by data row (row_count), arbin by Data_Point (last_source_datapoint_num) through the existing data_points filter (.res) or one extra WHERE clause (SQL Server). Other loaders stay full-read. Tests cover protocol membership, chunk equality with a full harmonize(parse()) read, marker rewind, the empty past-the-end chunk, parse-cache hygiene, and the #778 equality oracle driven by a real chunk. Closes #780 Co-authored-by: Cursor --- .../03-solved-issues/issue780_original.md | 14 + .issueflows/03-solved-issues/issue780_plan.md | 90 + .../03-solved-issues/issue780_status.md | 24 + .../incremental-load-protocol.md | 36 +- .../04-designs-and-guides/test-registry.md | 11 + HISTORY.md | 5 + cellpy/readers/instruments/arbin_res.py | 3024 +++++++++-------- cellpy/readers/instruments/arbin_sql.py | 1205 +++---- cellpy/readers/instruments/base.py | 1983 +++++------ cellpy/readers/instruments/incremental.py | 51 + cellpy/readers/instruments/maccor_txt.py | 1042 +++--- cellpy/readers/instruments/neware_txt.py | 256 +- tests/test_load_since.py | 249 ++ 13 files changed, 4314 insertions(+), 3676 deletions(-) create mode 100644 .issueflows/03-solved-issues/issue780_original.md create mode 100644 .issueflows/03-solved-issues/issue780_plan.md create mode 100644 .issueflows/03-solved-issues/issue780_status.md create mode 100644 cellpy/readers/instruments/incremental.py create mode 100644 tests/test_load_since.py diff --git a/.issueflows/03-solved-issues/issue780_original.md b/.issueflows/03-solved-issues/issue780_original.md new file mode 100644 index 00000000..6b944810 --- /dev/null +++ b/.issueflows/03-solved-issues/issue780_original.md @@ -0,0 +1,14 @@ +# Issue #780: L2: load_since() for cheap-partial loaders (arbin_res/sql, neware_txt, maccor_txt) + +Source: https://github.com/jepegit/cellpy/issues/780 + +## Original issue text + +Epic L of **cellpy 2.2 (Stage 5)**. Design: [live-incremental](https://github.com/cellpy/architecture-plan/blob/main/cellpy2-live-incremental-design.md) §7 item 2. Depends on **L1**. + +Implement `load_since(source, marker)` for the sources that can re-read cheaply from a marker first: `arbin_res`, `arbin_sql`, `neware_txt`, `maccor_txt`. Return native-schema raw rows appended since the marker (may overlap the tail) + the new marker. Other loaders stay full-read and fall back. + +## Epic context + +Epic #783 stage 2 ("Cell update from a marker"). Depends on #779 (merged in +PR #1100). `CellpyCell.update()` is #164 and consumes the chunks produced here. diff --git a/.issueflows/03-solved-issues/issue780_plan.md b/.issueflows/03-solved-issues/issue780_plan.md new file mode 100644 index 00000000..f022eeb0 --- /dev/null +++ b/.issueflows/03-solved-issues/issue780_plan.md @@ -0,0 +1,90 @@ +# Issue #780 plan — load_since for arbin_res, arbin_sql, neware_txt, maccor_txt + +## Goal + +Four shipped loaders advertise `SupportsIncrementalLoad` (#779) and return an +`IncrementalChunk` of harmonized native raw rows read since a `LoadMarker`, +plus the marker for the next call. Every other loader is unchanged and does +not match the protocol. + +## Constraints + +- Cellpy only. cellpycore unchanged; overlap trimming stays in `update_data`. +- `new_raw` is the same frame `harmonize(parse())` produces for a full load + (the frame `CellpyCell.from_raw` uses as `data.raw` on the harmonized + path), so a chunk is directly consumable by `update_core_data`. +- Cycle-local normalisation in `harmonize.normalize_reset_granularity` + (PER_STEP re-accumulation, PER_TEST/PER_CYCLE rebase to 0 at cycle start) + needs the whole cycle. A chunk that starts mid-cycle would be rebased + wrongly. **Policy:** every `load_since` re-reads from the first row of the + last cycle already seen. The marker therefore points at that cycle start + ("consumed" = committed complete cycles); the trailing overlap is allowed + by the contract and `update_data` keeps the new rows for it. +- The marker field is the one the source seeks on: `row_count` for the text + loaders (data rows before the re-read position), `last_source_datapoint_num` + for arbin (rows with a larger `Data_Point` are returned). `complete` stays + `False`; none of these sources can tell that a test has ended. +- `load_since(source, None)` (or an all-`None` marker) returns every row. + A marker past the end of the file returns an empty `new_raw` and the same + marker (empty `new_raw` is a no-op for `update_core_data`). +- `load_since` must not leave a partial frame in the loader's parse cache + (`_parsed_frame` / `_parsed_data`), or a later `loader()` would reuse it. +- Out of scope: `CellpyCell.update()` (#164), marker persistence, poll loop. + +### Prior art + +- `contract.py` — `LoadMarker`, `IncrementalChunk`, `SupportsIncrementalLoad` + (#779). Structural; adding a `load_since` method is enough to match. +- `AutoLoader.parse()` / `declarations()` + `harmonize()` — the two-stage + read every loader here already has. `TxtLoader.query_file` is the single + `pd.read_csv` call; `parse_loader_parameters` resolves `sep` / `skiprows` / + `header` (auto-formatter reads only the first 200 lines). +- `arbin_res.parse(source, **kwargs)` forwards `data_points=(d1, d2)` to + `_loader_win` / `_loader_posix`, which already filter `Data_Point >= d1`. +- `arbin_sql.parse()` → `_query_sql(name)`; the SQL gets one extra + `AND ... Data_Point > N` clause. No live server in-repo: mock via + `mock_data_001.xlsx` sheet `arbin_sql` like `tests/test_sql.py`. +- `tests/incremental_support.py` (#778) — `incremental_update` + + `assert_cell_frames_equal` are the oracle: head cell + real chunk must + equal a full load. + +## Approach + +1. New helper module `cellpy/readers/instruments/incremental.py`: + `last_cycle_start(frame, cycle_column) -> int` (row index of the first + row of the last cycle; 0 when the column is missing or the frame is + empty) and `vendor_column(declarations, native_name)` (inverse lookup in + `column_map`). +2. `TxtLoader`: + - `query_file(name, skiprows=None)` gains an optional override so a + partial read reuses the same `pd.read_csv` call. + - `_load_since_rows(source, marker)`: resolve formatter parameters the + way `parse()` does, read data rows from `marker.row_count` on with a + callable `skiprows` that keeps the header, harmonize, compute the + rewind row on the vendor frame, return + `IncrementalChunk(new_raw, LoadMarker(row_count=start + rewind))`. + Clear `_parsed_frame` afterwards. + - `neware_txt.DataLoader.load_since` and `maccor_txt.DataLoader.load_since` + delegate to it. `TxtLoader` itself does not get `load_since`, so + `local_instrument` / `batmo_bdf` / custom stay non-incremental. +3. `arbin_res.DataLoader.load_since`: `parse(source, data_points=(N + 1, + None))`, harmonize, rewind on the vendor `Cycle_Index` / `Data_Point` + columns, marker `last_source_datapoint_num = cycle_start_datapoint - 1`. + Clear `_parsed_data`. +4. `arbin_sql`: `_query_sql(name, since_data_point=None)` adds the clause; + `parse(source, since_data_point=…)`; `load_since` mirrors arbin_res. +5. Tests (`tests/test_load_since.py`, essential): + - protocol membership: the four match, `pec_csv` / `biologics_mpr` / + `local_instrument` do not. + - neware: `load_since(file, None)` equals the full harmonized frame; a + head file then the full file returns rows from the head's last cycle + start; head cell + chunk through `incremental_update` equals + `cellpy.get` of the whole file (the #778 oracle with a real chunk). + - maccor: `load_since(file, None)` equals full; marker round-trip. + - arbin_res (skip if `mdb-export` missing): full equality and the marker + re-read from a mid-test datapoint. + - arbin_sql: monkeypatched `_query_sql` records `since_data_point` and + serves the mock sheet. + - marker past end → empty `new_raw`, same marker. +6. Docs: extend `incremental-load-protocol.md` with the rewind policy and + per-loader marker field; registry rows; HISTORY bullet at close. diff --git a/.issueflows/03-solved-issues/issue780_status.md b/.issueflows/03-solved-issues/issue780_status.md new file mode 100644 index 00000000..8e1eb00d --- /dev/null +++ b/.issueflows/03-solved-issues/issue780_status.md @@ -0,0 +1,24 @@ +# Issue #780 status + +- [x] Done + +## What's done + +- `cellpy/readers/instruments/incremental.py`: `last_cycle_start`, + `vendor_column` (shared rewind-to-cycle-start policy, documented there). +- `TxtLoader.query_file(name, skiprows=None)` + `TxtLoader._load_since_rows`; + `neware_txt` and `maccor_txt` `DataLoader.load_since` delegate to it. + `TxtLoader` itself does not advertise the protocol. +- `arbin_res.DataLoader.load_since` via `parse(..., data_points=(N+1, None))`. +- `arbin_sql._query_sql(name, since_data_point=None)` adds a `Data_Point >` + clause on the fully qualified table; `parse(since_data_point=)`; + `DataLoader.load_since`. +- Both arbin loaders clear their parse cache; text path clears + `_parsed_frame`. +- `tests/test_load_since.py` (11 essential tests, incl. the #778 oracle + driven by a real chunk; arbin_res skips without mdbtools). +- `incremental-load-protocol.md` extended; registry rows; HISTORY bullet. + +## Remaining work + +- None in this issue. `CellpyCell.update()` consuming these chunks is #164. diff --git a/.issueflows/04-designs-and-guides/incremental-load-protocol.md b/.issueflows/04-designs-and-guides/incremental-load-protocol.md index 2522a1b3..a5b38801 100644 --- a/.issueflows/04-designs-and-guides/incremental-load-protocol.md +++ b/.issueflows/04-designs-and-guides/incremental-load-protocol.md @@ -27,7 +27,41 @@ cellpycore type. - Use pandas for `new_raw` because some cellpy frames are still pandas. Rejected: the core merge primitive and the loader result are polars. +## Loaders that implement it (#780) + +`arbin_res`, `arbin_sql`, `neware_txt`, `maccor_txt`. Every other shipped +loader stays full-read and does **not** match the protocol (`TxtLoader` +itself has the shared `_load_since_rows`, but only those two text loaders +expose `load_since`). + +- `new_raw` is the frame `harmonize(parse())` yields for the rows read, so + it has the same columns as `data.raw` on the harmonized load path and can + go straight into `update_core_data`. `test_id` is stamped 0 by + `harmonize`; the caller (`update()`, #164) re-stamps `active_test_id`. +- **Rewind to a cycle start.** `harmonize.normalize_reset_granularity` + re-accumulates per-step capacity and rebases each cycle to start at 0 + using the *first row of the cycle*. A chunk that begins mid-cycle would be + rebased against the wrong row without raising. So the marker a loader + returns points at the first row of the last cycle it read, and the next + call re-reads that cycle whole. The overlap is allowed by the contract and + `update_data` keeps the new rows for it (`kept_raw = raw < r2_start`). + Helper: `cellpy/readers/instruments/incremental.py::last_cycle_start`. +- **Marker field per source:** text loaders set `row_count` (file data rows + before the re-read position; header lines are not counted); arbin sets + `last_source_datapoint_num` (rows with a larger `Data_Point` are returned, + via the existing `data_points` filter for `.res` and one extra `WHERE` + clause for SQL Server). `byte_offset` is unused. `complete` is always + `False`; none of these sources can tell a test has ended. +- `marker=None` or an all-`None` marker reads everything. A marker past + the end gives an empty `new_raw` and the marker back unchanged (empty + `new_raw` is a no-op for `update_core_data`). +- `load_since` clears the loader's parse cache (`_parsed_frame` / + `_parsed_data`) so a later `loader()` never reuses a partial frame. +- A caller-made marker mid-cycle still reads the right rows, but the + cycle-local rebase may differ from a full load. Only loader-made markers + carry the equality guarantee. + ## Link Design §3 in `cellpy-design-and-development/active/cellpy2-live-incremental-design.md`. -Loaders that implement `load_since` are #780. `CellpyCell.update()` is #164. +`CellpyCell.update()` is #164. Tests: `tests/test_load_since.py`. diff --git a/.issueflows/04-designs-and-guides/test-registry.md b/.issueflows/04-designs-and-guides/test-registry.md index 2fbba38e..c8e5d666 100644 --- a/.issueflows/04-designs-and-guides/test-registry.md +++ b/.issueflows/04-designs-and-guides/test-registry.md @@ -189,6 +189,17 @@ current issue**. `/iflow-doctor` may audit the whole suite against this table. | tests/test_incremental_protocol.py::test_loader_can_match_both_protocols | yes | yes | instruments.contract.SupportsIncrementalLoad | #779 | optional second protocol | | tests/test_incremental_protocol.py::test_marker_and_chunk_are_frozen | yes | yes | instruments.contract.LoadMarker | #779 | frozen dataclasses | | tests/test_incremental_protocol.py::test_conformance_kit_ignores_a_missing_load_since | yes | yes | instruments.testing.check_loader | #779 | full-read kit unchanged | +| tests/test_load_since.py::test_only_the_four_cheap_partial_loaders_are_incremental | yes | yes | neware_txt / maccor_txt / arbin_res / arbin_sql load_since | #780 | protocol membership; others stay full-read | +| tests/test_load_since.py::test_last_cycle_start_finds_the_trailing_run | yes | yes | instruments.incremental.last_cycle_start | #780 | rewind helper | +| tests/test_load_since.py::test_neware_load_since_none_equals_full_harmonized_read | yes | yes | TxtLoader._load_since_rows | #780 | chunk == harmonize(parse()) | +| tests/test_load_since.py::test_neware_load_since_marker_rereads_from_last_cycle_start | yes | yes | TxtLoader._load_since_rows | #780 | row_count marker rewinds to cycle start | +| tests/test_load_since.py::test_neware_head_cell_plus_chunk_equals_full_load | yes | yes | load_since + update_core_data | #780 | #778 oracle with a real chunk | +| tests/test_load_since.py::test_marker_past_end_of_file_gives_empty_chunk_and_same_marker | yes | yes | TxtLoader._load_since_rows | #780 | empty new_raw contract | +| tests/test_load_since.py::test_load_since_does_not_poison_the_parse_cache | yes | yes | TxtLoader._load_since_rows / loader | #780 | `_parsed_frame` cleared | +| tests/test_load_since.py::test_maccor_load_since_matches_full_read_and_row_marker | yes | yes | maccor_txt.load_since | #780 | multi-line header; mid-cycle caller marker caveat | +| tests/test_load_since.py::test_arbin_res_load_since_seeks_on_datapoint_and_rewinds_to_cycle_start | yes | yes | arbin_res.load_since | #780 | skip without mdbtools | +| tests/test_load_since.py::test_arbin_sql_load_since_filters_on_datapoint | yes | yes | arbin_sql.load_since | #780 | mocked `_query_sql` | +| tests/test_load_since.py::test_arbin_sql_query_gets_a_datapoint_clause | yes | yes | arbin_sql._query_sql | #780 | SQL clause on the fully qualified table | | tests/test_dbreader.py::test_missing_column_warns_once | yes | yes | readers.dbreader.Reader._pick_info | #1008 | warn-once per missing header | | tests/test_dbreader.py::test_nom_cap_specifics_column_reaches_pages | yes | yes | batch._dbengine._create_pages_dict | #1008 | db value → pages | | tests/test_dbreader.py::test_simple_db_engine_skip_file_search_excel_reader | yes | yes | batch._dbengine.simple_db_engine / find_files | #1017 | skip_file_search frames one row per cell | diff --git a/HISTORY.md b/HISTORY.md index 80a46ca7..f4b0b801 100644 --- a/HISTORY.md +++ b/HISTORY.md @@ -6,6 +6,11 @@ `LoadMarker` and `IncrementalChunk`, hosted in cellpy. Shipped loaders stay full-read until they opt in. (#779) +* `load_since()` on `arbin_res`, `arbin_sql`, `neware_txt`, and + `maccor_txt`: returns the harmonized rows read since a marker plus the + next marker, which rewinds to the start of the last cycle so cycle-local + capacity normalisation stays correct. Other loaders stay full-read. (#780) + ## [2.1.5.post6] - 2026-09-25 * `summary_collector(...).plot()` keeps a lone charge or discharge series diff --git a/cellpy/readers/instruments/arbin_res.py b/cellpy/readers/instruments/arbin_res.py index 550194ec..925026d8 100644 --- a/cellpy/readers/instruments/arbin_res.py +++ b/cellpy/readers/instruments/arbin_res.py @@ -1,1494 +1,1530 @@ -"""arbin res-type data files. - -Vendor metadata mapping ("no silent drops"): the Arbin -``Global_Table`` columns and what cellpy does with them: - -=============================== ============================================== -Global_Table column destination -=============================== ============================================== -Channel_Index ``meta_test_dependent.channel_index`` -Creator ``meta_test_dependent.creator`` -Item_ID ``meta_test_dependent.test_ID`` (tester id; - provenance — the compact grouping key is - ``Data.active_test_id``) -Schedule_File_Name ``meta_test_dependent.schedule_file_name`` -Start_DateTime ``meta_common.start_datetime`` -Test_Name ``Data.test_name`` (orphan attribute; no home - in the metadata model yet) -Comments ``meta_common.comment`` (when non-empty and - the box still holds its default) -Test_ID raw ``test_id`` column during intra-file - multi-test stitching (``_merge``); overwritten - with the compact key after load -Applications_Path, deliberately dropped (tester housekeeping, -Channel_Number, Channel_Type, no metadata value) -DAQ_Index, Log_* flags, -Mapped_Aux_* columns -=============================== ============================================== -""" -import cellpy.config as config - -import logging -import os -import pathlib -import platform -import shutil -import sys -import tempfile -import time -import warnings - -import numpy as np -import pandas as pd -import sqlalchemy as sa - -from cellpy import prms -from cellpy.exceptions import LoaderError, NullData, OptionalDependencyError -from cellpy.parameters.internal_settings import HeaderDict, get_headers_normal -from cellpy.readers.data_structures import ( - Data, - FileID, - check64bit, - humanize_bytes, - xldate_as_datetime, -) -from cellpy.readers.instruments.base import MINIMUM_SELECTION, BaseLoader - -# TODO: use InstrumentSettings (dataclass) from internal_settings instead of HeaderDict. - -DEBUG_MODE = config.reader.diagnostics -ALLOW_MULTI_TEST_FILE = False -USE_SQLALCHEMY_ACCESS_ENGINE = True - -# Select odbc module -ODBC = prms._odbc -SEARCH_FOR_ODBC_DRIVERS = prms._search_for_odbc_driver - -_use_subprocess = config.instruments.Arbin.use_subprocess -_detect_subprocess_need = config.instruments.Arbin.detect_subprocess_need - -_is_posix = False -_is_macos = False -if os.name == "posix": - _is_posix = True -current_platform = platform.system() -if current_platform == "Darwin": - _is_macos = True - -if DEBUG_MODE: - logging.debug("DEBUG_MODE") - logging.debug(f"ODBC: {ODBC}") - logging.debug(f"SEARCH_FOR_ODBC_DRIVERS: {SEARCH_FOR_ODBC_DRIVERS}") - logging.debug(f"use_subprocess: {_use_subprocess}") - logging.debug(f"detect_subprocess_need: {_detect_subprocess_need}") - logging.debug(f"current_platform: {current_platform}") - -if _detect_subprocess_need: - logging.debug("detect_subprocess_need is True: checking versions") - python_version, os_version = platform.architecture() - if python_version == "64bit" and config.instruments.Arbin.office_version == "32bit": - logging.debug("python 64bit and office 32bit -> setting use_subprocess to True") - _use_subprocess = True - -if _use_subprocess and not _is_posix: - # The Windows users most likely have a strange custom path to mdbtools etc. - logging.debug( - "using subprocess (most likely mdbtools) on non-posix (most likely windows)" - ) - if not config.instruments.Arbin.sub_process_path: - _sub_process_path = str(prms.sub_process_path) - else: - _sub_process_path = str(config.instruments.Arbin.sub_process_path) - -if _is_posix: - _sub_process_path = "mdb-export" - -_MDB_EXPORT_COMMAND = "mdb-export" - - -def mdb_export_unavailable_reason(sub_process_path: str | None = None) -> str | None: - """Return a reason if the posix ``mdb-export`` command is missing. - - Windows bundled ``mdb-export.exe`` paths are left alone. Callers that - need a hard failure should use ``require_mdb_export``. - """ - if sub_process_path is None: - if not _is_posix: - return None - path = _sub_process_path - else: - path = sub_process_path - if str(path) != _MDB_EXPORT_COMMAND: - return None - if shutil.which(_MDB_EXPORT_COMMAND): - return None - return ( - "Reading Arbin .res on Linux/macOS needs mdbtools (provides `mdb-export`). " - "Debian/Ubuntu: apt install mdbtools. macOS: brew install mdbtools." - ) - - -def require_mdb_export(sub_process_path: str | None = None) -> None: - """Raise when posix loads need ``mdb-export`` and it is not on PATH.""" - reason = mdb_export_unavailable_reason(sub_process_path) - if reason is None: - return - raise OptionalDependencyError(reason) - - -try: - driver_dll = config.instruments.Arbin.odbc_driver -except AttributeError: - driver_dll = None - -if ODBC == "pyodbc": - try: - import pyodbc as dbloader - except ImportError: - warnings.warn("COULD NOT LOAD DBLOADER!", ImportWarning) - dbloader = None - -elif ODBC == "pypyodbc": - try: - import pypyodbc as dbloader - except ImportError: - warnings.warn("COULD NOT LOAD DBLOADER!", ImportWarning) - dbloader = None - -if DEBUG_MODE: - logging.debug(f"dbloader: {dbloader}") - - -# Names of the tables in the .res db that is used by cellpy -TABLE_NAMES = { - "normal": "Channel_Normal_Table", - "global": "Global_Table", - "statistic": "Channel_Statistic_Table", - "aux_global": "Aux_Global_Data_Table", - "aux": "Auxiliary_Table", -} - -SUMMARY_HEADERS_RENAMING_DICT = { - "test_id_txt": "Test_ID", - "data_point_txt": "Data_Point", - "vmax_on_cycle_txt": "Vmax_On_Cycle", - "charge_time_txt": "Charge_Time", - "discharge_time_txt": "Discharge_Time", -} - -NORMAL_HEADERS_RENAMING_DICT = { - "aci_phase_angle_txt": "ACI_Phase_Angle", - "ref_aci_phase_angle_txt": "Reference_ACI_Phase_Angle", - "ac_impedance_txt": "AC_Impedance", - "ref_ac_impedance_txt": "Reference_AC_Impedance", - "charge_capacity_txt": "Charge_Capacity", - "charge_energy_txt": "Charge_Energy", - "current_txt": "Current", - "cycle_index_txt": "Cycle_Index", - "data_point_txt": "Data_Point", - "datetime_txt": "DateTime", - "discharge_capacity_txt": "Discharge_Capacity", - "discharge_energy_txt": "Discharge_Energy", - "internal_resistance_txt": "Internal_Resistance", - "is_fc_data_txt": "Is_FC_Data", - "step_index_txt": "Step_Index", - "sub_step_index_txt": "Sub_Step_Index", # new - "step_time_txt": "Step_Time", - "sub_step_time_txt": "Sub_Step_Time", # new - "test_id_txt": "Test_ID", - "test_time_txt": "Test_Time", - "voltage_txt": "Voltage", - "ref_voltage_txt": "Reference_Voltage", # new - "dv_dt_txt": "dV/dt", - "frequency_txt": "Frequency", # new - "amplitude_txt": "Amplitude", # new -} - - -class DataLoader(BaseLoader): - """Class for loading arbin-data from res-files. - - Parameters from configuration (`config.instruments.Arbin`):: - - - max_res_filesize: break if file size exceeds this limit. - - chunk_size: size of chunks to load. - - max_chunks: max number of chunks to load. - - use_subprocess: use mdbtools or not. - - detect_subprocess_need: detect if mdbtools is needed. - - sub_process_path: path to mdbtools (or similar). - - office_version: version of office (32 or 64 bit). - - """ - - instrument_name = "arbin_res" - raw_ext = "res" - - def __init__(self, *args, **kwargs): - # could use __init__(self, cellpydata_object) and - # set self.logger = cellpydata_object.logger etc. - # then remember to include that as prm in "out of class" functions - # self.prms = prms - self.raw_ext = "res" - self.logger = logging.getLogger(__name__) - # use the following prm to limit to loading only - # one cycle or from cycle>x to cycle reports to log (debug)""" - - if not any([DEBUG_MODE]): - return run_data - - if DEBUG_MODE: - new_cols = run_data.raw.columns - for col in self.arbin_headers_normal: - if col not in new_cols: - logging.debug(f"Missing col: {col}") - # data.raw[col] = np.nan - return run_data - - def repair(self, file_name): - """try to repair a broken/corrupted file""" - raise NotImplemented - - def _query_table(self, table_name, conn, sql=None): - from sqlalchemy import create_engine, text - - self.logger.debug(f"reading {table_name}") - if sql is None: - sql = f"select * from {table_name}" - self.logger.debug(f"sql statement: {sql}") - with conn.connect() as connection: - df = pd.read_sql_query(sql=sa.text(sql), con=connection) - return df - - def _make_name_from_frame(self, df, aux_index, data_type, dx_dt=False): - df_names = df.loc[ - (df[self.arbin_headers_aux_global.aux_index_txt] == aux_index) - & (df[self.arbin_headers_aux_global.data_type_txt] == data_type), - :, - ] - unit = df_names[self.arbin_headers_aux_global.aux_unit_txt].values[0] - nick = ( - df_names[self.arbin_headers_aux_global.aux_name_txt].values[0] or aux_index - ) - if dx_dt: - name = f"aux_d_{nick}_dt_u_d{unit}_dt" - else: - name = f"aux_{nick}_u_{unit}" - return name - - def _loader_win( - self, - file_name, - temp_filename, - *args, - bad_steps=None, - dataset_number=None, - data_points=None, - **kwargs, - ): - conn = None - - table_name_global = TABLE_NAMES["global"] - table_name_aux_global = TABLE_NAMES["aux_global"] - table_name_aux = TABLE_NAMES["aux"] - - table_name_normal = TABLE_NAMES["normal"] - - if DEBUG_MODE: - time_0 = time.time() - - conn = self._get_connection_or_engine(temp_filename) - try: - self.logger.debug("reading global data table") - - global_data_df = self._query_table(table_name=table_name_global, conn=conn) - tests = global_data_df[self.arbin_headers_normal.test_id_txt] - number_of_sets = len(tests) - self.logger.debug(f"number of datasets: {number_of_sets}") - - if dataset_number is not None: - self.logger.info(f"Dataset number given: {dataset_number}") - self.logger.info(f"Available dataset numbers: {tests}") - # check if dataset_number is valid - # - - else: - dataset_number = None - - data = self._init_data(file_name, global_data_df, dataset_number) - self.logger.debug("reading raw-data") - test_id = data._internal_test_number - - # --------- read stats-data (summary-data) --------------------- - - # --------- read raw-data (normal-data) ------------------------ - length_of_test, normal_df = self._load_res_normal_table( - conn, test_id, bad_steps, data_points - ) - # --------- read auxiliary data (aux-data) --------------------- - normal_df = self._load_win_res_auxiliary_table( - conn, normal_df, table_name_aux, table_name_aux_global, test_id - ) - # FIX: error in order by since datetime is not accurate enough (also need sorting on test-time) - # sorting dataframe: - normal_df = normal_df.sort_values( - by=[ - self.arbin_headers_normal.datetime_txt, - self.arbin_headers_normal.test_time_txt, - ], - ascending=True, - ) - # TODO 216: add order by on test_time as well in sql query - summary_df = self._load_res_summary_table(conn, test_id) - if summary_df.empty and config.reader.use_cellpy_stat_file: - txt = "\nCould not find any summary (stats-file)!" - txt += "\n -> issue make_summary(use_cellpy_stat_file=False)" - logging.debug(txt) - # TODO: Enforce creating a summary df or modify renaming summary df (post process part) - # normal_df = normal_df.set_index("Data_Point") - - data.summary = summary_df - if DEBUG_MODE: - mem_usage = normal_df.memory_usage() - logging.debug( - f"memory usage for " - f"loaded data: \n{mem_usage}" - f"\ntotal: {humanize_bytes(mem_usage.sum())}" - ) - logging.debug(f"time used: {(time.time() - time_0):2.4f} s") - - data.raw = normal_df - data.raw_data_files_length.append(length_of_test) - return data - finally: - if conn is not None: - conn.dispose() - - def _load_win_res_auxiliary_table( - self, conn, normal_df, table_name_aux, table_name_aux_global, test_id - ): - aux_global_data_df = self._query_table(table_name_aux_global, conn) - if not aux_global_data_df.empty: - aux_df = self._get_aux_df(conn, test_id, table_name_aux) - aux_df, aux_global_data_df = self._aux_to_wide(aux_df, aux_global_data_df) - aux_df = self._rename_aux_cols(aux_df, aux_global_data_df) - - if not aux_df.empty: - normal_df = self._join_aux_to_normal(aux_df, normal_df) - return normal_df - - def _load_posix_res_auxiliary_table(self, aux_global_data_df, aux_df, normal_df): - if not aux_global_data_df.empty: - aux_df, aux_global_data_df = self._aux_to_wide(aux_df, aux_global_data_df) - aux_df = self._rename_aux_cols(aux_df, aux_global_data_df) - - if not aux_df.empty: - normal_df = self._join_aux_to_normal(aux_df, normal_df) - return normal_df - - def _join_aux_to_normal(self, aux_df, normal_df): - # TODO: clean up setting index (Data_Point). This is currently done in _post_process after - # the column names are changed to cellpy-column names ("data_point"). - # It also keeps a copy of the "data_point" - # column. And is that really necessary. - normal_df.set_index(self.arbin_headers_normal.data_point_txt, inplace=True) - normal_df = normal_df.join(aux_df, how="left") - normal_df.reset_index(inplace=True) - return normal_df - - def _rename_aux_cols(self, aux_df, aux_global_data_df): - aux_dfs = [] - if self.arbin_headers_aux.x_value_txt in aux_df.columns: - aux_df_x = aux_df[self.arbin_headers_aux.x_value_txt].copy() - aux_df_x.columns = [ - self._make_name_from_frame(aux_global_data_df, z[1], z[0]) - for z in aux_df_x.columns - ] - aux_dfs.append(aux_df_x) - if self.arbin_headers_aux.x_dt_value in aux_df.columns: - aux_df_dx_dt = aux_df[self.arbin_headers_aux.x_dt_value].copy() - aux_df_dx_dt.columns = [ - self._make_name_from_frame(aux_global_data_df, z[1], z[0], True) - for z in aux_df_dx_dt.columns - ] - aux_dfs.append(aux_df_dx_dt) - aux_df = pd.concat(aux_dfs, axis=1) - return aux_df - - def _aux_to_wide(self, aux_df, aux_global_data_df): - aux_df = aux_df.drop(self.arbin_headers_aux.test_id_txt, axis=1) - keys = [ - self.arbin_headers_aux.data_point_txt, - self.arbin_headers_aux.aux_index_txt, - self.arbin_headers_aux.data_type_txt, - ] - aux_df = aux_df.set_index(keys=keys) - aux_df = aux_df.unstack(2).unstack(1).dropna(axis=1) - aux_global_data_df = aux_global_data_df.fillna(0) - return aux_df, aux_global_data_df - - def _get_aux_df(self, conn, test_id, table_name_aux): - columns_txt = "*" - test_numbers = "(" + ",".join([str(tn) for tn in test_id]) + ")" - sql_1 = "select %s " % columns_txt - sql_2 = "from %s " % table_name_aux - sql_3 = f"where {self.arbin_headers_normal.test_id_txt} in {test_numbers}" - sql_4 = "" - sql_aux = sql_1 + sql_2 + sql_3 + sql_4 - aux_df = self._query_table(table_name_aux, conn, sql=sql_aux) - return aux_df - - def _loader_posix( - self, - file_name, - temp_filename, - temp_dir, - *args, - bad_steps=None, - dataset_number=None, - data_points=None, - **kwargs, - ): - # NOTE: this is the main loader for posix systems (macos and linux), but is also used for windows - # if the parameter use_subprocess is set to True (e.g. mdbtools' mdb-export.exe). - # TODO: auxiliary channels (table) - - table_name_global = TABLE_NAMES["global"] - table_name_stats = TABLE_NAMES["statistic"] - table_name_normal = TABLE_NAMES["normal"] - table_name_aux_global = TABLE_NAMES["aux_global"] - table_name_aux = TABLE_NAMES["aux"] - - if _is_posix: - if _is_macos: - self.logger.debug("MAC OSX USING MDBTOOLS") - else: - self.logger.debug("POSIX USING MDBTOOLS") - else: - self.logger.debug("WINDOWS USING SUBPROCESS (probably mdb-export.exe)") - - if DEBUG_MODE: - time_0 = time.time() - - ( - tmp_name_global, - tmp_name_raw, - tmp_name_stats, - tmp_name_aux_global, - tmp_name_aux, - ) = self._create_tmp_files( - table_name_global, - table_name_normal, - table_name_stats, - table_name_aux_global, - table_name_aux, - temp_dir, - temp_filename, - ) - - # use pandas to load in the data - global_data_df = pd.read_csv(tmp_name_global) - tests = global_data_df[self.arbin_headers_normal.test_id_txt] - number_of_sets = len(tests) - self.logger.debug("number of datasets: %i" % number_of_sets) - - if dataset_number is not None: - self.logger.info(f"Dataset number given: {dataset_number}") - self.logger.info(f"Available dataset numbers: {tests}") - else: - dataset_number = None - - data = self._init_data(file_name, global_data_df, dataset_number) - - self.logger.debug("reading raw-data") - - ( - length_of_test, - normal_df, - summary_df, - aux_global_data_df, - aux_df, - ) = self._load_from_tmp_files( - data, - tmp_name_global, - tmp_name_raw, - tmp_name_stats, - tmp_name_aux_global, - tmp_name_aux, - temp_filename, - bad_steps, - data_points, - ) - - # --------- read auxiliary data (aux-data) --------------------- - normal_df = self._load_posix_res_auxiliary_table( - aux_global_data_df, aux_df, normal_df - ) - - if summary_df.empty and config.reader.use_cellpy_stat_file: - txt = "\nCould not find any summary (stats-file)!" - txt += "\n -> issue make_summary(use_cellpy_stat_file=False)" - logging.debug(txt) - # normal_df = normal_df.set_index("Data_Point") - - data.summary = summary_df - if DEBUG_MODE: - mem_usage = normal_df.memory_usage() - logging.debug( - f"memory usage for " - f"loaded data: \n{mem_usage}" - f"\ntotal: {humanize_bytes(mem_usage.sum())}" - ) - logging.debug(f"time used: {(time.time() - time_0):2.4f} s") - - data.raw = normal_df - data.raw_data_files_length.append(length_of_test) - return data - - def _check_size(self): - file_size = os.path.getsize(self.temp_file_path) - hfilesize = humanize_bytes(file_size) - txt = f"File size: {file_size} ({hfilesize})" - self.logger.debug(txt) - if file_size > config.instruments.Arbin.max_res_filesize: - error_message = "\nERROR (loader):\n" - error_message += ( - f"{hfilesize} > {humanize_bytes(config.instruments.Arbin.max_res_filesize)} " - f"- File is too big!\n" - ) - error_message += "(edit config.instruments.Arbin ['max_res_filesize'])\n" - logging.critical(error_message) - return False - return True - - def parse(self, source, **kwargs): - """Vendor stage: read the .res into a frame with **Arbin** names. - - The two-stage counterpart to `loader`, and the point where the - vendor part of arbin ends: reading the Access database (ODBC on Windows, - mdbtools on posix) and assembling the normal data table. Everything - after — the rename to native columns, the datetime conversion — is - declared and handled by ``harmonize()``. - - This is exactly what ``loader()`` produces before ``_post_process``, so - the two share one read path and cannot drift. Scoped to a single test: - multi-test ``.res`` files are the switchover's concern, not the vendor - stage's (``loader()`` still splits and merges them). - - Returns: - The normal table with Arbin column names — ``DateTime`` still an - Excel serial, capacities as the file stores them. - """ - import polars as pl - - self.name = source - self.copy_to_temporary() - - use_mdbtools = _use_subprocess or _is_posix - if use_mdbtools: - data = self._loader_posix( - self.name, - self.temp_file_path, - self.temp_file_path.parent, - **kwargs, - ) - else: - data = self._loader_win(self.name, self.temp_file_path, **kwargs) - - # Keep the Data shell so loader() can finish post_process/merge without - # opening the Access file a second time (#560 Phase C). - self._parsed_data = data - frame = pl.from_pandas(data.raw.reset_index(drop=True)) - # Remember the vendor columns so declarations() can build aux_map / - # dropped without re-reading the Access file. - self._parsed_vendor_columns = tuple(frame.columns) - self._parsed = True - return frame - - def declarations(self): - """Declarations for arbin's normal table. - - Derived, not hand-written: ``get_headers_normal()`` is already a - ``{cellpy attr → Arbin column}`` map — the same shape the configuration - loaders carry as ``normal_headers_renaming_dict`` — so inverting it and - composing with cellpy-core's legacy→native map (via - ``derive_column_maps``) gives Arbin → native and the provenance rule for - free (``Test_ID`` is not mapped onto the framework ``test_id``). - - Wide-aux columns (``aux__u_``) are already on the vendor - frame after the Access read; they are declared via ``aux_map`` so - ``harmonize()`` keeps them under the native - ``aux__`` scheme instead of warn-and-dropping them - (Phase C). - """ - from cellpycore.units import CellpyUnits - - from cellpy.readers.instruments._aux_map import aux_map_from_columns - from cellpy.readers.instruments.config_declarations import derive_column_maps - from cellpy.readers.instruments.declarations import LoaderDeclarations - - if not getattr(self, "_parsed", False): - raise LoaderError( - "arbin_res.declarations() was called before parse()." - ) - - # cellpy attr -> Arbin column is what get_headers_normal() returns; - # derive_column_maps wants attr -> vendor, which is the same thing. - renaming = dict(self.arbin_headers_normal) - column_map, passthrough, _ = derive_column_maps(renaming) - - vendor_columns = list(getattr(self, "_parsed_vendor_columns", ())) - claimed = set(column_map) | set(passthrough) - aux_map = aux_map_from_columns(vendor_columns, already_declared=claimed) - # Provenance / tester housekeeping that the legacy path also drops - # from the "keep" set when renaming — silence the unrecognised warning. - dropped = tuple( - column - for column in ("Test_ID",) - if column in vendor_columns and column not in claimed and column not in aux_map - ) - # Anything still unrecognised after aux_map is a deliberate discard - # (PulseStage*, TC_Counter*, ACR, …) — list it so the flip does not - # spam warnings for every Arbin file with those columns. - still_open = [ - column - for column in vendor_columns - if column not in claimed - and column not in aux_map - and column not in dropped - ] - dropped = dropped + tuple(still_open) - - raw_units = { - key: value - for key, value in self.get_raw_units().items() - if hasattr(CellpyUnits(), key) - } - return LoaderDeclarations( - column_map=column_map, - raw_units=CellpyUnits(**raw_units), - passthrough=passthrough, - aux_map=aux_map, - dropped=dropped, - # Arbin stores DateTime as an Excel serial (days since 1899-12-30); - # harmonize() derives epoch_time_utc from it. This replaces the - # xldate conversion the legacy _post_process ran on the same column. - datetime_kind="excel_serial", - ) - - def loader( - self, - name, - *args, - bad_steps=None, - dataset_number=None, - data_points=None, - increment_cycle_index=True, - **kwargs, - ): - """Loads data from arbin .res files. - - Args: - name (str): path to .res file. - bad_steps (list of tuples): (c, s) tuples of steps s (in cycle c) - to skip loading. - dataset_number (int): the data set number ('Test-ID') to select if you are dealing - with arbin files with more than one data-set. - Defaults to selecting all data-sets and merging them. - data_points (tuple of ints): load only data from data_point[0] to - data_point[1] (use None for infinite). - increment_cycle_index (bool): increment the cycle index if merging several datasets (default True). - - Returns: - new data (Data) - """ - # TODO: @jepe - insert kwargs - current chunk, only normal data, etc - if dataset_number is not None: - self.logger.info(f"Dataset number given: {dataset_number}") - merge = False - else: - merge = True - - cached = getattr(self, "_parsed_data", None) - if cached is not None: - # parse() already ran the Access/mdbtools read. - new_data = cached - self._parsed_data = None - else: - try: - not_too_big = self._check_size() - if not not_too_big: - return None - except Exception as e: - self.logger.debug(f"could not get file size: {e}") - - use_mdbtools = False - if _use_subprocess: - use_mdbtools = True - if _is_posix: - use_mdbtools = True - - if use_mdbtools: - new_data = self._loader_posix( - self.name, - self.temp_file_path, - self.temp_file_path.parent, - *args, - bad_steps=bad_steps, - dataset_number=dataset_number, - data_points=data_points, - **kwargs, - ) - else: - new_data = self._loader_win( - self.name, - self.temp_file_path, - *args, - bad_steps=bad_steps, - dataset_number=dataset_number, - data_points=data_points, - **kwargs, - ) - - new_data = self._post_process(new_data) - if merge: - new_data = self._merge( - new_data, increment_cycle_index=increment_cycle_index - ) - - new_data = self.identify_last_data_point(new_data) - new_data = self._inspect(new_data) - - return new_data - - def _merge(self, data, increment_cycle_index=True): - """Merge data from different data-sets (Test-ID) into one data-set.""" - test_ids = data._internal_test_number - if len(test_ids) == 1: - logging.debug("Only one data-set - no need to merge") - return data - if data.raw.empty: - raise ValueError("No data to merge") - - logging.debug("Merging data (only the normal/raw data)") - grouped = data.raw.groupby(self.cellpy_headers_normal.test_id_txt) - groups = [] - last_data_point = 0 - last_test_time = 0.0 - last_cycle_index = 0 - for test_id, df in grouped: - last = df.iloc[-1] - df[self.cellpy_headers_normal.data_point_txt] += last_data_point - df[self.cellpy_headers_normal.test_time_txt] += last_test_time - if increment_cycle_index: - df[self.cellpy_headers_normal.cycle_index_txt] += last_cycle_index - last_data_point = last[self.cellpy_headers_normal.data_point_txt] - last_test_time = last[self.cellpy_headers_normal.test_time_txt] - last_cycle_index = last[self.cellpy_headers_normal.cycle_index_txt] - groups.append(df) - data.raw = pd.concat(groups, ignore_index=True) - return data - - @staticmethod - def _create_tmp_files( - table_name_global, - table_name_normal, - table_name_stats, - table_name_aux_global, - table_name_aux, - temp_dir, - temp_filename, - ): - import subprocess - - require_mdb_export(_sub_process_path) - # creating tmp-filenames - temp_csv_filename_global = os.path.join(temp_dir, "global_tmp.csv") - temp_csv_filename_normal = os.path.join(temp_dir, "normal_tmp.csv") - temp_csv_filename_stats = os.path.join(temp_dir, "stats_tmp.csv") - temp_csv_filename_aux_global = os.path.join(temp_dir, "aux_global_tmp.csv") - temp_csv_filename_aux = os.path.join(temp_dir, "aux_tmp.csv") - # making the cmds - mdb_prms = [ - (table_name_global, temp_csv_filename_global), - (table_name_normal, temp_csv_filename_normal), - (table_name_stats, temp_csv_filename_stats), - (table_name_aux_global, temp_csv_filename_aux_global), - (table_name_aux, temp_csv_filename_aux), - ] - # executing cmds - for table_name, tmp_file in mdb_prms: - with open(tmp_file, "w") as f: - try: - subprocess.call( - [_sub_process_path, temp_filename, table_name], stdout=f - ) - logging.debug(f"ran mdb-export {str(f)} {table_name}") - except FileNotFoundError: - require_mdb_export(_sub_process_path) - logging.critical( - f"Could not run {_sub_process_path} on {temp_filename}" - ) - raise - return ( - temp_csv_filename_global, - temp_csv_filename_normal, - temp_csv_filename_stats, - temp_csv_filename_aux_global, - temp_csv_filename_aux, - ) - - def _load_from_tmp_files( - self, - data, - temp_csv_filename_global, - temp_csv_filename_normal, - temp_csv_filename_stats, - temp_csv_filename_aux_global, - temp_csv_filename_aux, - temp_filename, - bad_steps, - data_points, - ): - """ - if bad_steps is not None: - if not isinstance(bad_steps, (list, tuple)): - bad_steps = [bad_steps] - for bad_cycle, bad_step in bad_steps: - self.logger.debug(f"bad_step def: [c={bad_cycle}, s={bad_step}]") - sql_4 += "AND NOT (%s=%i " % ( - self.headers_normal.cycle_index_txt, - bad_cycle, - ) - sql_4 += "AND %s=%i) " % (self.headers_normal.step_index_txt, bad_step) - - """ - # should include a more efficient to load the csv (maybe a loop where - # we load only chunks and only keep the parts that fulfill the - # filters (e.g. bad_steps, data_points,...) - normal_df = pd.read_csv(temp_csv_filename_normal) - # filter on test ID - if data._internal_test_number is not None: - normal_df = normal_df[ - normal_df[self.arbin_headers_normal.test_id_txt].isin( - data._internal_test_number - ) - ] - # sort on data point - if prms._sort_if_subprocess: - normal_df = normal_df.sort_values(self.arbin_headers_normal.data_point_txt) - - if bad_steps is not None: - logging.debug("removing bad steps") - if not isinstance(bad_steps, (list, tuple)): - bad_steps = [bad_steps] - if not isinstance(bad_steps[0], (list, tuple)): - bad_steps = [bad_steps] - for bad_cycle, bad_step in bad_steps: - self.logger.debug(f"bad_step def: [c={bad_cycle}, s={bad_step}]") - - selector = ( - normal_df[self.arbin_headers_normal.cycle_index_txt] == bad_cycle - ) & (normal_df[self.arbin_headers_normal.step_index_txt] == bad_step) - - normal_df = normal_df.loc[~selector, :] - - if config.reader.limit_loaded_cycles: - logging.debug("Not yet tested for aux data") - if len(config.reader.limit_loaded_cycles) > 1: - c1, c2 = config.reader.limit_loaded_cycles - selector = ( - normal_df[self.arbin_headers_normal.cycle_index_txt] > c1 - ) & (normal_df[self.arbin_headers_normal.cycle_index_txt] < c2) - - else: - c1 = config.reader.limit_loaded_cycles[0] - selector = normal_df[self.arbin_headers_normal.cycle_index_txt] == c1 - - normal_df = normal_df.loc[selector, :] - - if data_points is not None: - logging.debug("selecting data-point range") - logging.debug("Not yet tested for aux data") - d1, d2 = data_points - - if d1 is not None: - selector = normal_df[self.arbin_headers_normal.data_point_txt] >= d1 - normal_df = normal_df.loc[selector, :] - - if d2 is not None: - selector = normal_df[self.arbin_headers_normal.data_point_txt] <= d2 - normal_df = normal_df.loc[selector, :] - - length_of_test = normal_df.shape[0] - summary_df = pd.read_csv(temp_csv_filename_stats) - aux_global_df = pd.read_csv(temp_csv_filename_aux_global) - aux_df = pd.read_csv(temp_csv_filename_aux) - - # clean up - for f in [ - temp_filename, - temp_csv_filename_stats, - temp_csv_filename_normal, - temp_csv_filename_global, - temp_csv_filename_aux_global, - temp_csv_filename_aux, - ]: - if os.path.isfile(f): - try: - os.remove(f) - except WindowsError as e: - logging.warning(f"could not remove tmp-file\n{f} {e}") - return length_of_test, normal_df, summary_df, aux_global_df, aux_df - - def _init_data(self, file_name, global_data_df, test_no=None): - data = Data() - data.loaded_from = file_name - self.generate_fid() - # name of the .res file it is loaded from: - # data.parent_filename = os.path.basename(file_name) - - if test_no is None: - selected_global_data_df = global_data_df - data._internal_test_number = selected_global_data_df[ - self.arbin_headers_global.test_id_txt - ].values - else: - if not isinstance(test_no, (tuple, list)): - test_no = [test_no] - - selector = global_data_df[self.arbin_headers_global.test_id_txt].isin( - test_no - ) - selected_global_data_df = global_data_df.loc[selector, :] - if selected_global_data_df.empty: - raise NoDataFound(f"Could not find any test with test-ID(s) {test_no}") - data._internal_test_number = test_no - - # only picking the first entry (assuming only one cell pr file and channel) - data.channel_index = int( - selected_global_data_df[self.arbin_headers_global.channel_index_txt].values[ - 0 - ] - ) - data.creator = selected_global_data_df[ - self.arbin_headers_global.creator_txt - ].values[0] - data.test_ID = global_data_df[self.arbin_headers_global.item_id_txt].values[0] - data.schedule_file_name = selected_global_data_df[ - self.arbin_headers_global.schedule_file_name_txt - ].values[0] - # TODO: convert to datetime: - data.start_datetime = selected_global_data_df[ - self.arbin_headers_global.start_datetime_txt - ].values[0] - data.test_name = selected_global_data_df[ - self.arbin_headers_global.test_name_txt - ].values[0] - - # vendor metadata mining (issue #508, V2-08): Comments -> comment, - # only when non-empty and the box still holds its default. - try: - comments = selected_global_data_df[ - self.arbin_headers_global.comments_txt - ].values[0] - except (KeyError, IndexError): - comments = None - # empty values differ by platform (Windows ODBC: '', Linux mdbtools: - # NaN) - only map a real, non-empty string - if ( - isinstance(comments, str) - and comments.strip() - and data.meta_common.comment in (None, "") - ): - data.meta_common.comment = comments - - data.raw_data_files.append(self.fid) - return data - - def _normal_table_generator(self, **kwargs): - pass - - def _load_res_summary_table(self, conn, test_ids): - table_name_stats = TABLE_NAMES["statistic"] - test_numbers = "(" + ",".join([str(tn) for tn in test_ids]) + ")" - sql = ( - f"select * from {table_name_stats} " - f"where {self.arbin_headers_normal.test_id_txt} in {test_numbers} " - f"order by {self.arbin_headers_normal.test_id_txt}, {self.arbin_headers_normal.data_point_txt}" - ) - summary_df = self._query_table(table_name_stats, conn, sql=sql) - return summary_df - - def _load_res_normal_table(self, conn, test_ids, bad_steps, data_points): - self.logger.debug("starting loading raw-data") - self.logger.debug(f"connection: {conn} internal test-ID: {test_ids}") - self.logger.debug(f"bad steps: {bad_steps}") - - table_name_normal = TABLE_NAMES["normal"] - - if config.reader.select_minimal: # SETTING - columns = MINIMUM_SELECTION - columns_txt = ", ".join(["%s"] * len(columns)) % tuple(columns) - else: - columns_txt = "*" - - sql_1 = f"select {columns_txt} " - sql_2 = f"from {table_name_normal} " - test_numbers = "(" + ",".join([str(tn) for tn in test_ids]) + ")" - sql_3 = f"where {self.arbin_headers_normal.test_id_txt} in {test_numbers}" - sql_4 = " " - - if bad_steps is not None: - if not isinstance(bad_steps, (list, tuple)): - bad_steps = [bad_steps] - if not isinstance(bad_steps[0], (list, tuple)): - bad_steps = [bad_steps] - for bad_cycle, bad_step in bad_steps: - self.logger.debug(f"bad_step def: [c={bad_cycle}, s={bad_step}]") - sql_4 += ( - f"AND NOT ({self.arbin_headers_normal.cycle_index_txt}={bad_cycle} " - ) - sql_4 += f"AND {self.arbin_headers_normal.step_index_txt}={bad_step}) " - - if config.reader.limit_loaded_cycles: - if len(config.reader.limit_loaded_cycles) > 1: - sql_4 += "AND %s>%i " % ( - self.arbin_headers_normal.cycle_index_txt, - config.reader.limit_loaded_cycles[0], - ) - sql_4 += "AND %s<%i " % ( - self.arbin_headers_normal.cycle_index_txt, - config.reader.limit_loaded_cycles[-1], - ) - else: - sql_4 = "AND %s=%i " % ( - self.arbin_headers_normal.cycle_index_txt, - config.reader.limit_loaded_cycles[0], - ) - - if data_points is not None: - d1, d2 = data_points - if d1 is not None: - sql_4 += "AND %s>=%i " % (self.arbin_headers_normal.data_point_txt, d1) - if d2 is not None: - sql_4 += "AND %s<=%i " % (self.arbin_headers_normal.data_point_txt, d2) - - sql_5 = f"order by {self.arbin_headers_normal.datetime_txt}" - sql = sql_1 + sql_2 + sql_3 + sql_4 + sql_5 - - self.logger.debug("INFO ABOUT LOAD RES NORMAL") - self.logger.debug("sql statement: %s" % sql) - - if DEBUG_MODE: - current_memory_usage = sys.getsizeof(self) - self.logger.debug(f"current memory usage: {current_memory_usage}") - - if not config.instruments.Arbin.chunk_size: - self.logger.debug("no chunk-size given") - # memory here - normal_df = pd.read_sql_query(sql=sa.text(sql), con=conn.connect()) - # memory here - length_of_test = normal_df.shape[0] - else: - self.logger.debug(f"chunk-size: {config.instruments.Arbin.chunk_size}") - self.logger.debug("creating a pd.read_sql_query generator") - - normal_df_reader = pd.read_sql_query( - sql=sa.text(sql), - con=conn.connect(), - chunksize=config.instruments.Arbin.chunk_size, - ) - normal_df = None - chunk_number = 0 - self.logger.debug("created pandas sql reader") - self.logger.debug("iterating chunk-wise") - for i, chunk in enumerate(normal_df_reader): - self.logger.debug(f"iteration number {i}") - if config.instruments.Arbin.max_chunks: - self.logger.debug( - f"max number of chunks mode " - f"({config.instruments.Arbin.max_chunks})" - ) - if chunk_number < config.instruments.Arbin.max_chunks: - normal_df = pd.concat([normal_df, chunk], ignore_index=True) - self.logger.debug( - f"chunk {i} of {config.instruments.Arbin.max_chunks}" - ) - else: - break - else: - try: - normal_df = pd.concat([normal_df, chunk], ignore_index=True) - self.logger.debug("concatenated new chunk") - except MemoryError: - self.logger.error( - " - Could not read complete file (MemoryError)." - ) - self.logger.error( - f"Last successfully loaded chunk number: {chunk_number}" - ) - self.logger.error( - f"Chunk size: {config.instruments.Arbin.chunk_size}" - ) - break - chunk_number += 1 - length_of_test = normal_df.shape[0] - self.logger.debug(f"finished iterating (#rows: {length_of_test})") - - self.logger.debug(f"loaded to normal_df (length = {length_of_test})") - self.logger.debug(f"Headers:\n{normal_df.columns}") - if normal_df is None: - default_headers = [v for v in self.arbin_headers_normal.values()] - normal_df = pd.DataFrame(columns=default_headers) - return length_of_test, normal_df - - -def _check_loader_aux(): - from pathlib import Path - - from cellpy import log - - log.setup_logging(default_level="CRITICAL") - p = Path(r"C:\scripts\cellpy_dev_resources\2020_jinpeng_aux_temperature") - f1 = p / "BIT_LFP5p12s_Pack02_CAP_Cyc200_T25_Nov23.res" - f2 = p / "BIT_LFP50_12S1P_SOP_0_97_T5_cyc200_3500W_20191231.res" - f3 = p / "TJP_LR1865SZ_OCV_19_Cyc150_T25_201105.res" - - n = DataLoader().loader(f1) - print(n[0].raw.tail()) - - -def _check_loader_empty_normal(): - from cellpy import log - - log.setup_logging(default_level="CRITICAL") - - a = DataLoader() - cols = a.arbin_headers_normal - df = pd.DataFrame(columns=cols.values()) - print(df) - print(df.empty) - - -def _check_multi(): - import pathlib - import cellpy - - f = r"C:\scripting\cellpy_dev_resources\dev_data\arbin_multi\20230531_NG27_02_cc_01.res" - out = r"C:\scripting\cellpy_dev_resources\dev_data\arbin_multi\20230531_NG27_02_cc_01.xlsx" - p = pathlib.Path(f) - c = cellpy.get(p) - c.to_excel(out, raw=True) - - -def _noodle(): - import pandas as pd - - df = pd.DataFrame( - { - "a": [1, 2, 3, 4, 5], - "b": [1, 2, 3, 4, 5], - "c": [1, 1, 1, 1, 2], - } - ) - print(df) - - df2 = df[df["c"].isin([1])] - print(" new ".center(80, "-")) - print(df2) - - -if __name__ == "__main__": - print(" arbin-res-py ".center(80, "=")) - _noodle() - print(" finished ".center(80, "=")) +"""arbin res-type data files. + +Vendor metadata mapping ("no silent drops"): the Arbin +``Global_Table`` columns and what cellpy does with them: + +=============================== ============================================== +Global_Table column destination +=============================== ============================================== +Channel_Index ``meta_test_dependent.channel_index`` +Creator ``meta_test_dependent.creator`` +Item_ID ``meta_test_dependent.test_ID`` (tester id; + provenance — the compact grouping key is + ``Data.active_test_id``) +Schedule_File_Name ``meta_test_dependent.schedule_file_name`` +Start_DateTime ``meta_common.start_datetime`` +Test_Name ``Data.test_name`` (orphan attribute; no home + in the metadata model yet) +Comments ``meta_common.comment`` (when non-empty and + the box still holds its default) +Test_ID raw ``test_id`` column during intra-file + multi-test stitching (``_merge``); overwritten + with the compact key after load +Applications_Path, deliberately dropped (tester housekeeping, +Channel_Number, Channel_Type, no metadata value) +DAQ_Index, Log_* flags, +Mapped_Aux_* columns +=============================== ============================================== +""" +import cellpy.config as config + +import logging +import os +import pathlib +import platform +import shutil +import sys +import tempfile +import time +import warnings + +import numpy as np +import pandas as pd +import sqlalchemy as sa + +from cellpy import prms +from cellpy.exceptions import LoaderError, NullData, OptionalDependencyError +from cellpy.parameters.internal_settings import HeaderDict, get_headers_normal +from cellpy.readers.data_structures import ( + Data, + FileID, + check64bit, + humanize_bytes, + xldate_as_datetime, +) +from cellpy.readers.instruments.base import MINIMUM_SELECTION, BaseLoader + +# TODO: use InstrumentSettings (dataclass) from internal_settings instead of HeaderDict. + +DEBUG_MODE = config.reader.diagnostics +ALLOW_MULTI_TEST_FILE = False +USE_SQLALCHEMY_ACCESS_ENGINE = True + +# Select odbc module +ODBC = prms._odbc +SEARCH_FOR_ODBC_DRIVERS = prms._search_for_odbc_driver + +_use_subprocess = config.instruments.Arbin.use_subprocess +_detect_subprocess_need = config.instruments.Arbin.detect_subprocess_need + +_is_posix = False +_is_macos = False +if os.name == "posix": + _is_posix = True +current_platform = platform.system() +if current_platform == "Darwin": + _is_macos = True + +if DEBUG_MODE: + logging.debug("DEBUG_MODE") + logging.debug(f"ODBC: {ODBC}") + logging.debug(f"SEARCH_FOR_ODBC_DRIVERS: {SEARCH_FOR_ODBC_DRIVERS}") + logging.debug(f"use_subprocess: {_use_subprocess}") + logging.debug(f"detect_subprocess_need: {_detect_subprocess_need}") + logging.debug(f"current_platform: {current_platform}") + +if _detect_subprocess_need: + logging.debug("detect_subprocess_need is True: checking versions") + python_version, os_version = platform.architecture() + if python_version == "64bit" and config.instruments.Arbin.office_version == "32bit": + logging.debug("python 64bit and office 32bit -> setting use_subprocess to True") + _use_subprocess = True + +if _use_subprocess and not _is_posix: + # The Windows users most likely have a strange custom path to mdbtools etc. + logging.debug( + "using subprocess (most likely mdbtools) on non-posix (most likely windows)" + ) + if not config.instruments.Arbin.sub_process_path: + _sub_process_path = str(prms.sub_process_path) + else: + _sub_process_path = str(config.instruments.Arbin.sub_process_path) + +if _is_posix: + _sub_process_path = "mdb-export" + +_MDB_EXPORT_COMMAND = "mdb-export" + + +def mdb_export_unavailable_reason(sub_process_path: str | None = None) -> str | None: + """Return a reason if the posix ``mdb-export`` command is missing. + + Windows bundled ``mdb-export.exe`` paths are left alone. Callers that + need a hard failure should use ``require_mdb_export``. + """ + if sub_process_path is None: + if not _is_posix: + return None + path = _sub_process_path + else: + path = sub_process_path + if str(path) != _MDB_EXPORT_COMMAND: + return None + if shutil.which(_MDB_EXPORT_COMMAND): + return None + return ( + "Reading Arbin .res on Linux/macOS needs mdbtools (provides `mdb-export`). " + "Debian/Ubuntu: apt install mdbtools. macOS: brew install mdbtools." + ) + + +def require_mdb_export(sub_process_path: str | None = None) -> None: + """Raise when posix loads need ``mdb-export`` and it is not on PATH.""" + reason = mdb_export_unavailable_reason(sub_process_path) + if reason is None: + return + raise OptionalDependencyError(reason) + + +try: + driver_dll = config.instruments.Arbin.odbc_driver +except AttributeError: + driver_dll = None + +if ODBC == "pyodbc": + try: + import pyodbc as dbloader + except ImportError: + warnings.warn("COULD NOT LOAD DBLOADER!", ImportWarning) + dbloader = None + +elif ODBC == "pypyodbc": + try: + import pypyodbc as dbloader + except ImportError: + warnings.warn("COULD NOT LOAD DBLOADER!", ImportWarning) + dbloader = None + +if DEBUG_MODE: + logging.debug(f"dbloader: {dbloader}") + + +# Names of the tables in the .res db that is used by cellpy +TABLE_NAMES = { + "normal": "Channel_Normal_Table", + "global": "Global_Table", + "statistic": "Channel_Statistic_Table", + "aux_global": "Aux_Global_Data_Table", + "aux": "Auxiliary_Table", +} + +SUMMARY_HEADERS_RENAMING_DICT = { + "test_id_txt": "Test_ID", + "data_point_txt": "Data_Point", + "vmax_on_cycle_txt": "Vmax_On_Cycle", + "charge_time_txt": "Charge_Time", + "discharge_time_txt": "Discharge_Time", +} + +NORMAL_HEADERS_RENAMING_DICT = { + "aci_phase_angle_txt": "ACI_Phase_Angle", + "ref_aci_phase_angle_txt": "Reference_ACI_Phase_Angle", + "ac_impedance_txt": "AC_Impedance", + "ref_ac_impedance_txt": "Reference_AC_Impedance", + "charge_capacity_txt": "Charge_Capacity", + "charge_energy_txt": "Charge_Energy", + "current_txt": "Current", + "cycle_index_txt": "Cycle_Index", + "data_point_txt": "Data_Point", + "datetime_txt": "DateTime", + "discharge_capacity_txt": "Discharge_Capacity", + "discharge_energy_txt": "Discharge_Energy", + "internal_resistance_txt": "Internal_Resistance", + "is_fc_data_txt": "Is_FC_Data", + "step_index_txt": "Step_Index", + "sub_step_index_txt": "Sub_Step_Index", # new + "step_time_txt": "Step_Time", + "sub_step_time_txt": "Sub_Step_Time", # new + "test_id_txt": "Test_ID", + "test_time_txt": "Test_Time", + "voltage_txt": "Voltage", + "ref_voltage_txt": "Reference_Voltage", # new + "dv_dt_txt": "dV/dt", + "frequency_txt": "Frequency", # new + "amplitude_txt": "Amplitude", # new +} + + +class DataLoader(BaseLoader): + """Class for loading arbin-data from res-files. + + Parameters from configuration (`config.instruments.Arbin`):: + + - max_res_filesize: break if file size exceeds this limit. + - chunk_size: size of chunks to load. + - max_chunks: max number of chunks to load. + - use_subprocess: use mdbtools or not. + - detect_subprocess_need: detect if mdbtools is needed. + - sub_process_path: path to mdbtools (or similar). + - office_version: version of office (32 or 64 bit). + + """ + + instrument_name = "arbin_res" + raw_ext = "res" + + def __init__(self, *args, **kwargs): + # could use __init__(self, cellpydata_object) and + # set self.logger = cellpydata_object.logger etc. + # then remember to include that as prm in "out of class" functions + # self.prms = prms + self.raw_ext = "res" + self.logger = logging.getLogger(__name__) + # use the following prm to limit to loading only + # one cycle or from cycle>x to cycle reports to log (debug)""" + + if not any([DEBUG_MODE]): + return run_data + + if DEBUG_MODE: + new_cols = run_data.raw.columns + for col in self.arbin_headers_normal: + if col not in new_cols: + logging.debug(f"Missing col: {col}") + # data.raw[col] = np.nan + return run_data + + def repair(self, file_name): + """try to repair a broken/corrupted file""" + raise NotImplemented + + def _query_table(self, table_name, conn, sql=None): + from sqlalchemy import create_engine, text + + self.logger.debug(f"reading {table_name}") + if sql is None: + sql = f"select * from {table_name}" + self.logger.debug(f"sql statement: {sql}") + with conn.connect() as connection: + df = pd.read_sql_query(sql=sa.text(sql), con=connection) + return df + + def _make_name_from_frame(self, df, aux_index, data_type, dx_dt=False): + df_names = df.loc[ + (df[self.arbin_headers_aux_global.aux_index_txt] == aux_index) + & (df[self.arbin_headers_aux_global.data_type_txt] == data_type), + :, + ] + unit = df_names[self.arbin_headers_aux_global.aux_unit_txt].values[0] + nick = ( + df_names[self.arbin_headers_aux_global.aux_name_txt].values[0] or aux_index + ) + if dx_dt: + name = f"aux_d_{nick}_dt_u_d{unit}_dt" + else: + name = f"aux_{nick}_u_{unit}" + return name + + def _loader_win( + self, + file_name, + temp_filename, + *args, + bad_steps=None, + dataset_number=None, + data_points=None, + **kwargs, + ): + conn = None + + table_name_global = TABLE_NAMES["global"] + table_name_aux_global = TABLE_NAMES["aux_global"] + table_name_aux = TABLE_NAMES["aux"] + + table_name_normal = TABLE_NAMES["normal"] + + if DEBUG_MODE: + time_0 = time.time() + + conn = self._get_connection_or_engine(temp_filename) + try: + self.logger.debug("reading global data table") + + global_data_df = self._query_table(table_name=table_name_global, conn=conn) + tests = global_data_df[self.arbin_headers_normal.test_id_txt] + number_of_sets = len(tests) + self.logger.debug(f"number of datasets: {number_of_sets}") + + if dataset_number is not None: + self.logger.info(f"Dataset number given: {dataset_number}") + self.logger.info(f"Available dataset numbers: {tests}") + # check if dataset_number is valid + # + + else: + dataset_number = None + + data = self._init_data(file_name, global_data_df, dataset_number) + self.logger.debug("reading raw-data") + test_id = data._internal_test_number + + # --------- read stats-data (summary-data) --------------------- + + # --------- read raw-data (normal-data) ------------------------ + length_of_test, normal_df = self._load_res_normal_table( + conn, test_id, bad_steps, data_points + ) + # --------- read auxiliary data (aux-data) --------------------- + normal_df = self._load_win_res_auxiliary_table( + conn, normal_df, table_name_aux, table_name_aux_global, test_id + ) + # FIX: error in order by since datetime is not accurate enough (also need sorting on test-time) + # sorting dataframe: + normal_df = normal_df.sort_values( + by=[ + self.arbin_headers_normal.datetime_txt, + self.arbin_headers_normal.test_time_txt, + ], + ascending=True, + ) + # TODO 216: add order by on test_time as well in sql query + summary_df = self._load_res_summary_table(conn, test_id) + if summary_df.empty and config.reader.use_cellpy_stat_file: + txt = "\nCould not find any summary (stats-file)!" + txt += "\n -> issue make_summary(use_cellpy_stat_file=False)" + logging.debug(txt) + # TODO: Enforce creating a summary df or modify renaming summary df (post process part) + # normal_df = normal_df.set_index("Data_Point") + + data.summary = summary_df + if DEBUG_MODE: + mem_usage = normal_df.memory_usage() + logging.debug( + f"memory usage for " + f"loaded data: \n{mem_usage}" + f"\ntotal: {humanize_bytes(mem_usage.sum())}" + ) + logging.debug(f"time used: {(time.time() - time_0):2.4f} s") + + data.raw = normal_df + data.raw_data_files_length.append(length_of_test) + return data + finally: + if conn is not None: + conn.dispose() + + def _load_win_res_auxiliary_table( + self, conn, normal_df, table_name_aux, table_name_aux_global, test_id + ): + aux_global_data_df = self._query_table(table_name_aux_global, conn) + if not aux_global_data_df.empty: + aux_df = self._get_aux_df(conn, test_id, table_name_aux) + aux_df, aux_global_data_df = self._aux_to_wide(aux_df, aux_global_data_df) + aux_df = self._rename_aux_cols(aux_df, aux_global_data_df) + + if not aux_df.empty: + normal_df = self._join_aux_to_normal(aux_df, normal_df) + return normal_df + + def _load_posix_res_auxiliary_table(self, aux_global_data_df, aux_df, normal_df): + if not aux_global_data_df.empty: + aux_df, aux_global_data_df = self._aux_to_wide(aux_df, aux_global_data_df) + aux_df = self._rename_aux_cols(aux_df, aux_global_data_df) + + if not aux_df.empty: + normal_df = self._join_aux_to_normal(aux_df, normal_df) + return normal_df + + def _join_aux_to_normal(self, aux_df, normal_df): + # TODO: clean up setting index (Data_Point). This is currently done in _post_process after + # the column names are changed to cellpy-column names ("data_point"). + # It also keeps a copy of the "data_point" + # column. And is that really necessary. + normal_df.set_index(self.arbin_headers_normal.data_point_txt, inplace=True) + normal_df = normal_df.join(aux_df, how="left") + normal_df.reset_index(inplace=True) + return normal_df + + def _rename_aux_cols(self, aux_df, aux_global_data_df): + aux_dfs = [] + if self.arbin_headers_aux.x_value_txt in aux_df.columns: + aux_df_x = aux_df[self.arbin_headers_aux.x_value_txt].copy() + aux_df_x.columns = [ + self._make_name_from_frame(aux_global_data_df, z[1], z[0]) + for z in aux_df_x.columns + ] + aux_dfs.append(aux_df_x) + if self.arbin_headers_aux.x_dt_value in aux_df.columns: + aux_df_dx_dt = aux_df[self.arbin_headers_aux.x_dt_value].copy() + aux_df_dx_dt.columns = [ + self._make_name_from_frame(aux_global_data_df, z[1], z[0], True) + for z in aux_df_dx_dt.columns + ] + aux_dfs.append(aux_df_dx_dt) + aux_df = pd.concat(aux_dfs, axis=1) + return aux_df + + def _aux_to_wide(self, aux_df, aux_global_data_df): + aux_df = aux_df.drop(self.arbin_headers_aux.test_id_txt, axis=1) + keys = [ + self.arbin_headers_aux.data_point_txt, + self.arbin_headers_aux.aux_index_txt, + self.arbin_headers_aux.data_type_txt, + ] + aux_df = aux_df.set_index(keys=keys) + aux_df = aux_df.unstack(2).unstack(1).dropna(axis=1) + aux_global_data_df = aux_global_data_df.fillna(0) + return aux_df, aux_global_data_df + + def _get_aux_df(self, conn, test_id, table_name_aux): + columns_txt = "*" + test_numbers = "(" + ",".join([str(tn) for tn in test_id]) + ")" + sql_1 = "select %s " % columns_txt + sql_2 = "from %s " % table_name_aux + sql_3 = f"where {self.arbin_headers_normal.test_id_txt} in {test_numbers}" + sql_4 = "" + sql_aux = sql_1 + sql_2 + sql_3 + sql_4 + aux_df = self._query_table(table_name_aux, conn, sql=sql_aux) + return aux_df + + def _loader_posix( + self, + file_name, + temp_filename, + temp_dir, + *args, + bad_steps=None, + dataset_number=None, + data_points=None, + **kwargs, + ): + # NOTE: this is the main loader for posix systems (macos and linux), but is also used for windows + # if the parameter use_subprocess is set to True (e.g. mdbtools' mdb-export.exe). + # TODO: auxiliary channels (table) + + table_name_global = TABLE_NAMES["global"] + table_name_stats = TABLE_NAMES["statistic"] + table_name_normal = TABLE_NAMES["normal"] + table_name_aux_global = TABLE_NAMES["aux_global"] + table_name_aux = TABLE_NAMES["aux"] + + if _is_posix: + if _is_macos: + self.logger.debug("MAC OSX USING MDBTOOLS") + else: + self.logger.debug("POSIX USING MDBTOOLS") + else: + self.logger.debug("WINDOWS USING SUBPROCESS (probably mdb-export.exe)") + + if DEBUG_MODE: + time_0 = time.time() + + ( + tmp_name_global, + tmp_name_raw, + tmp_name_stats, + tmp_name_aux_global, + tmp_name_aux, + ) = self._create_tmp_files( + table_name_global, + table_name_normal, + table_name_stats, + table_name_aux_global, + table_name_aux, + temp_dir, + temp_filename, + ) + + # use pandas to load in the data + global_data_df = pd.read_csv(tmp_name_global) + tests = global_data_df[self.arbin_headers_normal.test_id_txt] + number_of_sets = len(tests) + self.logger.debug("number of datasets: %i" % number_of_sets) + + if dataset_number is not None: + self.logger.info(f"Dataset number given: {dataset_number}") + self.logger.info(f"Available dataset numbers: {tests}") + else: + dataset_number = None + + data = self._init_data(file_name, global_data_df, dataset_number) + + self.logger.debug("reading raw-data") + + ( + length_of_test, + normal_df, + summary_df, + aux_global_data_df, + aux_df, + ) = self._load_from_tmp_files( + data, + tmp_name_global, + tmp_name_raw, + tmp_name_stats, + tmp_name_aux_global, + tmp_name_aux, + temp_filename, + bad_steps, + data_points, + ) + + # --------- read auxiliary data (aux-data) --------------------- + normal_df = self._load_posix_res_auxiliary_table( + aux_global_data_df, aux_df, normal_df + ) + + if summary_df.empty and config.reader.use_cellpy_stat_file: + txt = "\nCould not find any summary (stats-file)!" + txt += "\n -> issue make_summary(use_cellpy_stat_file=False)" + logging.debug(txt) + # normal_df = normal_df.set_index("Data_Point") + + data.summary = summary_df + if DEBUG_MODE: + mem_usage = normal_df.memory_usage() + logging.debug( + f"memory usage for " + f"loaded data: \n{mem_usage}" + f"\ntotal: {humanize_bytes(mem_usage.sum())}" + ) + logging.debug(f"time used: {(time.time() - time_0):2.4f} s") + + data.raw = normal_df + data.raw_data_files_length.append(length_of_test) + return data + + def _check_size(self): + file_size = os.path.getsize(self.temp_file_path) + hfilesize = humanize_bytes(file_size) + txt = f"File size: {file_size} ({hfilesize})" + self.logger.debug(txt) + if file_size > config.instruments.Arbin.max_res_filesize: + error_message = "\nERROR (loader):\n" + error_message += ( + f"{hfilesize} > {humanize_bytes(config.instruments.Arbin.max_res_filesize)} " + f"- File is too big!\n" + ) + error_message += "(edit config.instruments.Arbin ['max_res_filesize'])\n" + logging.critical(error_message) + return False + return True + + def parse(self, source, **kwargs): + """Vendor stage: read the .res into a frame with **Arbin** names. + + The two-stage counterpart to `loader`, and the point where the + vendor part of arbin ends: reading the Access database (ODBC on Windows, + mdbtools on posix) and assembling the normal data table. Everything + after — the rename to native columns, the datetime conversion — is + declared and handled by ``harmonize()``. + + This is exactly what ``loader()`` produces before ``_post_process``, so + the two share one read path and cannot drift. Scoped to a single test: + multi-test ``.res`` files are the switchover's concern, not the vendor + stage's (``loader()`` still splits and merges them). + + Returns: + The normal table with Arbin column names — ``DateTime`` still an + Excel serial, capacities as the file stores them. + """ + import polars as pl + + self.name = source + self.copy_to_temporary() + + use_mdbtools = _use_subprocess or _is_posix + if use_mdbtools: + data = self._loader_posix( + self.name, + self.temp_file_path, + self.temp_file_path.parent, + **kwargs, + ) + else: + data = self._loader_win(self.name, self.temp_file_path, **kwargs) + + # Keep the Data shell so loader() can finish post_process/merge without + # opening the Access file a second time (#560 Phase C). + self._parsed_data = data + frame = pl.from_pandas(data.raw.reset_index(drop=True)) + # Remember the vendor columns so declarations() can build aux_map / + # dropped without re-reading the Access file. + self._parsed_vendor_columns = tuple(frame.columns) + self._parsed = True + return frame + + def load_since(self, source, marker=None): + """Rows with ``Data_Point`` above the marker (`SupportsIncrementalLoad`, #780). + + Seeks on ``LoadMarker.last_source_datapoint_num`` through the + ``data_points`` filter the normal-table read already has. The + returned marker is one below the first ``Data_Point`` of the last + cycle read, so the next call re-reads that cycle whole (see + ``cellpy.readers.instruments.incremental``). Scoped to a single test + like ``parse()``. + """ + import polars as pl + + from cellpy.readers.instruments.contract import IncrementalChunk, LoadMarker + from cellpy.readers.instruments.harmonize import harmonize + from cellpy.readers.instruments.incremental import last_cycle_start, vendor_column + + since = None if marker is None else marker.last_source_datapoint_num + if marker is None: + marker = LoadMarker() + data_points = None if since is None else (int(since) + 1, None) + + vendor = self.parse(source, data_points=data_points) + self._parsed_data = None + if vendor.height == 0: + return IncrementalChunk(new_raw=pl.DataFrame(), marker=marker) + + declarations = self.declarations() + new_raw = harmonize(vendor, declarations, strict=False) + rewind = last_cycle_start(vendor, vendor_column(declarations, "cycle_num")) + datapoint_column = vendor_column(declarations, "datapoint_num") + cycle_start = int(vendor.get_column(datapoint_column)[rewind]) + return IncrementalChunk( + new_raw=new_raw, + marker=LoadMarker(last_source_datapoint_num=cycle_start - 1), + ) + + def declarations(self): + """Declarations for arbin's normal table. + + Derived, not hand-written: ``get_headers_normal()`` is already a + ``{cellpy attr → Arbin column}`` map — the same shape the configuration + loaders carry as ``normal_headers_renaming_dict`` — so inverting it and + composing with cellpy-core's legacy→native map (via + ``derive_column_maps``) gives Arbin → native and the provenance rule for + free (``Test_ID`` is not mapped onto the framework ``test_id``). + + Wide-aux columns (``aux__u_``) are already on the vendor + frame after the Access read; they are declared via ``aux_map`` so + ``harmonize()`` keeps them under the native + ``aux__`` scheme instead of warn-and-dropping them + (Phase C). + """ + from cellpycore.units import CellpyUnits + + from cellpy.readers.instruments._aux_map import aux_map_from_columns + from cellpy.readers.instruments.config_declarations import derive_column_maps + from cellpy.readers.instruments.declarations import LoaderDeclarations + + if not getattr(self, "_parsed", False): + raise LoaderError( + "arbin_res.declarations() was called before parse()." + ) + + # cellpy attr -> Arbin column is what get_headers_normal() returns; + # derive_column_maps wants attr -> vendor, which is the same thing. + renaming = dict(self.arbin_headers_normal) + column_map, passthrough, _ = derive_column_maps(renaming) + + vendor_columns = list(getattr(self, "_parsed_vendor_columns", ())) + claimed = set(column_map) | set(passthrough) + aux_map = aux_map_from_columns(vendor_columns, already_declared=claimed) + # Provenance / tester housekeeping that the legacy path also drops + # from the "keep" set when renaming — silence the unrecognised warning. + dropped = tuple( + column + for column in ("Test_ID",) + if column in vendor_columns and column not in claimed and column not in aux_map + ) + # Anything still unrecognised after aux_map is a deliberate discard + # (PulseStage*, TC_Counter*, ACR, …) — list it so the flip does not + # spam warnings for every Arbin file with those columns. + still_open = [ + column + for column in vendor_columns + if column not in claimed + and column not in aux_map + and column not in dropped + ] + dropped = dropped + tuple(still_open) + + raw_units = { + key: value + for key, value in self.get_raw_units().items() + if hasattr(CellpyUnits(), key) + } + return LoaderDeclarations( + column_map=column_map, + raw_units=CellpyUnits(**raw_units), + passthrough=passthrough, + aux_map=aux_map, + dropped=dropped, + # Arbin stores DateTime as an Excel serial (days since 1899-12-30); + # harmonize() derives epoch_time_utc from it. This replaces the + # xldate conversion the legacy _post_process ran on the same column. + datetime_kind="excel_serial", + ) + + def loader( + self, + name, + *args, + bad_steps=None, + dataset_number=None, + data_points=None, + increment_cycle_index=True, + **kwargs, + ): + """Loads data from arbin .res files. + + Args: + name (str): path to .res file. + bad_steps (list of tuples): (c, s) tuples of steps s (in cycle c) + to skip loading. + dataset_number (int): the data set number ('Test-ID') to select if you are dealing + with arbin files with more than one data-set. + Defaults to selecting all data-sets and merging them. + data_points (tuple of ints): load only data from data_point[0] to + data_point[1] (use None for infinite). + increment_cycle_index (bool): increment the cycle index if merging several datasets (default True). + + Returns: + new data (Data) + """ + # TODO: @jepe - insert kwargs - current chunk, only normal data, etc + if dataset_number is not None: + self.logger.info(f"Dataset number given: {dataset_number}") + merge = False + else: + merge = True + + cached = getattr(self, "_parsed_data", None) + if cached is not None: + # parse() already ran the Access/mdbtools read. + new_data = cached + self._parsed_data = None + else: + try: + not_too_big = self._check_size() + if not not_too_big: + return None + except Exception as e: + self.logger.debug(f"could not get file size: {e}") + + use_mdbtools = False + if _use_subprocess: + use_mdbtools = True + if _is_posix: + use_mdbtools = True + + if use_mdbtools: + new_data = self._loader_posix( + self.name, + self.temp_file_path, + self.temp_file_path.parent, + *args, + bad_steps=bad_steps, + dataset_number=dataset_number, + data_points=data_points, + **kwargs, + ) + else: + new_data = self._loader_win( + self.name, + self.temp_file_path, + *args, + bad_steps=bad_steps, + dataset_number=dataset_number, + data_points=data_points, + **kwargs, + ) + + new_data = self._post_process(new_data) + if merge: + new_data = self._merge( + new_data, increment_cycle_index=increment_cycle_index + ) + + new_data = self.identify_last_data_point(new_data) + new_data = self._inspect(new_data) + + return new_data + + def _merge(self, data, increment_cycle_index=True): + """Merge data from different data-sets (Test-ID) into one data-set.""" + test_ids = data._internal_test_number + if len(test_ids) == 1: + logging.debug("Only one data-set - no need to merge") + return data + if data.raw.empty: + raise ValueError("No data to merge") + + logging.debug("Merging data (only the normal/raw data)") + grouped = data.raw.groupby(self.cellpy_headers_normal.test_id_txt) + groups = [] + last_data_point = 0 + last_test_time = 0.0 + last_cycle_index = 0 + for test_id, df in grouped: + last = df.iloc[-1] + df[self.cellpy_headers_normal.data_point_txt] += last_data_point + df[self.cellpy_headers_normal.test_time_txt] += last_test_time + if increment_cycle_index: + df[self.cellpy_headers_normal.cycle_index_txt] += last_cycle_index + last_data_point = last[self.cellpy_headers_normal.data_point_txt] + last_test_time = last[self.cellpy_headers_normal.test_time_txt] + last_cycle_index = last[self.cellpy_headers_normal.cycle_index_txt] + groups.append(df) + data.raw = pd.concat(groups, ignore_index=True) + return data + + @staticmethod + def _create_tmp_files( + table_name_global, + table_name_normal, + table_name_stats, + table_name_aux_global, + table_name_aux, + temp_dir, + temp_filename, + ): + import subprocess + + require_mdb_export(_sub_process_path) + # creating tmp-filenames + temp_csv_filename_global = os.path.join(temp_dir, "global_tmp.csv") + temp_csv_filename_normal = os.path.join(temp_dir, "normal_tmp.csv") + temp_csv_filename_stats = os.path.join(temp_dir, "stats_tmp.csv") + temp_csv_filename_aux_global = os.path.join(temp_dir, "aux_global_tmp.csv") + temp_csv_filename_aux = os.path.join(temp_dir, "aux_tmp.csv") + # making the cmds + mdb_prms = [ + (table_name_global, temp_csv_filename_global), + (table_name_normal, temp_csv_filename_normal), + (table_name_stats, temp_csv_filename_stats), + (table_name_aux_global, temp_csv_filename_aux_global), + (table_name_aux, temp_csv_filename_aux), + ] + # executing cmds + for table_name, tmp_file in mdb_prms: + with open(tmp_file, "w") as f: + try: + subprocess.call( + [_sub_process_path, temp_filename, table_name], stdout=f + ) + logging.debug(f"ran mdb-export {str(f)} {table_name}") + except FileNotFoundError: + require_mdb_export(_sub_process_path) + logging.critical( + f"Could not run {_sub_process_path} on {temp_filename}" + ) + raise + return ( + temp_csv_filename_global, + temp_csv_filename_normal, + temp_csv_filename_stats, + temp_csv_filename_aux_global, + temp_csv_filename_aux, + ) + + def _load_from_tmp_files( + self, + data, + temp_csv_filename_global, + temp_csv_filename_normal, + temp_csv_filename_stats, + temp_csv_filename_aux_global, + temp_csv_filename_aux, + temp_filename, + bad_steps, + data_points, + ): + """ + if bad_steps is not None: + if not isinstance(bad_steps, (list, tuple)): + bad_steps = [bad_steps] + for bad_cycle, bad_step in bad_steps: + self.logger.debug(f"bad_step def: [c={bad_cycle}, s={bad_step}]") + sql_4 += "AND NOT (%s=%i " % ( + self.headers_normal.cycle_index_txt, + bad_cycle, + ) + sql_4 += "AND %s=%i) " % (self.headers_normal.step_index_txt, bad_step) + + """ + # should include a more efficient to load the csv (maybe a loop where + # we load only chunks and only keep the parts that fulfill the + # filters (e.g. bad_steps, data_points,...) + normal_df = pd.read_csv(temp_csv_filename_normal) + # filter on test ID + if data._internal_test_number is not None: + normal_df = normal_df[ + normal_df[self.arbin_headers_normal.test_id_txt].isin( + data._internal_test_number + ) + ] + # sort on data point + if prms._sort_if_subprocess: + normal_df = normal_df.sort_values(self.arbin_headers_normal.data_point_txt) + + if bad_steps is not None: + logging.debug("removing bad steps") + if not isinstance(bad_steps, (list, tuple)): + bad_steps = [bad_steps] + if not isinstance(bad_steps[0], (list, tuple)): + bad_steps = [bad_steps] + for bad_cycle, bad_step in bad_steps: + self.logger.debug(f"bad_step def: [c={bad_cycle}, s={bad_step}]") + + selector = ( + normal_df[self.arbin_headers_normal.cycle_index_txt] == bad_cycle + ) & (normal_df[self.arbin_headers_normal.step_index_txt] == bad_step) + + normal_df = normal_df.loc[~selector, :] + + if config.reader.limit_loaded_cycles: + logging.debug("Not yet tested for aux data") + if len(config.reader.limit_loaded_cycles) > 1: + c1, c2 = config.reader.limit_loaded_cycles + selector = ( + normal_df[self.arbin_headers_normal.cycle_index_txt] > c1 + ) & (normal_df[self.arbin_headers_normal.cycle_index_txt] < c2) + + else: + c1 = config.reader.limit_loaded_cycles[0] + selector = normal_df[self.arbin_headers_normal.cycle_index_txt] == c1 + + normal_df = normal_df.loc[selector, :] + + if data_points is not None: + logging.debug("selecting data-point range") + logging.debug("Not yet tested for aux data") + d1, d2 = data_points + + if d1 is not None: + selector = normal_df[self.arbin_headers_normal.data_point_txt] >= d1 + normal_df = normal_df.loc[selector, :] + + if d2 is not None: + selector = normal_df[self.arbin_headers_normal.data_point_txt] <= d2 + normal_df = normal_df.loc[selector, :] + + length_of_test = normal_df.shape[0] + summary_df = pd.read_csv(temp_csv_filename_stats) + aux_global_df = pd.read_csv(temp_csv_filename_aux_global) + aux_df = pd.read_csv(temp_csv_filename_aux) + + # clean up + for f in [ + temp_filename, + temp_csv_filename_stats, + temp_csv_filename_normal, + temp_csv_filename_global, + temp_csv_filename_aux_global, + temp_csv_filename_aux, + ]: + if os.path.isfile(f): + try: + os.remove(f) + except WindowsError as e: + logging.warning(f"could not remove tmp-file\n{f} {e}") + return length_of_test, normal_df, summary_df, aux_global_df, aux_df + + def _init_data(self, file_name, global_data_df, test_no=None): + data = Data() + data.loaded_from = file_name + self.generate_fid() + # name of the .res file it is loaded from: + # data.parent_filename = os.path.basename(file_name) + + if test_no is None: + selected_global_data_df = global_data_df + data._internal_test_number = selected_global_data_df[ + self.arbin_headers_global.test_id_txt + ].values + else: + if not isinstance(test_no, (tuple, list)): + test_no = [test_no] + + selector = global_data_df[self.arbin_headers_global.test_id_txt].isin( + test_no + ) + selected_global_data_df = global_data_df.loc[selector, :] + if selected_global_data_df.empty: + raise NoDataFound(f"Could not find any test with test-ID(s) {test_no}") + data._internal_test_number = test_no + + # only picking the first entry (assuming only one cell pr file and channel) + data.channel_index = int( + selected_global_data_df[self.arbin_headers_global.channel_index_txt].values[ + 0 + ] + ) + data.creator = selected_global_data_df[ + self.arbin_headers_global.creator_txt + ].values[0] + data.test_ID = global_data_df[self.arbin_headers_global.item_id_txt].values[0] + data.schedule_file_name = selected_global_data_df[ + self.arbin_headers_global.schedule_file_name_txt + ].values[0] + # TODO: convert to datetime: + data.start_datetime = selected_global_data_df[ + self.arbin_headers_global.start_datetime_txt + ].values[0] + data.test_name = selected_global_data_df[ + self.arbin_headers_global.test_name_txt + ].values[0] + + # vendor metadata mining (issue #508, V2-08): Comments -> comment, + # only when non-empty and the box still holds its default. + try: + comments = selected_global_data_df[ + self.arbin_headers_global.comments_txt + ].values[0] + except (KeyError, IndexError): + comments = None + # empty values differ by platform (Windows ODBC: '', Linux mdbtools: + # NaN) - only map a real, non-empty string + if ( + isinstance(comments, str) + and comments.strip() + and data.meta_common.comment in (None, "") + ): + data.meta_common.comment = comments + + data.raw_data_files.append(self.fid) + return data + + def _normal_table_generator(self, **kwargs): + pass + + def _load_res_summary_table(self, conn, test_ids): + table_name_stats = TABLE_NAMES["statistic"] + test_numbers = "(" + ",".join([str(tn) for tn in test_ids]) + ")" + sql = ( + f"select * from {table_name_stats} " + f"where {self.arbin_headers_normal.test_id_txt} in {test_numbers} " + f"order by {self.arbin_headers_normal.test_id_txt}, {self.arbin_headers_normal.data_point_txt}" + ) + summary_df = self._query_table(table_name_stats, conn, sql=sql) + return summary_df + + def _load_res_normal_table(self, conn, test_ids, bad_steps, data_points): + self.logger.debug("starting loading raw-data") + self.logger.debug(f"connection: {conn} internal test-ID: {test_ids}") + self.logger.debug(f"bad steps: {bad_steps}") + + table_name_normal = TABLE_NAMES["normal"] + + if config.reader.select_minimal: # SETTING + columns = MINIMUM_SELECTION + columns_txt = ", ".join(["%s"] * len(columns)) % tuple(columns) + else: + columns_txt = "*" + + sql_1 = f"select {columns_txt} " + sql_2 = f"from {table_name_normal} " + test_numbers = "(" + ",".join([str(tn) for tn in test_ids]) + ")" + sql_3 = f"where {self.arbin_headers_normal.test_id_txt} in {test_numbers}" + sql_4 = " " + + if bad_steps is not None: + if not isinstance(bad_steps, (list, tuple)): + bad_steps = [bad_steps] + if not isinstance(bad_steps[0], (list, tuple)): + bad_steps = [bad_steps] + for bad_cycle, bad_step in bad_steps: + self.logger.debug(f"bad_step def: [c={bad_cycle}, s={bad_step}]") + sql_4 += ( + f"AND NOT ({self.arbin_headers_normal.cycle_index_txt}={bad_cycle} " + ) + sql_4 += f"AND {self.arbin_headers_normal.step_index_txt}={bad_step}) " + + if config.reader.limit_loaded_cycles: + if len(config.reader.limit_loaded_cycles) > 1: + sql_4 += "AND %s>%i " % ( + self.arbin_headers_normal.cycle_index_txt, + config.reader.limit_loaded_cycles[0], + ) + sql_4 += "AND %s<%i " % ( + self.arbin_headers_normal.cycle_index_txt, + config.reader.limit_loaded_cycles[-1], + ) + else: + sql_4 = "AND %s=%i " % ( + self.arbin_headers_normal.cycle_index_txt, + config.reader.limit_loaded_cycles[0], + ) + + if data_points is not None: + d1, d2 = data_points + if d1 is not None: + sql_4 += "AND %s>=%i " % (self.arbin_headers_normal.data_point_txt, d1) + if d2 is not None: + sql_4 += "AND %s<=%i " % (self.arbin_headers_normal.data_point_txt, d2) + + sql_5 = f"order by {self.arbin_headers_normal.datetime_txt}" + sql = sql_1 + sql_2 + sql_3 + sql_4 + sql_5 + + self.logger.debug("INFO ABOUT LOAD RES NORMAL") + self.logger.debug("sql statement: %s" % sql) + + if DEBUG_MODE: + current_memory_usage = sys.getsizeof(self) + self.logger.debug(f"current memory usage: {current_memory_usage}") + + if not config.instruments.Arbin.chunk_size: + self.logger.debug("no chunk-size given") + # memory here + normal_df = pd.read_sql_query(sql=sa.text(sql), con=conn.connect()) + # memory here + length_of_test = normal_df.shape[0] + else: + self.logger.debug(f"chunk-size: {config.instruments.Arbin.chunk_size}") + self.logger.debug("creating a pd.read_sql_query generator") + + normal_df_reader = pd.read_sql_query( + sql=sa.text(sql), + con=conn.connect(), + chunksize=config.instruments.Arbin.chunk_size, + ) + normal_df = None + chunk_number = 0 + self.logger.debug("created pandas sql reader") + self.logger.debug("iterating chunk-wise") + for i, chunk in enumerate(normal_df_reader): + self.logger.debug(f"iteration number {i}") + if config.instruments.Arbin.max_chunks: + self.logger.debug( + f"max number of chunks mode " + f"({config.instruments.Arbin.max_chunks})" + ) + if chunk_number < config.instruments.Arbin.max_chunks: + normal_df = pd.concat([normal_df, chunk], ignore_index=True) + self.logger.debug( + f"chunk {i} of {config.instruments.Arbin.max_chunks}" + ) + else: + break + else: + try: + normal_df = pd.concat([normal_df, chunk], ignore_index=True) + self.logger.debug("concatenated new chunk") + except MemoryError: + self.logger.error( + " - Could not read complete file (MemoryError)." + ) + self.logger.error( + f"Last successfully loaded chunk number: {chunk_number}" + ) + self.logger.error( + f"Chunk size: {config.instruments.Arbin.chunk_size}" + ) + break + chunk_number += 1 + length_of_test = normal_df.shape[0] + self.logger.debug(f"finished iterating (#rows: {length_of_test})") + + self.logger.debug(f"loaded to normal_df (length = {length_of_test})") + self.logger.debug(f"Headers:\n{normal_df.columns}") + if normal_df is None: + default_headers = [v for v in self.arbin_headers_normal.values()] + normal_df = pd.DataFrame(columns=default_headers) + return length_of_test, normal_df + + +def _check_loader_aux(): + from pathlib import Path + + from cellpy import log + + log.setup_logging(default_level="CRITICAL") + p = Path(r"C:\scripts\cellpy_dev_resources\2020_jinpeng_aux_temperature") + f1 = p / "BIT_LFP5p12s_Pack02_CAP_Cyc200_T25_Nov23.res" + f2 = p / "BIT_LFP50_12S1P_SOP_0_97_T5_cyc200_3500W_20191231.res" + f3 = p / "TJP_LR1865SZ_OCV_19_Cyc150_T25_201105.res" + + n = DataLoader().loader(f1) + print(n[0].raw.tail()) + + +def _check_loader_empty_normal(): + from cellpy import log + + log.setup_logging(default_level="CRITICAL") + + a = DataLoader() + cols = a.arbin_headers_normal + df = pd.DataFrame(columns=cols.values()) + print(df) + print(df.empty) + + +def _check_multi(): + import pathlib + import cellpy + + f = r"C:\scripting\cellpy_dev_resources\dev_data\arbin_multi\20230531_NG27_02_cc_01.res" + out = r"C:\scripting\cellpy_dev_resources\dev_data\arbin_multi\20230531_NG27_02_cc_01.xlsx" + p = pathlib.Path(f) + c = cellpy.get(p) + c.to_excel(out, raw=True) + + +def _noodle(): + import pandas as pd + + df = pd.DataFrame( + { + "a": [1, 2, 3, 4, 5], + "b": [1, 2, 3, 4, 5], + "c": [1, 1, 1, 1, 2], + } + ) + print(df) + + df2 = df[df["c"].isin([1])] + print(" new ".center(80, "-")) + print(df2) + + +if __name__ == "__main__": + print(" arbin-res-py ".center(80, "=")) + _noodle() + print(" finished ".center(80, "=")) diff --git a/cellpy/readers/instruments/arbin_sql.py b/cellpy/readers/instruments/arbin_sql.py index b87c87c8..137272d7 100644 --- a/cellpy/readers/instruments/arbin_sql.py +++ b/cellpy/readers/instruments/arbin_sql.py @@ -1,581 +1,624 @@ -"""arbin MS SQL Server data""" -import cellpy.config as config - -import datetime -import logging -import os -import platform -import shutil -import sys -import tempfile -import time -import warnings - -import numpy as np -import pandas as pd -import pyodbc -from dateutil.parser import parse - -from cellpy import prms -from cellpy.parameters.internal_settings import HeaderDict, get_headers_normal -from cellpy.readers.data_structures import ( - Data, - FileID, - check64bit, - humanize_bytes, - xldate_as_datetime, -) -from cellpy.readers.instruments.base import BaseLoader -from cellpy.readers.instruments.arbin_sql_config import arbin_sql_value - -# TODO: @muhammad - get more meta data from the SQL db -# TODO: @jepe - update the batch functionality (including filefinder) -# TODO: @muhammad - make routine for "setting up the SQL Server" so that it is accessible and document it - -DEBUG_MODE = config.reader.diagnostics # not used -ALLOW_MULTI_TEST_FILE = prms._allow_multi_test_file # not used -ODBC = prms._odbc -SEARCH_FOR_ODBC_DRIVERS = prms._search_for_odbc_driver # not used -DATE_TIME_FORMAT = prms._date_time_format - -# Names of the tables in the SQL Server db that is used by cellpy - -# Not used anymore - maybe use a similar dict for the SQL table names (they are hard-coded at the moment) -TABLE_NAMES = { - "normal": "Channel_Normal_Table", - "global": "Global_Table", - "statistic": "Channel_Statistic_Table", - "aux_global": "Aux_Global_Data_Table", - "aux": "Auxiliary_Table", -} - -# Contains several headers not encountered yet in the Arbin SQL Server tables -summary_headers_renaming_dict = { - "test_id_txt": "Test_ID", - "data_point_txt": "Data_Point", - "vmax_on_cycle_txt": "Vmax_On_Cycle", - "charge_time_txt": "Charge_Time", - "discharge_time_txt": "Discharge_Time", -} - -# Contains several headers not encountered yet in the Arbin SQL Server tables -normal_headers_renaming_dict = { - "aci_phase_angle_txt": "ACI_Phase_Angle", - "ref_aci_phase_angle_txt": "Reference_ACI_Phase_Angle", - "ac_impedance_txt": "AC_Impedance", - "ref_ac_impedance_txt": "Reference_AC_Impedance", - "charge_capacity_txt": "Charge_Capacity", - "charge_energy_txt": "Charge_Energy", - "current_txt": "Current", - "cycle_index_txt": "Cycle_ID", - "data_point_txt": "Data_Point", - "datetime_txt": "Date_Time", - "discharge_capacity_txt": "Discharge_Capacity", - "discharge_energy_txt": "Discharge_Energy", - "internal_resistance_txt": "Internal_Resistance", - "is_fc_data_txt": "Is_FC_Data", - "step_index_txt": "Step_ID", - "sub_step_index_txt": "Sub_Step_Index", # new - "step_time_txt": "Step_Time", - "sub_step_time_txt": "Sub_Step_Time", # new - "test_id_txt": "Test_ID", - "test_time_txt": "Test_Time", - "voltage_txt": "Voltage", - "ref_voltage_txt": "Reference_Voltage", # new - "dv_dt_txt": "dV/dt", - "frequency_txt": "Frequency", # new - "amplitude_txt": "Amplitude", # new - "channel_id_txt": "Channel_ID", # new Arbin SQL Server - "data_flag_txt": "Data_Flags", # new Arbin SQL Server - "test_name_txt": "Test_Name", # new Arbin SQL Server -} - - -# Arbin SQL Server table headers (for both data df and stats df) -# -------------------------------------------------------------- -# Test_ID -# Channel_ID -# Date_Time -# Data_Point -# Test_Time -# Step_Time -# Cycle_ID -# Step_ID -# Current -# Voltage -# Charge_Capacity -# Discharge_Capacity -# Charge_Energy -# Discharge_Energy -# Data_Flags -# Test_Name -# (all these headers are now implemented in the internal_settings - - -def from_arbin_to_datetime(n): - if isinstance(n, int): - n = str(n) - ms_component = n[-7:] - date_time_component = n[:-7] - temp = f"{date_time_component}.{ms_component}" - datetime_object = datetime.datetime.fromtimestamp(float(temp)) - time_in_str = datetime_object.strftime(DATE_TIME_FORMAT) - return time_in_str - - -class DataLoader(BaseLoader): - """Class for loading arbin-data from MS SQL server.""" - - instrument_name = "arbin_sql" - _is_db = True - - def __init__(self, *args, **kwargs): - """initiates the ArbinSQLLoader class""" - self.arbin_headers_normal = ( - get_headers_normal() - ) # the column headers defined by Arbin - self.cellpy_headers_normal = ( - get_headers_normal() - ) # the column headers defined by cellpy - self.arbin_headers_global = self.get_headers_global() - self.arbin_headers_aux_global = self.get_headers_aux_global() - self.arbin_headers_aux = self.get_headers_aux() - self.current_chunk = 0 # use this to set chunks to load - self.server = arbin_sql_value("SQL_server") - - @staticmethod - def get_headers_normal(): - """Defines the so-called normal column headings for Arbin SQL Server""" - # covered by cellpy at the moment - return get_headers_normal() - - @staticmethod - def get_headers_aux(): - """Defines the so-called auxiliary table column headings for Arbin SQL Server""" - # not in use (yet) - headers = HeaderDict() - # - aux column headings (specific for Arbin) - headers["test_id_txt"] = "Test_ID" - headers["data_point_txt"] = "Data_Point" - headers["aux_index_txt"] = "Auxiliary_Index" - headers["data_type_txt"] = "Data_Type" - headers["x_value_txt"] = "X" - headers["x_dt_value"] = "dX_dt" - return headers - - @staticmethod - def get_headers_aux_global(): - """Defines the so-called auxiliary global column headings for Arbin SQL Server""" - # not in use yet - headers = HeaderDict() - # - aux global column headings (specific for Arbin) - headers["channel_index_txt"] = "Channel_Index" - headers["aux_index_txt"] = "Auxiliary_Index" - headers["data_type_txt"] = "Data_Type" - headers["aux_name_txt"] = "Nickname" - headers["aux_unit_txt"] = "Unit" - return headers - - @staticmethod - def get_headers_global(): - """Defines the so-called global column headings for Arbin SQL Server""" - # not in use yet - headers = HeaderDict() - # - global column headings (specific for Arbin) - headers["applications_path_txt"] = "Applications_Path" - headers["channel_index_txt"] = "Channel_Index" - headers["channel_number_txt"] = "Channel_Number" - headers["channel_type_txt"] = "Channel_Type" - headers["comments_txt"] = "Comments" - headers["creator_txt"] = "Creator" - headers["daq_index_txt"] = "DAQ_Index" - headers["item_id_txt"] = "Item_ID" - headers["log_aux_data_flag_txt"] = "Log_Aux_Data_Flag" - headers["log_chanstat_data_flag_txt"] = "Log_ChanStat_Data_Flag" - headers["log_event_data_flag_txt"] = "Log_Event_Data_Flag" - headers["log_smart_battery_data_flag_txt"] = "Log_Smart_Battery_Data_Flag" - headers["mapped_aux_conc_cnumber_txt"] = "Mapped_Aux_Conc_CNumber" - headers["mapped_aux_di_cnumber_txt"] = "Mapped_Aux_DI_CNumber" - headers["mapped_aux_do_cnumber_txt"] = "Mapped_Aux_DO_CNumber" - headers["mapped_aux_flow_rate_cnumber_txt"] = "Mapped_Aux_Flow_Rate_CNumber" - headers["mapped_aux_ph_number_txt"] = "Mapped_Aux_PH_Number" - headers["mapped_aux_pressure_number_txt"] = "Mapped_Aux_Pressure_Number" - headers["mapped_aux_temperature_number_txt"] = "Mapped_Aux_Temperature_Number" - headers["mapped_aux_voltage_number_txt"] = "Mapped_Aux_Voltage_Number" - headers["schedule_file_name_txt"] = ( - "Schedule_File_Name" # KEEP FOR CELLPY FILE FORMAT - ) - headers["start_datetime_txt"] = "Start_DateTime" - headers["test_id_txt"] = "Test_ID" # KEEP FOR CELLPY FILE FORMAT - headers["test_name_txt"] = "Test_Name" # KEEP FOR CELLPY FILE FORMAT - return headers - - @staticmethod - def get_raw_units(): - raw_units = dict() - raw_units["current"] = "A" - raw_units["charge"] = "Ah" - raw_units["mass"] = "g" - raw_units["voltage"] = "V" - return raw_units - - @staticmethod - def get_raw_limits(): - """returns a dictionary with resolution limits""" - raw_limits = dict() - raw_limits["current_hard"] = 0.000_000_000_000_1 - raw_limits["current_soft"] = 0.000_01 - raw_limits["stable_current_hard"] = 2.0 - raw_limits["stable_current_soft"] = 4.0 - raw_limits["stable_voltage_hard"] = 2.0 - raw_limits["stable_voltage_soft"] = 4.0 - raw_limits["stable_charge_hard"] = 0.001 - raw_limits["stable_charge_soft"] = 5.0 - raw_limits["ir_change"] = 0.00001 - return raw_limits - - def parse(self, source, **kwargs): - """Vendor stage: query the SQL Server into an Arbin-named frame. - - The two-stage counterpart to `loader`, tapping the same - ``_query_sql`` read. It stops before ``_post_process`` — the rename, the - Arbin-integer datetime decode and the dtype coercion are declared and - handled by ``harmonize()``. No in-repo fixture exists (this needs a live - SQL Server), so the port is exercised at the declaration level; the - column map and ``datetime_kind`` mirror the parity-verified - ``arbin_sql_h5``, which shares the vendor's integer-tick timestamp. - """ - import polars as pl - - self.name = source - self.is_db = True - data_df, _ = self._query_sql(self.name) - self._parsed = True - return pl.from_pandas(data_df.reset_index(drop=True)) - - def declarations(self): - """Declarations for Arbin SQL Server. - - Derived from the module ``normal_headers_renaming_dict`` restricted to - real cellpy headers — the same set the legacy ``_post_process`` renames - by iterating ``arbin_headers_normal`` — so ``derive_column_maps`` gives - Arbin → native with ``Test_ID`` dropped as provenance. ``datetime_kind`` - is ``"arbin_epoch"``: the vendor ``Date_Time`` is integer 100 ns ticks - since the Unix epoch. - """ - from cellpy.readers.instruments.config_declarations import ( - declarations_from_renaming, - ) - - return declarations_from_renaming( - self.get_raw_units(), - normal_headers_renaming_dict, - "arbin_epoch", - getattr(self, "_parsed", False), - "arbin_sql", - ) - - # TODO: rename this (for all instruments) to e.g. load - # TODO: implement more options (bad_cycles, ...) - def loader(self, name, **kwargs): - """returns a Data object with loaded data. - - Loads data from arbin SQL server db. - - Args: - name (str): name of the test - - Returns: - new_tests (list of data objects) - """ - # self.name = name - self.is_db = True - data_df, stat_df = self._query_sql(self.name) - aux_data_df = None # Needs to be implemented - meta_data = None # Should be implemented - - # init data - - # selecting only one value (might implement multi-channel/id use later) - test_id = data_df["Test_ID"].iloc[0] - id_name = f"{arbin_sql_value('SQL_server')}:{name}:{test_id}" - - channel_id = data_df["Channel_ID"].iloc[0] - - data = Data() - data.loaded_from = id_name - data.channel_index = channel_id - data.test_ID = test_id - data.test_name = name - - # The following metadata is not implemented yet for SQL loader: - data.creator = None - data.schedule_file_name = None - data.start_datetime = None # REMARK! convert to datetime when implementing - - # Generating a FileID project: - self.generate_fid() - data.raw_data_files.append(self.fid) - - data.raw = data_df - data.raw_data_files_length.append(len(data_df)) - data.summary = stat_df - data = self._post_process(data) - data = self.identify_last_data_point(data) - - return data - - def _post_process(self, data, **kwargs): - # TODO: move this to parent - - fix_datetime = kwargs.pop("fix_datetime", True) - set_index = kwargs.pop("set_index", True) - rename_headers = kwargs.pop("rename_headers", True) - extract_start_datetime = kwargs.pop("extract_start_datetime", True) - set_dtypes = kwargs.pop("set_dtypes", True) - - # TODO: insert post-processing and div tests here - # - check dtypes - - # Remark that we also set index during saving the file to hdf5 if - # it is not set. - from pprint import pprint - - if rename_headers: - columns = {} - for key in self.arbin_headers_normal: - old_header = normal_headers_renaming_dict.get(key, None) - new_header = self.cellpy_headers_normal[key] - if old_header: - columns[old_header] = new_header - logging.debug( - f"processing cellpy normal header key '{key}':" - f" old_header='{old_header}' -> new_header='{new_header}'" - ) - logging.debug(f"renaming dict: {columns}") - data.raw.rename(index=str, columns=columns, inplace=True) - try: - columns = {} - for key, old_header in summary_headers_renaming_dict.items(): - try: - columns[old_header] = self.cellpy_headers_normal[key] - except KeyError: - columns[old_header] = old_header.lower() - data.summary.rename(index=str, columns=columns, inplace=True) - except Exception as e: - logging.debug(f"Could not rename summary df ::\n{e}") - - if fix_datetime: - h_datetime = self.cellpy_headers_normal.datetime_txt - data.raw[h_datetime] = data.raw[h_datetime].apply(from_arbin_to_datetime) - - if h_datetime in data.summary: - data.summary[h_datetime] = data.summary[h_datetime].apply( - from_arbin_to_datetime - ) - - if set_index: - hdr_data_point = self.cellpy_headers_normal.data_point_txt - if data.raw.index.name != hdr_data_point: - data.raw = data.raw.set_index(hdr_data_point, drop=False) - - if extract_start_datetime: - hdr_date_time = self.arbin_headers_normal.datetime_txt - data.start_datetime = parse(data.raw[hdr_date_time].iat[0][:-7]) - - if set_dtypes: - logging.debug("setting data types") - # test_time_txt = self.cellpy_headers_normal.test_time_txt - # step_time_txt = self.cellpy_headers_normal.step_time_txt - date_time_txt = self.cellpy_headers_normal.datetime_txt - logging.debug("converting to datetime format") - try: - # data.raw[test_time_txt] = pd.to_timedelta(data.raw[test_time_txt]) # cellpy is not ready for this - # data.raw[step_time_txt] = pd.to_timedelta(data.raw[step_time_txt]) # cellpy is not ready for this - data.raw[date_time_txt] = pd.to_datetime( - data.raw[date_time_txt], format=DATE_TIME_FORMAT - ) - except ValueError: - logging.debug("could not convert to datetime format") - - return data - - def _query_sql(self, name): - # TODO: refactor and include optional SQL arguments - name_str = f"('{name}', '')" - con_str = ( - f"Driver={{{arbin_sql_value('SQL_Driver')}}};" - + f"Server={arbin_sql_value('SQL_server')};Trusted_Connection=yes;" - ) - - # TODO: use variable for the name of the main db (ArbinPro8....) - # TODO: consider making a function that searches for correct ArbinPro version - master_q = ( - "SELECT Database_Name, Test_Name FROM " - "ArbinPro8MasterInfo.dbo.TestList_Table WHERE " - f"ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name IN {name_str}" - ) - - conn = pyodbc.connect(con_str) - sql_query = pd.read_sql_query(master_q, conn) - - datas_df = [] - stats_df = [] - - for index, row in sql_query.iterrows(): - # TODO: use variables - see above - # TODO: consider to use f-strings - data_query = ( - "SELECT " - + str(row["Database_Name"]) - + ".dbo.IV_Basic_Table.*, ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name " - "FROM " + str(row["Database_Name"]) + ".dbo.IV_Basic_Table " - "JOIN ArbinPro8MasterInfo.dbo.TestList_Table " - "ON " - + str(row["Database_Name"]) - + ".dbo.IV_Basic_Table.Test_ID = ArbinPro8MasterInfo.dbo.TestList_Table.Test_ID " - "WHERE ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name IN " - + str(name_str) - ) - - stat_query = ( - "SELECT " - + str(row["Database_Name"]) - + ".dbo.StatisticData_Table.*, ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name " - "FROM " + str(row["Database_Name"]) + ".dbo.StatisticData_Table " - "JOIN ArbinPro8MasterInfo.dbo.TestList_Table " - "ON " - + str(row["Database_Name"]) - + ".dbo.StatisticData_Table.Test_ID = ArbinPro8MasterInfo.dbo.TestList_Table.Test_ID " - "WHERE ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name IN " - + str(name_str) - ) - - datas_df.append(pd.read_sql_query(data_query, conn)) - stats_df.append(pd.read_sql_query(stat_query, conn)) - - data_df = pd.concat(datas_df, axis=0) - stat_df = pd.concat(stats_df, axis=0) - - return data_df, stat_df - - -def _check_sql_loader(server: str = None, tests: list = None): - test_name = tuple(tests) + ("",) # neat trick :-) - print(f"** test str: {test_name}") - con_str = "Driver={SQL Server};Server=" + server + ";Trusted_Connection=yes;" - master_q = ( - "SELECT Database_Name, Test_Name FROM " - "ArbinPro8MasterInfo.dbo.TestList_Table WHERE " - f"ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name IN {test_name}" - ) - - conn = pyodbc.connect(con_str) - print("** connected to server") - sql_query = pd.read_sql_query(master_q, conn) - print("** SQL query:") - print(sql_query) - for index, row in sql_query.iterrows(): - # Muhammad, why is it a loop here? - print(f"** index: {index}") - print(f"** row: {row}") - data_query = ( - "SELECT " - + str(row["Database_Name"]) - + ".dbo.IV_Basic_Table.*, ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name " - "FROM " + str(row["Database_Name"]) + ".dbo.IV_Basic_Table " - "JOIN ArbinPro8MasterInfo.dbo.TestList_Table " - "ON " - + str(row["Database_Name"]) - + ".dbo.IV_Basic_Table.Test_ID = ArbinPro8MasterInfo.dbo.TestList_Table.Test_ID " - "WHERE ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name IN " - + str(test_name) - ) - - stat_query = ( - "SELECT " - + str(row["Database_Name"]) - + ".dbo.StatisticData_Table.*, ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name " - "FROM " + str(row["Database_Name"]) + ".dbo.StatisticData_Table " - "JOIN ArbinPro8MasterInfo.dbo.TestList_Table " - "ON " - + str(row["Database_Name"]) - + ".dbo.StatisticData_Table.Test_ID = ArbinPro8MasterInfo.dbo.TestList_Table.Test_ID " - "WHERE ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name IN " - + str(test_name) - ) - print(f"** data query: {data_query}") - print(f"** stat query: {stat_query}") - - # if looping, maybe these should be concatenated? - data_df = pd.read_sql_query(data_query, conn) - stat_df = pd.read_sql_query(stat_query, conn) - - return data_df, stat_df - - -def _check_query(): - import pathlib - - name = ["20201106_HC03B1W_1_cc_01"] - dd, ds = check_sql_loader(arbin_sql_value("SQL_server"), name) - out = pathlib.Path(r"C:\scripts\notebooks\Div") - input("x") - - -def _check_loader(): - print(" Testing connection to arbin sql server ".center(80, "-")) - - sql_loader = DataLoader() - name = "20201106_HC03B1W_1_cc_01" - cell = sql_loader.loader(name) - - return cell - - -def _check_loader_from_outside(): - import matplotlib.pyplot as plt - - from cellpy import cellreader - - name = "20200820_CoFBAT_slurry07B_01_cc_01" - c = cellreader.CellpyCell() - c.set_instrument("arbin_sql") - # print(c) - c.from_raw(name) - # print(c) - c.make_step_table() - c.make_summary() - # print(c) - raw = c.data.raw - steps = c.data.steps - summary = c.data.summary - raw.to_csv(r"C:\scripting\trash\raw.csv", sep=";") - steps.to_csv(r"C:\scripting\trash\steps.csv", sep=";") - summary.to_csv(r"C:\scripting\trash\summary.csv", sep=";") - - n = c.get_number_of_cycles() - print(f"number of cycles: {n}") - - cycle = c.get_cap(1, method="forth") - print(cycle.head()) - # get_cap returns a native CurveCols frame (capacity / potential). - from cellpycore.config import CurveCols - - curve_cols = CurveCols() - cycle.plot(x=curve_cols.capacity, y=curve_cols.potential) - plt.show() - - -def _check_get(): - import cellpy - - name = "20200820_CoFBAT_slurry07B_01_cc_01" - c = cellpy.get(name, instrument="arbin_sql") - print(c) - - -if __name__ == "__main__": - # test_query() - # cell = test_loader() - _check_get() +"""arbin MS SQL Server data""" +import cellpy.config as config + +import datetime +import logging +import os +import platform +import shutil +import sys +import tempfile +import time +import warnings + +import numpy as np +import pandas as pd +import pyodbc +from dateutil.parser import parse + +from cellpy import prms +from cellpy.parameters.internal_settings import HeaderDict, get_headers_normal +from cellpy.readers.data_structures import ( + Data, + FileID, + check64bit, + humanize_bytes, + xldate_as_datetime, +) +from cellpy.readers.instruments.base import BaseLoader +from cellpy.readers.instruments.arbin_sql_config import arbin_sql_value + +# TODO: @muhammad - get more meta data from the SQL db +# TODO: @jepe - update the batch functionality (including filefinder) +# TODO: @muhammad - make routine for "setting up the SQL Server" so that it is accessible and document it + +DEBUG_MODE = config.reader.diagnostics # not used +ALLOW_MULTI_TEST_FILE = prms._allow_multi_test_file # not used +ODBC = prms._odbc +SEARCH_FOR_ODBC_DRIVERS = prms._search_for_odbc_driver # not used +DATE_TIME_FORMAT = prms._date_time_format + +# Names of the tables in the SQL Server db that is used by cellpy + +# Not used anymore - maybe use a similar dict for the SQL table names (they are hard-coded at the moment) +TABLE_NAMES = { + "normal": "Channel_Normal_Table", + "global": "Global_Table", + "statistic": "Channel_Statistic_Table", + "aux_global": "Aux_Global_Data_Table", + "aux": "Auxiliary_Table", +} + +# Contains several headers not encountered yet in the Arbin SQL Server tables +summary_headers_renaming_dict = { + "test_id_txt": "Test_ID", + "data_point_txt": "Data_Point", + "vmax_on_cycle_txt": "Vmax_On_Cycle", + "charge_time_txt": "Charge_Time", + "discharge_time_txt": "Discharge_Time", +} + +# Contains several headers not encountered yet in the Arbin SQL Server tables +normal_headers_renaming_dict = { + "aci_phase_angle_txt": "ACI_Phase_Angle", + "ref_aci_phase_angle_txt": "Reference_ACI_Phase_Angle", + "ac_impedance_txt": "AC_Impedance", + "ref_ac_impedance_txt": "Reference_AC_Impedance", + "charge_capacity_txt": "Charge_Capacity", + "charge_energy_txt": "Charge_Energy", + "current_txt": "Current", + "cycle_index_txt": "Cycle_ID", + "data_point_txt": "Data_Point", + "datetime_txt": "Date_Time", + "discharge_capacity_txt": "Discharge_Capacity", + "discharge_energy_txt": "Discharge_Energy", + "internal_resistance_txt": "Internal_Resistance", + "is_fc_data_txt": "Is_FC_Data", + "step_index_txt": "Step_ID", + "sub_step_index_txt": "Sub_Step_Index", # new + "step_time_txt": "Step_Time", + "sub_step_time_txt": "Sub_Step_Time", # new + "test_id_txt": "Test_ID", + "test_time_txt": "Test_Time", + "voltage_txt": "Voltage", + "ref_voltage_txt": "Reference_Voltage", # new + "dv_dt_txt": "dV/dt", + "frequency_txt": "Frequency", # new + "amplitude_txt": "Amplitude", # new + "channel_id_txt": "Channel_ID", # new Arbin SQL Server + "data_flag_txt": "Data_Flags", # new Arbin SQL Server + "test_name_txt": "Test_Name", # new Arbin SQL Server +} + + +# Arbin SQL Server table headers (for both data df and stats df) +# -------------------------------------------------------------- +# Test_ID +# Channel_ID +# Date_Time +# Data_Point +# Test_Time +# Step_Time +# Cycle_ID +# Step_ID +# Current +# Voltage +# Charge_Capacity +# Discharge_Capacity +# Charge_Energy +# Discharge_Energy +# Data_Flags +# Test_Name +# (all these headers are now implemented in the internal_settings + + +def from_arbin_to_datetime(n): + if isinstance(n, int): + n = str(n) + ms_component = n[-7:] + date_time_component = n[:-7] + temp = f"{date_time_component}.{ms_component}" + datetime_object = datetime.datetime.fromtimestamp(float(temp)) + time_in_str = datetime_object.strftime(DATE_TIME_FORMAT) + return time_in_str + + +class DataLoader(BaseLoader): + """Class for loading arbin-data from MS SQL server.""" + + instrument_name = "arbin_sql" + _is_db = True + + def __init__(self, *args, **kwargs): + """initiates the ArbinSQLLoader class""" + self.arbin_headers_normal = ( + get_headers_normal() + ) # the column headers defined by Arbin + self.cellpy_headers_normal = ( + get_headers_normal() + ) # the column headers defined by cellpy + self.arbin_headers_global = self.get_headers_global() + self.arbin_headers_aux_global = self.get_headers_aux_global() + self.arbin_headers_aux = self.get_headers_aux() + self.current_chunk = 0 # use this to set chunks to load + self.server = arbin_sql_value("SQL_server") + + @staticmethod + def get_headers_normal(): + """Defines the so-called normal column headings for Arbin SQL Server""" + # covered by cellpy at the moment + return get_headers_normal() + + @staticmethod + def get_headers_aux(): + """Defines the so-called auxiliary table column headings for Arbin SQL Server""" + # not in use (yet) + headers = HeaderDict() + # - aux column headings (specific for Arbin) + headers["test_id_txt"] = "Test_ID" + headers["data_point_txt"] = "Data_Point" + headers["aux_index_txt"] = "Auxiliary_Index" + headers["data_type_txt"] = "Data_Type" + headers["x_value_txt"] = "X" + headers["x_dt_value"] = "dX_dt" + return headers + + @staticmethod + def get_headers_aux_global(): + """Defines the so-called auxiliary global column headings for Arbin SQL Server""" + # not in use yet + headers = HeaderDict() + # - aux global column headings (specific for Arbin) + headers["channel_index_txt"] = "Channel_Index" + headers["aux_index_txt"] = "Auxiliary_Index" + headers["data_type_txt"] = "Data_Type" + headers["aux_name_txt"] = "Nickname" + headers["aux_unit_txt"] = "Unit" + return headers + + @staticmethod + def get_headers_global(): + """Defines the so-called global column headings for Arbin SQL Server""" + # not in use yet + headers = HeaderDict() + # - global column headings (specific for Arbin) + headers["applications_path_txt"] = "Applications_Path" + headers["channel_index_txt"] = "Channel_Index" + headers["channel_number_txt"] = "Channel_Number" + headers["channel_type_txt"] = "Channel_Type" + headers["comments_txt"] = "Comments" + headers["creator_txt"] = "Creator" + headers["daq_index_txt"] = "DAQ_Index" + headers["item_id_txt"] = "Item_ID" + headers["log_aux_data_flag_txt"] = "Log_Aux_Data_Flag" + headers["log_chanstat_data_flag_txt"] = "Log_ChanStat_Data_Flag" + headers["log_event_data_flag_txt"] = "Log_Event_Data_Flag" + headers["log_smart_battery_data_flag_txt"] = "Log_Smart_Battery_Data_Flag" + headers["mapped_aux_conc_cnumber_txt"] = "Mapped_Aux_Conc_CNumber" + headers["mapped_aux_di_cnumber_txt"] = "Mapped_Aux_DI_CNumber" + headers["mapped_aux_do_cnumber_txt"] = "Mapped_Aux_DO_CNumber" + headers["mapped_aux_flow_rate_cnumber_txt"] = "Mapped_Aux_Flow_Rate_CNumber" + headers["mapped_aux_ph_number_txt"] = "Mapped_Aux_PH_Number" + headers["mapped_aux_pressure_number_txt"] = "Mapped_Aux_Pressure_Number" + headers["mapped_aux_temperature_number_txt"] = "Mapped_Aux_Temperature_Number" + headers["mapped_aux_voltage_number_txt"] = "Mapped_Aux_Voltage_Number" + headers["schedule_file_name_txt"] = ( + "Schedule_File_Name" # KEEP FOR CELLPY FILE FORMAT + ) + headers["start_datetime_txt"] = "Start_DateTime" + headers["test_id_txt"] = "Test_ID" # KEEP FOR CELLPY FILE FORMAT + headers["test_name_txt"] = "Test_Name" # KEEP FOR CELLPY FILE FORMAT + return headers + + @staticmethod + def get_raw_units(): + raw_units = dict() + raw_units["current"] = "A" + raw_units["charge"] = "Ah" + raw_units["mass"] = "g" + raw_units["voltage"] = "V" + return raw_units + + @staticmethod + def get_raw_limits(): + """returns a dictionary with resolution limits""" + raw_limits = dict() + raw_limits["current_hard"] = 0.000_000_000_000_1 + raw_limits["current_soft"] = 0.000_01 + raw_limits["stable_current_hard"] = 2.0 + raw_limits["stable_current_soft"] = 4.0 + raw_limits["stable_voltage_hard"] = 2.0 + raw_limits["stable_voltage_soft"] = 4.0 + raw_limits["stable_charge_hard"] = 0.001 + raw_limits["stable_charge_soft"] = 5.0 + raw_limits["ir_change"] = 0.00001 + return raw_limits + + def parse(self, source, **kwargs): + """Vendor stage: query the SQL Server into an Arbin-named frame. + + The two-stage counterpart to `loader`, tapping the same + ``_query_sql`` read. It stops before ``_post_process`` — the rename, the + Arbin-integer datetime decode and the dtype coercion are declared and + handled by ``harmonize()``. No in-repo fixture exists (this needs a live + SQL Server), so the port is exercised at the declaration level; the + column map and ``datetime_kind`` mirror the parity-verified + ``arbin_sql_h5``, which shares the vendor's integer-tick timestamp. + """ + import polars as pl + + self.name = source + self.is_db = True + data_df, _ = self._query_sql(self.name, since_data_point=kwargs.pop("since_data_point", None)) + self._parsed = True + return pl.from_pandas(data_df.reset_index(drop=True)) + + def load_since(self, source, marker=None): + """Rows with ``Data_Point`` above the marker (`SupportsIncrementalLoad`, #780). + + Seeks on ``LoadMarker.last_source_datapoint_num`` with one extra + ``WHERE`` clause on the normal-table query. The returned marker is one + below the first ``Data_Point`` of the last cycle read, so the next + call re-reads that cycle whole (see + ``cellpy.readers.instruments.incremental``). + """ + import polars as pl + + from cellpy.readers.instruments.contract import IncrementalChunk, LoadMarker + from cellpy.readers.instruments.harmonize import harmonize + from cellpy.readers.instruments.incremental import last_cycle_start, vendor_column + + since = None if marker is None else marker.last_source_datapoint_num + if marker is None: + marker = LoadMarker() + + vendor = self.parse(source, since_data_point=since) + if vendor.height == 0: + return IncrementalChunk(new_raw=pl.DataFrame(), marker=marker) + + declarations = self.declarations() + new_raw = harmonize(vendor, declarations, strict=False) + rewind = last_cycle_start(vendor, vendor_column(declarations, "cycle_num")) + datapoint_column = vendor_column(declarations, "datapoint_num") + cycle_start = int(vendor.get_column(datapoint_column)[rewind]) + return IncrementalChunk( + new_raw=new_raw, + marker=LoadMarker(last_source_datapoint_num=cycle_start - 1), + ) + + def declarations(self): + """Declarations for Arbin SQL Server. + + Derived from the module ``normal_headers_renaming_dict`` restricted to + real cellpy headers — the same set the legacy ``_post_process`` renames + by iterating ``arbin_headers_normal`` — so ``derive_column_maps`` gives + Arbin → native with ``Test_ID`` dropped as provenance. ``datetime_kind`` + is ``"arbin_epoch"``: the vendor ``Date_Time`` is integer 100 ns ticks + since the Unix epoch. + """ + from cellpy.readers.instruments.config_declarations import ( + declarations_from_renaming, + ) + + return declarations_from_renaming( + self.get_raw_units(), + normal_headers_renaming_dict, + "arbin_epoch", + getattr(self, "_parsed", False), + "arbin_sql", + ) + + # TODO: rename this (for all instruments) to e.g. load + # TODO: implement more options (bad_cycles, ...) + def loader(self, name, **kwargs): + """returns a Data object with loaded data. + + Loads data from arbin SQL server db. + + Args: + name (str): name of the test + + Returns: + new_tests (list of data objects) + """ + # self.name = name + self.is_db = True + data_df, stat_df = self._query_sql(self.name) + aux_data_df = None # Needs to be implemented + meta_data = None # Should be implemented + + # init data + + # selecting only one value (might implement multi-channel/id use later) + test_id = data_df["Test_ID"].iloc[0] + id_name = f"{arbin_sql_value('SQL_server')}:{name}:{test_id}" + + channel_id = data_df["Channel_ID"].iloc[0] + + data = Data() + data.loaded_from = id_name + data.channel_index = channel_id + data.test_ID = test_id + data.test_name = name + + # The following metadata is not implemented yet for SQL loader: + data.creator = None + data.schedule_file_name = None + data.start_datetime = None # REMARK! convert to datetime when implementing + + # Generating a FileID project: + self.generate_fid() + data.raw_data_files.append(self.fid) + + data.raw = data_df + data.raw_data_files_length.append(len(data_df)) + data.summary = stat_df + data = self._post_process(data) + data = self.identify_last_data_point(data) + + return data + + def _post_process(self, data, **kwargs): + # TODO: move this to parent + + fix_datetime = kwargs.pop("fix_datetime", True) + set_index = kwargs.pop("set_index", True) + rename_headers = kwargs.pop("rename_headers", True) + extract_start_datetime = kwargs.pop("extract_start_datetime", True) + set_dtypes = kwargs.pop("set_dtypes", True) + + # TODO: insert post-processing and div tests here + # - check dtypes + + # Remark that we also set index during saving the file to hdf5 if + # it is not set. + from pprint import pprint + + if rename_headers: + columns = {} + for key in self.arbin_headers_normal: + old_header = normal_headers_renaming_dict.get(key, None) + new_header = self.cellpy_headers_normal[key] + if old_header: + columns[old_header] = new_header + logging.debug( + f"processing cellpy normal header key '{key}':" + f" old_header='{old_header}' -> new_header='{new_header}'" + ) + logging.debug(f"renaming dict: {columns}") + data.raw.rename(index=str, columns=columns, inplace=True) + try: + columns = {} + for key, old_header in summary_headers_renaming_dict.items(): + try: + columns[old_header] = self.cellpy_headers_normal[key] + except KeyError: + columns[old_header] = old_header.lower() + data.summary.rename(index=str, columns=columns, inplace=True) + except Exception as e: + logging.debug(f"Could not rename summary df ::\n{e}") + + if fix_datetime: + h_datetime = self.cellpy_headers_normal.datetime_txt + data.raw[h_datetime] = data.raw[h_datetime].apply(from_arbin_to_datetime) + + if h_datetime in data.summary: + data.summary[h_datetime] = data.summary[h_datetime].apply( + from_arbin_to_datetime + ) + + if set_index: + hdr_data_point = self.cellpy_headers_normal.data_point_txt + if data.raw.index.name != hdr_data_point: + data.raw = data.raw.set_index(hdr_data_point, drop=False) + + if extract_start_datetime: + hdr_date_time = self.arbin_headers_normal.datetime_txt + data.start_datetime = parse(data.raw[hdr_date_time].iat[0][:-7]) + + if set_dtypes: + logging.debug("setting data types") + # test_time_txt = self.cellpy_headers_normal.test_time_txt + # step_time_txt = self.cellpy_headers_normal.step_time_txt + date_time_txt = self.cellpy_headers_normal.datetime_txt + logging.debug("converting to datetime format") + try: + # data.raw[test_time_txt] = pd.to_timedelta(data.raw[test_time_txt]) # cellpy is not ready for this + # data.raw[step_time_txt] = pd.to_timedelta(data.raw[step_time_txt]) # cellpy is not ready for this + data.raw[date_time_txt] = pd.to_datetime( + data.raw[date_time_txt], format=DATE_TIME_FORMAT + ) + except ValueError: + logging.debug("could not convert to datetime format") + + return data + + def _query_sql(self, name, since_data_point=None): + """Query the normal and statistics tables for test ``name``. + + Args: + name: the Arbin test name. + since_data_point: when given, only normal-table rows with + ``Data_Point`` strictly above this value are returned + (incremental read, #780). The statistics table is unfiltered. + """ + # TODO: refactor and include optional SQL arguments + name_str = f"('{name}', '')" + con_str = ( + f"Driver={{{arbin_sql_value('SQL_Driver')}}};" + + f"Server={arbin_sql_value('SQL_server')};Trusted_Connection=yes;" + ) + + # TODO: use variable for the name of the main db (ArbinPro8....) + # TODO: consider making a function that searches for correct ArbinPro version + master_q = ( + "SELECT Database_Name, Test_Name FROM " + "ArbinPro8MasterInfo.dbo.TestList_Table WHERE " + f"ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name IN {name_str}" + ) + + conn = pyodbc.connect(con_str) + sql_query = pd.read_sql_query(master_q, conn) + + datas_df = [] + stats_df = [] + + for index, row in sql_query.iterrows(): + # TODO: use variables - see above + # TODO: consider to use f-strings + data_query = ( + "SELECT " + + str(row["Database_Name"]) + + ".dbo.IV_Basic_Table.*, ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name " + "FROM " + str(row["Database_Name"]) + ".dbo.IV_Basic_Table " + "JOIN ArbinPro8MasterInfo.dbo.TestList_Table " + "ON " + + str(row["Database_Name"]) + + ".dbo.IV_Basic_Table.Test_ID = ArbinPro8MasterInfo.dbo.TestList_Table.Test_ID " + "WHERE ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name IN " + + str(name_str) + ) + if since_data_point is not None: + data_query += f" AND {row['Database_Name']}.dbo.IV_Basic_Table.Data_Point > {int(since_data_point)}" + + stat_query = ( + "SELECT " + + str(row["Database_Name"]) + + ".dbo.StatisticData_Table.*, ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name " + "FROM " + str(row["Database_Name"]) + ".dbo.StatisticData_Table " + "JOIN ArbinPro8MasterInfo.dbo.TestList_Table " + "ON " + + str(row["Database_Name"]) + + ".dbo.StatisticData_Table.Test_ID = ArbinPro8MasterInfo.dbo.TestList_Table.Test_ID " + "WHERE ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name IN " + + str(name_str) + ) + + datas_df.append(pd.read_sql_query(data_query, conn)) + stats_df.append(pd.read_sql_query(stat_query, conn)) + + data_df = pd.concat(datas_df, axis=0) + stat_df = pd.concat(stats_df, axis=0) + + return data_df, stat_df + + +def _check_sql_loader(server: str = None, tests: list = None): + test_name = tuple(tests) + ("",) # neat trick :-) + print(f"** test str: {test_name}") + con_str = "Driver={SQL Server};Server=" + server + ";Trusted_Connection=yes;" + master_q = ( + "SELECT Database_Name, Test_Name FROM " + "ArbinPro8MasterInfo.dbo.TestList_Table WHERE " + f"ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name IN {test_name}" + ) + + conn = pyodbc.connect(con_str) + print("** connected to server") + sql_query = pd.read_sql_query(master_q, conn) + print("** SQL query:") + print(sql_query) + for index, row in sql_query.iterrows(): + # Muhammad, why is it a loop here? + print(f"** index: {index}") + print(f"** row: {row}") + data_query = ( + "SELECT " + + str(row["Database_Name"]) + + ".dbo.IV_Basic_Table.*, ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name " + "FROM " + str(row["Database_Name"]) + ".dbo.IV_Basic_Table " + "JOIN ArbinPro8MasterInfo.dbo.TestList_Table " + "ON " + + str(row["Database_Name"]) + + ".dbo.IV_Basic_Table.Test_ID = ArbinPro8MasterInfo.dbo.TestList_Table.Test_ID " + "WHERE ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name IN " + + str(test_name) + ) + + stat_query = ( + "SELECT " + + str(row["Database_Name"]) + + ".dbo.StatisticData_Table.*, ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name " + "FROM " + str(row["Database_Name"]) + ".dbo.StatisticData_Table " + "JOIN ArbinPro8MasterInfo.dbo.TestList_Table " + "ON " + + str(row["Database_Name"]) + + ".dbo.StatisticData_Table.Test_ID = ArbinPro8MasterInfo.dbo.TestList_Table.Test_ID " + "WHERE ArbinPro8MasterInfo.dbo.TestList_Table.Test_Name IN " + + str(test_name) + ) + print(f"** data query: {data_query}") + print(f"** stat query: {stat_query}") + + # if looping, maybe these should be concatenated? + data_df = pd.read_sql_query(data_query, conn) + stat_df = pd.read_sql_query(stat_query, conn) + + return data_df, stat_df + + +def _check_query(): + import pathlib + + name = ["20201106_HC03B1W_1_cc_01"] + dd, ds = check_sql_loader(arbin_sql_value("SQL_server"), name) + out = pathlib.Path(r"C:\scripts\notebooks\Div") + input("x") + + +def _check_loader(): + print(" Testing connection to arbin sql server ".center(80, "-")) + + sql_loader = DataLoader() + name = "20201106_HC03B1W_1_cc_01" + cell = sql_loader.loader(name) + + return cell + + +def _check_loader_from_outside(): + import matplotlib.pyplot as plt + + from cellpy import cellreader + + name = "20200820_CoFBAT_slurry07B_01_cc_01" + c = cellreader.CellpyCell() + c.set_instrument("arbin_sql") + # print(c) + c.from_raw(name) + # print(c) + c.make_step_table() + c.make_summary() + # print(c) + raw = c.data.raw + steps = c.data.steps + summary = c.data.summary + raw.to_csv(r"C:\scripting\trash\raw.csv", sep=";") + steps.to_csv(r"C:\scripting\trash\steps.csv", sep=";") + summary.to_csv(r"C:\scripting\trash\summary.csv", sep=";") + + n = c.get_number_of_cycles() + print(f"number of cycles: {n}") + + cycle = c.get_cap(1, method="forth") + print(cycle.head()) + # get_cap returns a native CurveCols frame (capacity / potential). + from cellpycore.config import CurveCols + + curve_cols = CurveCols() + cycle.plot(x=curve_cols.capacity, y=curve_cols.potential) + plt.show() + + +def _check_get(): + import cellpy + + name = "20200820_CoFBAT_slurry07B_01_cc_01" + c = cellpy.get(name, instrument="arbin_sql") + print(c) + + +if __name__ == "__main__": + # test_query() + # cell = test_loader() + _check_get() diff --git a/cellpy/readers/instruments/base.py b/cellpy/readers/instruments/base.py index f8097c4e..f6a13767 100644 --- a/cellpy/readers/instruments/base.py +++ b/cellpy/readers/instruments/base.py @@ -1,959 +1,1024 @@ -""" -When you make a new loader you have to subclass the Loader class. -Remember also to register it in cellpy.cellreader. - -(for future development, not used very efficiently yet). -""" - -import abc -import logging -import pathlib -import shutil -import tempfile -from abc import ABC -from typing import List, Union - -import pandas as pd - -from cellpy.exceptions import LoaderError, WrongFileVersion -import cellpy.internals.connections -import cellpy.readers.data_structures as core -from cellpy.parameters.internal_settings import headers_normal, merge_raw_units -from cellpy.readers.instruments.configurations import ( - ModelParameters, - register_configuration_from_module, -) -from cellpy.readers.instruments.processors import post_processors, pre_processors -from cellpy.readers.instruments.processors.post_processors import ( - ORDERED_POST_PROCESSING_STEPS, -) - -MINIMUM_SELECTION = [ - "Data_Point", - "Test_Time", - "Step_Time", - "DateTime", - "Step_Index", - "Cycle_Index", - "Current", - "Voltage", - "Charge_Capacity", - "Discharge_Capacity", - "Internal_Resistance", -] - - -# TODO: move this to another module (e.g. inside processors): -def find_delimiter_and_start( - file_name, - separators=None, - checking_length_header=30, - checking_length_whole=200, - check_encoding=True, -): - """Function to automatically detect the delimiter and what line the first data appears on. - - This function is fairly stupid. It reads a window of up to - ``checking_length_whole`` non-empty lines and treats up to - ``checking_length_header`` of the leading ones as a possible header. It - counts the appearances of the different possible delimiters in the data - rows past that header and selects a delimiter if its per-row count is both - uniform and positive. - - The header window is clamped to the actual number of lines read, so a short - file - down to a header line and a single data row - is inspected correctly - rather than running off the end of the sample. - - The first line is defined as where the delimiter is used same number of times (probably a header line). - - Args: - file_name: path to the file. - separators: list of possible delimiters. - checking_length_header: number of lines to check for header. - checking_length_whole: number of lines to check for delimiter. - check_encoding: check encoding. - - Returns: - separator: the delimiter. - first_index: the index of the first line with data. - encoding: the encoding (None if not found or checked). - - Raises: - LoaderError: if the file has no inspectable content, no candidate - delimiter fits the data rows, or the header row cannot be located. - """ - - if separators is None: - separators = [";", "\t", "|", ","] - logging.debug(f"checking internals of the file {file_name}") - - encoding = None - - if check_encoding: - import charset_normalizer - - results = charset_normalizer.from_path( - file_name, - steps=10, # Number of steps/block to extract from my_byte_str - chunk_size=512, # Set block size of each extraction - ) - if results: - r = results.best() - encoding = r.encoding - - with open(file_name, "r") as fin: - lines = [] - for j in range(checking_length_whole): - line = fin.readline() - if not line: - break - if len(line.strip()): - lines.append(line) - - if not lines: - raise LoaderError( - f"could not detect a delimiter in {file_name}: " - "the file has no non-empty lines to inspect" - ) - - separator, number_of_hits = _find_separator( - lines, separators, checking_length_header - ) - - if separator is None: - candidates = ", ".join(repr(s) for s in separators) - raise LoaderError( - f"could not detect a delimiter in {file_name}: none of the " - f"candidate delimiters ({candidates}) appears a consistent, " - "positive number of times across the data rows" - ) - - if separator == "\t": - logging.debug("seperator = TAB") - elif separator == " ": - logging.debug("seperator = SPACE") - else: - logging.debug(f"seperator = {separator}") - - first_index = _find_first_line_whit_delimiter( - checking_length_header, lines, number_of_hits, separator - ) - if first_index is None: - raise LoaderError( - f"detected delimiter {separator!r} in {file_name} but could not " - f"locate the header row within the first {checking_length_header} " - "lines" - ) - logging.debug(f"First line with delimiter: {first_index}") - return separator, first_index, encoding - - -def _find_first_line_whit_delimiter( - checking_length_header, lines, number_of_hits, separator -): - """Return the index of the first line that carries the data-row delimiter count. - - Only the first ``checking_length_header`` lines are searched (the header is - assumed to live there). Returns ``None`` if no line matches, so the caller - can raise a delimiter error that names the file. - """ - first_part = lines[:checking_length_header] - for line_number, line in enumerate(first_part): - if line.count(separator) == number_of_hits: - return line_number - return None - - -def _find_separator(lines, separators, checking_length_header): - """Pick the delimiter from the data rows that follow a possible header. - - The header may be up to ``checking_length_header`` lines, but a short file - cannot hold that many header lines - so the header window is clamped to the - number of lines actually read, always leaving at least one data row to - inspect. The delimiter is the highest-priority candidate whose per-row - count is both uniform and positive across those data rows. - - Returns ``(None, None)`` when no candidate qualifies. - """ - logging.debug("searching for separators") - n = len(lines) - - # reserve up to checking_length_header leading lines as a possible header, - # but never so many that no data row is left to inspect. - header_window = min(checking_length_header, n - 1) if n > 1 else 0 - data_lines = lines[header_window:] - - # the final line can be truncated mid-write, so drop it - but only when a - # data row can still be spared (a header + single data row must keep it). - if len(data_lines) > 1: - data_lines = data_lines[:-1] - - for candidate in separators: - counts = {line.count(candidate) for line in data_lines} - if len(counts) == 1: - number_of_hits = counts.pop() - if number_of_hits > 0: - return candidate, number_of_hits - - return None, None - - -def query_csv( - self, - name, - sep=None, - skiprows=None, - header=None, - encoding=None, - decimal=None, - thousands=None, -): - """function to query a csv file using pandas.read_csv. - - - Args: - name: path to the file. - sep: delimiter. - skiprows: number of lines to skip. - header: number of the header lines. - encoding: encoding. - decimal: character used for decimal in the raw data, defaults to '.'. - thousands: character used for thousands in the raw data, defaults to ','. - - Returns: - pandas.DataFrame - - """ - logging.debug(f"parsing with pandas.read_csv: {name}") - sep = sep or self.sep - skiprows = skiprows or self.skiprows - header = header or self.header - encoding = encoding or self.encoding - decimal = decimal or self.decimal - thousands = thousands or self.thousands - logging.critical(f"{sep=}, {skiprows=}, {header=}, {encoding=}, {decimal=}") - data_df = pd.read_csv( - name, - sep=sep, - skiprows=skiprows, - header=header, - encoding=encoding, - decimal=decimal, - thousands=thousands, - ) - return data_df - - -class AtomicLoad: - """Atomic loading class""" - - instrument_name = "atomic_loader" - - _name = None - _temp_file_path = None - _fid = None - _is_db: bool = False - _copy_also_local: bool = True - _refuse_copying: bool = False - - @property - def is_db(self): - """Is the file stored in the database""" - return self._is_db - - @is_db.setter - def is_db(self, value: bool): - """Is the file stored in the database""" - self._is_db = value - - @property - def refuse_copying(self): - """Should the file be copied to a temporary file""" - return self._refuse_copying - - @refuse_copying.setter - def refuse_copying(self, value: bool): - """Should the file be copied to a temporary file""" - self._refuse_copying = value - - @property - def name(self): - """The name of the file to be loaded""" - return self._name - - @name.setter - def name(self, value): - """The name of the file to be loaded""" - if not self.is_db and not isinstance(value, cellpy.internals.connections.OtherPath): - logging.debug("converting to OtherPath") - value = cellpy.internals.connections.OtherPath(value) - self._name = value - - @property - def temp_file_path(self): # -> Union[cellpy.internals.connections.OtherPath, pathlib.Path] - """The name of the file to be loaded if copied to a temporary file""" - return self._temp_file_path - - @temp_file_path.setter - def temp_file_path(self, value): - """The name of the file to be loaded if copied to a temporary file""" - self._temp_file_path = value - - @property - def fid(self): - """The unique file id""" - if self._fid is None: - self.generate_fid() - return self._fid - - def generate_fid(self, value=None): - """Generate a unique file id""" - if self.is_db: - self._fid = core.FileID(self.name, is_db=True) - elif self._temp_file_path is not None: - self._fid = core.FileID(self.name) - elif self._name is not None: - self._fid = core.FileID(self.name) - elif value is not None: - self._fid = core.FileID(value) - else: - raise ValueError("could not generate fid") - - def copy_to_temporary(self): - """Copy file to a temporary file""" - - logging.debug(f"external file received? {self.name.is_external=}") - if self.name is None: - raise ValueError("no file name given to loader class (self.name is None)") - - if self._refuse_copying: - logging.debug("refusing copying") - self._temp_file_path = self.name - return - - if not self._copy_also_local and not self.name.is_external: - self._temp_file_path = self.name - return - - from cellpy.internals.progress import emit - - emit("copy") - self._temp_file_path = self.name.copy() - - def loader_executor(self, *args, **kwargs): - """Load the file""" - name = args[0] - self.refuse_copying = kwargs.pop("refuse_copying", False) - self.name = name - if not self.is_db: - self.copy_to_temporary() - cellpy_data = self.loader(*args, **kwargs) - return cellpy_data - - def loader(self, *args, **kwargs): - """The method that does the actual loading. - - This method should be overwritten by the specific loader class. - """ - ... - - -class BaseLoader(AtomicLoad, metaclass=abc.ABCMeta): - """Main loading class""" - - instrument_name = "base_loader" - - # TODO: should also include the functions for getting cellpy headers etc - # here - - @staticmethod - @abc.abstractmethod - def get_raw_units() -> dict: - """Units used by the instrument. - - The internal cellpy units are given in the ``cellpy_units`` attribute. - - Returns: - dictionary of units (str) - - Example: - A minimum viable implementation could look like this:: - - @staticmethod - def get_raw_units(): - raw_units = dict() - raw_units["current"] = "A" - raw_units["charge"] = "Ah" - raw_units["mass"] = "g" - raw_units["voltage"] = "V" - return raw_units - - """ - # This is needed for example when converting the capacity to a specific capacity. - # So far, it has been difficult to get any kind of consensus on what the most optimal - # units are for storing cycling data. Therefore, cellpy implements three levels of units: - # 1) the raw units that the data is loaded in already has and 2) the cellpy units used by cellpy - # when generating summaries and related information, and 3) output units that can be set to get the data - # in a specif unit when exporting or creating specific outputs such as ICA. - # - # Comment 2022.09.11:: - # - # still not sure if we should use raw units or cellpy units in the cellpy-files (.h5/ .cellpy). - # Currently, the summary is in cellpy units and the raw and step data is in raw units. If - # you have any input on this topic, let us know. - - pass - - @abc.abstractmethod - def get_raw_limits(self) -> dict: - """Limits used to identify type of step. - - The raw limits are 'epsilons' used to check if the current and/or voltage is stable (for example - for galvanostatic steps, one would expect that the current is stable (constant) and non-zero). - If the (accumulated) change is less than 'epsilon', then cellpy interpret it to be stable. - It is expected that different instruments (with different resolution etc.) have different - resolutions and noice levels, thus different 'epsilons'. - - Returns: - the raw limits (dict) - - """ - pass - - @classmethod - def get_params(cls, parameter: Union[str, None]) -> dict: - """Retrieves parameters needed for facilitating working with the - instrument without registering it. - - Typically, it should include the name and raw_ext. - - Return: parameters or a selected parameter - """ - - return getattr(cls, parameter) - - @abc.abstractmethod - def loader(self, *args, **kwargs) -> list: - """Loads data into a Data object and returns it""" - # This method is used by cellreader through the AtomicLoad.loader_executor method. - # It should be overwritten by the specific loader class. - # - # Notice that it is highly recommended that you don't try to implement .loader_executor yourself - # in your subclass! - pass - - @staticmethod - def identify_last_data_point(data: core.Data) -> core.Data: - """This method is used to find the last record in the data.""" - return core.identify_last_data_point(data) - - -class AutoLoader(BaseLoader): - """Main autoload class. - - This class can be sub-classed if you want to make a data-reader for different type of "easily parsed" files - (for example csv-files). The subclass needs to have at least one - associated CONFIGURATION_MODULE defined and must have the following attributes as minimum:: - - default_model: str = NICK_NAME_OF_DEFAULT_CONFIGURATION_MODULE - supported_models: dict = SUPPORTED_MODELS - - where SUPPORTED_MODELS is a dictionary with ``{"NICK_NAME" : "CONFIGURATION_MODULE_NAME"}`` key-value pairs. - Remark! the NICK_NAME must be in upper-case! - - It is also possible to set these in a custom pre_init method:: - - @classmethod - def pre_init(cls): - cls.default_model: str = NICK_NAME_OF_DEFAULT_CONFIGURATION_MODULE - cls.supported_models: dict = SUPPORTED_MODELS - - or turn off automatic registering of configuration:: - - @classmethod - def pre_init(cls): - cls.auto_register_config = False # defaults to True - - During initialisation of the class, if ``auto_register_config == True``, it will dynamically load the definitions - provided in the CONFIGURATION_MODULE.py located in the ``cellpy.readers.instruments.configurations`` - folder/package. - - Attributes can be set during initialisation of the class as **kwargs that are then handled by the - ``parse_formatter_parameters`` method. - - Remark that some also can be provided as arguments to the ``loader`` method and will then automatically - be "transparent" to the ``cellpy.get`` function. So if you would like to give the user access to modify - these arguments, you should implement them in the ``parse_loader_parameters`` method. - - """ - - instrument_name = "auto_loader" - - def __init__(self, *args, **kwargs): - self.auto_register_config = True - #: Whether `parse()` has run. Guards `declarations()`, whose answer is - #: only correct once the file's own units have been read (see there). - self._parsed = False - self.pre_init() - - if not hasattr(self, "supported_models"): - raise AttributeError( - "missing attribute in sub-class of AutoLoader: supported_models" - ) - if not hasattr(self, "default_model"): - raise AttributeError( - "missing attribute in sub-class of AutoLoader: default_model" - ) - - # in case model is given as argument - self.model = kwargs.pop("model", self.default_model) - - if self.auto_register_config: - self.config_params = self.register_configuration() - - self.parse_formatter_parameters(**kwargs) - self.override_config_params(**kwargs) - - self.pre_processors = self.config_params.pre_processors - additional_pre_processor_args = kwargs.pop( - "pre_processors", None - ) # could replace None with an empty dict to get rid of the if-clause: - if additional_pre_processor_args: - for key in additional_pre_processor_args: - self.pre_processors[key] = additional_pre_processor_args[key] - - self.post_processors = self.config_params.post_processors - additional_post_processor_args = kwargs.pop( - "post_processors", None - ) # could replace None with an empty dict to get rid of the if-clause: - if additional_post_processor_args: - for key in additional_post_processor_args: - self.post_processors[key] = additional_post_processor_args[key] - - self.include_aux = kwargs.pop("include_aux", False) - self.keep_all_columns = kwargs.pop("keep_all_columns", False) - self.cellpy_headers_normal = ( - headers_normal # the column headers defined by cellpy - ) - - def __str__(self): - txt = f"{self.__class__.__name__}\n" - txt += f" instrument_name: {self.instrument_name}\n" - txt += f" model: {self.model}\n" - return txt - - @abc.abstractmethod - def parse_formatter_parameters(self, **kwargs) -> None: ... - - @abc.abstractmethod - def parse_loader_parameters(self, **kwargs): ... - - @abc.abstractmethod - def query_file(self, file_path: Union[str, pathlib.Path]) -> pd.DataFrame: ... - - def pre_init(self) -> None: ... - - def register_configuration(self) -> ModelParameters: - """Register and load model configuration""" - if ( - self.model is None - ): # in case None was given as argument (model=None in initialisation) - self.model = self.default_model - model_module_name = self.supported_models.get(self.model.upper(), None) - if model_module_name is None: - raise Exception( - f"The model {self.model} does not have any defined configuration." - f"\nCurrent supported models are {[*self.supported_models.keys()]}" - ) - return register_configuration_from_module(self.model, model_module_name) - - def override_config_params(self, **kwargs) -> None: - """Override configuration parameters""" - pass - - def get_raw_units(self): - return self.config_params.raw_units - - def get_raw_limits(self): - return self.config_params.raw_limits - - @staticmethod - def get_headers_aux(raw: pd.DataFrame) -> dict: - raise NotImplementedError( - "missing method in sub-class of TxtLoader: get_headers_aux" - ) - - def _pre_process(self): - for processor_name in self.pre_processors: - if self.pre_processors[processor_name]: - if hasattr(pre_processors, processor_name): - logging.critical(f"running pre-processor: {processor_name}") - processor = getattr(pre_processors, processor_name) - self.temp_file_path = processor(self.temp_file_path) - else: - raise NotImplementedError( - f"{processor_name} is not currently supported - aborting!" - ) - - def parse(self, source: Union[str, pathlib.Path], **kwargs) -> pd.DataFrame: - """Vendor stage: read the file into a frame with **vendor** column names. - - The first half of the two-stage design: everything after this is - declared rather than coded, and handled by ``harmonize()``. This is the - same work ``loader()`` does before it starts building a ``Data`` — the - pre-processors, the formatter parameters, and ``query_file`` — exposed - on its own so the two stages can be driven, tested and compared - separately. It does not change how ``loader()`` behaves. - - Args: - source: path to the vendor file. - **kwargs: loader knobs, as ``loader()`` takes them. - - Returns: - The parsed vendor frame, before any renaming. - """ - self.refuse_copying = kwargs.pop("refuse_copying", False) - self.name = source - if not self.is_db: - self.copy_to_temporary() - if self.pre_processors: - self._pre_process() - self.parse_loader_parameters(**kwargs) - frame = self.query_file(self.temp_file_path) - # Cache for loader() so a follow-up legacy shell build does not - # re-query the same file (#560 Phase C — avoid double vendor read). - self._parsed_frame = frame - self._parsed = True - return frame - - def declarations(self): - """The `LoaderDeclarations` for the file most recently parsed. - - **Call this after `parse`, not before** — and it raises if you do - not, which is the point. Declarations are *not* a static property of the - loader class for every instrument: neware writes its units into the - column names (``Current(A)``), and which units those are is read from - the file, so the configuration's defaults (``mA``) are corrected during - parsing. Reading declarations first would hand back vendor column names - no file contains, and those columns would be silently unmapped rather - than raising — the failure mode this whole arc keeps running into. - - Deriving from ``config_params`` after the parse is what makes the - declarations per-file without any loader having to opt in. - - Returns: - A validated ``LoaderDeclarations`` derived from this loader's - configuration. - - Raises: - LoaderError: if called before ``parse()``, or if the configuration - does not carry a renaming dict to derive from. - """ - from cellpy.exceptions import LoaderError - from cellpy.readers.instruments.config_declarations import ( - declarations_from_configuration, - ) - - if not getattr(self, "_parsed", False): - raise LoaderError( - f"{type(self).__name__}.declarations() was called before " - f"parse(); for instruments whose column names carry units read " - f"from the file, the declarations are only correct once the " - f"file has been parsed." - ) - return declarations_from_configuration(self.config_params) - - def loader(self, name: Union[str, pathlib.Path], **kwargs: str) -> core.Data: - """returns a Data object with loaded data. - - Loads data from a txt file (csv-ish). - - Args: - name (str, pathlib.Path): name of the file. - kwargs (dict): key-word arguments from raw_loader. - - Returns: - new_tests (list of data objects) - - """ - pre_processor_hook = kwargs.pop("pre_processor_hook", None) - - cached = getattr(self, "_parsed_frame", None) - if cached is not None: - # parse() already ran query_file (+ pre-processors); reuse it. - data_df = cached - self._parsed_frame = None - else: - if self.pre_processors: - self._pre_process() - - self.parse_loader_parameters(**kwargs) - - data_df = self.query_file(self.temp_file_path) - - if pre_processor_hook is not None: - logging.debug("running pre-processing-hook") - data_df = pre_processor_hook(data_df) - - data = core.Data() - - # metadata - meta = self.parse_meta() - data.loaded_from = name - data.channel_index = meta.get("channel_index", None) - data.test_ID = meta.get("test_ID", None) - data.test_name = meta.get("test_name", None) - data.creator = meta.get("creator", None) - data.schedule_file_name = meta.get("schedule_file_name", None) - # TODO: convert to datetime: - data.start_datetime = meta.get("start_datetime", None) - - # Generating a FileID project: - self.generate_fid() - data.raw_data_files.append(self.fid) - - data.raw = data_df - data.raw_data_files_length.append(len(data_df)) - # stamp instrument units by value so a Data obtained directly from the - # loader carries correct raw_units (issue #508); CellpyCell.from_raw - # re-applies the same merge (idempotent). - data.raw_units = merge_raw_units(self.get_raw_units()) - data.summary = ( - pd.DataFrame() - ) # creating an empty frame - loading summary is not implemented - data = self._post_process(data) - data = self.identify_last_data_point(data) - if data.start_datetime is None: - # TODO: convert to datetime: - data.start_datetime = data.raw[headers_normal.datetime_txt].iat[0] - - data = self.validate(data) - return data - - def validate(self, data: core.Data) -> core.Data: - """Validation of the loaded data, should raise an appropriate exception if it fails.""" - - logging.debug("no validation of defined in this sub-class of TxtLoader") - return data - - def parse_meta(self) -> dict: - """Method that parses the data for meta-data (e.g. start-time, channel number, ...)""" - - logging.debug( - "no parsing method for meta-data defined in this sub-class of TxtLoader" - ) - return dict() - - def _post_rename_headers(self, data): - if self.include_aux: - new_aux_headers = self.get_headers_aux(data.raw) - data.raw.rename(index=str, columns=new_aux_headers, inplace=True) - return data - - def _post_process(self, data): - # ordered post-processing steps: - for processor_name in ORDERED_POST_PROCESSING_STEPS: - if processor_name in self.post_processors: - try: - data = self._perform_post_process_step(data, processor_name) - except Exception as e: - logging.error(f"failed to run {processor_name}: {e}") - raise WrongFileVersion(f"failed to run {processor_name}: {e}") - - # non-ordered post-processing steps - for processor_name in self.post_processors: - if processor_name not in ORDERED_POST_PROCESSING_STEPS: - try: - data = self._perform_post_process_step(data, processor_name) - except Exception as e: - logging.error(f"failed to run {processor_name}: {e}") - raise WrongFileVersion(f"failed to run {processor_name}: {e}") - return data - - def _perform_post_process_step(self, data, processor_name): - if self.post_processors[processor_name]: - if hasattr(post_processors, processor_name): - logging.critical(f"running post-processor: {processor_name}") - processor = getattr(post_processors, processor_name) - data = processor(data, self.config_params) - if hasattr(self, f"_post_{processor_name}"): # internal addon-function - _processor = getattr(self, f"_post_{processor_name}") - data = _processor(data) - else: - raise NotImplementedError( - f"{processor_name} is not currently supported - aborting!" - ) - return data - - -class TxtLoader(AutoLoader, ABC): - """Main txt loading class (for sub-classing). - - The subclass of a ``TxtLoader`` gets its information by loading model specifications from its respective module - (``cellpy.readers.instruments.configurations.``) or configuration file (yaml). - - Remark that if you implement automatic loading of the formatter, the module / yaml-file must include all - the required formatter parameters (sep, skiprows, header, encoding, decimal, thousands). - - If you need more flexibility, try using the ``CustomTxtLoader`` or subclass directly - from ``AutoLoader`` or ``Loader``. - - Attributes: - model (str): short name of the (already implemented) sub-model. - sep (str): delimiter. - skiprows (int): number of lines to skip. - header (int): number of the header lines. - encoding (str): encoding. - decimal (str): character used for decimal in the raw data, defaults to '.'. - processors (dict): pre-processing steps to take (before loading with pandas). - post_processors (dict): post-processing steps to make after loading the data, but before - returning them to the caller. - include_aux (bool): also parse so-called auxiliary columns / data. Defaults to False. - keep_all_columns (bool): load all columns, also columns that are not 100% necessary for ``cellpy`` to work. - - Remark that the configuration settings for the sub-model must include a list of column header names - that should be kept if keep_all_columns is False (default). - - Args: - sep (str): the delimiter (also works as a switch to turn on/off automatic detection of delimiter and - start of data (skiprows)). - - """ - - instrument_name = "txt_loader" - raw_ext = "*" - - # override this if needed - def __init__(self, *args, **kwargs): - super().__init__(*args, **kwargs) - - def __str__(self): - txt = f"{type(self)}\n" - txt += f" instrument_name: {self.instrument_name}\n" - txt += f" model: {self.model}\n" - return txt - - def parse_formatter_parameters(self, **kwargs): - """Parse the formatter parameters.""" - - logging.debug(f"model: {self.model}") - if not self.config_params.formatters: - # Setting defaults if formatter is not loaded - logging.debug("No formatter given - using default values.") - self.sep = kwargs.pop("sep", None) - self.skiprows = kwargs.pop("skiprows", 0) - self.header = kwargs.pop("header", 0) - self.encoding = kwargs.pop("encoding", "utf-8") - self.decimal = kwargs.pop("decimal", ".") - self.thousands = kwargs.pop("thousands", None) - - else: - # Remark! This will break if one of these parameters are missing - # (not a keyword argument and not within the configuration): - self.sep = kwargs.pop("sep", self.config_params.formatters["sep"]) - self.skiprows = kwargs.pop( - "skiprows", self.config_params.formatters["skiprows"] - ) - self.header = kwargs.pop("header", self.config_params.formatters["header"]) - self.encoding = kwargs.pop( - "encoding", self.config_params.formatters["encoding"] - ) - self.decimal = kwargs.pop( - "decimal", self.config_params.formatters["decimal"] - ) - self.thousands = kwargs.pop( - "thousands", self.config_params.formatters["thousands"] - ) - logging.debug( - f"Formatters: self.sep={self.sep} self.skiprows={self.skiprows} self.header={self.header} self.encoding={self.encoding}" - ) - logging.debug( - f"Formatters (cont.): self.decimal={self.decimal} self.thousands={self.thousands}" - ) - - # override this if needed - def parse_loader_parameters(self, auto_formatter=None, **kwargs): - """Parse the loader parameters. - - Args: - auto_formatter: if True, the formatter will be set to auto-formatting. - **kwargs: keyword arguments. - """ - if auto_formatter: - self._auto_formatter() - else: - # backup option - do auto-formatting if sep is not given - sep = kwargs.get("sep", None) - if sep is not None: - self.sep = sep - if self.sep is None: - self._auto_formatter() - - if raw_units := kwargs.get("raw_units", None): - logging.critical(f"overriding raw_units: {raw_units}") - self.config_params.raw_units.update(raw_units) - - if unit_labels := kwargs.get("unit_labels", None): - logging.critical(f"overriding unit_labels: {unit_labels}") - self.config_params.unit_labels.update(unit_labels) - - if raw_limits := kwargs.get("raw_limits", None): - logging.critical(f"overriding raw_limits: {raw_limits}") - self.config_params.raw_limits.update(raw_limits) - - if encoding := kwargs.get("encoding", None): - logging.critical(f"overriding encoding: {encoding}") - self.encoding = encoding - - if decimal := kwargs.get("decimal", None): - logging.critical(f"overriding decimal: {decimal}") - self.decimal = decimal - - if thousands := kwargs.get("thousands", None): - logging.critical(f"overriding thousands: {thousands}") - self.thousands = thousands - - if skiprows := kwargs.get("skiprows", None): - logging.critical(f"overriding skiprows: {skiprows}") - self.skiprows = skiprows - - if header := kwargs.get("header", None): - logging.critical(f"overriding header: {header}") - self.header = header - - if sep := kwargs.get("sep", None): - logging.critical(f"overriding sep: {sep}") - self.sep = sep - - def _auto_formatter(self): - separator, first_index, encoding = find_delimiter_and_start( - self.name, - separators=None, - checking_length_header=100, - checking_length_whole=200, - ) - self.encoding = encoding or "UTF-8" - self.sep = separator - self.skiprows = first_index - 1 - self.header = 0 - - logging.critical( - f"auto-formatting: {self.sep=}, {self.skiprows=}, {self.header=}, {self.encoding=}, {self.decimal=}" - ) - - # override this if using other query functions - def query_file(self, name): - logging.critical(f"parsing with pandas.read_csv: {name}") - logging.critical( - f"parameters: {self.sep=}, {self.skiprows=}, {self.header=}, {self.encoding=}, {self.decimal=}" - ) - data_df = pd.read_csv( - name, - sep=self.sep, - skiprows=self.skiprows, - header=self.header, - encoding=self.encoding, - decimal=self.decimal, - thousands=self.thousands, - ) - return data_df +""" +When you make a new loader you have to subclass the Loader class. +Remember also to register it in cellpy.cellreader. + +(for future development, not used very efficiently yet). +""" + +import abc +import logging +import pathlib +import shutil +import tempfile +from abc import ABC +from typing import List, Union + +import pandas as pd + +from cellpy.exceptions import LoaderError, WrongFileVersion +import cellpy.internals.connections +import cellpy.readers.data_structures as core +from cellpy.parameters.internal_settings import headers_normal, merge_raw_units +from cellpy.readers.instruments.configurations import ( + ModelParameters, + register_configuration_from_module, +) +from cellpy.readers.instruments.processors import post_processors, pre_processors +from cellpy.readers.instruments.processors.post_processors import ( + ORDERED_POST_PROCESSING_STEPS, +) + +MINIMUM_SELECTION = [ + "Data_Point", + "Test_Time", + "Step_Time", + "DateTime", + "Step_Index", + "Cycle_Index", + "Current", + "Voltage", + "Charge_Capacity", + "Discharge_Capacity", + "Internal_Resistance", +] + + +# TODO: move this to another module (e.g. inside processors): +def find_delimiter_and_start( + file_name, + separators=None, + checking_length_header=30, + checking_length_whole=200, + check_encoding=True, +): + """Function to automatically detect the delimiter and what line the first data appears on. + + This function is fairly stupid. It reads a window of up to + ``checking_length_whole`` non-empty lines and treats up to + ``checking_length_header`` of the leading ones as a possible header. It + counts the appearances of the different possible delimiters in the data + rows past that header and selects a delimiter if its per-row count is both + uniform and positive. + + The header window is clamped to the actual number of lines read, so a short + file - down to a header line and a single data row - is inspected correctly + rather than running off the end of the sample. + + The first line is defined as where the delimiter is used same number of times (probably a header line). + + Args: + file_name: path to the file. + separators: list of possible delimiters. + checking_length_header: number of lines to check for header. + checking_length_whole: number of lines to check for delimiter. + check_encoding: check encoding. + + Returns: + separator: the delimiter. + first_index: the index of the first line with data. + encoding: the encoding (None if not found or checked). + + Raises: + LoaderError: if the file has no inspectable content, no candidate + delimiter fits the data rows, or the header row cannot be located. + """ + + if separators is None: + separators = [";", "\t", "|", ","] + logging.debug(f"checking internals of the file {file_name}") + + encoding = None + + if check_encoding: + import charset_normalizer + + results = charset_normalizer.from_path( + file_name, + steps=10, # Number of steps/block to extract from my_byte_str + chunk_size=512, # Set block size of each extraction + ) + if results: + r = results.best() + encoding = r.encoding + + with open(file_name, "r") as fin: + lines = [] + for j in range(checking_length_whole): + line = fin.readline() + if not line: + break + if len(line.strip()): + lines.append(line) + + if not lines: + raise LoaderError( + f"could not detect a delimiter in {file_name}: " + "the file has no non-empty lines to inspect" + ) + + separator, number_of_hits = _find_separator( + lines, separators, checking_length_header + ) + + if separator is None: + candidates = ", ".join(repr(s) for s in separators) + raise LoaderError( + f"could not detect a delimiter in {file_name}: none of the " + f"candidate delimiters ({candidates}) appears a consistent, " + "positive number of times across the data rows" + ) + + if separator == "\t": + logging.debug("seperator = TAB") + elif separator == " ": + logging.debug("seperator = SPACE") + else: + logging.debug(f"seperator = {separator}") + + first_index = _find_first_line_whit_delimiter( + checking_length_header, lines, number_of_hits, separator + ) + if first_index is None: + raise LoaderError( + f"detected delimiter {separator!r} in {file_name} but could not " + f"locate the header row within the first {checking_length_header} " + "lines" + ) + logging.debug(f"First line with delimiter: {first_index}") + return separator, first_index, encoding + + +def _find_first_line_whit_delimiter( + checking_length_header, lines, number_of_hits, separator +): + """Return the index of the first line that carries the data-row delimiter count. + + Only the first ``checking_length_header`` lines are searched (the header is + assumed to live there). Returns ``None`` if no line matches, so the caller + can raise a delimiter error that names the file. + """ + first_part = lines[:checking_length_header] + for line_number, line in enumerate(first_part): + if line.count(separator) == number_of_hits: + return line_number + return None + + +def _find_separator(lines, separators, checking_length_header): + """Pick the delimiter from the data rows that follow a possible header. + + The header may be up to ``checking_length_header`` lines, but a short file + cannot hold that many header lines - so the header window is clamped to the + number of lines actually read, always leaving at least one data row to + inspect. The delimiter is the highest-priority candidate whose per-row + count is both uniform and positive across those data rows. + + Returns ``(None, None)`` when no candidate qualifies. + """ + logging.debug("searching for separators") + n = len(lines) + + # reserve up to checking_length_header leading lines as a possible header, + # but never so many that no data row is left to inspect. + header_window = min(checking_length_header, n - 1) if n > 1 else 0 + data_lines = lines[header_window:] + + # the final line can be truncated mid-write, so drop it - but only when a + # data row can still be spared (a header + single data row must keep it). + if len(data_lines) > 1: + data_lines = data_lines[:-1] + + for candidate in separators: + counts = {line.count(candidate) for line in data_lines} + if len(counts) == 1: + number_of_hits = counts.pop() + if number_of_hits > 0: + return candidate, number_of_hits + + return None, None + + +def query_csv( + self, + name, + sep=None, + skiprows=None, + header=None, + encoding=None, + decimal=None, + thousands=None, +): + """function to query a csv file using pandas.read_csv. + + + Args: + name: path to the file. + sep: delimiter. + skiprows: number of lines to skip. + header: number of the header lines. + encoding: encoding. + decimal: character used for decimal in the raw data, defaults to '.'. + thousands: character used for thousands in the raw data, defaults to ','. + + Returns: + pandas.DataFrame + + """ + logging.debug(f"parsing with pandas.read_csv: {name}") + sep = sep or self.sep + skiprows = skiprows or self.skiprows + header = header or self.header + encoding = encoding or self.encoding + decimal = decimal or self.decimal + thousands = thousands or self.thousands + logging.critical(f"{sep=}, {skiprows=}, {header=}, {encoding=}, {decimal=}") + data_df = pd.read_csv( + name, + sep=sep, + skiprows=skiprows, + header=header, + encoding=encoding, + decimal=decimal, + thousands=thousands, + ) + return data_df + + +class AtomicLoad: + """Atomic loading class""" + + instrument_name = "atomic_loader" + + _name = None + _temp_file_path = None + _fid = None + _is_db: bool = False + _copy_also_local: bool = True + _refuse_copying: bool = False + + @property + def is_db(self): + """Is the file stored in the database""" + return self._is_db + + @is_db.setter + def is_db(self, value: bool): + """Is the file stored in the database""" + self._is_db = value + + @property + def refuse_copying(self): + """Should the file be copied to a temporary file""" + return self._refuse_copying + + @refuse_copying.setter + def refuse_copying(self, value: bool): + """Should the file be copied to a temporary file""" + self._refuse_copying = value + + @property + def name(self): + """The name of the file to be loaded""" + return self._name + + @name.setter + def name(self, value): + """The name of the file to be loaded""" + if not self.is_db and not isinstance(value, cellpy.internals.connections.OtherPath): + logging.debug("converting to OtherPath") + value = cellpy.internals.connections.OtherPath(value) + self._name = value + + @property + def temp_file_path(self): # -> Union[cellpy.internals.connections.OtherPath, pathlib.Path] + """The name of the file to be loaded if copied to a temporary file""" + return self._temp_file_path + + @temp_file_path.setter + def temp_file_path(self, value): + """The name of the file to be loaded if copied to a temporary file""" + self._temp_file_path = value + + @property + def fid(self): + """The unique file id""" + if self._fid is None: + self.generate_fid() + return self._fid + + def generate_fid(self, value=None): + """Generate a unique file id""" + if self.is_db: + self._fid = core.FileID(self.name, is_db=True) + elif self._temp_file_path is not None: + self._fid = core.FileID(self.name) + elif self._name is not None: + self._fid = core.FileID(self.name) + elif value is not None: + self._fid = core.FileID(value) + else: + raise ValueError("could not generate fid") + + def copy_to_temporary(self): + """Copy file to a temporary file""" + + logging.debug(f"external file received? {self.name.is_external=}") + if self.name is None: + raise ValueError("no file name given to loader class (self.name is None)") + + if self._refuse_copying: + logging.debug("refusing copying") + self._temp_file_path = self.name + return + + if not self._copy_also_local and not self.name.is_external: + self._temp_file_path = self.name + return + + from cellpy.internals.progress import emit + + emit("copy") + self._temp_file_path = self.name.copy() + + def loader_executor(self, *args, **kwargs): + """Load the file""" + name = args[0] + self.refuse_copying = kwargs.pop("refuse_copying", False) + self.name = name + if not self.is_db: + self.copy_to_temporary() + cellpy_data = self.loader(*args, **kwargs) + return cellpy_data + + def loader(self, *args, **kwargs): + """The method that does the actual loading. + + This method should be overwritten by the specific loader class. + """ + ... + + +class BaseLoader(AtomicLoad, metaclass=abc.ABCMeta): + """Main loading class""" + + instrument_name = "base_loader" + + # TODO: should also include the functions for getting cellpy headers etc + # here + + @staticmethod + @abc.abstractmethod + def get_raw_units() -> dict: + """Units used by the instrument. + + The internal cellpy units are given in the ``cellpy_units`` attribute. + + Returns: + dictionary of units (str) + + Example: + A minimum viable implementation could look like this:: + + @staticmethod + def get_raw_units(): + raw_units = dict() + raw_units["current"] = "A" + raw_units["charge"] = "Ah" + raw_units["mass"] = "g" + raw_units["voltage"] = "V" + return raw_units + + """ + # This is needed for example when converting the capacity to a specific capacity. + # So far, it has been difficult to get any kind of consensus on what the most optimal + # units are for storing cycling data. Therefore, cellpy implements three levels of units: + # 1) the raw units that the data is loaded in already has and 2) the cellpy units used by cellpy + # when generating summaries and related information, and 3) output units that can be set to get the data + # in a specif unit when exporting or creating specific outputs such as ICA. + # + # Comment 2022.09.11:: + # + # still not sure if we should use raw units or cellpy units in the cellpy-files (.h5/ .cellpy). + # Currently, the summary is in cellpy units and the raw and step data is in raw units. If + # you have any input on this topic, let us know. + + pass + + @abc.abstractmethod + def get_raw_limits(self) -> dict: + """Limits used to identify type of step. + + The raw limits are 'epsilons' used to check if the current and/or voltage is stable (for example + for galvanostatic steps, one would expect that the current is stable (constant) and non-zero). + If the (accumulated) change is less than 'epsilon', then cellpy interpret it to be stable. + It is expected that different instruments (with different resolution etc.) have different + resolutions and noice levels, thus different 'epsilons'. + + Returns: + the raw limits (dict) + + """ + pass + + @classmethod + def get_params(cls, parameter: Union[str, None]) -> dict: + """Retrieves parameters needed for facilitating working with the + instrument without registering it. + + Typically, it should include the name and raw_ext. + + Return: parameters or a selected parameter + """ + + return getattr(cls, parameter) + + @abc.abstractmethod + def loader(self, *args, **kwargs) -> list: + """Loads data into a Data object and returns it""" + # This method is used by cellreader through the AtomicLoad.loader_executor method. + # It should be overwritten by the specific loader class. + # + # Notice that it is highly recommended that you don't try to implement .loader_executor yourself + # in your subclass! + pass + + @staticmethod + def identify_last_data_point(data: core.Data) -> core.Data: + """This method is used to find the last record in the data.""" + return core.identify_last_data_point(data) + + +class AutoLoader(BaseLoader): + """Main autoload class. + + This class can be sub-classed if you want to make a data-reader for different type of "easily parsed" files + (for example csv-files). The subclass needs to have at least one + associated CONFIGURATION_MODULE defined and must have the following attributes as minimum:: + + default_model: str = NICK_NAME_OF_DEFAULT_CONFIGURATION_MODULE + supported_models: dict = SUPPORTED_MODELS + + where SUPPORTED_MODELS is a dictionary with ``{"NICK_NAME" : "CONFIGURATION_MODULE_NAME"}`` key-value pairs. + Remark! the NICK_NAME must be in upper-case! + + It is also possible to set these in a custom pre_init method:: + + @classmethod + def pre_init(cls): + cls.default_model: str = NICK_NAME_OF_DEFAULT_CONFIGURATION_MODULE + cls.supported_models: dict = SUPPORTED_MODELS + + or turn off automatic registering of configuration:: + + @classmethod + def pre_init(cls): + cls.auto_register_config = False # defaults to True + + During initialisation of the class, if ``auto_register_config == True``, it will dynamically load the definitions + provided in the CONFIGURATION_MODULE.py located in the ``cellpy.readers.instruments.configurations`` + folder/package. + + Attributes can be set during initialisation of the class as **kwargs that are then handled by the + ``parse_formatter_parameters`` method. + + Remark that some also can be provided as arguments to the ``loader`` method and will then automatically + be "transparent" to the ``cellpy.get`` function. So if you would like to give the user access to modify + these arguments, you should implement them in the ``parse_loader_parameters`` method. + + """ + + instrument_name = "auto_loader" + + def __init__(self, *args, **kwargs): + self.auto_register_config = True + #: Whether `parse()` has run. Guards `declarations()`, whose answer is + #: only correct once the file's own units have been read (see there). + self._parsed = False + self.pre_init() + + if not hasattr(self, "supported_models"): + raise AttributeError( + "missing attribute in sub-class of AutoLoader: supported_models" + ) + if not hasattr(self, "default_model"): + raise AttributeError( + "missing attribute in sub-class of AutoLoader: default_model" + ) + + # in case model is given as argument + self.model = kwargs.pop("model", self.default_model) + + if self.auto_register_config: + self.config_params = self.register_configuration() + + self.parse_formatter_parameters(**kwargs) + self.override_config_params(**kwargs) + + self.pre_processors = self.config_params.pre_processors + additional_pre_processor_args = kwargs.pop( + "pre_processors", None + ) # could replace None with an empty dict to get rid of the if-clause: + if additional_pre_processor_args: + for key in additional_pre_processor_args: + self.pre_processors[key] = additional_pre_processor_args[key] + + self.post_processors = self.config_params.post_processors + additional_post_processor_args = kwargs.pop( + "post_processors", None + ) # could replace None with an empty dict to get rid of the if-clause: + if additional_post_processor_args: + for key in additional_post_processor_args: + self.post_processors[key] = additional_post_processor_args[key] + + self.include_aux = kwargs.pop("include_aux", False) + self.keep_all_columns = kwargs.pop("keep_all_columns", False) + self.cellpy_headers_normal = ( + headers_normal # the column headers defined by cellpy + ) + + def __str__(self): + txt = f"{self.__class__.__name__}\n" + txt += f" instrument_name: {self.instrument_name}\n" + txt += f" model: {self.model}\n" + return txt + + @abc.abstractmethod + def parse_formatter_parameters(self, **kwargs) -> None: ... + + @abc.abstractmethod + def parse_loader_parameters(self, **kwargs): ... + + @abc.abstractmethod + def query_file(self, file_path: Union[str, pathlib.Path]) -> pd.DataFrame: ... + + def pre_init(self) -> None: ... + + def register_configuration(self) -> ModelParameters: + """Register and load model configuration""" + if ( + self.model is None + ): # in case None was given as argument (model=None in initialisation) + self.model = self.default_model + model_module_name = self.supported_models.get(self.model.upper(), None) + if model_module_name is None: + raise Exception( + f"The model {self.model} does not have any defined configuration." + f"\nCurrent supported models are {[*self.supported_models.keys()]}" + ) + return register_configuration_from_module(self.model, model_module_name) + + def override_config_params(self, **kwargs) -> None: + """Override configuration parameters""" + pass + + def get_raw_units(self): + return self.config_params.raw_units + + def get_raw_limits(self): + return self.config_params.raw_limits + + @staticmethod + def get_headers_aux(raw: pd.DataFrame) -> dict: + raise NotImplementedError( + "missing method in sub-class of TxtLoader: get_headers_aux" + ) + + def _pre_process(self): + for processor_name in self.pre_processors: + if self.pre_processors[processor_name]: + if hasattr(pre_processors, processor_name): + logging.critical(f"running pre-processor: {processor_name}") + processor = getattr(pre_processors, processor_name) + self.temp_file_path = processor(self.temp_file_path) + else: + raise NotImplementedError( + f"{processor_name} is not currently supported - aborting!" + ) + + def parse(self, source: Union[str, pathlib.Path], **kwargs) -> pd.DataFrame: + """Vendor stage: read the file into a frame with **vendor** column names. + + The first half of the two-stage design: everything after this is + declared rather than coded, and handled by ``harmonize()``. This is the + same work ``loader()`` does before it starts building a ``Data`` — the + pre-processors, the formatter parameters, and ``query_file`` — exposed + on its own so the two stages can be driven, tested and compared + separately. It does not change how ``loader()`` behaves. + + Args: + source: path to the vendor file. + **kwargs: loader knobs, as ``loader()`` takes them. + + Returns: + The parsed vendor frame, before any renaming. + """ + self.refuse_copying = kwargs.pop("refuse_copying", False) + self.name = source + if not self.is_db: + self.copy_to_temporary() + if self.pre_processors: + self._pre_process() + self.parse_loader_parameters(**kwargs) + frame = self.query_file(self.temp_file_path) + # Cache for loader() so a follow-up legacy shell build does not + # re-query the same file (#560 Phase C — avoid double vendor read). + self._parsed_frame = frame + self._parsed = True + return frame + + def declarations(self): + """The `LoaderDeclarations` for the file most recently parsed. + + **Call this after `parse`, not before** — and it raises if you do + not, which is the point. Declarations are *not* a static property of the + loader class for every instrument: neware writes its units into the + column names (``Current(A)``), and which units those are is read from + the file, so the configuration's defaults (``mA``) are corrected during + parsing. Reading declarations first would hand back vendor column names + no file contains, and those columns would be silently unmapped rather + than raising — the failure mode this whole arc keeps running into. + + Deriving from ``config_params`` after the parse is what makes the + declarations per-file without any loader having to opt in. + + Returns: + A validated ``LoaderDeclarations`` derived from this loader's + configuration. + + Raises: + LoaderError: if called before ``parse()``, or if the configuration + does not carry a renaming dict to derive from. + """ + from cellpy.exceptions import LoaderError + from cellpy.readers.instruments.config_declarations import ( + declarations_from_configuration, + ) + + if not getattr(self, "_parsed", False): + raise LoaderError( + f"{type(self).__name__}.declarations() was called before " + f"parse(); for instruments whose column names carry units read " + f"from the file, the declarations are only correct once the " + f"file has been parsed." + ) + return declarations_from_configuration(self.config_params) + + def loader(self, name: Union[str, pathlib.Path], **kwargs: str) -> core.Data: + """returns a Data object with loaded data. + + Loads data from a txt file (csv-ish). + + Args: + name (str, pathlib.Path): name of the file. + kwargs (dict): key-word arguments from raw_loader. + + Returns: + new_tests (list of data objects) + + """ + pre_processor_hook = kwargs.pop("pre_processor_hook", None) + + cached = getattr(self, "_parsed_frame", None) + if cached is not None: + # parse() already ran query_file (+ pre-processors); reuse it. + data_df = cached + self._parsed_frame = None + else: + if self.pre_processors: + self._pre_process() + + self.parse_loader_parameters(**kwargs) + + data_df = self.query_file(self.temp_file_path) + + if pre_processor_hook is not None: + logging.debug("running pre-processing-hook") + data_df = pre_processor_hook(data_df) + + data = core.Data() + + # metadata + meta = self.parse_meta() + data.loaded_from = name + data.channel_index = meta.get("channel_index", None) + data.test_ID = meta.get("test_ID", None) + data.test_name = meta.get("test_name", None) + data.creator = meta.get("creator", None) + data.schedule_file_name = meta.get("schedule_file_name", None) + # TODO: convert to datetime: + data.start_datetime = meta.get("start_datetime", None) + + # Generating a FileID project: + self.generate_fid() + data.raw_data_files.append(self.fid) + + data.raw = data_df + data.raw_data_files_length.append(len(data_df)) + # stamp instrument units by value so a Data obtained directly from the + # loader carries correct raw_units (issue #508); CellpyCell.from_raw + # re-applies the same merge (idempotent). + data.raw_units = merge_raw_units(self.get_raw_units()) + data.summary = ( + pd.DataFrame() + ) # creating an empty frame - loading summary is not implemented + data = self._post_process(data) + data = self.identify_last_data_point(data) + if data.start_datetime is None: + # TODO: convert to datetime: + data.start_datetime = data.raw[headers_normal.datetime_txt].iat[0] + + data = self.validate(data) + return data + + def validate(self, data: core.Data) -> core.Data: + """Validation of the loaded data, should raise an appropriate exception if it fails.""" + + logging.debug("no validation of defined in this sub-class of TxtLoader") + return data + + def parse_meta(self) -> dict: + """Method that parses the data for meta-data (e.g. start-time, channel number, ...)""" + + logging.debug( + "no parsing method for meta-data defined in this sub-class of TxtLoader" + ) + return dict() + + def _post_rename_headers(self, data): + if self.include_aux: + new_aux_headers = self.get_headers_aux(data.raw) + data.raw.rename(index=str, columns=new_aux_headers, inplace=True) + return data + + def _post_process(self, data): + # ordered post-processing steps: + for processor_name in ORDERED_POST_PROCESSING_STEPS: + if processor_name in self.post_processors: + try: + data = self._perform_post_process_step(data, processor_name) + except Exception as e: + logging.error(f"failed to run {processor_name}: {e}") + raise WrongFileVersion(f"failed to run {processor_name}: {e}") + + # non-ordered post-processing steps + for processor_name in self.post_processors: + if processor_name not in ORDERED_POST_PROCESSING_STEPS: + try: + data = self._perform_post_process_step(data, processor_name) + except Exception as e: + logging.error(f"failed to run {processor_name}: {e}") + raise WrongFileVersion(f"failed to run {processor_name}: {e}") + return data + + def _perform_post_process_step(self, data, processor_name): + if self.post_processors[processor_name]: + if hasattr(post_processors, processor_name): + logging.critical(f"running post-processor: {processor_name}") + processor = getattr(post_processors, processor_name) + data = processor(data, self.config_params) + if hasattr(self, f"_post_{processor_name}"): # internal addon-function + _processor = getattr(self, f"_post_{processor_name}") + data = _processor(data) + else: + raise NotImplementedError( + f"{processor_name} is not currently supported - aborting!" + ) + return data + + +class TxtLoader(AutoLoader, ABC): + """Main txt loading class (for sub-classing). + + The subclass of a ``TxtLoader`` gets its information by loading model specifications from its respective module + (``cellpy.readers.instruments.configurations.``) or configuration file (yaml). + + Remark that if you implement automatic loading of the formatter, the module / yaml-file must include all + the required formatter parameters (sep, skiprows, header, encoding, decimal, thousands). + + If you need more flexibility, try using the ``CustomTxtLoader`` or subclass directly + from ``AutoLoader`` or ``Loader``. + + Attributes: + model (str): short name of the (already implemented) sub-model. + sep (str): delimiter. + skiprows (int): number of lines to skip. + header (int): number of the header lines. + encoding (str): encoding. + decimal (str): character used for decimal in the raw data, defaults to '.'. + processors (dict): pre-processing steps to take (before loading with pandas). + post_processors (dict): post-processing steps to make after loading the data, but before + returning them to the caller. + include_aux (bool): also parse so-called auxiliary columns / data. Defaults to False. + keep_all_columns (bool): load all columns, also columns that are not 100% necessary for ``cellpy`` to work. + + Remark that the configuration settings for the sub-model must include a list of column header names + that should be kept if keep_all_columns is False (default). + + Args: + sep (str): the delimiter (also works as a switch to turn on/off automatic detection of delimiter and + start of data (skiprows)). + + """ + + instrument_name = "txt_loader" + raw_ext = "*" + + # override this if needed + def __init__(self, *args, **kwargs): + super().__init__(*args, **kwargs) + + def __str__(self): + txt = f"{type(self)}\n" + txt += f" instrument_name: {self.instrument_name}\n" + txt += f" model: {self.model}\n" + return txt + + def parse_formatter_parameters(self, **kwargs): + """Parse the formatter parameters.""" + + logging.debug(f"model: {self.model}") + if not self.config_params.formatters: + # Setting defaults if formatter is not loaded + logging.debug("No formatter given - using default values.") + self.sep = kwargs.pop("sep", None) + self.skiprows = kwargs.pop("skiprows", 0) + self.header = kwargs.pop("header", 0) + self.encoding = kwargs.pop("encoding", "utf-8") + self.decimal = kwargs.pop("decimal", ".") + self.thousands = kwargs.pop("thousands", None) + + else: + # Remark! This will break if one of these parameters are missing + # (not a keyword argument and not within the configuration): + self.sep = kwargs.pop("sep", self.config_params.formatters["sep"]) + self.skiprows = kwargs.pop( + "skiprows", self.config_params.formatters["skiprows"] + ) + self.header = kwargs.pop("header", self.config_params.formatters["header"]) + self.encoding = kwargs.pop( + "encoding", self.config_params.formatters["encoding"] + ) + self.decimal = kwargs.pop( + "decimal", self.config_params.formatters["decimal"] + ) + self.thousands = kwargs.pop( + "thousands", self.config_params.formatters["thousands"] + ) + logging.debug( + f"Formatters: self.sep={self.sep} self.skiprows={self.skiprows} self.header={self.header} self.encoding={self.encoding}" + ) + logging.debug( + f"Formatters (cont.): self.decimal={self.decimal} self.thousands={self.thousands}" + ) + + # override this if needed + def parse_loader_parameters(self, auto_formatter=None, **kwargs): + """Parse the loader parameters. + + Args: + auto_formatter: if True, the formatter will be set to auto-formatting. + **kwargs: keyword arguments. + """ + if auto_formatter: + self._auto_formatter() + else: + # backup option - do auto-formatting if sep is not given + sep = kwargs.get("sep", None) + if sep is not None: + self.sep = sep + if self.sep is None: + self._auto_formatter() + + if raw_units := kwargs.get("raw_units", None): + logging.critical(f"overriding raw_units: {raw_units}") + self.config_params.raw_units.update(raw_units) + + if unit_labels := kwargs.get("unit_labels", None): + logging.critical(f"overriding unit_labels: {unit_labels}") + self.config_params.unit_labels.update(unit_labels) + + if raw_limits := kwargs.get("raw_limits", None): + logging.critical(f"overriding raw_limits: {raw_limits}") + self.config_params.raw_limits.update(raw_limits) + + if encoding := kwargs.get("encoding", None): + logging.critical(f"overriding encoding: {encoding}") + self.encoding = encoding + + if decimal := kwargs.get("decimal", None): + logging.critical(f"overriding decimal: {decimal}") + self.decimal = decimal + + if thousands := kwargs.get("thousands", None): + logging.critical(f"overriding thousands: {thousands}") + self.thousands = thousands + + if skiprows := kwargs.get("skiprows", None): + logging.critical(f"overriding skiprows: {skiprows}") + self.skiprows = skiprows + + if header := kwargs.get("header", None): + logging.critical(f"overriding header: {header}") + self.header = header + + if sep := kwargs.get("sep", None): + logging.critical(f"overriding sep: {sep}") + self.sep = sep + + def _auto_formatter(self): + separator, first_index, encoding = find_delimiter_and_start( + self.name, + separators=None, + checking_length_header=100, + checking_length_whole=200, + ) + self.encoding = encoding or "UTF-8" + self.sep = separator + self.skiprows = first_index - 1 + self.header = 0 + + logging.critical( + f"auto-formatting: {self.sep=}, {self.skiprows=}, {self.header=}, {self.encoding=}, {self.decimal=}" + ) + + # override this if using other query functions + def query_file(self, name, skiprows=None): + """Read the file with ``pd.read_csv`` using the resolved formatter parameters. + + Args: + name: path to read. + skiprows: override for ``self.skiprows`` (an int, or a callable on + the 0-based line index as ``pd.read_csv`` accepts). Used by the + incremental read to skip already-consumed data rows while + keeping the header line. + """ + if skiprows is None: + skiprows = self.skiprows + logging.critical(f"parsing with pandas.read_csv: {name}") + logging.critical( + f"parameters: {self.sep=}, {skiprows=}, {self.header=}, {self.encoding=}, {self.decimal=}" + ) + data_df = pd.read_csv( + name, + sep=self.sep, + skiprows=skiprows, + header=self.header, + encoding=self.encoding, + decimal=self.decimal, + thousands=self.thousands, + ) + return data_df + + def _load_since_rows(self, source, marker=None): + """Incremental read for text sources: data rows from ``marker.row_count`` on. + + Shared implementation behind the ``load_since`` of the text loaders + that opt into `SupportsIncrementalLoad` (neware_txt, maccor_txt). + Formatter parameters are resolved exactly as ``parse()`` resolves + them, the header line is kept, and the rows are harmonized with this + loader's declarations so ``new_raw`` is the same frame a full + ``harmonize(parse())`` yields for those rows. + + The returned marker's ``row_count`` is the file data-row index of the + first row of the last cycle read (see + ``cellpy.readers.instruments.incremental``), so the next call re-reads + that cycle whole. A marker at or past the end of the file gives an + empty ``new_raw`` and the marker back unchanged. + """ + import polars as pl + + from cellpy.readers.instruments.contract import IncrementalChunk, LoadMarker + from cellpy.readers.instruments.harmonize import harmonize + from cellpy.readers.instruments.incremental import last_cycle_start, vendor_column + + start = 0 if marker is None or marker.row_count is None else int(marker.row_count) + if marker is None: + marker = LoadMarker() + + self.name = source + if not self.is_db: + self.copy_to_temporary() + if self.pre_processors: + self._pre_process() + self.parse_loader_parameters() + + leading = max(int(self.skiprows or 0), 0) + header_lines = (int(self.header) + 1) if isinstance(self.header, int) else 0 + first_data_line = leading + header_lines + + def skip(line_index: int) -> bool: + if line_index < leading: + return True + return first_data_line <= line_index < first_data_line + start + + vendor = self.query_file(self.temp_file_path, skiprows=skip) + self._parsed_frame = None + self._parsed = True + if len(vendor) == 0: + return IncrementalChunk(new_raw=pl.DataFrame(), marker=marker) + + vendor_pl = pl.from_pandas(vendor) + declarations = self.declarations() + new_raw = harmonize(vendor_pl, declarations, strict=False) + rewind = last_cycle_start(vendor_pl, vendor_column(declarations, "cycle_num")) + return IncrementalChunk(new_raw=new_raw, marker=LoadMarker(row_count=start + rewind)) diff --git a/cellpy/readers/instruments/incremental.py b/cellpy/readers/instruments/incremental.py new file mode 100644 index 00000000..3024f5fc --- /dev/null +++ b/cellpy/readers/instruments/incremental.py @@ -0,0 +1,51 @@ +"""Shared pieces for loaders that implement `SupportsIncrementalLoad` (#780). + +Why every ``load_since`` rewinds to a cycle start +-------------------------------------------------- +``harmonize.normalize_reset_granularity`` re-accumulates per-step capacity +and rebases each cycle so it starts at 0. Both look at the *first row of the +cycle*. A chunk that begins mid-cycle would be rebased against the wrong row, +and the corruption would not raise anything. So a loader does not re-read +from the last row it saw; it re-reads from the first row of the last cycle it +saw. The marker points there. The trailing overlap is allowed by the +contract, and core ``update_data`` keeps the new rows for the overlapping +range, so the result equals a full load. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +if TYPE_CHECKING: # pragma: no cover - typing only + import polars as pl + + from cellpy.readers.instruments.declarations import LoaderDeclarations + + +def vendor_column(declarations: "LoaderDeclarations", native_name: str) -> str | None: + """The vendor column that ``declarations.column_map`` sends to ``native_name``.""" + for vendor, native in declarations.column_map.items(): + if native == native_name: + return vendor + return None + + +def last_cycle_start(frame: "pl.DataFrame", cycle_column: str | None) -> int: + """Row index of the first row of the last cycle in ``frame``. + + Returns 0 when the frame is empty or has no ``cycle_column``, so a caller + that cannot find cycle boundaries re-reads everything rather than + guessing. + """ + import polars as pl + + if cycle_column is None or frame.height == 0 or cycle_column not in frame.columns: + return 0 + cycles = frame.get_column(cycle_column) + last = cycles[-1] + if last is None: + return 0 + earlier = frame.with_row_index("_row").filter(pl.col(cycle_column) != last) + if earlier.height == 0: + return 0 + return int(earlier.get_column("_row").max()) + 1 diff --git a/cellpy/readers/instruments/maccor_txt.py b/cellpy/readers/instruments/maccor_txt.py index 72a84084..38678207 100644 --- a/cellpy/readers/instruments/maccor_txt.py +++ b/cellpy/readers/instruments/maccor_txt.py @@ -1,517 +1,525 @@ -"""Maccor txt data""" -import cellpy.config as config - -import pandas as pd - -from cellpy import prms -from cellpy import exceptions -from cellpy.parameters.internal_settings import ( - HeaderDict, - base_columns_float, - base_columns_int, - headers_normal, -) -from cellpy.readers.instruments.base import TxtLoader - - -SUPPORTED_MODELS = { - "ZERO": "maccor_txt_zero", # needs to be updated (does no postprocessing at the moment) - "ONE": "maccor_txt_one", - "TWO": "maccor_txt_two", - "THREE": "maccor_txt_three", - "S4000-UBHAM": "maccor_txt_one", - "S4000-KIT": "maccor_txt_two", - "S4000-WMG": "maccor_txt_three", -} - - -MUST_HAVE_RAW_COLUMNS = [ - headers_normal.test_time_txt, - headers_normal.step_time_txt, - headers_normal.current_txt, - headers_normal.voltage_txt, - headers_normal.step_index_txt, - headers_normal.cycle_index_txt, - headers_normal.charge_capacity_txt, - headers_normal.discharge_capacity_txt, -] - - -class DataLoader(TxtLoader): - """Class for loading data from Maccor txt files.""" - - instrument_name = "maccor_txt" - raw_ext = "txt" - - default_model = config.instruments.Maccor.default_model # Required - supported_models = SUPPORTED_MODELS # Required - - @staticmethod - def get_headers_aux(raw): - """Defines the so-called auxiliary table column headings""" - - headers = HeaderDict() - for col in raw.columns: - if col.startswith("Aux_"): - ncol = col.replace("/", "_") - ncol = "".join(ncol.split("(")[0]) - headers[col] = ncol.lower() - - return headers - - def validate(self, data): - """A simple check that all the needed columns has been successfully - loaded and that they get the correct type""" - missing_must_have_columns = [] - - # validating the float-type raw data - for col in base_columns_float: - if col in data.raw.columns: - data.raw[col] = pd.to_numeric(data.raw[col], errors="coerce") - else: - if col in MUST_HAVE_RAW_COLUMNS: - missing_must_have_columns.append(col) - - # validating the integer-type raw data - for col in base_columns_int: - if col in data.raw.columns: - data.raw[col] = pd.to_numeric( - data.raw[col], errors="coerce", downcast="integer" - ) - else: - if col in MUST_HAVE_RAW_COLUMNS: - missing_must_have_columns.append(col) - - if missing_must_have_columns: - raise exceptions.IOError( - f"Missing needed columns: {missing_must_have_columns}\nAborting!" - ) - return data - - -def _check_retrieve_file(n=1): - import pathlib - - pd.options.display.max_columns = 100 - # config.reader.sep = "\t" - data_root = pathlib.Path(r"C:\scripting\cellpy_dev_resources") - data_dir = data_root / r"2021_leafs_data\Charge-Discharge\Maccor series 4000" - if n == 2: - name = data_dir / "KIT-Full-cell-PW-HC-CT-cell016.txt" - else: - name = data_dir / "01_UBham_M50_Validation_0deg_01.txt" - print(name) - print(f"Exists? {name.is_file()}") - if name.is_file(): - return name - else: - raise IOError(f"could not locate the file {name}") - - -def _check_dev_loader(name=None, model=None): - if name is None: - name = check_retrieve_file() - - pd.options.display.max_columns = 100 - # config.reader.sep = "\t" - - sep = "\t" - loader1 = DataLoader(sep=sep, model=model) - loader2 = DataLoader(model="one") - loader3 = DataLoader(model="zero") - loader4 = DataLoader(model="zero") - dd = loader1.loader(name) - dd = loader2.loader(name) - dd = loader3.loader(name) - dd = loader4.loader(name) - raw = dd[0].raw - print(len(raw)) - - -def _check_dev_loader2(name=None, model=None, sep=None, number=2): - if name is None: - name = check_retrieve_file(number) - - pd.options.display.max_columns = 100 - - if sep is not None and sep != "none": - loader3 = DataLoader(sep=sep, model=model) - elif sep == "none": - loader3 = DataLoader(sep=None, model=model) - else: - loader3 = DataLoader(model=model) - - dd = loader3.loader(name) - - raw = dd[0].raw - print(len(raw)) - print(raw) - - -def c_heck_loader(name=None, number=1, model="one"): - import matplotlib.pyplot as plt - - if name is None: - name = check_retrieve_file(number) - print(name) - pd.options.display.max_columns = 100 - # config.reader.sep = "\t" - - loader = DataLoader(sep="\t", model=model) - dd = loader.loader(name) - raw = dd[0].raw - caps = [headers_normal.charge_capacity_txt, headers_normal.discharge_capacity_txt] - raw.plot( - x=headers_normal.data_point_txt, - y=headers_normal.current_txt, - title="current vs data-point", - ) - raw.plot( - x=headers_normal.data_point_txt, - y=caps, - title="capacity vs data-point", - ) - raw.plot( - x=headers_normal.test_time_txt, - y=caps, - title="capacity vs test-time", - ) - raw.plot( - x=headers_normal.step_time_txt, - y=caps, - title="capacity vs step-time", - ) - print(raw.head()) - plt.show() - - -def _check_loader_from_outside(): - # NOT EDITED YET!!! - import pathlib - - import matplotlib.pyplot as plt - - from cellpy import cellreader - - pd.options.display.max_columns = 100 - datadir = pathlib.Path( - r"C:\scripts\cellpy_dev_resources\2021_leafs_data\Charge-Discharge\Maccor series 4000" - ) - name = datadir / "01_UBham_M50_Validation_0deg_01.txt" - out = pathlib.Path(r"C:\scripts\notebooks\Div") - print(f"Exists? {name.is_file()}") - - c = cellreader.CellpyCell() - c.set_instrument("maccor_txt", sep="\t") - - c.from_raw(name) - c.set_mass(1000) - - c.make_step_table() - c.make_summary() - - raw = c.data.raw - steps = c.data.steps - summary = c.data.summary - raw.to_csv(r"C:\scripts\notebooks\Div\trash\raw.csv", sep=";") - steps.to_csv(r"C:\scripts\notebooks\Div\trash\steps.csv", sep=";") - summary.to_csv(r"C:\scripts\notebooks\Div\trash\summary.csv", sep=";") - - hdr = c.schema.raw - fig_1, (ax1, ax2, ax3) = plt.subplots(3, 1, figsize=(6, 10)) - raw.plot(x=hdr.test_time, y=hdr.potential, ax=ax1) - raw.plot( - x=hdr.test_time, - y=[hdr.cumulative_charge_capacity, hdr.cumulative_discharge_capacity], - ax=ax3, - ) - raw.plot(x=hdr.test_time, y=hdr.current, ax=ax2) - - n = c.get_number_of_cycles() - print(f"number of cycles: {n}") - - cycle = c.get_cap(1, method="forth-and-forth") - print(cycle.head()) - - # get_cap returns a native CurveCols frame (capacity / potential). - from cellpycore.config import CurveCols - - curve_cols = CurveCols() - fig_2, (ax4, ax5, ax6) = plt.subplots(1, 3) - cycle.plot(x=curve_cols.capacity, y=curve_cols.potential, ax=ax4) - s = c.get_step_numbers() - t = c.sget_timestamp(1, s[1]) - v = c.sget_voltage(1, s[1]) - steps = c.sget_step_numbers(1, s[1]) - - print("step numbers:") - print(s) - print("sget step numbers:") - print(steps) - print("\ntesttime:") - print(t) - print("\nvoltage") - print(v) - - ax5.plot(t, v, label="voltage") - ax6.plot(t, steps, label="steps") - - fig_3, (ax7, ax8) = plt.subplots(2, sharex=True) - raw.plot(x=hdr.test_time, y=hdr.potential, ax=ax7) - raw.plot(x=hdr.test_time, y=hdr.step_num, ax=ax8) - - plt.legend() - plt.show() - - outfile = out / "test_out" - c.save(outfile) - - -def _check_loader_from_outside_with_get(): - import pathlib - - import matplotlib.pyplot as plt - - import cellpy - - pd.options.display.max_columns = 100 - datadir = pathlib.Path( - r"C:\scripting\cellpy_dev_resources\2021_leafs_data\Charge-Discharge\Maccor series 4000" - ) - name = datadir / "01_UBham_M50_Validation_0deg_01.txt" - out = pathlib.Path(r"C:\scripting\trash") - print(f"File exists? {name.is_file()}") - if not name.is_file(): - print(f"could not find {name} ") - return - - c = cellpy.get(filename=name, instrument="maccor_txt", model="one", mass=1.0) - print("loaded") - raw = c.data.raw - steps = c.data.steps - summary = c.data.summary - - raw.to_csv(r"C:\scripting\trash\raw.csv", sep=";") - steps.to_csv(r"C:\scripting\trash\steps.csv", sep=";") - summary.to_csv(r"C:\scripting\trash\summary.csv", sep=";") - - hdr = c.schema.raw - fig_1, (ax1, ax2, ax3) = plt.subplots(3, 1, figsize=(6, 10)) - raw.plot(x=hdr.test_time, y=hdr.potential, ax=ax1, title="voltage") - raw.plot( - x=hdr.test_time, - y=[hdr.cumulative_charge_capacity, hdr.cumulative_discharge_capacity], - ax=ax3, - title="caps", - ) - raw.plot(x=hdr.test_time, y=hdr.current, ax=ax2, title="current") - - n = c.get_number_of_cycles() - print(f"number of cycles: {n}") - - cycle = c.get_cap(1, method="forth-and-forth") - - fig_2, (ax4, ax5, ax6) = plt.subplots(1, 3) - # cycle.plot(x="capacity", y="voltage", ax=ax4) - s = c.get_step_numbers() - t = c.sget_timestamp(1, s[1]) - v = c.sget_voltage(1, s[1]) - steps = c.sget_step_numbers(1, s[1]) - - print("step numbers:") - print(s) - print("sget step numbers:") - print(steps) - print("\ntesttime:") - print(t) - print("\nvoltage") - print(v) - - ax5.plot(t, v, label="voltage") - ax6.plot(t, steps, label="steps") - - fig_3, (ax7, ax8) = plt.subplots(2, sharex=True) - raw.plot(x=hdr.test_time, y=hdr.potential, ax=ax7, title="voltage") - raw.plot(x=hdr.test_time, y=hdr.step_num, ax=ax8, title="step index") - - plt.legend() - plt.show() - - outfile = out / "test_out" - c.save(outfile) - - -def _check_loader_from_outside_with_get2(): - import pathlib - - import matplotlib.pyplot as plt - - import cellpy - from cellpy.parameters.internal_settings import headers_normal - - keep = [ - headers_normal.data_point_txt, - headers_normal.test_time_txt, - headers_normal.step_time_txt, - headers_normal.step_index_txt, - headers_normal.cycle_index_txt, - headers_normal.current_txt, - headers_normal.voltage_txt, - headers_normal.ref_voltage_txt, - headers_normal.charge_capacity_txt, - headers_normal.discharge_capacity_txt, - headers_normal.internal_resistance_txt, - # "ir_pct_change" - ] - - # zero: auto - # one: - # two: KIT - # three | WMB_SIMBA: WMG new - INSTRUMENT = "maccor_txt" - # MODEL = "WMG_SIMBA" - # MODEL = "two" - # MODEL = None - # MODEL = "KIT_SIMBA" - MODEL = "S4000-WMG" - # FILENAME = "1044-CT-MaccorExport.txt" # WMG_SIMBA - # FILENAME = "01_UBham_M50_Validation_0deg_01.txt" # WMG_SIMBA and NONE - # FILENAME = "KIT-Full-cell-PW-HC-CT-cell002.txt" - # FILENAME = "KIT-Full-cell-PW-HC-CT-cell016.txt" - # FILENAME = "KIT-Full-cell-PW-HC-CT-cell013.txt" # Two - # FILENAME = "KIT-Full-cell-PW-HC-CT-cell009.txt" # FAILS in convert_date_time_to_datetime! - # FILENAME = "SIM-A7-1039 - 073.txt" - # FILENAME = "SIM-A7-1047-ET - 079.txt" - FILENAME = "1039_Data from File.txt" - # FILENAME = "1044_Data from File.txt" - # FILENAME = "1475_Data from File.txt" - - DATADIR = r"C:\scripting\cellpy_dev_resources\2021_leafs_data\Data from File" - - pd.options.display.max_columns = 100 - datadir = pathlib.Path(DATADIR) - name = datadir / FILENAME - out = pathlib.Path(r"C:\scripting\trash") - print(f"File exists? {name.is_file()}") - if not name.is_file(): - print(f"could not find {name} ") - return - - c = cellpy.get( - filename=name, instrument=INSTRUMENT, model=MODEL, mass=1.0, auto_summary=False - ) - print(f"loaded the file - now lets see what we got") - raw = c.data.raw - hdr = c.schema.raw - print(raw.head()) - c.make_step_table() - - steps = c.data.steps - summary = c.data.summary - - raw.to_csv(out / "raw.csv", sep=";") - steps.to_csv(out / "steps.csv", sep=";") - summary.to_csv(out / "summary.csv", sep=";") - - fig_1, (ax1, ax2, ax3, ax4) = plt.subplots( - 4, - 1, - figsize=(6, 10), - constrained_layout=True, - sharex=True, - ) - raw.plot(x=hdr.test_time, y=hdr.potential, ax=ax1, xlabel="") - raw.plot(x=hdr.test_time, y=hdr.current, ax=ax2, xlabel="") - raw.plot( - x=hdr.test_time, - y=[hdr.cumulative_charge_capacity, hdr.cumulative_discharge_capacity], - ax=ax3, - xlabel="", - ) - raw.plot(x=hdr.test_time, y=hdr.cycle_num, ax=ax4) - fig_1.suptitle(f"{name.name}", fontsize=16) - - n = c.get_number_of_cycles() - print(f"Number of cycles: {n}") - - plt.legend() - plt.show() - - outfile = out / "test_out" - c.save(outfile) - - -def _fix_bugs_now(): - import pathlib - - import cellpy - import pandas as pd - import matplotlib.pyplot as plt - - DATADIR = r"C:\scripting\cellpy_dev_resources\dev_data\simba_Maccor\S4000" - FILENAME = "KIT-CW-SIMBA-ET-FC-NAPF6-007.037-Full_data.txt" - INSTRUMENT = "maccor_txt" - MODEL = "TWO" - - pd.options.display.max_columns = 100 - datadir = pathlib.Path(DATADIR) - name = datadir / FILENAME - out = pathlib.Path(r"C:\scripting\trash") - print(f"File exists? {name.is_file()}") - if not name.is_file(): - print(f"could not find {name} ") - return - - c = cellpy.get( - filename=name, - instrument=INSTRUMENT, - model=MODEL, - mass=1.0, - auto_summary=False, - # auto_formatter=True, # does not work for UBHAM (finds only two comment rows, but should be three) - # post_processors={"another_param": False}, # This parameter will be available in the post-processors - ) - print(f"loaded the file - now lets see what we got") - raw = c.data.raw - hdr = c.schema.raw - print(raw.head()) - c.make_step_table() - steps = c.data.steps - summary = c.data.summary - - raw.to_csv(out / "raw.csv", sep=";") - steps.to_csv(out / "steps.csv", sep=";") - summary.to_csv(out / "summary.csv", sep=";") - - fig_1, (ax1, ax2, ax3, ax4) = plt.subplots( - 4, - 1, - figsize=(6, 10), - constrained_layout=True, - sharex=True, - ) - raw.plot(x=hdr.test_time, y=hdr.potential, ax=ax1, xlabel="") - raw.plot(x=hdr.test_time, y=hdr.current, ax=ax2, xlabel="") - raw.plot( - x=hdr.test_time, - y=[hdr.cumulative_charge_capacity, hdr.cumulative_discharge_capacity], - ax=ax3, - xlabel="", - ) - raw.plot(x=hdr.test_time, y=hdr.cycle_num, ax=ax4) - fig_1.suptitle(f"{name.name}", fontsize=16) - - n = c.get_number_of_cycles() - print(f"Number of cycles: {n}") - - plt.legend() - plt.show() - - outfile = out / "test_out" - c.save(outfile) - - -if __name__ == "__main__": - # check_dev_loader2(model="two") - # check_loader(number=2, model="two") - _fix_bugs_now() +"""Maccor txt data""" +import cellpy.config as config + +import pandas as pd + +from cellpy import prms +from cellpy import exceptions +from cellpy.parameters.internal_settings import ( + HeaderDict, + base_columns_float, + base_columns_int, + headers_normal, +) +from cellpy.readers.instruments.base import TxtLoader + + +SUPPORTED_MODELS = { + "ZERO": "maccor_txt_zero", # needs to be updated (does no postprocessing at the moment) + "ONE": "maccor_txt_one", + "TWO": "maccor_txt_two", + "THREE": "maccor_txt_three", + "S4000-UBHAM": "maccor_txt_one", + "S4000-KIT": "maccor_txt_two", + "S4000-WMG": "maccor_txt_three", +} + + +MUST_HAVE_RAW_COLUMNS = [ + headers_normal.test_time_txt, + headers_normal.step_time_txt, + headers_normal.current_txt, + headers_normal.voltage_txt, + headers_normal.step_index_txt, + headers_normal.cycle_index_txt, + headers_normal.charge_capacity_txt, + headers_normal.discharge_capacity_txt, +] + + +class DataLoader(TxtLoader): + """Class for loading data from Maccor txt files.""" + + instrument_name = "maccor_txt" + raw_ext = "txt" + + default_model = config.instruments.Maccor.default_model # Required + supported_models = SUPPORTED_MODELS # Required + + def load_since(self, source, marker=None): + """Rows appended since ``marker`` (`SupportsIncrementalLoad`, #780). + + Seeks by data row (``LoadMarker.row_count``) and re-reads from the + start of the last cycle seen; see ``TxtLoader._load_since_rows``. + """ + return self._load_since_rows(source, marker) + + @staticmethod + def get_headers_aux(raw): + """Defines the so-called auxiliary table column headings""" + + headers = HeaderDict() + for col in raw.columns: + if col.startswith("Aux_"): + ncol = col.replace("/", "_") + ncol = "".join(ncol.split("(")[0]) + headers[col] = ncol.lower() + + return headers + + def validate(self, data): + """A simple check that all the needed columns has been successfully + loaded and that they get the correct type""" + missing_must_have_columns = [] + + # validating the float-type raw data + for col in base_columns_float: + if col in data.raw.columns: + data.raw[col] = pd.to_numeric(data.raw[col], errors="coerce") + else: + if col in MUST_HAVE_RAW_COLUMNS: + missing_must_have_columns.append(col) + + # validating the integer-type raw data + for col in base_columns_int: + if col in data.raw.columns: + data.raw[col] = pd.to_numeric( + data.raw[col], errors="coerce", downcast="integer" + ) + else: + if col in MUST_HAVE_RAW_COLUMNS: + missing_must_have_columns.append(col) + + if missing_must_have_columns: + raise exceptions.IOError( + f"Missing needed columns: {missing_must_have_columns}\nAborting!" + ) + return data + + +def _check_retrieve_file(n=1): + import pathlib + + pd.options.display.max_columns = 100 + # config.reader.sep = "\t" + data_root = pathlib.Path(r"C:\scripting\cellpy_dev_resources") + data_dir = data_root / r"2021_leafs_data\Charge-Discharge\Maccor series 4000" + if n == 2: + name = data_dir / "KIT-Full-cell-PW-HC-CT-cell016.txt" + else: + name = data_dir / "01_UBham_M50_Validation_0deg_01.txt" + print(name) + print(f"Exists? {name.is_file()}") + if name.is_file(): + return name + else: + raise IOError(f"could not locate the file {name}") + + +def _check_dev_loader(name=None, model=None): + if name is None: + name = check_retrieve_file() + + pd.options.display.max_columns = 100 + # config.reader.sep = "\t" + + sep = "\t" + loader1 = DataLoader(sep=sep, model=model) + loader2 = DataLoader(model="one") + loader3 = DataLoader(model="zero") + loader4 = DataLoader(model="zero") + dd = loader1.loader(name) + dd = loader2.loader(name) + dd = loader3.loader(name) + dd = loader4.loader(name) + raw = dd[0].raw + print(len(raw)) + + +def _check_dev_loader2(name=None, model=None, sep=None, number=2): + if name is None: + name = check_retrieve_file(number) + + pd.options.display.max_columns = 100 + + if sep is not None and sep != "none": + loader3 = DataLoader(sep=sep, model=model) + elif sep == "none": + loader3 = DataLoader(sep=None, model=model) + else: + loader3 = DataLoader(model=model) + + dd = loader3.loader(name) + + raw = dd[0].raw + print(len(raw)) + print(raw) + + +def c_heck_loader(name=None, number=1, model="one"): + import matplotlib.pyplot as plt + + if name is None: + name = check_retrieve_file(number) + print(name) + pd.options.display.max_columns = 100 + # config.reader.sep = "\t" + + loader = DataLoader(sep="\t", model=model) + dd = loader.loader(name) + raw = dd[0].raw + caps = [headers_normal.charge_capacity_txt, headers_normal.discharge_capacity_txt] + raw.plot( + x=headers_normal.data_point_txt, + y=headers_normal.current_txt, + title="current vs data-point", + ) + raw.plot( + x=headers_normal.data_point_txt, + y=caps, + title="capacity vs data-point", + ) + raw.plot( + x=headers_normal.test_time_txt, + y=caps, + title="capacity vs test-time", + ) + raw.plot( + x=headers_normal.step_time_txt, + y=caps, + title="capacity vs step-time", + ) + print(raw.head()) + plt.show() + + +def _check_loader_from_outside(): + # NOT EDITED YET!!! + import pathlib + + import matplotlib.pyplot as plt + + from cellpy import cellreader + + pd.options.display.max_columns = 100 + datadir = pathlib.Path( + r"C:\scripts\cellpy_dev_resources\2021_leafs_data\Charge-Discharge\Maccor series 4000" + ) + name = datadir / "01_UBham_M50_Validation_0deg_01.txt" + out = pathlib.Path(r"C:\scripts\notebooks\Div") + print(f"Exists? {name.is_file()}") + + c = cellreader.CellpyCell() + c.set_instrument("maccor_txt", sep="\t") + + c.from_raw(name) + c.set_mass(1000) + + c.make_step_table() + c.make_summary() + + raw = c.data.raw + steps = c.data.steps + summary = c.data.summary + raw.to_csv(r"C:\scripts\notebooks\Div\trash\raw.csv", sep=";") + steps.to_csv(r"C:\scripts\notebooks\Div\trash\steps.csv", sep=";") + summary.to_csv(r"C:\scripts\notebooks\Div\trash\summary.csv", sep=";") + + hdr = c.schema.raw + fig_1, (ax1, ax2, ax3) = plt.subplots(3, 1, figsize=(6, 10)) + raw.plot(x=hdr.test_time, y=hdr.potential, ax=ax1) + raw.plot( + x=hdr.test_time, + y=[hdr.cumulative_charge_capacity, hdr.cumulative_discharge_capacity], + ax=ax3, + ) + raw.plot(x=hdr.test_time, y=hdr.current, ax=ax2) + + n = c.get_number_of_cycles() + print(f"number of cycles: {n}") + + cycle = c.get_cap(1, method="forth-and-forth") + print(cycle.head()) + + # get_cap returns a native CurveCols frame (capacity / potential). + from cellpycore.config import CurveCols + + curve_cols = CurveCols() + fig_2, (ax4, ax5, ax6) = plt.subplots(1, 3) + cycle.plot(x=curve_cols.capacity, y=curve_cols.potential, ax=ax4) + s = c.get_step_numbers() + t = c.sget_timestamp(1, s[1]) + v = c.sget_voltage(1, s[1]) + steps = c.sget_step_numbers(1, s[1]) + + print("step numbers:") + print(s) + print("sget step numbers:") + print(steps) + print("\ntesttime:") + print(t) + print("\nvoltage") + print(v) + + ax5.plot(t, v, label="voltage") + ax6.plot(t, steps, label="steps") + + fig_3, (ax7, ax8) = plt.subplots(2, sharex=True) + raw.plot(x=hdr.test_time, y=hdr.potential, ax=ax7) + raw.plot(x=hdr.test_time, y=hdr.step_num, ax=ax8) + + plt.legend() + plt.show() + + outfile = out / "test_out" + c.save(outfile) + + +def _check_loader_from_outside_with_get(): + import pathlib + + import matplotlib.pyplot as plt + + import cellpy + + pd.options.display.max_columns = 100 + datadir = pathlib.Path( + r"C:\scripting\cellpy_dev_resources\2021_leafs_data\Charge-Discharge\Maccor series 4000" + ) + name = datadir / "01_UBham_M50_Validation_0deg_01.txt" + out = pathlib.Path(r"C:\scripting\trash") + print(f"File exists? {name.is_file()}") + if not name.is_file(): + print(f"could not find {name} ") + return + + c = cellpy.get(filename=name, instrument="maccor_txt", model="one", mass=1.0) + print("loaded") + raw = c.data.raw + steps = c.data.steps + summary = c.data.summary + + raw.to_csv(r"C:\scripting\trash\raw.csv", sep=";") + steps.to_csv(r"C:\scripting\trash\steps.csv", sep=";") + summary.to_csv(r"C:\scripting\trash\summary.csv", sep=";") + + hdr = c.schema.raw + fig_1, (ax1, ax2, ax3) = plt.subplots(3, 1, figsize=(6, 10)) + raw.plot(x=hdr.test_time, y=hdr.potential, ax=ax1, title="voltage") + raw.plot( + x=hdr.test_time, + y=[hdr.cumulative_charge_capacity, hdr.cumulative_discharge_capacity], + ax=ax3, + title="caps", + ) + raw.plot(x=hdr.test_time, y=hdr.current, ax=ax2, title="current") + + n = c.get_number_of_cycles() + print(f"number of cycles: {n}") + + cycle = c.get_cap(1, method="forth-and-forth") + + fig_2, (ax4, ax5, ax6) = plt.subplots(1, 3) + # cycle.plot(x="capacity", y="voltage", ax=ax4) + s = c.get_step_numbers() + t = c.sget_timestamp(1, s[1]) + v = c.sget_voltage(1, s[1]) + steps = c.sget_step_numbers(1, s[1]) + + print("step numbers:") + print(s) + print("sget step numbers:") + print(steps) + print("\ntesttime:") + print(t) + print("\nvoltage") + print(v) + + ax5.plot(t, v, label="voltage") + ax6.plot(t, steps, label="steps") + + fig_3, (ax7, ax8) = plt.subplots(2, sharex=True) + raw.plot(x=hdr.test_time, y=hdr.potential, ax=ax7, title="voltage") + raw.plot(x=hdr.test_time, y=hdr.step_num, ax=ax8, title="step index") + + plt.legend() + plt.show() + + outfile = out / "test_out" + c.save(outfile) + + +def _check_loader_from_outside_with_get2(): + import pathlib + + import matplotlib.pyplot as plt + + import cellpy + from cellpy.parameters.internal_settings import headers_normal + + keep = [ + headers_normal.data_point_txt, + headers_normal.test_time_txt, + headers_normal.step_time_txt, + headers_normal.step_index_txt, + headers_normal.cycle_index_txt, + headers_normal.current_txt, + headers_normal.voltage_txt, + headers_normal.ref_voltage_txt, + headers_normal.charge_capacity_txt, + headers_normal.discharge_capacity_txt, + headers_normal.internal_resistance_txt, + # "ir_pct_change" + ] + + # zero: auto + # one: + # two: KIT + # three | WMB_SIMBA: WMG new + INSTRUMENT = "maccor_txt" + # MODEL = "WMG_SIMBA" + # MODEL = "two" + # MODEL = None + # MODEL = "KIT_SIMBA" + MODEL = "S4000-WMG" + # FILENAME = "1044-CT-MaccorExport.txt" # WMG_SIMBA + # FILENAME = "01_UBham_M50_Validation_0deg_01.txt" # WMG_SIMBA and NONE + # FILENAME = "KIT-Full-cell-PW-HC-CT-cell002.txt" + # FILENAME = "KIT-Full-cell-PW-HC-CT-cell016.txt" + # FILENAME = "KIT-Full-cell-PW-HC-CT-cell013.txt" # Two + # FILENAME = "KIT-Full-cell-PW-HC-CT-cell009.txt" # FAILS in convert_date_time_to_datetime! + # FILENAME = "SIM-A7-1039 - 073.txt" + # FILENAME = "SIM-A7-1047-ET - 079.txt" + FILENAME = "1039_Data from File.txt" + # FILENAME = "1044_Data from File.txt" + # FILENAME = "1475_Data from File.txt" + + DATADIR = r"C:\scripting\cellpy_dev_resources\2021_leafs_data\Data from File" + + pd.options.display.max_columns = 100 + datadir = pathlib.Path(DATADIR) + name = datadir / FILENAME + out = pathlib.Path(r"C:\scripting\trash") + print(f"File exists? {name.is_file()}") + if not name.is_file(): + print(f"could not find {name} ") + return + + c = cellpy.get( + filename=name, instrument=INSTRUMENT, model=MODEL, mass=1.0, auto_summary=False + ) + print(f"loaded the file - now lets see what we got") + raw = c.data.raw + hdr = c.schema.raw + print(raw.head()) + c.make_step_table() + + steps = c.data.steps + summary = c.data.summary + + raw.to_csv(out / "raw.csv", sep=";") + steps.to_csv(out / "steps.csv", sep=";") + summary.to_csv(out / "summary.csv", sep=";") + + fig_1, (ax1, ax2, ax3, ax4) = plt.subplots( + 4, + 1, + figsize=(6, 10), + constrained_layout=True, + sharex=True, + ) + raw.plot(x=hdr.test_time, y=hdr.potential, ax=ax1, xlabel="") + raw.plot(x=hdr.test_time, y=hdr.current, ax=ax2, xlabel="") + raw.plot( + x=hdr.test_time, + y=[hdr.cumulative_charge_capacity, hdr.cumulative_discharge_capacity], + ax=ax3, + xlabel="", + ) + raw.plot(x=hdr.test_time, y=hdr.cycle_num, ax=ax4) + fig_1.suptitle(f"{name.name}", fontsize=16) + + n = c.get_number_of_cycles() + print(f"Number of cycles: {n}") + + plt.legend() + plt.show() + + outfile = out / "test_out" + c.save(outfile) + + +def _fix_bugs_now(): + import pathlib + + import cellpy + import pandas as pd + import matplotlib.pyplot as plt + + DATADIR = r"C:\scripting\cellpy_dev_resources\dev_data\simba_Maccor\S4000" + FILENAME = "KIT-CW-SIMBA-ET-FC-NAPF6-007.037-Full_data.txt" + INSTRUMENT = "maccor_txt" + MODEL = "TWO" + + pd.options.display.max_columns = 100 + datadir = pathlib.Path(DATADIR) + name = datadir / FILENAME + out = pathlib.Path(r"C:\scripting\trash") + print(f"File exists? {name.is_file()}") + if not name.is_file(): + print(f"could not find {name} ") + return + + c = cellpy.get( + filename=name, + instrument=INSTRUMENT, + model=MODEL, + mass=1.0, + auto_summary=False, + # auto_formatter=True, # does not work for UBHAM (finds only two comment rows, but should be three) + # post_processors={"another_param": False}, # This parameter will be available in the post-processors + ) + print(f"loaded the file - now lets see what we got") + raw = c.data.raw + hdr = c.schema.raw + print(raw.head()) + c.make_step_table() + steps = c.data.steps + summary = c.data.summary + + raw.to_csv(out / "raw.csv", sep=";") + steps.to_csv(out / "steps.csv", sep=";") + summary.to_csv(out / "summary.csv", sep=";") + + fig_1, (ax1, ax2, ax3, ax4) = plt.subplots( + 4, + 1, + figsize=(6, 10), + constrained_layout=True, + sharex=True, + ) + raw.plot(x=hdr.test_time, y=hdr.potential, ax=ax1, xlabel="") + raw.plot(x=hdr.test_time, y=hdr.current, ax=ax2, xlabel="") + raw.plot( + x=hdr.test_time, + y=[hdr.cumulative_charge_capacity, hdr.cumulative_discharge_capacity], + ax=ax3, + xlabel="", + ) + raw.plot(x=hdr.test_time, y=hdr.cycle_num, ax=ax4) + fig_1.suptitle(f"{name.name}", fontsize=16) + + n = c.get_number_of_cycles() + print(f"Number of cycles: {n}") + + plt.legend() + plt.show() + + outfile = out / "test_out" + c.save(outfile) + + +if __name__ == "__main__": + # check_dev_loader2(model="two") + # check_loader(number=2, model="two") + _fix_bugs_now() diff --git a/cellpy/readers/instruments/neware_txt.py b/cellpy/readers/instruments/neware_txt.py index 7783d5b6..be7c17be 100644 --- a/cellpy/readers/instruments/neware_txt.py +++ b/cellpy/readers/instruments/neware_txt.py @@ -1,124 +1,132 @@ -"""Neware txt data - with explanations how it was implemented. - -1. Update SUPPORTED_MODELS, raw_ext and default_model -2. Add instrument to prms.py - a. create the settings item: - - Neware = {"default_model": "UIO"} - Neware = AttrDict(Neware) - - ... - b. add it to Instruments: - Instruments = InstrumentsClass( - ... - Neware=Neware - ) - - c. Update the dataclass in prms.py: - - @dataclass - class InstrumentsClass(CellPyConfig): - tester: str - custom_instrument_definitions_file: Union[str, None] - Arbin: AttrDict - Maccor: AttrDict - Neware: AttrDict - -3. (optionally) add Neware defaults to .cellpy_prms_default.conf - -4. Create instrument configuration file in readers/instruments/configurations - - formatters - states - normal_headers_renaming_dict - file_info - raw_units - post_processors - -5. Put a file in test_data and create at least one test. -""" -import cellpy.config as config - -import pandas as pd - -from cellpy import prms -from cellpy import exceptions -from cellpy.parameters.internal_settings import ( - HeaderDict, - base_columns_float, - base_columns_int, - headers_normal, -) -from cellpy.readers.instruments.base import TxtLoader - - -SUPPORTED_MODELS = { - "ONE": "neware_txt_zero", - "UIO": "neware_txt_zero", - "UIO_AGA": "neware_txt_one", - "TIOTECH": "neware_txt_two", -} - - -MUST_HAVE_RAW_COLUMNS = [ - headers_normal.test_time_txt, - headers_normal.step_time_txt, - headers_normal.current_txt, - headers_normal.voltage_txt, - headers_normal.step_index_txt, - headers_normal.cycle_index_txt, - headers_normal.charge_capacity_txt, - headers_normal.discharge_capacity_txt, -] - - -class DataLoader(TxtLoader): - """Class for loading data from Neware txt files.""" - - instrument_name = "neware_txt" - raw_ext = "csv" - - default_model = config.instruments.Neware.default_model # Required - supported_models = SUPPORTED_MODELS # Required - - @staticmethod - def get_headers_aux(raw): - """Defines the so-called auxiliary table column headings""" - - headers = HeaderDict() - for col in raw.columns: - if col.startswith("Aux_"): - ncol = col.replace("/", "_") - ncol = "".join(ncol.split("(")[0]) - headers[col] = ncol.lower() - - return headers - - def validate(self, data): - """A simple check that all the needed columns has been successfully - loaded and that they get the correct type""" - missing_must_have_columns = [] - - # validating the float-type raw data - for col in base_columns_float: - if col in data.raw.columns: - data.raw[col] = pd.to_numeric(data.raw[col], errors="coerce") - else: - if col in MUST_HAVE_RAW_COLUMNS: - missing_must_have_columns.append(col) - - # validating the integer-type raw data - for col in base_columns_int: - if col in data.raw.columns: - data.raw[col] = pd.to_numeric( - data.raw[col], errors="coerce", downcast="integer" - ) - else: - if col in MUST_HAVE_RAW_COLUMNS: - missing_must_have_columns.append(col) - - if missing_must_have_columns: - raise exceptions.IOError( - f"Missing needed columns: {missing_must_have_columns}\nAborting!" - ) - return data +"""Neware txt data - with explanations how it was implemented. + +1. Update SUPPORTED_MODELS, raw_ext and default_model +2. Add instrument to prms.py + a. create the settings item: + + Neware = {"default_model": "UIO"} + Neware = AttrDict(Neware) + + ... + b. add it to Instruments: + Instruments = InstrumentsClass( + ... + Neware=Neware + ) + + c. Update the dataclass in prms.py: + + @dataclass + class InstrumentsClass(CellPyConfig): + tester: str + custom_instrument_definitions_file: Union[str, None] + Arbin: AttrDict + Maccor: AttrDict + Neware: AttrDict + +3. (optionally) add Neware defaults to .cellpy_prms_default.conf + +4. Create instrument configuration file in readers/instruments/configurations + + formatters + states + normal_headers_renaming_dict + file_info + raw_units + post_processors + +5. Put a file in test_data and create at least one test. +""" +import cellpy.config as config + +import pandas as pd + +from cellpy import prms +from cellpy import exceptions +from cellpy.parameters.internal_settings import ( + HeaderDict, + base_columns_float, + base_columns_int, + headers_normal, +) +from cellpy.readers.instruments.base import TxtLoader + + +SUPPORTED_MODELS = { + "ONE": "neware_txt_zero", + "UIO": "neware_txt_zero", + "UIO_AGA": "neware_txt_one", + "TIOTECH": "neware_txt_two", +} + + +MUST_HAVE_RAW_COLUMNS = [ + headers_normal.test_time_txt, + headers_normal.step_time_txt, + headers_normal.current_txt, + headers_normal.voltage_txt, + headers_normal.step_index_txt, + headers_normal.cycle_index_txt, + headers_normal.charge_capacity_txt, + headers_normal.discharge_capacity_txt, +] + + +class DataLoader(TxtLoader): + """Class for loading data from Neware txt files.""" + + instrument_name = "neware_txt" + raw_ext = "csv" + + default_model = config.instruments.Neware.default_model # Required + supported_models = SUPPORTED_MODELS # Required + + def load_since(self, source, marker=None): + """Rows appended since ``marker`` (`SupportsIncrementalLoad`, #780). + + Seeks by data row (``LoadMarker.row_count``) and re-reads from the + start of the last cycle seen; see ``TxtLoader._load_since_rows``. + """ + return self._load_since_rows(source, marker) + + @staticmethod + def get_headers_aux(raw): + """Defines the so-called auxiliary table column headings""" + + headers = HeaderDict() + for col in raw.columns: + if col.startswith("Aux_"): + ncol = col.replace("/", "_") + ncol = "".join(ncol.split("(")[0]) + headers[col] = ncol.lower() + + return headers + + def validate(self, data): + """A simple check that all the needed columns has been successfully + loaded and that they get the correct type""" + missing_must_have_columns = [] + + # validating the float-type raw data + for col in base_columns_float: + if col in data.raw.columns: + data.raw[col] = pd.to_numeric(data.raw[col], errors="coerce") + else: + if col in MUST_HAVE_RAW_COLUMNS: + missing_must_have_columns.append(col) + + # validating the integer-type raw data + for col in base_columns_int: + if col in data.raw.columns: + data.raw[col] = pd.to_numeric( + data.raw[col], errors="coerce", downcast="integer" + ) + else: + if col in MUST_HAVE_RAW_COLUMNS: + missing_must_have_columns.append(col) + + if missing_must_have_columns: + raise exceptions.IOError( + f"Missing needed columns: {missing_must_have_columns}\nAborting!" + ) + return data diff --git a/tests/test_load_since.py b/tests/test_load_since.py new file mode 100644 index 00000000..15daa73c --- /dev/null +++ b/tests/test_load_since.py @@ -0,0 +1,249 @@ +"""load_since on the cheap-partial loaders (issue #780, L2). + +Four shipped loaders match ``SupportsIncrementalLoad``; the rest do not. A +chunk is the same harmonized frame a full ``harmonize(parse())`` yields for +those rows, the marker rewinds to the start of the last cycle read, and a +head cell plus a real chunk equals a full load (the #778 oracle). +""" + +from __future__ import annotations + +import pandas as pd +import polars as pl +import pytest + +import cellpy +from cellpy.readers.instruments import ( + arbin_res, + arbin_sql, + biologics_mpr, + local_instrument, + maccor_txt, + neware_txt, + pec_csv, +) +from cellpy.readers.instruments.contract import IncrementalChunk, LoadMarker, SupportsIncrementalLoad +from cellpy.readers.instruments.harmonize import harmonize +from cellpy.readers.instruments.incremental import last_cycle_start +from tests import fdv +from tests.incremental_support import ( + NEWARE_KWARGS, + NEWARE_UIO, + assert_cell_frames_equal, + incremental_update, + truncate_text_file, +) + +MACCOR = NEWARE_UIO.parent / "maccor_001.txt" +RES = NEWARE_UIO.parent / "20160805_test001_45_cc_01.res" + +needs_neware = pytest.mark.skipif(not NEWARE_UIO.is_file(), reason="neware_uio.csv fixture missing") +needs_maccor = pytest.mark.skipif(not MACCOR.is_file(), reason="maccor_001.txt fixture missing") +needs_res = pytest.mark.skipif( + not RES.is_file() or arbin_res.mdb_export_unavailable_reason() is not None, + reason="res fixture or mdbtools missing", +) + + +def _full(loader, path) -> pl.DataFrame: + return harmonize(loader.parse(path), loader.declarations(), strict=False) + + +@pytest.mark.essential +def test_only_the_four_cheap_partial_loaders_are_incremental(): + for module in (neware_txt, maccor_txt, arbin_res, arbin_sql): + assert issubclass(module.DataLoader, SupportsIncrementalLoad), module.__name__ + for module in (pec_csv, biologics_mpr, local_instrument): + assert not issubclass(module.DataLoader, SupportsIncrementalLoad), module.__name__ + + +@pytest.mark.essential +def test_last_cycle_start_finds_the_trailing_run(): + frame = pl.DataFrame({"c": [1, 1, 2, 2, 2, 3]}) + assert last_cycle_start(frame, "c") == 5 + assert last_cycle_start(pl.DataFrame({"c": [4, 4]}), "c") == 0 + assert last_cycle_start(pl.DataFrame({"c": []}), "c") == 0 + assert last_cycle_start(frame, None) == 0 + assert last_cycle_start(frame, "missing") == 0 + + +@needs_neware +@pytest.mark.essential +def test_neware_load_since_none_equals_full_harmonized_read(): + chunk = neware_txt.DataLoader(model="UIO").load_since(NEWARE_UIO, None) + full = _full(neware_txt.DataLoader(model="UIO"), NEWARE_UIO) + assert isinstance(chunk, IncrementalChunk) + assert chunk.new_raw.equals(full) + assert chunk.complete is False + # marker = data-row index of the first row of the last cycle + assert chunk.marker.row_count == last_cycle_start(full, "cycle_num") + assert chunk.marker.last_source_datapoint_num is None + + +@needs_neware +@pytest.mark.essential +def test_neware_load_since_marker_rereads_from_last_cycle_start(tmp_path): + full = _full(neware_txt.DataLoader(model="UIO"), NEWARE_UIO) + # head = everything up to the middle of cycle 3 + cycle3 = full.filter(pl.col("cycle_num") == 3) + cut = int((cycle3["datapoint_num"].min() + cycle3["datapoint_num"].max()) // 2) + head_path = truncate_text_file(NEWARE_UIO, tmp_path / "head.csv", cut) + + head = neware_txt.DataLoader(model="UIO").load_since(head_path, None) + assert head.new_raw.height == cut + cycle3_start = int(cycle3["datapoint_num"].min()) + assert head.marker.row_count == cycle3_start - 1 # datapoint_num is 1-based + + tail = neware_txt.DataLoader(model="UIO").load_since(NEWARE_UIO, head.marker) + assert int(tail.new_raw["datapoint_num"].min()) == cycle3_start + assert tail.new_raw.equals(full.slice(head.marker.row_count)) + assert tail.marker.row_count == last_cycle_start(full, "cycle_num") + + +@needs_neware +@pytest.mark.essential +def test_neware_head_cell_plus_chunk_equals_full_load(tmp_path): + """The #778 oracle driven by a real load_since chunk.""" + full_cell = cellpy.get(NEWARE_UIO, testing=True, **NEWARE_KWARGS) + steps = full_cell.data.steps + st = full_cell.schema.steps + later = steps[steps[st.cycle_num] == 3].iloc[1] + cut = (int(later[st.datapoint_num_first]) + int(later[st.datapoint_num_last])) // 2 + head_path = truncate_text_file(NEWARE_UIO, tmp_path / "head.csv", cut) + head_cell = cellpy.get(head_path, testing=True, **NEWARE_KWARGS) + + marker = neware_txt.DataLoader(model="UIO").load_since(head_path, None).marker + chunk = neware_txt.DataLoader(model="UIO").load_since(NEWARE_UIO, marker) + new_raw = chunk.new_raw.to_pandas() + new_raw[full_cell.schema.raw.test_id] = head_cell.data.active_test_id + + updated = incremental_update(head_cell, new_raw) + assert_cell_frames_equal(updated, full_cell) + + +@needs_neware +@pytest.mark.essential +def test_marker_past_end_of_file_gives_empty_chunk_and_same_marker(): + marker = LoadMarker(row_count=10**7) + chunk = neware_txt.DataLoader(model="UIO").load_since(NEWARE_UIO, marker) + assert chunk.new_raw.height == 0 + assert chunk.marker == marker + + +@needs_neware +@pytest.mark.essential +def test_load_since_does_not_poison_the_parse_cache(): + loader = neware_txt.DataLoader(model="UIO") + loader.load_since(NEWARE_UIO, LoadMarker(row_count=9000)) + assert getattr(loader, "_parsed_frame", None) is None + data = loader.loader(NEWARE_UIO) + assert len(data.raw) == 9065 + + +@needs_maccor +@pytest.mark.essential +def test_maccor_load_since_matches_full_read_and_row_marker(): + full = _full(maccor_txt.DataLoader(), MACCOR) + chunk = maccor_txt.DataLoader().load_since(MACCOR, None) + assert chunk.new_raw.equals(full) + # single-cycle fixture: the rewind lands on row 0 + assert chunk.marker.row_count == 0 + + again = maccor_txt.DataLoader().load_since(MACCOR, chunk.marker) + assert again.new_raw.equals(full) + + # A caller-made marker mid-cycle still reads the right rows; only the + # cycle-local capacity rebase differs, which is why loader-made markers + # rewind to a cycle start. + since = maccor_txt.DataLoader().load_since(MACCOR, LoadMarker(row_count=100)) + assert since.new_raw.height == full.height - 100 + keys = ["datapoint_num", "cycle_num", "step_num", "current", "potential", "test_time"] + assert since.new_raw.select(keys).equals(full.slice(100).select(keys)) + + +@needs_res +@pytest.mark.essential +def test_arbin_res_load_since_seeks_on_datapoint_and_rewinds_to_cycle_start(): + full = _full(arbin_res.DataLoader(), RES) + chunk = arbin_res.DataLoader().load_since(RES, None) + assert chunk.new_raw.equals(full) + last_start_row = last_cycle_start(full, "cycle_num") + last_start_dp = int(full["datapoint_num"][last_start_row]) + assert chunk.marker.last_source_datapoint_num == last_start_dp - 1 + assert chunk.marker.row_count is None + + mid = int(full["datapoint_num"].max()) // 2 + since = arbin_res.DataLoader().load_since(RES, LoadMarker(last_source_datapoint_num=mid)) + assert int(since.new_raw["datapoint_num"].min()) == mid + 1 + assert since.new_raw.equals(full.filter(pl.col("datapoint_num") > mid)) + + tail = arbin_res.DataLoader().load_since(RES, chunk.marker) + assert int(tail.new_raw["datapoint_num"].min()) == last_start_dp + assert tail.new_raw["cycle_num"].n_unique() == 1 + + +def _two_cycle_mock() -> pd.DataFrame: + """The 29-row arbin_sql mock sheet repeated as cycles 1 and 2.""" + one = pd.read_excel(fdv.mock_file_path, sheet_name="arbin_sql") + one["Date_Time"] = one["Date_Time"].astype("int64") + two = one.copy() + two["Data_Point"] = two["Data_Point"] + len(one) + two["Cycle_ID"] = 2 + return pd.concat([one, two], ignore_index=True) + + +@pytest.mark.essential +def test_arbin_sql_load_since_filters_on_datapoint(monkeypatch): + mock = _two_cycle_mock() + seen = [] + + def fake_query(self, name, since_data_point=None): + seen.append(since_data_point) + frame = mock if since_data_point is None else mock[mock["Data_Point"] > since_data_point] + return frame.reset_index(drop=True), pd.DataFrame() + + monkeypatch.setattr(arbin_sql.DataLoader, "_query_sql", fake_query) + + chunk = arbin_sql.DataLoader().load_since("some_test", None) + assert seen == [None] + assert chunk.new_raw.height == len(mock) + # cycle 2 starts at Data_Point 30 -> marker 29 + assert chunk.marker.last_source_datapoint_num == 29 + + tail = arbin_sql.DataLoader().load_since("some_test", chunk.marker) + assert seen[-1] == 29 + assert tail.new_raw.height == 29 + assert int(tail.new_raw["datapoint_num"].min()) == 30 + assert tail.marker == chunk.marker + + empty = arbin_sql.DataLoader().load_since("some_test", LoadMarker(last_source_datapoint_num=10**6)) + assert empty.new_raw.height == 0 + + +@pytest.mark.essential +def test_arbin_sql_query_gets_a_datapoint_clause(monkeypatch): + queries = [] + + class FakeConn: + pass + + def fake_connect(*_args, **_kwargs): + return FakeConn() + + def fake_read_sql_query(sql, _conn): + queries.append(sql) + if "TestList_Table WHERE" in sql and "IV_Basic_Table" not in sql and "StatisticData_Table" not in sql: + return pd.DataFrame({"Database_Name": ["ArbinPro8Data"], "Test_Name": ["t"]}) + return pd.DataFrame() + + monkeypatch.setattr(arbin_sql.pyodbc, "connect", fake_connect) + monkeypatch.setattr(arbin_sql.pd, "read_sql_query", fake_read_sql_query) + + arbin_sql.DataLoader()._query_sql("t", since_data_point=123) + data_queries = [q for q in queries if "IV_Basic_Table" in q] + assert len(data_queries) == 1 + assert "ArbinPro8Data.dbo.IV_Basic_Table.Data_Point > 123" in data_queries[0] + + queries.clear() + arbin_sql.DataLoader()._query_sql("t") + assert all("Data_Point >" not in q for q in queries) From 38e9c9c07525f1ce4b4758c052fcc5fbbf98f9da Mon Sep 17 00:00:00 2001 From: jepegit Date: Sat, 26 Sep 2026 00:41:36 +0200 Subject: [PATCH 3/3] Add CellpyCell.update() for refreshing a cell from a grown raw source (#164) Change detection from file size/mtime, incremental append through load_since + cellpycore update_core_data for arbin_res, arbin_sql, neware_txt and maccor_txt, and a full-reload fallback that keeps cell metadata. Works on cells loaded from a cellpy-file. The #778 test oracle now delegates to the shipped engine. Co-authored-by: Cursor --- .../03-solved-issues/issue164_original.md | 21 ++ .issueflows/03-solved-issues/issue164_plan.md | 43 +++ .../03-solved-issues/issue164_status.md | 34 +++ .../incremental-load-protocol.md | 47 +++- .../04-designs-and-guides/test-registry.md | 9 + AGENTS.md | 3 + HISTORY.md | 7 + cellpy/readers/cellreader.py | 256 ++++++++++++++++++ docs/agents/index.md | 9 + tests/incremental_support.py | 39 +-- tests/test_cell_update.py | 145 ++++++++++ 11 files changed, 578 insertions(+), 35 deletions(-) create mode 100644 .issueflows/03-solved-issues/issue164_original.md create mode 100644 .issueflows/03-solved-issues/issue164_plan.md create mode 100644 .issueflows/03-solved-issues/issue164_status.md create mode 100644 tests/test_cell_update.py diff --git a/.issueflows/03-solved-issues/issue164_original.md b/.issueflows/03-solved-issues/issue164_original.md new file mode 100644 index 00000000..6d4db864 --- /dev/null +++ b/.issueflows/03-solved-issues/issue164_original.md @@ -0,0 +1,21 @@ +# Issue #164: Allow for c.update() + +- GitHub: https://github.com/jepegit/cellpy/issues/164 +- Labels: enhancement, v2, cellpy2-stage4, cellpy2-stage5 +- Epic: #783 (Epic L, live/incremental), Stage 2. Depends on: #780. + +## Original description + +After loading a file (e.g. a cellpyfile), it should be possible to update it by + +```python +c = cellpy.get("cellpyfile.h5) +# the cellpy file contains the paths to the original raw files +c.update() +# only the new data will be loaded and processed +``` + +To achieve this, cellpy will need to have an easy way to find the last loaded +data and load from that. And update summaries from starting from the first new +step (or the last old if it is not complete) etc. This means that a smart way +of lookup and merging if several raw files are used is needed. diff --git a/.issueflows/03-solved-issues/issue164_plan.md b/.issueflows/03-solved-issues/issue164_plan.md new file mode 100644 index 00000000..091fc216 --- /dev/null +++ b/.issueflows/03-solved-issues/issue164_plan.md @@ -0,0 +1,43 @@ +# Plan: #164 `CellpyCell.update()` + +Confirmed approach (autonomous run under #783 stage 2; user asked to +"process the issues"). Builds on #778 (`update_core_data`), #779 (protocol), +#780 (`load_since` on four loaders). + +## Approach + +1. `CellpyCell.update(force=False, **loader_kwargs) -> bool` in + `cellpy/readers/cellreader.py`, before `merge`. +2. Change detection: fresh `FileID` size/mtime vs stored `raw_data_files`; + db sources always "changed". +3. Loader recovery from `data._provenance["source_type"]` via + `set_instrument` (cellpy-file loads carry the default tester). +4. Marker derived from `data.raw` (`_marker_from_raw`: both `row_count` and + `last_source_datapoint_num`, rewound to the last cycle start). No new + cellpy-file field. +5. Incremental path (single fid + native schema + harmonized raw + loader + matches `SupportsIncrementalLoad`): `load_since` → stamp `test_id`, align + dtypes → `_update_from_raw_rows` (`core.update_core_data` + + `_add_summary_extras` + `_refresh_scaled_summary_columns`). +6. Fallback full reload (`ValueError`/`LoaderError`, multi-file, + non-incremental loader): `from_raw` on all sources, restore + `meta_common` / `cycle_mode` / `cell_name`, `make_step_table`, + `make_summary(find_ir=...)`. +7. `_refresh_fid`: size/mtimes, `last_data_point`, `raw_data_files_length`. +8. `tests/incremental_support.incremental_update` delegates to the new engine. + +## Tests (`tests/test_cell_update.py`, essential) + +no-op, growth == full load, two growths (marker), FileID refresh, cellpy-file +round trip, single-cycle fallback keeps meta, force, no source raises, +non-incremental loader routes to full reload. + +## Docs + +`incremental-load-protocol.md` section for #164, `docs/agents/index.md`, +root `AGENTS.md` quick facts, HISTORY bullet, test-registry rows. + +## Out of scope + +Poll loop (#781), batch live refresh (#782), multi-file incremental merge +(falls back to full reload). diff --git a/.issueflows/03-solved-issues/issue164_status.md b/.issueflows/03-solved-issues/issue164_status.md new file mode 100644 index 00000000..3c373fe0 --- /dev/null +++ b/.issueflows/03-solved-issues/issue164_status.md @@ -0,0 +1,34 @@ +# Status: #164 `CellpyCell.update()` + +- [x] Done + +## Done + +- `CellpyCell.update()` + helpers (`_raw_sources_changed`, + `_ensure_loader_for_update`, `_marker_from_raw`, `_update_incremental`, + `_update_from_raw_rows`, `_summary_has_ir`, `_refresh_fid`, + `_update_full_reload`) and module helpers `_frame_to_pandas`, + `_align_dtypes` in `cellpy/readers/cellreader.py`; `_load_marker` attribute + initialised in `__init__`. +- `tests/incremental_support.incremental_update` now delegates to + `CellpyCell._update_from_raw_rows` (the #778 oracle covers the shipped engine). +- `tests/test_cell_update.py`: 9 essential tests. +- Docs: `incremental-load-protocol.md` (#164 section), `docs/agents/index.md`, + root `AGENTS.md`, HISTORY `[Unreleased]`, test-registry. + +## Verification + +- `uv run pytest tests/test_cell_update.py tests/test_incremental_update.py tests/test_load_since.py` green. +- `uv run pytest -m essential` — see PR. + +## Notes + +- Branch `164-cell-update` stacked on `780-load-since` (PR #1101); PR base + is `780-load-since` until #1101 merges. +- Multi-file cells and non-incremental loaders take the full-reload path + (documented; the smart multi-file merge from the original issue is not + needed for the live use case, one file per running test). + +## Remaining + +- None for this issue. Stage 3: #781 (poll loop), #782 (batch live refresh). diff --git a/.issueflows/04-designs-and-guides/incremental-load-protocol.md b/.issueflows/04-designs-and-guides/incremental-load-protocol.md index a5b38801..bd978122 100644 --- a/.issueflows/04-designs-and-guides/incremental-load-protocol.md +++ b/.issueflows/04-designs-and-guides/incremental-load-protocol.md @@ -61,7 +61,52 @@ expose `load_since`). cycle-local rebase may differ from a full load. Only loader-made markers carry the equality guarantee. +## `CellpyCell.update()` (#164) + +The public consumer of the protocol, in `cellpy/readers/cellreader.py` +(section "incremental refresh"). Returns `bool` (frames changed). + +- **Change detection** compares a fresh `FileID(fid.full_name)` size and + mtime with the stored `raw_data_files` entry (same stats + `check_file_ids` uses). Databases (`is_db`) and unreadable stats always + count as changed. `force=True` skips the check. +- **Loader recovery.** A cell loaded from a cellpy-file carries the config + default tester; `data._provenance["source_type"]` (persisted) names the + loader that read the raw. `update()` calls `set_instrument` from it (and + forwards `**loader_kwargs`, e.g. `model=`, since the model is not + persisted). Decision: no new cellpy-file field. +- **Marker without state.** The marker is not persisted either. When the + cell has no in-memory `_load_marker`, `_marker_from_raw` derives one from + `data.raw` with **both** seek fields filled (`row_count` = index of the + first row of the last cycle; `last_source_datapoint_num` = the datapoint + before it), so text and arbin loaders each find their field. Rewinding to + the cycle start mirrors the loaders' own policy above. Alternative + rejected: storing the marker in the cellpy-file (extra schema, and the + derived one is exact for loader-made markers anyway). +- **Incremental path** only when: one raw file, `native_schema`, + `config.reader.use_harmonized_raw`, and + `isinstance(loader_class, SupportsIncrementalLoad)`. The chunk gets + `test_id = active_test_id` and its dtypes cast to the existing raw's + (`_align_dtypes`; harmonize can yield Int32 where raw has Int64), then + goes through `_update_from_raw_rows` → `core.update_core_data` with the + by-value inputs cellpy owns (`nom_cap_abs`, current factor, raw limits), + `_add_summary_extras`, and `_refresh_scaled_summary_columns`. + `find_ir` follows whether the current summary has `ir_charge`. +- **Fallback = full reload** on `ValueError` / `LoaderError` from the + incremental path (typically core refusing a chunk whose start is at or + before the first kept row, i.e. a single-cycle head), for multi-file + cells, and for non-incremental loaders. `from_raw` on all recorded + sources, then `meta_common` (deep copy), `cycle_mode`, and `cell_name` + are restored before `make_step_table()` / `make_summary(find_ir=...)`. +- **FileID refresh** after either path: size / mtimes, `last_data_point` + (max datapoint), `raw_data_files_length[-1]`. A second `update()` on the + same file is then a no-op. +- `tests/incremental_support.incremental_update` (the #778 oracle) now + delegates to `CellpyCell._update_from_raw_rows`, so the equality tests + cover the shipped engine. + ## Link Design §3 in `cellpy-design-and-development/active/cellpy2-live-incremental-design.md`. -`CellpyCell.update()` is #164. Tests: `tests/test_load_since.py`. +Tests: `tests/test_load_since.py` (#780), `tests/test_cell_update.py` (#164). +Consumers: `live.py` poll loop (#781), batch live refresh (#782). diff --git a/.issueflows/04-designs-and-guides/test-registry.md b/.issueflows/04-designs-and-guides/test-registry.md index c8e5d666..9f5fdcd9 100644 --- a/.issueflows/04-designs-and-guides/test-registry.md +++ b/.issueflows/04-designs-and-guides/test-registry.md @@ -200,6 +200,15 @@ current issue**. `/iflow-doctor` may audit the whole suite against this table. | tests/test_load_since.py::test_arbin_res_load_since_seeks_on_datapoint_and_rewinds_to_cycle_start | yes | yes | arbin_res.load_since | #780 | skip without mdbtools | | tests/test_load_since.py::test_arbin_sql_load_since_filters_on_datapoint | yes | yes | arbin_sql.load_since | #780 | mocked `_query_sql` | | tests/test_load_since.py::test_arbin_sql_query_gets_a_datapoint_clause | yes | yes | arbin_sql._query_sql | #780 | SQL clause on the fully qualified table | +| tests/test_cell_update.py::test_update_on_unchanged_source_is_a_noop | yes | yes | CellpyCell.update / _raw_sources_changed | #164 | size+mtime unchanged → False | +| tests/test_cell_update.py::test_update_after_growth_equals_full_load | yes | yes | CellpyCell.update (incremental) | #164 | head + grown tail == full cellpy.get | +| tests/test_cell_update.py::test_update_twice_tracks_the_marker | yes | yes | CellpyCell._marker_from_raw / _load_marker | #164 | two growths; marker rewinds | +| tests/test_cell_update.py::test_update_refreshes_file_id | yes | yes | CellpyCell._refresh_fid | #164 | size / last_data_point / lengths; second update no-op | +| tests/test_cell_update.py::test_update_after_cellpy_file_round_trip | yes | yes | CellpyCell._ensure_loader_for_update | #164 | loader from provenance; mass kept | +| tests/test_cell_update.py::test_update_falls_back_to_full_reload_and_keeps_meta | yes | yes | CellpyCell._update_full_reload | #164 | single-cycle head → core rejects → reload | +| tests/test_cell_update.py::test_update_force_reloads_an_unchanged_source | yes | yes | CellpyCell.update(force=True) | #164 | | +| tests/test_cell_update.py::test_update_without_raw_source_raises | yes | yes | CellpyCell.update | #164 | NoDataFound | +| tests/test_cell_update.py::test_update_uses_full_reload_for_non_incremental_loader | yes | yes | CellpyCell.update (protocol gate) | #164 | loader without load_since → full reload | | tests/test_dbreader.py::test_missing_column_warns_once | yes | yes | readers.dbreader.Reader._pick_info | #1008 | warn-once per missing header | | tests/test_dbreader.py::test_nom_cap_specifics_column_reaches_pages | yes | yes | batch._dbengine._create_pages_dict | #1008 | db value → pages | | tests/test_dbreader.py::test_simple_db_engine_skip_file_search_excel_reader | yes | yes | batch._dbengine.simple_db_engine / find_files | #1017 | skip_file_search frames one row per cell | diff --git a/AGENTS.md b/AGENTS.md index 596ecb77..722bb490 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -374,6 +374,9 @@ Quick facts: `kind=raw` lists raw and sets `needs_metadata`). - Metadata peek (no frames): `cellpy.read_meta(path)` → dict with `cell` / `tests`. - Ingestion form fields: `cellpy.instrument_meta_schema(instrument)` → `fields` / `units`. +- Live/running test: `c.update()` re-reads only the new raw rows (incremental + loaders) or reloads fully, returns `True` when frames changed; `False` if + the raw file did not change on disk. Also works after `cellpy.get(".cellpy")`. - Frames: `c.data.raw` / `.steps` / `.summary`; columns via `c.schema.*`. After a raw load, each cycle's raw capacity starts at 0. A forgotten tester reset that 1.x plotted as doubled capacity is rebased on load for every diff --git a/HISTORY.md b/HISTORY.md index f4b0b801..05f5dc20 100644 --- a/HISTORY.md +++ b/HISTORY.md @@ -11,6 +11,13 @@ next marker, which rewinds to the start of the last cycle so cycle-local capacity normalisation stays correct. Other loaders stay full-read. (#780) +* `CellpyCell.update()`: refresh a cell from a raw file that grew. Detects + changes from file size/mtime, reads only the new rows for incremental + loaders (arbin_res, arbin_sql, neware_txt, maccor_txt) and appends them + through cellpy-core, otherwise reloads fully while keeping mass, area, + nominal capacity, cycle mode and cell name. Works on cells loaded from a + cellpy-file. Returns `True` when the frames changed. (#164) + ## [2.1.5.post6] - 2026-09-25 * `summary_collector(...).plot()` keeps a lone charge or discharge series diff --git a/cellpy/readers/cellreader.py b/cellpy/readers/cellreader.py index 603f1e52..8a06c4fe 100644 --- a/cellpy/readers/cellreader.py +++ b/cellpy/readers/cellreader.py @@ -37,6 +37,7 @@ from cellpy.exceptions import ( DeprecatedFeature, + LoaderError, MixedCycleModesError, NoDataFound, ) @@ -123,6 +124,29 @@ } +def _frame_to_pandas(frame): + """pandas view of a core (polars) frame; pandas passes through.""" + return frame.to_pandas() if hasattr(frame, "to_pandas") else frame + + +def _align_dtypes(chunk, existing): + """Cast the shared columns of a polars ``chunk`` to ``existing``'s dtypes. + + A freshly harmonized chunk can carry narrower integer types (Int32 for a + literal ``test_id``, for example) than the raw frame it is appended to; + core's vertical concat requires an exact dtype match. + """ + import polars as pl + + target = existing if isinstance(existing, pl.DataFrame) else pl.from_pandas(existing) + casts = [ + pl.col(name).cast(dtype) + for name, dtype in target.schema.items() + if name in chunk.columns and chunk.schema[name] != dtype + ] + return chunk.with_columns(casts) if casts else chunk + + def normalize_summary_meta_fields(fields=None): """Normalize meta field names for ``SUMMARY_META_DEPENDENCIES`` / ``refresh_after``. @@ -338,6 +362,9 @@ def __init__( self.tester = tester self.loader = None # this will be set in the function set_instrument + #: Incremental-load position for ``update()`` (#164); derived from + #: the raw frame when None, so it is never persisted. + self._load_marker = None self.debug = debug logging.debug("created CellpyCell instance") @@ -1870,6 +1897,235 @@ def _convert2fid_list(tbl): # -------------------- cellpy file handling end ---------------------- + # -------------------- incremental refresh (#164) -------------------- + + def update(self, force=False, **loader_kwargs): + """Refresh this cell from its raw source(s) if they have grown. + + The headline live/incremental feature (Epic L, cellpy 2.2). Works on a + cell loaded from raw and on a cell loaded from a cellpy-file (the + raw-file ids stored there name the source). + + Flow: + + 1. Change detection on the recorded raw files (size and mtime, as in + ``check_file_ids``). Unchanged and not ``force`` → no-op. + 2. Single source whose loader implements ``SupportsIncrementalLoad`` + (arbin_res, arbin_sql, neware_txt, maccor_txt): read only the rows + since the load marker and append them through core + ``update_core_data`` (overlap trimmed, affected steps rebuilt, + summary refreshed). The marker is derived from the loaded raw when + this cell has none yet, so a cellpy-file round trip needs no extra + state. + 3. Otherwise, or when the incremental path rejects the chunk (for + example the head held a single cycle): full reload of every raw + file, then step table and summary. Cell metadata (mass, nominal + capacity, area, cycle mode, name) is kept. + + Args: + force: refresh even when the file stats did not change. + **loader_kwargs: forwarded to ``set_instrument`` when the loader has + to be (re)created from the stored provenance, e.g. + ``model="UIO"`` for a neware export. + + Returns: + bool: ``True`` when the frames changed, ``False`` for a no-op. + + Raises: + NoDataFound: if the cell has no recorded raw source. + + Examples: + ```python + c = cellpy.get("running_test.csv", instrument="neware_txt") + ... # the tester keeps writing + if c.update(): + print(c.data.summary.tail(1)) + ``` + """ + data = self.data + fids = [f for f in (data.raw_data_files or []) if f is not None] + if not fids: + raise NoDataFound("cannot update: no raw source recorded on this cell") + + if not force and not self._raw_sources_changed(fids): + logging.info("update: raw source(s) unchanged") + return False + + self._ensure_loader_for_update(**loader_kwargs) + + from cellpy.readers.instruments.contract import SupportsIncrementalLoad + + loader = self.loader_class + can_go_incremental = ( + len(fids) == 1 + and self.native_schema + and getattr(config.reader, "use_harmonized_raw", True) + and isinstance(loader, SupportsIncrementalLoad) + ) + if can_go_incremental: + try: + return self._update_incremental(fids[0], loader) + except (ValueError, LoaderError) as exc: + logging.info(f"update: incremental path declined ({exc}); reloading") + return self._update_full_reload(fids, **loader_kwargs) + + def _raw_sources_changed(self, fids) -> bool: + """True if any recorded raw file differs from disk in size or mtime. + + Sources without file stats (databases, missing files) count as changed + so the caller still tries to refresh them. + """ + for fid in fids: + if getattr(fid, "is_db", False) or not fid.full_name: + return True + current = ds.FileID(fid.full_name) + if current.name is None: + return True + if fid.size is None or fid.last_modified is None: + return True + if int(current.size) != int(fid.size): + return True + if float(current.last_modified) != float(fid.last_modified): + return True + return False + + def _ensure_loader_for_update(self, **loader_kwargs): + """Recreate the loader from stored provenance when the tester changed. + + A cell loaded from a cellpy-file carries the default instrument, not + the one that read the raw file; ``_provenance['source_type']`` + remembers it. + """ + provenance = getattr(self.data, "_provenance", None) or {} + source_type = provenance.get("source_type") + if loader_kwargs or (source_type and source_type != self.tester): + instrument = source_type or self.tester + logging.debug(f"update: setting instrument {instrument} ({loader_kwargs})") + self.set_instrument(instrument=instrument, **loader_kwargs) + self.tester = instrument + + def _marker_from_raw(self): + """Derive a `LoadMarker` from the loaded raw when none is stored. + + Both seek fields are filled so any implementing loader can read the + one it uses: ``row_count`` = row index of the first row of the last + cycle (text loaders); ``last_source_datapoint_num`` = the datapoint + just before that row (arbin). Rewinding to the cycle start mirrors the + loaders' own marker policy (see ``instruments/incremental.py``). + """ + import polars as pl + + from cellpy.readers.instruments.contract import LoadMarker + from cellpy.readers.instruments.incremental import last_cycle_start + + raw = self.data.raw + frame = raw if isinstance(raw, pl.DataFrame) else pl.from_pandas(raw) + if frame.height == 0: + return None + cycle_column = self.schema.raw.cycle_num + start_row = last_cycle_start(frame, cycle_column) + datapoint_column = self.schema.raw.datapoint_num + datapoint = int(frame.get_column(datapoint_column)[start_row]) + return LoadMarker(last_source_datapoint_num=datapoint - 1, row_count=start_row) + + def _update_incremental(self, fid, loader) -> bool: + import polars as pl + + marker = getattr(self, "_load_marker", None) or self._marker_from_raw() + if marker is None: + raise ValueError("no raw rows to derive a load marker from") + source = fid.full_name if not getattr(fid, "is_db", False) else fid.name + chunk = loader.load_since(source, marker) + self._load_marker = chunk.marker + if chunk.new_raw is None or chunk.new_raw.height == 0: + self._refresh_fid(fid) + return False + + test_id_column = self.schema.raw.test_id + new_raw = chunk.new_raw.with_columns(pl.lit(int(self.data.active_test_id)).alias(test_id_column)) + new_raw = _align_dtypes(new_raw, self.data.raw) + self._update_from_raw_rows(new_raw, find_ir=self._summary_has_ir()) + self._refresh_fid(fid) + logging.info(f"update: appended {chunk.new_raw.height} raw rows (incremental)") + return True + + def _update_from_raw_rows(self, new_raw, find_ir=True): + """Append ``new_raw`` (native schema) through core ``update_core_data``. + + Mirrors the cellpy-side orchestration in ``make_step_table`` / + ``make_summary`` (by-value nominal capacity, current factor, raw + limits) and re-applies the summary extras and the scaled (mass/area) + columns, so the result compares with a full ``cellpy.get``. Raises + ``ValueError`` when core rejects the chunk (full reload territory). + """ + from cellpy.readers.native_core import _add_summary_extras + + data = self.data + factor = core_units.calculate_current_conversion_factor( + data.raw_units["current"], to_units=self.cellpy_units + ) + nom_cap_abs = self._resolve_nom_cap_abs(data) + out = self.core.update_core_data( + data, + new_raw, + nom_cap_abs=nom_cap_abs, + current_conversion_factor=factor, + find_ir=find_ir, + raw_limits=self.raw_limits, + ) + # ``update_core_data`` returns a bare cellpycore ``Data``; copy the + # frames back so cellpy's metadata-bearing ``Data`` stays the owner. + data.raw = _frame_to_pandas(out.raw) + data.steps = _frame_to_pandas(out.steps) + data.summary = _add_summary_extras(_frame_to_pandas(out.summary), self.core.schema) + self._refresh_scaled_summary_columns() + return self + + def _summary_has_ir(self) -> bool: + summary = getattr(self.data, "summary", None) + if summary is None or getattr(summary, "empty", True): + return True + return self.schema.summary.ir_charge in summary.columns + + def _refresh_fid(self, fid): + """Re-stat the raw source and record the new tail on its `FileID`.""" + if not getattr(fid, "is_db", False) and fid.full_name: + current = ds.FileID(fid.full_name) + if current.name is not None: + fid.size = current.size + fid.last_modified = current.last_modified + fid.last_accessed = current.last_accessed + fid.last_info_changed = current.last_info_changed + raw = self.data.raw + datapoint_column = self.schema.raw.datapoint_num + if datapoint_column in raw.columns and len(raw): + fid.last_data_point = int(raw[datapoint_column].max()) + if self.data.raw_data_files_length: + self.data.raw_data_files_length[-1] = len(raw) + + def _update_full_reload(self, fids, **loader_kwargs) -> bool: + sources = [f.name if getattr(f, "is_db", False) else f.full_name for f in fids] + old = self.data + keep_meta = copy.deepcopy(old.meta_common) + keep_cycle_mode = self.cycle_mode + keep_name = self._cell_name + find_ir = self._summary_has_ir() + is_a_file = self.tester not in DB_READER_INSTRUMENTS + + self.from_raw(file_names=sources, is_a_file=is_a_file) + self.data.meta_common = keep_meta + if keep_cycle_mode is not None: + self.cycle_mode = keep_cycle_mode + if keep_name is not None: + self._cell_name = keep_name + self.make_step_table() + self.make_summary(find_ir=find_ir) + self._load_marker = None + logging.info("update: full reload") + return True + + # -------------------- incremental refresh end ----------------------- + def merge(self, cells, mode="campaign", renumber_cycles=True, **kwargs): """Merge other cells/datasets into this one. diff --git a/docs/agents/index.md b/docs/agents/index.md index e59187f4..ffffd19e 100644 --- a/docs/agents/index.md +++ b/docs/agents/index.md @@ -169,6 +169,15 @@ Useful methods on `CellpyCell` (non-exhaustive): meta-dependent columns (cheaper than a full `make_summary()`). See `cellpy.readers.cellreader.SUMMARY_META_DEPENDENCIES` for the map GUIs can use for messaging. +- `update()` — refresh a cell from a raw file that is still being written + (a running test). Returns `True` if the frames changed, `False` when the + file's size/mtime are unchanged (`force=True` overrides). With + `arbin_res` / `arbin_sql` / `neware_txt` / `maccor_txt` only the new rows + are read and appended (the last cycle is re-read whole); other loaders, + or a head too short to append to, fall back to a full reload that keeps + mass / area / nominal capacity / cycle mode. Works on a cell loaded from a + `.cellpy` file too (the raw path is stored in it); pass loader kwargs + such as `model="UIO"` when the instrument needs them. - `save` / `to_csv` / Excel helpers — persist for the user's workflow Deeper shape docs: [Data structure](../fundamentals/data_structure.md). diff --git a/tests/incremental_support.py b/tests/incremental_support.py index 52e4736e..c4c0733c 100644 --- a/tests/incremental_support.py +++ b/tests/incremental_support.py @@ -4,11 +4,9 @@ ``update_core_data`` must produce the same ``raw`` / ``steps`` / ``summary`` as a single full load. -``incremental_update`` is a test-side prototype of what L3 (#164) will expose as -``CellpyCell.update()``: it drives ``update_core_data`` with the cellpy-owned -by-value inputs (nominal capacity, current factor, instrument raw limits) and then -re-applies the cellpy-side summary extras and the scaled (mass/area) columns. -Replace this helper with the public API once L3 lands. +``incremental_update`` feeds a tail frame through the same code path L3 (#164) +uses inside ``CellpyCell.update()`` (``CellpyCell._update_from_raw_rows``), so +these tests stay the oracle for the shipped implementation. """ from __future__ import annotations @@ -17,9 +15,6 @@ import pandas as pd import pandas.testing as pdt -from cellpycore import units as core_units - -from cellpy.readers.native_core import _add_summary_extras REPO_ROOT = Path(__file__).resolve().parents[1] NEWARE_UIO = REPO_ROOT / "testdata" / "data" / "neware_uio.csv" @@ -39,33 +34,9 @@ def tail_rows(raw: pd.DataFrame, datapoint_col: str, since: int, overlap: int = return raw[raw[datapoint_col] > since - overlap] -def _to_pandas(frame): - return frame.to_pandas() if hasattr(frame, "to_pandas") else frame - - def incremental_update(cell, new_raw: pd.DataFrame, find_ir: bool = True): - """Append ``new_raw`` to ``cell`` in place via ``update_core_data``. - - Mirrors the cellpy-side orchestration in ``make_step_table`` / - ``make_summary`` so the result is comparable with a full ``cellpy.get``. - """ - factor = core_units.calculate_current_conversion_factor(cell.data.raw_units["current"], to_units=cell.cellpy_units) - nom_cap_abs = cell._resolve_nom_cap_abs(cell.data) - out = cell.core.update_core_data( - cell.data, - new_raw, - nom_cap_abs=nom_cap_abs, - current_conversion_factor=factor, - find_ir=find_ir, - raw_limits=cell.raw_limits, - ) - # ``update_core_data`` returns a bare cellpycore ``Data``; copy the frames back - # so cellpy's metadata-bearing ``Data`` stays the owner. - cell.data.raw = _to_pandas(out.raw) - cell.data.steps = _to_pandas(out.steps) - cell.data.summary = _add_summary_extras(_to_pandas(out.summary), cell.core.schema) - cell._refresh_scaled_summary_columns() - return cell + """Append ``new_raw`` to ``cell`` in place through ``CellpyCell.update()``'s engine.""" + return cell._update_from_raw_rows(new_raw, find_ir=find_ir) def _normalize(frame: pd.DataFrame, sort_by) -> pd.DataFrame: diff --git a/tests/test_cell_update.py b/tests/test_cell_update.py new file mode 100644 index 00000000..ecb8972d --- /dev/null +++ b/tests/test_cell_update.py @@ -0,0 +1,145 @@ +"""``CellpyCell.update()`` (#164): refresh a cell from a raw source that grew. + +The oracle is a full ``cellpy.get`` of the complete file: a cell loaded from a +truncated copy, then ``update()``-ed after the copy grew, must carry the same +raw / steps / summary frames. +""" + +from __future__ import annotations + +import shutil + +import pytest + +import cellpy +from cellpy.exceptions import NoDataFound +from cellpy.readers.cellreader import CellpyCell +from tests.incremental_support import ( + NEWARE_KWARGS, + NEWARE_UIO, + assert_cell_frames_equal, + truncate_text_file, +) + +pytestmark = pytest.mark.essential + +MID_CYCLE_3 = 6000 # rows; well inside the third of four cycles +MID_CYCLE_4 = 8800 +SINGLE_CYCLE = 1000 # rows; still inside the first cycle + + +@pytest.fixture(scope="module") +def full_cell(): + return cellpy.get(NEWARE_UIO, testing=True, **NEWARE_KWARGS) + + +@pytest.fixture +def live_file(tmp_path): + return truncate_text_file(NEWARE_UIO, tmp_path / "live.csv", MID_CYCLE_3) + + +def _grow(path, n_rows=None): + if n_rows is None: + shutil.copyfile(NEWARE_UIO, path) + else: + truncate_text_file(NEWARE_UIO, path, n_rows) + + +def test_update_on_unchanged_source_is_a_noop(live_file): + c = cellpy.get(live_file, testing=True, **NEWARE_KWARGS) + n_raw = len(c.data.raw) + assert c.update() is False + assert len(c.data.raw) == n_raw + + +def test_update_after_growth_equals_full_load(live_file, full_cell): + c = cellpy.get(live_file, testing=True, **NEWARE_KWARGS) + _grow(live_file) + assert c.update() is True + assert_cell_frames_equal(c, full_cell) + + +def test_update_twice_tracks_the_marker(live_file, full_cell): + c = cellpy.get(live_file, testing=True, **NEWARE_KWARGS) + _grow(live_file, MID_CYCLE_4) + assert c.update() is True + assert len(c.data.raw) == MID_CYCLE_4 + marker = c._load_marker + assert marker is not None and marker.row_count is not None + assert marker.row_count < MID_CYCLE_4 # rewound to the last cycle start + _grow(live_file) + assert c.update() is True + assert_cell_frames_equal(c, full_cell) + + +def test_update_refreshes_file_id(live_file): + c = cellpy.get(live_file, testing=True, **NEWARE_KWARGS) + fid = c.data.raw_data_files[0] + old_size, old_last = fid.size, fid.last_data_point + _grow(live_file) + c.update() + assert fid.size > old_size + assert fid.last_data_point > old_last + assert fid.last_data_point == int(c.data.raw[c.schema.raw.datapoint_num].max()) + assert c.data.raw_data_files_length[-1] == len(c.data.raw) + assert c.update() is False # stats now match the file again + + +def test_update_after_cellpy_file_round_trip(live_file, tmp_path): + c = cellpy.get(live_file, testing=True, mass=1.3, **NEWARE_KWARGS) + cellpy_file = tmp_path / "live.cellpy" + c.save(cellpy_file) + reloaded = cellpy.get(cellpy_file, testing=True) + assert reloaded.tester != "neware_txt" # the file does not carry the loader + _grow(live_file) + assert reloaded.update() is True + assert reloaded.tester == "neware_txt" + assert reloaded.mass == pytest.approx(1.3) + expected = cellpy.get(NEWARE_UIO, testing=True, mass=1.3, **NEWARE_KWARGS) + assert_cell_frames_equal(reloaded, expected) + + +def test_update_falls_back_to_full_reload_and_keeps_meta(tmp_path): + live = truncate_text_file(NEWARE_UIO, tmp_path / "live.csv", SINGLE_CYCLE) + c = cellpy.get(live, testing=True, mass=2.5, **NEWARE_KWARGS) + c.cell_name = "keep-me" + _grow(live) + # A single-cycle head leaves nothing before the rewind point, so core + # rejects the chunk and update() reloads the whole file instead. + assert c.update() is True + assert c.mass == pytest.approx(2.5) + assert c.cell_name == "keep-me" + expected = cellpy.get(NEWARE_UIO, testing=True, mass=2.5, **NEWARE_KWARGS) + assert_cell_frames_equal(c, expected) + + +def test_update_force_reloads_an_unchanged_source(live_file): + c = cellpy.get(live_file, testing=True, **NEWARE_KWARGS) + before = c.data.raw.copy() + assert c.update(force=True) is True + assert len(c.data.raw) == len(before) + + +def test_update_without_raw_source_raises(): + c = CellpyCell(initialize=True) + with pytest.raises(NoDataFound): + c.update() + + +def test_update_uses_full_reload_for_non_incremental_loader(live_file, full_cell, monkeypatch): + c = cellpy.get(live_file, testing=True, **NEWARE_KWARGS) + calls = [] + + def _no_incremental(*args, **kwargs): + calls.append(1) + raise AssertionError("incremental path must not run") + + monkeypatch.setattr(c, "_update_incremental", _no_incremental) + # Pretend the loader lacks load_since (a fresh subclass sidesteps the ABC + # isinstance cache): the protocol check must route to full reload. + not_incremental = type("NotIncremental", (type(c.loader_class),), {"load_since": None}) + c.loader_class.__class__ = not_incremental + _grow(live_file) + assert c.update() is True + assert calls == [] + assert_cell_frames_equal(c, full_cell)