"""Reusable validator functions and composable pipelines for value normalization.
Provides functions for composable inbound/outbound value normalization
(e.g., converting between pandas DataFrames and dictionaries).
"""
from collections.abc import Callable, Iterable
from dataclasses import dataclass, field
from datetime import UTC, datetime
from enum import StrEnum
import math
import os
import pathlib
from typing import Any, Literal
import warnings
import numpy as np
import pandas as pd
import polars as pl
from scipy.integrate import cumulative_trapezoid, trapezoid
from scipy.optimize import lsq_linear
from .errors import IonworksError
# --- DataFrame Backend Configuration ---------------------------------------- #
# Type alias for DataFrame (pandas or polars)
DataFrame = pd.DataFrame | pl.DataFrame
def _get_default_backend() -> str:
"""Get default backend from environment variable or fall back to 'polars'."""
env_val = os.getenv("IONWORKS_DATAFRAME_BACKEND", "polars").lower()
if env_val not in ("polars", "pandas"):
return "polars"
return env_val
# Module-level configuration for DataFrame return type
# Initialized from IONWORKS_DATAFRAME_BACKEND env var, defaults to "polars"
_dataframe_backend: str = _get_default_backend()
[docs]
def set_dataframe_backend(backend: str) -> None:
"""Set the default DataFrame backend for data fetching.
This overrides the IONWORKS_DATAFRAME_BACKEND environment variable.
Parameters
----------
backend : str
DataFrame backend to use: "polars" or "pandas".
Raises
------
ValueError
If backend is not "polars" or "pandas".
"""
global _dataframe_backend
if backend not in ("polars", "pandas"):
raise ValueError(f"backend must be 'polars' or 'pandas', got '{backend}'")
_dataframe_backend = backend
[docs]
def get_dataframe_backend() -> str:
"""Get the current DataFrame backend setting.
Returns
-------
str
Current backend: "polars" or "pandas".
"""
return _dataframe_backend
# --- Measurement Data Validators -------------------------------------------- #
Severity = Literal["error", "warning"]
#: Default relative-error tolerance for capacity/energy integral checks.
#: Used by :func:`validate_capacity_energy_from_current_power` and by
#: ``ionworksdata.transform.fix_swapped_charge_discharge_columns`` so both
#: agree on the same definition of "within tolerance".
CAPACITY_ENERGY_INTEGRAL_TOLERANCE: float = 0.10
[docs]
class IssueCode(StrEnum):
"""Stable identifiers for measurement validation findings.
Use these — not substrings of the human-readable ``message`` — when
branching on issue identity. Members are ``str``-compatible.
"""
CURRENT_SIGN_REVERSED = "current_sign_reversed"
CURRENT_SIGN_INDETERMINATE = "current_sign_indeterminate"
CURRENT_SIGN_UNSIGNED = "current_sign_unsigned"
CUMULATIVE_VALUE_NOT_RESET = "cumulative_value_not_reset"
CUMULATIVE_VALUE_DECREASED = "cumulative_value_decreased"
STEP_TOO_FEW_POINTS = "step_too_few_points"
TIME_DOES_NOT_START_AT_ZERO = "time_does_not_start_at_zero"
TIME_NOT_MONOTONIC = "time_not_monotonic"
STEP_COUNT_MISSING = "step_count_missing"
STEP_COUNT_DOES_NOT_START_AT_ZERO = "step_count_does_not_start_at_zero"
STEP_COUNT_NON_SEQUENTIAL = "step_count_non_sequential"
CYCLE_CHANGES_WITHIN_STEP = "cycle_changes_within_step"
OCP_VOLTAGE_COLUMN_MISSING = "ocp_voltage_column_missing"
OCP_X_AXIS_COLUMN_MISSING = "ocp_x_axis_column_missing"
TIME_SERIES_ROW_COUNT_EXCEEDED = "time_series_row_count_exceeded"
TIME_GAP_TOO_LARGE = "time_gap_too_large"
VOLTAGE_CONTINUITY = "voltage_continuity"
CONSECUTIVE_SAME_DIRECTION_FULL_STEPS = "consecutive_same_direction_full_steps"
STEP_CAPACITY_EXCEEDS_RATED = "step_capacity_exceeds_rated"
CHARGE_DISCHARGE_CAPACITY_COLUMNS_SWAPPED = (
"charge_discharge_capacity_columns_swapped"
)
CHARGE_DISCHARGE_ENERGY_COLUMNS_SWAPPED = "charge_discharge_energy_columns_swapped"
DISCHARGE_CAPACITY_INTEGRAL_MISMATCH = "discharge_capacity_integral_mismatch"
CHARGE_CAPACITY_INTEGRAL_MISMATCH = "charge_capacity_integral_mismatch"
DISCHARGE_ENERGY_INTEGRAL_MISMATCH = "discharge_energy_integral_mismatch"
CHARGE_ENERGY_INTEGRAL_MISMATCH = "charge_energy_integral_mismatch"
EIS_COLUMNS_MISSING = "eis_columns_missing"
EIS_ZIM_SIGN_REVERSED = "eis_zim_sign_reversed"
EIS_IMPEDANCE_MAGNITUDE_IMPLAUSIBLE = "eis_impedance_magnitude_implausible"
START_TIME_IN_FUTURE = "start_time_in_future"
END_TIME_IN_FUTURE = "end_time_in_future"
END_TIME_BEFORE_START_TIME = "end_time_before_start_time"
DURATION_MISMATCH = "duration_mismatch"
[docs]
@dataclass(frozen=True)
class ValidationIssue:
"""Structured measurement validation finding.
Returned by the ``validate_*`` functions and carried by
:class:`MeasurementValidationError`. Branch on ``code`` rather than
parsing ``message``; messages are for display only and may be reworded
between releases.
Parameters
----------
code : IssueCode
Stable identifier of the finding.
message : str
Human-readable description of the failure, including any fix hint.
severity : {"error", "warning"}, optional
``"error"`` (the default) causes :func:`validate_measurement_data`
to raise. ``"warning"`` is emitted via :func:`warnings.warn` and is
**not** attached to the resulting :class:`MeasurementValidationError`
— callers who need programmatic access to warning-severity issues
must invoke the underlying ``validate_*`` producer directly rather
than relying on ``e.has_code(...)``.
payload : dict, optional
Structured details: step indices, column names, observed values,
thresholds. Keys depend on ``code``.
"""
code: IssueCode
message: str
severity: Severity = "error"
payload: dict[str, Any] = field(default_factory=dict)
def __str__(self) -> str:
return self.message
[docs]
class MeasurementValidationError(IonworksError):
"""Exception raised when measurement data validation fails.
``errors`` is a list of structured :class:`ValidationIssue` records.
Branch on ``issue.code`` rather than parsing ``issue.message``.
"""
[docs]
def __init__(
self,
message: str,
errors: list[ValidationIssue] | None = None,
) -> None:
super().__init__(message)
self.errors: list[ValidationIssue] = errors or []
[docs]
def has_code(self, code: IssueCode) -> bool:
"""Return True if any issue matches ``code``."""
return any(issue.code == code for issue in self.errors)
def _to_py_scalar(value: Any) -> Any:
"""Return a native Python scalar for ``value`` if it has ``.item()``.
``np.int64(0).item()`` returns ``0`` (Python ``int``); ``0.item()``
doesn't exist. Use this when copying a single element out of a numpy
array into a payload that will be JSON-serialised, so callers see
plain ``int``/``float``/``bool`` rather than numpy scalar types.
"""
return value.item() if hasattr(value, "item") else value
def _get_column(df: DataFrame, col: str) -> np.ndarray:
"""
Extract a column as a numpy array from either pandas or polars DataFrame.
Parameters
----------
df : DataFrame
pandas or polars DataFrame.
col : str
Column name.
Returns
-------
np.ndarray
Column values as numpy array.
"""
if isinstance(df, pl.DataFrame):
return df.get_column(col).to_numpy()
return df[col].to_numpy()
def _has_column(df: DataFrame, col: str) -> bool:
"""Check if a column exists in the DataFrame."""
return col in df.columns
def _get_step_group_indices(step_data: np.ndarray) -> np.ndarray:
"""Compute step group indices for each row (0-indexed, based on contiguous groups).
Parameters
----------
step_data : np.ndarray
Array of step numbers/identifiers.
Returns
-------
np.ndarray
Array where each element is the step group index (0, 1, 2, ...) for
that row.
"""
changes = np.concatenate([[True], np.diff(step_data) != 0])
return np.cumsum(changes) - 1
def _fit_ecm_convention(
t: np.ndarray,
I_data: np.ndarray,
voltage: np.ndarray,
) -> tuple[float, float, float] | None:
"""Fit an OCV-R ECM under one sign convention hypothesis.
Model: ``V = v0 + delta * soc - I_data * R0`` with ``delta >= 0``
(monotone OCV) and ``R0 >= 0`` (positive resistance).
Parameters
----------
t : np.ndarray
Time values [s].
I_data : np.ndarray
Signed current [A] for this convention hypothesis.
voltage : np.ndarray
Measured terminal voltage [V].
Returns
-------
tuple[float, float, float] | None
``(delta, sse, R0)`` if the fit succeeded, or ``None`` if SOC
range is degenerate.
"""
# SOC: charge entering the cell increases SOC. In ECM convention
# discharge current (I_data > 0) *decreases* SOC, so integrate -I.
charge = cumulative_trapezoid(-I_data, t, initial=0)
# Normalize to [0, 1] using min/max so SOC is always in range
c_min, c_max = charge.min(), charge.max()
span = c_max - c_min
if span < 1e-12:
return None
soc = (charge - c_min) / span
# Design matrix: V = v0*1 + delta*soc + R0*(-I_data)
# After increment transform, columns are [1, soc, -I_data]
neg_I = -I_data
v_min, v_max = float(voltage.min()), float(voltage.max())
v_range = max(v_max - v_min, 0.01)
lb = np.array([v_min - v_range, 0.0, 0.0], dtype=np.float64)
ub = np.array([v_max + v_range, 2.0 * v_range, np.inf], dtype=np.float64)
A = np.column_stack((np.ones_like(soc), soc, neg_I))
b = np.asarray(voltage, dtype=np.float64).ravel()
result = lsq_linear(A, b, bounds=(lb, ub), method="bvls")
v0, delta, R0 = result.x
sse = float(result.cost)
return delta, sse, R0
[docs]
def positive_current_is_charge(
t: np.ndarray,
current: np.ndarray,
voltage: np.ndarray,
) -> tuple[bool, float]:
"""Determine whether positive current corresponds to charging.
Fits an OCV-R equivalent-circuit model ``V = OCV(SOC) - I * R0`` under
two sign-convention hypotheses (positive = discharge vs. positive =
charge). OCV is linear in SOC, constrained to a non-negative slope
(monotonically increasing); R0 is a single positive scalar. When the
sign convention is wrong the SOC axis is inverted, forcing the OCV
slope toward zero. The convention producing the larger OCV slope is
selected.
Parameters
----------
t : np.ndarray
Time values [s].
current : np.ndarray
Current values [A].
voltage : np.ndarray
Voltage values [V].
Returns
-------
is_charge : bool
``True`` if positive current is charging, ``False`` if
discharging. Returns ``False`` when there is insufficient data.
p_value : float
Confidence metric in [0, 1]. Lower values indicate higher confidence.
Computed as the sum of two ratios clipped to [0, 1]: the ratio of the
smaller to the larger OCV slope (``delta_ratio``) and the ratio of the
winner's SSE to the loser's SSE (``sse_ratio``). Returns 1.0 when the
result is ambiguous or data is insufficient.
"""
t = np.asarray(t, dtype=float)
current = np.asarray(current, dtype=float)
voltage = np.asarray(voltage, dtype=float)
if t.size < 2 or t.max() == t.min() or np.linalg.norm(current, ord=np.inf) < 1e-12:
return False, 1.0
# 3 dof
dof = len(t) - 3
if dof <= 0:
Q = cumulative_trapezoid(y=current, x=t, initial=0)
if len(t) < 2 or Q[1] == Q[0]:
return False, 1.0
slope = (voltage[1] - voltage[0]) / (Q[1] - Q[0])
return bool(slope >= 0), 1.0
# Flat voltage: OCV slope is meaningless; treat as ambiguous (match ECM tie-break).
v_span = float(np.ptp(voltage))
v_ref = max(float(np.max(np.abs(voltage))), 1.0)
if v_span <= max(1e-9, 1e-6 * v_ref):
return True, 1.0
# Fit under both sign conventions
# Case A: positive current = discharge (I_data = +current)
# Case B: positive current = charge (I_data = -current)
fit_dis = _fit_ecm_convention(t, current, voltage)
fit_chg = _fit_ecm_convention(t, -current, voltage)
if fit_dis is None and fit_chg is None:
return False, 1.0
if fit_dis is None:
return True, 1.0
if fit_chg is None:
return False, 1.0
delta_dis, sse_dis, _ = fit_dis
delta_chg, sse_chg, _ = fit_chg
# Both deltas essentially zero → ambiguous (e.g. flat voltage)
eps = 1e-10
delta_max = max(delta_dis, delta_chg)
if delta_max < eps:
# Fall back: same as flat-voltage case, default to charge=True
return True, 1.0
# The convention with the larger OCV slope wins
is_charge = bool(delta_chg > delta_dis)
# Confidence from two signals:
# 1) delta_ratio: how much the loser's slope collapsed (near 0 = clear)
# 2) sse_ratio: how well the winner explains the data vs the loser
delta_ratio = min(delta_dis, delta_chg) / max(delta_dis, delta_chg, eps)
sse_winner = sse_chg if is_charge else sse_dis
sse_loser = sse_dis if is_charge else sse_chg
sse_ratio = sse_winner / max(sse_loser, eps) if sse_loser > eps else 0.0
# p_value: product of both ratios. Low when the winner clearly
# dominates on both OCV slope and goodness-of-fit.
p_value = float(np.clip(delta_ratio + sse_ratio, 0.0, 1.0))
return is_charge, p_value
[docs]
def validate_positive_current_is_discharge( # noqa: PLR0913
df: DataFrame,
current_col: str = "Current [A]",
voltage_col: str = "Voltage [V]",
time_col: str = "Time [s]",
step_col: str | None = None,
rest_tol: float = 1e-3,
relative_rest_tol_frac: float = 0.02,
relative_rest_tol_percentile: float = 95.0,
) -> list[ValidationIssue]:
"""
Validate that positive current corresponds to discharge.
Discharge should cause voltage to decrease. This function analyzes the
relationship between current direction and voltage change to verify the
sign convention is correct.
Fits an OCV-R ECM per step, then uses a confidence vote across steps
weighted by the trapezoidal integral of ``|I(t)| dt`` over each step
so that steps actually moving charge dominate the decision and long
near-zero-current voltage holds contribute negligible weight.
Parameters
----------
df : DataFrame
Time series data with current and voltage columns (pandas or polars).
current_col : str
Name of the current column.
voltage_col : str
Name of the voltage column.
time_col : str
Name of the time column.
step_col : str, optional
Name of the step column. If provided, analyzes per-step. Otherwise,
infers steps from current sign changes.
rest_tol : float
Tolerance for considering current as zero (rest).
relative_rest_tol_frac : float
Fraction of a robust current scale used to set a relative rest threshold.
The robust scale is the percentile of ``abs(current)`` given by
``relative_rest_tol_percentile``.
relative_rest_tol_percentile : float
Percentile in [0, 100] used to estimate the robust current scale for the
relative threshold.
The effective rest threshold is ``min(rest_tol, relative_rest_tol)`` so the
stricter (smaller) threshold takes precedence.
Returns
-------
list[ValidationIssue]
List of validation issues. Empty if validation passes.
"""
if not _has_column(df, current_col) or not _has_column(df, voltage_col):
return []
if not _has_column(df, time_col):
return []
current = _get_column(df, current_col)
voltage = _get_column(df, voltage_col)
time = _get_column(df, time_col)
if len(current) == 0:
return []
# Use a robust relative threshold so low-current datasets don't get
# treated as all-rest simply because they are below a fixed absolute floor.
abs_current = np.abs(current)
robust_scale = float(np.percentile(abs_current, relative_rest_tol_percentile))
relative_rest_tol = max(relative_rest_tol_frac * robust_scale, 1e-12)
# Precedence rule: use the stricter (smaller) rest threshold.
effective_rest_tol = min(rest_tol, relative_rest_tol)
# Determine step groups
if step_col and _has_column(df, step_col):
step_data = _get_column(df, step_col)
else:
# Infer steps from current sign changes
max_abs = np.max(np.abs(current))
if max_abs == 0:
return []
normalized = current / max_abs
step_data = np.sign(normalized * (np.abs(normalized) > effective_rest_tol))
step_groups = _get_step_group_indices(step_data)
num_steps = step_groups[-1] + 1
# Mean current per step (for rest filtering)
step_current_sum = np.bincount(step_groups, weights=current, minlength=num_steps)
step_counts = np.bincount(step_groups, minlength=num_steps).astype(float)
step_counts[step_counts == 0] = 1
mean_current = step_current_sum / step_counts
# Identify non-rest steps
non_rest_steps = set(np.where(np.abs(mean_current) >= effective_rest_tol)[0])
if not non_rest_steps:
return []
# Classify each non-rest step using an OCV-R ECM fit. Weight each
# step's vote by the trapezoidal integral of ``|I(t)| dt`` over the
# step so that steps actually moving charge dominate the decision and
# long near-zero-current voltage holds contribute negligible weight.
# Flag a sign-convention error when at least 75% of the weighted
# evidence points to charge, and return an ambiguous error when it
# is between 25% and 75%.
charge_weight = 0.0
discharge_weight = 0.0
for step_id in non_rest_steps:
mask = step_groups == step_id
t_step = time[mask]
i_step = current[mask]
if t_step.size < 2:
continue
charge_passed = float(trapezoid(np.abs(i_step), t_step))
if charge_passed <= 0:
continue
is_charge, p_value = positive_current_is_charge(t_step, i_step, voltage[mask])
confidence = 1.0 - p_value
weight = confidence * charge_passed
if is_charge:
charge_weight += weight
else:
discharge_weight += weight
total_weight = charge_weight + discharge_weight
if total_weight <= 0:
return []
charge_fraction = charge_weight / total_weight
if charge_fraction >= 0.75:
return [
ValidationIssue(
IssueCode.CURRENT_SIGN_REVERSED,
"Current sign convention error: positive current appears to "
"be charge, not discharge. Voltage increases when current is "
"positive, but for discharge, voltage should decrease. Use "
"ionworksdata.transform.set_positive_current_for_discharge"
"(data) to fix this.",
payload={"charge_fraction": charge_fraction},
)
]
elif charge_fraction >= 0.25:
# All-positive non-rest current + a split vote = unsigned (magnitude-only)
# data spanning both charge and discharge steps, not an ambiguous global
# convention. It has a concrete fix, so flag it distinctly. Restricted to
# all-positive because unsigned cyclers record magnitudes as positive; an
# all-negative split vote stays the honest "ambiguous" below.
non_rest_current = current[abs_current > effective_rest_tol]
all_positive = non_rest_current.size > 0 and bool((non_rest_current > 0).all())
if all_positive:
return [
ValidationIssue(
IssueCode.CURRENT_SIGN_UNSIGNED,
"Current sign convention error: the current appears to be "
"unsigned (magnitude-only) but contains both charge and "
"discharge steps. Sign it using the cycler mode column via "
"ionworksdata.transform.set_positive_current_for_discharge"
"(data) before validation.",
payload={"charge_fraction": charge_fraction},
)
]
return [
ValidationIssue(
IssueCode.CURRENT_SIGN_INDETERMINATE,
"Current sign convention error: the sign convention is "
"ambiguous. Check if all currents are positive or negative.",
payload={"charge_fraction": charge_fraction},
)
]
else:
return []
[docs]
def validate_cumulative_values_reset_per_step(
df: DataFrame,
step_col: str = "Step count",
cumulative_cols: list[str] | None = None,
tolerance: float = 1e-6,
) -> list[ValidationIssue]:
"""Validate cumulative values reset to ~0 at each step and only increase.
Parameters
----------
df : DataFrame
Time series data (pandas or polars).
step_col : str
Name of the column containing step numbers.
cumulative_cols : list[str], optional
List of cumulative column names to validate. If None, checks for common
capacity and energy columns.
tolerance : float
Tolerance for considering a value as "zero" at step start.
Returns
-------
list[ValidationIssue]
List of validation issues. Empty if validation passes.
"""
errors: list[ValidationIssue] = []
if not _has_column(df, step_col):
return []
if cumulative_cols is None:
cumulative_cols = [
"Discharge capacity [A.h]",
"Charge capacity [A.h]",
"Discharge energy [W.h]",
"Charge energy [W.h]",
]
cols_to_check = [col for col in cumulative_cols if _has_column(df, col)]
if not cols_to_check:
return []
fix_hint = (
"Use ionworksdata.transform.set_capacity(data) and/or "
"ionworksdata.transform.set_energy(data) to fix this."
)
step_data = _get_column(df, step_col)
if len(step_data) == 0:
return []
step_groups = _get_step_group_indices(step_data)
# Find step boundaries (first index of each step)
step_boundaries = np.where(np.diff(step_groups, prepend=-1) != 0)[0]
for col in cols_to_check:
values = _get_column(df, col)
# Check 1: Values at step starts should be ~0
start_values = values[step_boundaries]
non_zero_mask = np.abs(start_values) > tolerance
non_zero_steps = np.where(non_zero_mask)[0]
for step_idx in non_zero_steps:
observed = float(start_values[step_idx])
errors.append(
ValidationIssue(
IssueCode.CUMULATIVE_VALUE_NOT_RESET,
f"Column '{col}' does not reset at start of "
f"step {step_idx}: expected ~0, got "
f"{observed:.6f}. Cumulative values "
f"should reset to 0 at the start of each step. "
f"{fix_hint}",
payload={
"column": col,
"step_index": int(step_idx),
"observed": observed,
"tolerance": tolerance,
},
)
)
# Check 2: Values should be monotonically non-decreasing within each step
# Compute diff and check where it's negative within same step
value_diffs = np.diff(values, prepend=values[0])
step_diffs = np.diff(step_groups, prepend=step_groups[0])
# Mask: same step (diff == 0) and value decreased
within_step = step_diffs == 0
decreased = value_diffs < -tolerance
# Find first decrease per step
problem_indices = np.where(within_step & decreased)[0]
if len(problem_indices) > 0:
# Group by step and report first decrease per step
problem_steps = step_groups[problem_indices]
unique_problem_steps = np.unique(problem_steps)
for step_idx in unique_problem_steps:
step_problem_indices = problem_indices[problem_steps == step_idx]
first_idx = int(step_problem_indices[0])
prev_value = float(values[first_idx - 1])
curr_value = float(values[first_idx])
errors.append(
ValidationIssue(
IssueCode.CUMULATIVE_VALUE_DECREASED,
f"Column '{col}' decreases within step "
f"{step_idx} at index {first_idx}: value went "
f"from {prev_value:.6f} to "
f"{curr_value:.6f}. Cumulative values "
f"should only increase within a step. "
f"{fix_hint}",
payload={
"column": col,
"step_index": int(step_idx),
"row_index": first_idx,
"previous_value": prev_value,
"current_value": curr_value,
},
)
)
return errors
[docs]
def validate_minimum_points_per_step(
df: DataFrame,
step_col: str = "Step count",
min_points: int = 2,
) -> list[ValidationIssue]:
"""
Validate that each step has at least a minimum number of data points.
Parameters
----------
df : DataFrame
Time series data (pandas or polars).
step_col : str
Name of the column containing step numbers.
min_points : int
Minimum number of points required per step.
Returns
-------
list[ValidationIssue]
List of validation issues. Empty if validation passes.
"""
if not _has_column(df, step_col):
return []
step_data = _get_column(df, step_col)
if len(step_data) == 0:
return []
step_groups = _get_step_group_indices(step_data)
num_steps = step_groups[-1] + 1
# Vectorized count per step
step_counts = np.bincount(step_groups, minlength=num_steps)
# Find steps with insufficient points
insufficient_mask = step_counts < min_points
insufficient_steps = np.where(insufficient_mask)[0]
errors: list[ValidationIssue] = []
for step_idx in insufficient_steps:
num_points = int(step_counts[step_idx])
errors.append(
ValidationIssue(
IssueCode.STEP_TOO_FEW_POINTS,
f"Step {step_idx} has only {num_points} data point(s), "
f"but at least {min_points} are required.",
payload={
"step_index": int(step_idx),
"num_points": num_points,
"min_points": min_points,
},
)
)
return errors
[docs]
def validate_time_starts_at_zero(
df: DataFrame,
tolerance: float = 1e-6,
) -> list[ValidationIssue]:
"""Validate that 'Time [s]' starts at 0.
Parameters
----------
df : DataFrame
Time series data (pandas or polars).
tolerance : float
Tolerance for considering the start value as zero.
Returns
-------
list[ValidationIssue]
List of validation issues. Empty if validation passes.
"""
time_col = "Time [s]"
if not _has_column(df, time_col):
return []
time_data = _get_column(df, time_col)
if len(time_data) == 0:
return []
first = float(time_data[0])
if abs(first) > tolerance:
return [
ValidationIssue(
IssueCode.TIME_DOES_NOT_START_AT_ZERO,
f"Column '{time_col}' must start at 0, but starts at "
f"{first}. Use ionworksdata.transform.reset_time"
f"(data) to fix this. To indicate the absolute time when a "
f"step starts, use the start_time field in the measurement "
f"metadata.",
payload={"observed": first, "tolerance": tolerance},
)
]
return []
[docs]
def validate_time_monotonic(
df: DataFrame,
time_col: str = "Time [s]",
tolerance: float = 1e-12,
) -> list[ValidationIssue]:
"""Validate that the time column is monotonically non-decreasing.
Parameters
----------
df : DataFrame
Time series data (pandas or polars).
time_col : str
Name of the time column.
tolerance : float
Numerical tolerance; time[i] must be >= time[i-1] - tolerance.
Returns
-------
list[ValidationIssue]
List of validation issues. Empty if validation passes.
"""
if not _has_column(df, time_col):
return []
time_data = _get_column(df, time_col)
if len(time_data) < 2:
return []
diffs = np.diff(time_data)
bad_mask = diffs < -tolerance
if not np.any(bad_mask):
return []
bad_indices = np.where(bad_mask)[0]
first_idx = int(bad_indices[0])
prev_t = float(time_data[first_idx])
next_t = float(time_data[first_idx + 1])
return [
ValidationIssue(
IssueCode.TIME_NOT_MONOTONIC,
f"Column '{time_col}' must be monotonically non-decreasing. "
f"At index {first_idx + 1}: {next_t:.6f}s < "
f"previous {prev_t:.6f}s. "
f"Use a cumulative time series (e.g. ionworksdata.transform or "
f"ensure per-step time is converted to global elapsed time).",
payload={
"row_index": first_idx + 1,
"previous_time": prev_t,
"current_time": next_t,
"num_violations": int(bad_mask.sum()),
},
)
]
[docs]
def validate_step_count_sequential(
df: DataFrame,
) -> list[ValidationIssue]:
"""Validate that 'Step count' exists, starts at 0, and increases by 1.
Parameters
----------
df : DataFrame
Time series data (pandas or polars).
Returns
-------
list[ValidationIssue]
List of validation issues. Empty if validation passes.
"""
step_col = "Step count"
fix_hint = "Use ionworksdata.transform.set_step_count(data) to fix this."
if not _has_column(df, step_col):
return [
ValidationIssue(
IssueCode.STEP_COUNT_MISSING,
f"Column '{step_col}' is required but was not found in "
f"the data. Available columns: {list(df.columns)}. "
f"{fix_hint}",
payload={
"column": step_col,
"available_columns": list(df.columns),
},
)
]
step_data = _get_column(df, step_col)
if len(step_data) == 0:
return []
errors: list[ValidationIssue] = []
if step_data[0] != 0:
first_val = _to_py_scalar(step_data[0])
errors.append(
ValidationIssue(
IssueCode.STEP_COUNT_DOES_NOT_START_AT_ZERO,
f"Column '{step_col}' must start at 0, but starts at "
f"{first_val}. {fix_hint}",
payload={"column": step_col, "observed_first": first_val},
)
)
raw_diffs = np.diff(step_data)
bad_mask = (raw_diffs != 0) & (raw_diffs != 1)
if np.any(bad_mask):
bad_indices = np.where(bad_mask)[0]
examples = []
# Use ``.item()`` (not ``int()``) so float Step counts like 0.5
# round-trip to the payload instead of being truncated to 0.
bad_transitions: list[dict[str, Any]] = []
for idx in bad_indices[:5]:
examples.append(
f"index {idx}: {step_data[idx]} -> "
f"{step_data[idx + 1]} (diff={raw_diffs[idx]})"
)
bad_transitions.append(
{
"row_index": int(idx),
"from": _to_py_scalar(step_data[idx]),
"to": _to_py_scalar(step_data[idx + 1]),
"diff": _to_py_scalar(raw_diffs[idx]),
}
)
more = ""
if len(bad_indices) > 5:
more = f" (and {len(bad_indices) - 5} more)"
errors.append(
ValidationIssue(
IssueCode.STEP_COUNT_NON_SEQUENTIAL,
f"Column '{step_col}' must increase by 1 at each step "
f"transition, but found {len(bad_indices)} invalid "
f"transition(s): " + "; ".join(examples) + f".{more} {fix_hint}",
payload={
"column": step_col,
"num_violations": int(len(bad_indices)),
"examples": bad_transitions,
},
)
)
return errors
[docs]
def validate_cycle_constant_within_step(
df: DataFrame,
step_col: str = "Step count",
cycle_col: str | None = None,
) -> list[ValidationIssue]:
"""
Validate that cycle number does not change within a step.
Parameters
----------
df : DataFrame
Time series data (pandas or polars).
step_col : str
Name of the column containing step numbers.
cycle_col : str, optional
Name of the column containing cycle numbers. If None, tries common names.
Returns
-------
list[ValidationIssue]
List of validation issues. Empty if validation passes.
"""
if not _has_column(df, step_col):
return []
# Find cycle column
if cycle_col is None:
for col in ["Cycle count", "Cycle number", "Cycle from cycler"]:
if _has_column(df, col):
cycle_col = col
break
if cycle_col is None or not _has_column(df, cycle_col):
return []
step_data = _get_column(df, step_col)
if len(step_data) == 0:
return []
cycle_data = _get_column(df, cycle_col)
step_groups = _get_step_group_indices(step_data)
# Detect cycle changes within steps:
# A cycle change within a step occurs when:
# - The cycle value differs from the previous row
# - AND we're in the same step group
cycle_diffs = np.diff(cycle_data, prepend=cycle_data[0])
step_diffs = np.diff(step_groups, prepend=step_groups[0])
# Within-step cycle change: same step (step_diff == 0) but cycle changed
within_step_cycle_change = (step_diffs == 0) & (cycle_diffs != 0)
problem_indices = np.where(within_step_cycle_change)[0]
if len(problem_indices) == 0:
return []
# Group by step and report
problem_steps = step_groups[problem_indices]
unique_problem_steps = np.unique(problem_steps)
errors: list[ValidationIssue] = []
for step_idx in unique_problem_steps:
step_mask = step_groups == step_idx
unique_cycles = np.unique(cycle_data[step_mask])
cycles_list = unique_cycles.tolist()
errors.append(
ValidationIssue(
IssueCode.CYCLE_CHANGES_WITHIN_STEP,
f"Cycle number changes within step {step_idx}: "
f"found cycles {cycles_list}. "
f"Each step should belong to a single cycle. "
f"Use ionworksdata.transform.set_cycle_count(data) "
f"to fix this.",
payload={
"step_index": int(step_idx),
"cycle_column": cycle_col,
"cycles": cycles_list,
},
)
)
return errors
[docs]
def validate_ocp_columns(df: DataFrame) -> list[ValidationIssue]:
"""Validate that OCP data has required columns.
Checks that the DataFrame contains:
1. A 'Voltage [V]' column
2. At least one x-axis column: 'Capacity [A.h]', 'Stoichiometry', or 'SOC'
Parameters
----------
df : DataFrame
Time series data (pandas or polars).
Returns
-------
list[ValidationIssue]
List of validation issues. Empty if validation passes.
"""
errors: list[ValidationIssue] = []
available = list(df.columns)
if not _has_column(df, "Voltage [V]"):
errors.append(
ValidationIssue(
IssueCode.OCP_VOLTAGE_COLUMN_MISSING,
"OCP data must contain a 'Voltage [V]' column. "
f"Available columns: {available}",
payload={"available_columns": available},
)
)
x_axis_columns = ["Capacity [A.h]", "Stoichiometry", "SOC"]
if not any(_has_column(df, col) for col in x_axis_columns):
errors.append(
ValidationIssue(
IssueCode.OCP_X_AXIS_COLUMN_MISSING,
"OCP data must contain at least one x-axis column: "
f"{', '.join(x_axis_columns)}. "
f"Available columns: {available}",
payload={
"candidates": x_axis_columns,
"available_columns": available,
},
)
)
return errors
#: Default EIS column names (BioLogic/Gamry reader convention).
EIS_FREQ_COL = "Frequency [Hz]"
EIS_ZRE_COL = "Z_Re [Ohm]"
EIS_ZIM_COL = "Z_Im [Ohm]"
#: Roughly cell-size-invariant ohmic resistance-capacity product [Ohm.A.h], used to
#: estimate a plausible ohmic-resistance scale from rated capacity alone
#: (``R0 ~ product / capacity``, since larger cells have proportionally more
#: electrode area in parallel). For Li-ion this clusters around 0.05-0.09 Ohm.A.h
#: (power cells lower, small high-energy cells higher; ~1 order of magnitude spread),
#: so 0.05 is a central-but-conservative anchor. It is only an order-of-magnitude
#: reference for the deliberately wide (default 100x) EIS magnitude sanity band, so
#: the exact value is not load-bearing.
EIS_RESISTANCE_CAPACITY_PRODUCT: float = 0.05
[docs]
def validate_eis_columns(
df: DataFrame,
freq_col: str = EIS_FREQ_COL,
zre_col: str = EIS_ZRE_COL,
zim_col: str = EIS_ZIM_COL,
) -> list[ValidationIssue]:
"""Validate that EIS data has the required frequency and impedance columns.
Parameters
----------
df : DataFrame
EIS spectrum data (pandas or polars).
freq_col : str, optional
Name of the frequency column. Defaults to ``"Frequency [Hz]"``.
zre_col : str, optional
Name of the real-impedance column. Defaults to ``"Z_Re [Ohm]"``.
zim_col : str, optional
Name of the imaginary-impedance column. Defaults to ``"Z_Im [Ohm]"``.
Returns
-------
list[ValidationIssue]
A single ``EIS_COLUMNS_MISSING`` issue listing the missing and available
columns, or an empty list when all required columns are present.
"""
required = [freq_col, zre_col, zim_col]
missing = [col for col in required if not _has_column(df, col)]
if not missing:
return []
available = list(df.columns)
return [
ValidationIssue(
IssueCode.EIS_COLUMNS_MISSING,
f"EIS data must contain the columns {required}. "
f"Missing: {missing}. Available columns: {available}",
payload={
"required": required,
"missing": missing,
"available_columns": available,
},
)
]
def _eis_capacitive_band(
freq: np.ndarray, z_re: np.ndarray, z_im: np.ndarray
) -> tuple[np.ndarray, float] | None:
"""Return the capacitive-band ``Z_Im`` and the ohmic intercept ``R0``.
Sorts by increasing frequency, locates the high-frequency real-axis intercept
``R0 = min(Z_Re)``, and trims the inductive tail — the points at frequencies
above the intercept. The retained points (intercept down to the lowest
frequency) are the capacitive band.
Parameters
----------
freq : np.ndarray
Frequency values [Hz].
z_re : np.ndarray
Real impedance values [Ohm].
z_im : np.ndarray
Imaginary impedance values [Ohm] (raw ``Im(Z)`` convention).
Returns
-------
tuple[np.ndarray, float] | None
``(z_im_capacitive, r0)`` or ``None`` when there are fewer than three
finite points to analyse.
"""
# Drop rows with any non-finite value so a single blank/NaN cell cannot
# corrupt the argsort, the ``min(Z_Re)`` intercept, or the sign vote.
finite = np.isfinite(freq) & np.isfinite(z_re) & np.isfinite(z_im)
freq, z_re, z_im = freq[finite], z_re[finite], z_im[finite]
if freq.size < 3:
return None
order = np.argsort(freq)
z_re_sorted = z_re[order]
z_im_sorted = z_im[order]
# R0 is the high-frequency real-axis intercept; points at higher frequency
# than it (larger sorted index) are the inductive tail and are dropped.
r0_idx = int(np.argmin(z_re_sorted))
r0 = float(z_re_sorted[r0_idx])
capacitive = z_im_sorted[: r0_idx + 1]
return capacitive, r0
[docs]
def validate_eis_zim_sign(
df: DataFrame,
freq_col: str = EIS_FREQ_COL,
zre_col: str = EIS_ZRE_COL,
zim_col: str = EIS_ZIM_COL,
reversed_fraction_threshold: float = 0.75,
) -> list[ValidationIssue]:
"""Validate that the EIS ``Z_Im`` sign convention is not reversed.
The database stores the raw imaginary part ``Z_Im = Im(Z)``, which is
**negative** across the capacitive arc; only a possible high-frequency
inductive loop is positive. This check sorts by frequency, trims the inductive
tail (points above the ``R0 = min(Z_Re)`` intercept), and flags the spectrum
when the capacitive band is predominantly positive — i.e. the whole spectrum
has been stored with a flipped sign. Points are weighted by ``|Z_Im|`` so the
near-zero points around the real-axis crossing do not dominate the vote.
Parameters
----------
df : DataFrame
EIS spectrum data (pandas or polars).
freq_col : str, optional
Name of the frequency column. Defaults to ``"Frequency [Hz]"``.
zre_col : str, optional
Name of the real-impedance column. Defaults to ``"Z_Re [Ohm]"``.
zim_col : str, optional
Name of the imaginary-impedance column. Defaults to ``"Z_Im [Ohm]"``.
reversed_fraction_threshold : float, optional
Minimum ``|Z_Im|``-weighted fraction of the capacitive band that must be
positive to flag a reversed sign. Defaults to ``0.75`` (75 %).
Returns
-------
list[ValidationIssue]
A single ``EIS_ZIM_SIGN_REVERSED`` issue when the sign appears flipped, or
an empty list otherwise (including when columns are missing or there are
too few points).
"""
if not all(_has_column(df, c) for c in (freq_col, zre_col, zim_col)):
return []
freq = np.asarray(_get_column(df, freq_col), dtype=float)
z_re = np.asarray(_get_column(df, zre_col), dtype=float)
z_im = np.asarray(_get_column(df, zim_col), dtype=float)
band = _eis_capacitive_band(freq, z_re, z_im)
if band is None:
return []
capacitive, r0 = band
weights = np.abs(capacitive)
total = float(weights.sum())
if total <= 0:
return []
positive_fraction = float(weights[capacitive > 0].sum()) / total
if positive_fraction < reversed_fraction_threshold:
return []
return [
ValidationIssue(
IssueCode.EIS_ZIM_SIGN_REVERSED,
f"EIS 'Z_Im [Ohm]' sign appears reversed: "
f"{positive_fraction * 100:.0f}% (by |Z_Im| weight) of the capacitive "
f"band is positive, but Z_Im stores the raw imaginary part, which is "
f"negative across the capacitive arc (only a high-frequency inductive "
f"loop is positive). Negate the column to fix this: Z_Im = -Z_Im.",
payload={
"positive_fraction": positive_fraction,
"r0": r0,
"n_capacitive_points": int(capacitive.size),
},
)
]
[docs]
def validate_eis_impedance_magnitude(
df: DataFrame,
rated_capacity: float | None,
factor: float = 100.0,
r_times_q: float = EIS_RESISTANCE_CAPACITY_PRODUCT,
zre_col: str = EIS_ZRE_COL,
) -> list[ValidationIssue]:
"""Check that the EIS ohmic resistance is within a wide plausibility band.
Estimates a plausible ohmic-resistance scale from the rated capacity alone via
the roughly cell-size-invariant product ``R0 ~ r_times_q / rated_capacity``,
then flags the spectrum only when the measured intercept ``R0 = min(Z_Re)`` is
more than ``factor`` times away from it in either direction. The band is
deliberately wide (default ``100x``) so it fires only on order-of-magnitude
errors — e.g. an impedance stored in the wrong unit — and never on a plausible
spectrum. The exact ``r_times_q`` value is unimportant at this tolerance.
Parameters
----------
df : DataFrame
EIS spectrum data (pandas or polars).
rated_capacity : float | None
Rated (nominal) cell capacity in A.h. The check is skipped when this is
``None`` or non-positive.
factor : float, optional
Half-width of the sanity band as a multiplicative factor around the
expected ohmic resistance. Defaults to ``100.0``.
r_times_q : float, optional
Order-of-magnitude resistance-capacity product [Ohm.A.h] used to estimate
the expected ohmic resistance. Defaults to
:data:`EIS_RESISTANCE_CAPACITY_PRODUCT`.
zre_col : str, optional
Name of the real-impedance column. Defaults to ``"Z_Re [Ohm]"``.
Returns
-------
list[ValidationIssue]
A single ``EIS_IMPEDANCE_MAGNITUDE_IMPLAUSIBLE`` issue when ``R0`` is
outside the band, or an empty list otherwise (including when the rated
capacity is unavailable, the column is missing, or ``R0`` is non-positive).
"""
if rated_capacity is None or rated_capacity <= 0:
return []
if not _has_column(df, zre_col):
return []
z_re = np.asarray(_get_column(df, zre_col), dtype=float)
z_re = z_re[np.isfinite(z_re)]
if z_re.size == 0:
return []
# R0, the ohmic intercept, is order-independent so no frequency sort is needed.
r0 = float(np.min(z_re))
if r0 <= 0:
return []
r_expected = r_times_q / rated_capacity
low = r_expected / factor
high = r_expected * factor
if low <= r0 <= high:
return []
return [
ValidationIssue(
IssueCode.EIS_IMPEDANCE_MAGNITUDE_IMPLAUSIBLE,
f"EIS ohmic resistance R0 = min(Z_Re) = {r0:.3e} Ohm is outside the "
f"{factor:g}x sanity band [{low:.3e}, {high:.3e}] Ohm around the "
f"expected {r_expected:.3e} Ohm (order-of-magnitude estimate for a "
f"{rated_capacity:g} A.h cell). This usually indicates a units or "
f"scale error (e.g. impedance stored in the wrong unit).",
payload={
"r0": r0,
"r_expected": r_expected,
"band_low": low,
"band_high": high,
"factor": float(factor),
"r_times_q": float(r_times_q),
"rated_capacity": float(rated_capacity),
},
severity="error",
)
]
[docs]
def validate_time_series_row_count(
df: DataFrame,
max_rows: int = 1000,
) -> list[ValidationIssue]:
"""Validate that the time series does not exceed the maximum row count.
Datasets larger than ``max_rows`` should be uploaded via the standard
upload flow and then referenced with ``"db:<measurement_id>"`` or
``iwdata.DataLoader.from_db(MEASUREMENT_ID)`` in pipeline configurations.
Parameters
----------
df : DataFrame
Time series data (pandas or polars).
max_rows : int
Maximum allowed number of rows.
Returns
-------
list[ValidationIssue]
List of validation issues. Empty if validation passes.
"""
n_rows = len(df)
if n_rows > max_rows:
return [
ValidationIssue(
IssueCode.TIME_SERIES_ROW_COUNT_EXCEEDED,
f"Time series has {n_rows} rows, which exceeds the maximum "
f"of {max_rows} rows for inline data. Upload the data as a "
f"measurement using client.cell_measurement.create() and "
f'reference it with "db:<measurement_id>" or '
f"iwdata.DataLoader.from_db(MEASUREMENT_ID) instead.",
payload={"n_rows": int(n_rows), "max_rows": int(max_rows)},
)
]
return []
[docs]
def validate_time_gaps(
df: DataFrame,
time_col: str = "Time [s]",
max_gap_seconds: float = 5 * 3600,
) -> list[ValidationIssue]:
"""Validate that there are no large gaps between consecutive time samples.
A gap longer than ``max_gap_seconds`` between two consecutive rows is almost
always a sign that a chunk of cycling was dropped — for example, a rest
period was recorded as elapsed time in the file but the intermediate rows
were stripped, leaving the capacity integral to count current across the
unrecorded interval.
Parameters
----------
df : DataFrame
Time series data with a time column (pandas or polars).
time_col : str, optional
Name of the time column. Defaults to ``"Time [s]"``.
max_gap_seconds : float, optional
Maximum allowed gap between consecutive time samples, in seconds.
Defaults to ``5 * 3600`` (5 hours).
Returns
-------
list[ValidationIssue]
List of validation issues. Empty if validation passes.
"""
if not _has_column(df, time_col):
return []
time_data = _get_column(df, time_col)
if len(time_data) < 2:
return []
diffs = np.diff(np.asarray(time_data, dtype=float))
bad_mask = diffs > max_gap_seconds
if not np.any(bad_mask):
return []
bad_indices = np.where(bad_mask)[0]
first_idx = int(bad_indices[0])
gap_seconds = float(diffs[first_idx])
gap_hours = gap_seconds / 3600.0
num_violations = int(bad_mask.sum())
more = f" (and {num_violations - 1} more)" if num_violations > 1 else ""
return [
ValidationIssue(
IssueCode.TIME_GAP_TOO_LARGE,
f"Time gap of {gap_hours:.2f} h between rows {first_idx} and "
f"{first_idx + 1} exceeds the maximum allowed gap of "
f"{max_gap_seconds / 3600:.1f} h.{more} Large time gaps "
f"typically mean rows were dropped during processing; the "
f"capacity integral will count current across the missing "
f"interval.",
payload={
"row_index": first_idx,
"gap_seconds": gap_seconds,
"max_gap_seconds": float(max_gap_seconds),
"num_violations": num_violations,
},
)
]
[docs]
def validate_voltage_continuity(
df: DataFrame,
voltage_window: tuple[float, float],
voltage_col: str = "Voltage [V]",
jump_fraction: float = 0.80,
max_bad_fraction: float = 0.05,
) -> list[ValidationIssue]:
"""Validate that consecutive rows do not exhibit unphysical voltage jumps.
Computes the absolute voltage difference between every pair of consecutive rows
and counts the fraction that exceed ``jump_fraction`` of the rated voltage window
``(V_max - V_min)``. If more than ``max_bad_fraction`` of row pairs exceed that
threshold, the data is flagged as likely being out of chronological order (for
example, after a faulty ``partition_by("Cycle_raw")`` grouping that interleaves
pulse and rest rows from different cycles).
Parameters
----------
df : DataFrame
Time series data with a voltage column (pandas or polars).
voltage_window : tuple[float, float]
``(V_min, V_max)`` rated voltage window of the cell, typically the
lower/upper cutoff voltages. Only the span ``V_max - V_min`` is used.
voltage_col : str, optional
Name of the voltage column. Defaults to ``"Voltage [V]"``.
jump_fraction : float, optional
Fraction of the voltage window considered the maximum physically plausible
single-row voltage change. Defaults to ``0.80`` (80 %).
max_bad_fraction : float, optional
Maximum fraction of consecutive row pairs allowed to exceed the jump
threshold before the check fails. Defaults to ``0.05`` (5 %).
Returns
-------
list[ValidationIssue]
List of validation issues. Empty if validation passes.
"""
if not _has_column(df, voltage_col):
return []
voltage = _get_column(df, voltage_col)
if len(voltage) < 2:
return []
v_min, v_max = float(voltage_window[0]), float(voltage_window[1])
window = v_max - v_min
if window <= 0:
return []
threshold = jump_fraction * window
diffs = np.abs(np.diff(voltage))
bad_count = int(np.sum(diffs > threshold))
total_pairs = int(diffs.size)
bad_fraction = bad_count / total_pairs
if bad_fraction <= max_bad_fraction:
return []
return [
ValidationIssue(
IssueCode.VOLTAGE_CONTINUITY,
f"Voltage continuity check failed: {bad_count}/{total_pairs} "
f"({bad_fraction * 100:.1f}%) of consecutive row pairs have a "
f"voltage jump greater than {jump_fraction * 100:.0f}% of the "
f"rated voltage window ({threshold:.3f} V). This usually means "
f"the time series is not in chronological order — check for "
f'grouping operations such as ``partition_by("Cycle_raw")`` '
f"that may destroy the natural row order.",
payload={
"bad_count": bad_count,
"total_pairs": total_pairs,
"bad_fraction": bad_fraction,
"threshold_volts": float(threshold),
"voltage_window": [v_min, v_max],
},
)
]
def _step_direction(step_type: str) -> str | None:
"""Map a ``Step type`` label to ``"charge"``, ``"discharge"``, or ``None``.
``None`` is returned for rest steps, EIS steps, and any unknown type so that
rests do not reset the same-direction streak tracked by the consecutive
full-step check.
"""
if not isinstance(step_type, str):
return None
lowered = step_type.lower()
if "discharge" in lowered:
return "discharge"
if "charge" in lowered:
return "charge"
return None
[docs]
def validate_consecutive_same_direction_full_steps(
steps_df: DataFrame,
rated_capacity: float | None,
step_type_col: str = "Step type",
discharge_capacity_col: str = "Discharge capacity [A.h]",
charge_capacity_col: str = "Charge capacity [A.h]",
step_count_col: str = "Step count",
full_step_fraction: float = 2.0,
) -> list[ValidationIssue]:
"""Validate that no two consecutive full-capacity steps share the same direction.
Walks the step summary in order. For each constant-current step (identified by
its ``Step type``) that delivers more than ``full_step_fraction`` of the rated
capacity, tracks the direction. If two such steps appear consecutively with the
same direction (ignoring rest / EIS / unknown steps, which do not reset the
streak), the check fails. This catches measurements that bundle multiple
independent experiments — e.g. a discharge-rate file concatenating several CC
discharges into a single measurement.
Parameters
----------
steps_df : DataFrame
Step summary dataframe (as produced by ``ionworksdata.steps.summarize``).
rated_capacity : float | None
Rated (nominal) cell capacity in A.h. Used together with
``full_step_fraction`` to decide whether a step counts as a full
charge / discharge. The check is skipped when this is ``None`` or
non-positive.
step_type_col : str, optional
Name of the column containing the step type labels (``"Rest"``,
``"Constant current discharge"``, etc.). Defaults to ``"Step type"``.
discharge_capacity_col : str, optional
Name of the per-step discharge capacity column.
charge_capacity_col : str, optional
Name of the per-step charge capacity column.
step_count_col : str, optional
Name of the per-step identifier column, reported in error messages.
full_step_fraction : float, optional
Multiple of rated capacity above which a step is considered a
full (and then some) charge / discharge. Defaults to ``2.0`` — i.e.
only steps delivering more than 2× rated capacity trigger the check,
which tolerates one or two full cycles being captured in a single
step while still catching runaway concatenations.
Returns
-------
list[ValidationIssue]
List of validation issues. Empty if validation passes.
"""
if rated_capacity is None or rated_capacity <= 0:
return []
if not _has_column(steps_df, step_type_col):
return []
if not _has_column(steps_df, discharge_capacity_col) or not _has_column(
steps_df, charge_capacity_col
):
return []
step_types = _get_column(steps_df, step_type_col)
discharge_caps = np.asarray(
_get_column(steps_df, discharge_capacity_col), dtype=float
)
charge_caps = np.asarray(_get_column(steps_df, charge_capacity_col), dtype=float)
step_counts = (
_get_column(steps_df, step_count_col)
if _has_column(steps_df, step_count_col)
else np.arange(len(step_types))
)
threshold = full_step_fraction * rated_capacity
errors: list[ValidationIssue] = []
prev_direction: str | None = None
prev_step_count: Any = None
for i, step_type in enumerate(step_types):
direction = _step_direction(step_type)
if direction is None:
continue
raw = discharge_caps[i] if direction == "discharge" else charge_caps[i]
delivered = 0.0 if np.isnan(raw) else abs(float(raw))
is_full_step = delivered > threshold
if is_full_step and direction == prev_direction:
curr_step = step_counts[i]
errors.append(
ValidationIssue(
IssueCode.CONSECUTIVE_SAME_DIRECTION_FULL_STEPS,
f"Consecutive full-capacity {direction} steps detected: "
f"step {prev_step_count} and step {curr_step} each "
f"delivered more than {full_step_fraction * 100:.0f}% of "
f"the rated capacity ({threshold:.3f} A.h). A single "
f"measurement should not contain multiple independent "
f"{direction} experiments; split them into separate "
f"measurements.",
payload={
"direction": direction,
"previous_step": _to_py_scalar(prev_step_count),
"current_step": _to_py_scalar(curr_step),
"threshold_capacity": float(threshold),
"rated_capacity": float(rated_capacity),
},
)
)
if is_full_step:
prev_direction = direction
prev_step_count = step_counts[i]
return errors
[docs]
def validate_step_capacity_within_rated(
steps_df: DataFrame,
rated_capacity: float | None,
discharge_capacity_col: str = "Discharge capacity [A.h]",
charge_capacity_col: str = "Charge capacity [A.h]",
step_count_col: str = "Step count",
max_ratio: float = 5.0,
) -> list[ValidationIssue]:
"""Soft-check that no single step exceeds ``max_ratio`` × rated capacity.
A single step accumulating more capacity than several full charges or
discharges of the cell usually indicates wrong step boundaries or that the
capacity integral was inflated across unrecorded time gaps. Returns a list
of warnings; callers may emit them via :func:`warnings.warn` instead of
raising.
Parameters
----------
steps_df : DataFrame
Step summary dataframe (as produced by ``ionworksdata.steps.summarize``).
rated_capacity : float | None
Rated (nominal) cell capacity in A.h. The check is skipped when this
is ``None`` or non-positive.
discharge_capacity_col : str, optional
Name of the per-step discharge capacity column.
charge_capacity_col : str, optional
Name of the per-step charge capacity column.
step_count_col : str, optional
Name of the per-step identifier column, reported in warning messages.
max_ratio : float, optional
Maximum allowed ratio of per-step capacity to rated capacity.
Defaults to ``5.0`` (500 %).
Returns
-------
list[ValidationIssue]
List of validation issues. Empty if no step exceeds the threshold.
"""
if rated_capacity is None or rated_capacity <= 0:
return []
if not _has_column(steps_df, discharge_capacity_col) and not _has_column(
steps_df, charge_capacity_col
):
return []
threshold = max_ratio * rated_capacity
step_counts = (
_get_column(steps_df, step_count_col)
if _has_column(steps_df, step_count_col)
else np.arange(len(steps_df))
)
total_count = 0
first: tuple[Any, str, float] | None = None
for col in (discharge_capacity_col, charge_capacity_col):
if not _has_column(steps_df, col):
continue
abs_values = np.abs(np.asarray(_get_column(steps_df, col), dtype=float))
over_mask = ~np.isnan(abs_values) & (abs_values > threshold)
col_count = int(over_mask.sum())
if col_count == 0:
continue
total_count += col_count
if first is None:
idx = int(np.argmax(over_mask))
first = (step_counts[idx], col, float(abs_values[idx]))
if first is None:
return []
first_step, first_col, first_value = first
more = f" (and {total_count - 1} more)" if total_count > 1 else ""
first_step_payload = _to_py_scalar(first_step)
return [
ValidationIssue(
IssueCode.STEP_CAPACITY_EXCEEDS_RATED,
f"{total_count} step(s) exceed {max_ratio * 100:.0f}% of the "
f"rated capacity ({threshold:.3f} A.h). First: step "
f"{first_step} has '{first_col}' = {first_value:.3f} A.h.{more} "
f"This usually means step boundaries are wrong or the capacity "
f"integral counted current across unrecorded time gaps.",
payload={
"total_count": total_count,
"first_step": first_step_payload,
"first_column": first_col,
"first_value": first_value,
"threshold_capacity": float(threshold),
"rated_capacity": float(rated_capacity),
},
severity="warning",
)
]
[docs]
def validate_charge_discharge_column_direction( # noqa: PLR0913
steps_df: DataFrame,
mean_current_col: str = "Mean current [A]",
min_current_col: str = "Min current [A]",
max_current_col: str = "Max current [A]",
std_current_col: str = "Std current [A]",
duration_col: str = "Duration [s]",
step_count_col: str = "Step count",
discharge_capacity_col: str = "Discharge capacity [A.h]",
charge_capacity_col: str = "Charge capacity [A.h]",
discharge_energy_col: str = "Discharge energy [W.h]",
charge_energy_col: str = "Charge energy [W.h]",
rest_tol: float = 1e-3,
min_duration_s: float = 60.0,
max_std_to_mean_ratio: float = 0.5,
dominance_ratio: float = 4.0,
swapped_fraction_threshold: float = 0.75,
) -> list[ValidationIssue]:
"""Detect swapped charge/discharge capacity (and energy) columns in a step summary.
With the platform sign convention (positive current = discharge), a step that
only discharges must accumulate its A.h in the discharge column, and a step that
only charges must accumulate in the charge column. A cycler export with the two
column labels inverted is symmetric under every cumulative-reset and magnitude
check, so this is the only signal that catches it: for each step whose current
direction is unambiguous, compare the sign of the mean current against which
column of the pair actually accumulated, and flag the pair when the clear
majority of such steps put their capacity in the opposite-direction column.
A step votes only when the evidence is unambiguous on every axis:
- its mean current is clearly non-zero (``|mean| >= rest_tol``),
- it lasted at least ``min_duration_s`` (when a duration column is available),
- its current did not change sign — the min/max currents stay on the same side
of zero (within ``rest_tol``), or the standard deviation is small relative to
``|mean|``; either piece of evidence qualifies, so a single opposite-sign
boundary sample inherited from the preceding step does not silence an
otherwise one-directional step. Bidirectional (drive-cycle) steps
legitimately accumulate in **both** columns and are excluded by this test,
and
- one column of the pair clearly dominates the other (``dominance_ratio``), so
a step that accumulated comparably in both columns never votes.
Votes are weighted by the capacity (or energy) the step moved, so a handful of
tiny glitch steps cannot outvote real cycling. The capacity pair and the energy
pair are evaluated independently; energy direction is also judged from the mean
current, since terminal voltage is positive and power therefore shares the
current's sign.
Parameters
----------
steps_df : DataFrame
Step summary dataframe (as produced by ``ionworksdata.steps.identify``),
pandas or polars.
mean_current_col : str, optional
Name of the per-step mean current column [A]. The check is skipped
entirely when this column is absent.
min_current_col, max_current_col : str, optional
Names of the per-step min/max current columns [A], used to establish that
a step's current never changed sign.
std_current_col : str, optional
Name of the per-step current standard deviation column [A]. Alternative
sign-unambiguity evidence: the step also qualifies when
``std <= max_std_to_mean_ratio * |mean|``, even when its min/max fail the
same-sign test (e.g. polluted by a boundary sample).
duration_col : str, optional
Name of the per-step duration column [s]. Steps shorter than
``min_duration_s`` (or with NaN duration) do not vote. No duration filter
is applied when the column is absent.
step_count_col : str, optional
Name of the per-step identifier column, reported in the issue payload.
discharge_capacity_col, charge_capacity_col : str, optional
Names of the per-step capacity pair [A.h]. The pair is skipped when either
column is absent.
discharge_energy_col, charge_energy_col : str, optional
Names of the per-step energy pair [W.h]. The pair is skipped when either
column is absent.
rest_tol : float, optional
Current magnitude [A] below which a step counts as rest and does not vote.
Also the tolerance for the min/max same-sign test, so a discharge step
whose current briefly touches zero still qualifies. Defaults to ``1e-3``.
min_duration_s : float, optional
Minimum step duration [s] for a step to vote. Defaults to ``60``.
max_std_to_mean_ratio : float, optional
Maximum ``std / |mean|`` for a step to qualify via the standard-deviation
fallback. Defaults to ``0.5``.
dominance_ratio : float, optional
Minimum ratio of the larger to the smaller column value for the step's
accumulation to count as clearly one-directional. Defaults to ``4.0``.
swapped_fraction_threshold : float, optional
Minimum weighted fraction of voting steps that must accumulate in the
wrong column to flag the pair as swapped. Defaults to ``0.75``.
Returns
-------
list[ValidationIssue]
At most one issue per pair (``CHARGE_DISCHARGE_CAPACITY_COLUMNS_SWAPPED``
and/or ``CHARGE_DISCHARGE_ENERGY_COLUMNS_SWAPPED``). Empty when neither
pair appears swapped or there is not enough unambiguous evidence.
"""
if not _has_column(steps_df, mean_current_col) or len(steps_df) == 0:
return []
n = len(steps_df)
mean = np.asarray(_get_column(steps_df, mean_current_col), dtype=float)
def _optional(col: str) -> np.ndarray:
if _has_column(steps_df, col):
return np.asarray(_get_column(steps_df, col), dtype=float)
return np.full(n, np.nan)
min_i = _optional(min_current_col)
max_i = _optional(max_current_col)
std_i = _optional(std_current_col)
abs_mean = np.abs(mean)
non_rest = np.isfinite(mean) & (abs_mean >= rest_tol)
# Sign-unambiguity: either piece of evidence qualifies a step — min/max
# currents on one side of zero, or a std small relative to |mean|. The two
# are alternatives (not min/max-first precedence) because a step's min/max
# can inherit a single opposite-sign boundary sample from the preceding
# step, which must not silence an otherwise clearly one-directional step.
# A step with neither piece of evidence never votes.
minmax_valid = np.isfinite(min_i) & np.isfinite(max_i)
minmax_ok = np.where(mean > 0, min_i >= -rest_tol, max_i <= rest_tol)
std_ok = np.isfinite(std_i) & (std_i <= max_std_to_mean_ratio * abs_mean)
unambiguous = (minmax_valid & minmax_ok) | std_ok
if _has_column(steps_df, duration_col):
duration = np.asarray(_get_column(steps_df, duration_col), dtype=float)
duration_ok = np.isfinite(duration) & (duration >= min_duration_s)
else:
duration_ok = np.ones(n, dtype=bool)
eligible = non_rest & unambiguous & duration_ok
if not np.any(eligible):
return []
step_counts = (
_get_column(steps_df, step_count_col)
if _has_column(steps_df, step_count_col)
else np.arange(n)
)
pairs: list[tuple[str, str, str, IssueCode]] = [
(
discharge_capacity_col,
charge_capacity_col,
"capacity",
IssueCode.CHARGE_DISCHARGE_CAPACITY_COLUMNS_SWAPPED,
),
(
discharge_energy_col,
charge_energy_col,
"energy",
IssueCode.CHARGE_DISCHARGE_ENERGY_COLUMNS_SWAPPED,
),
]
errors: list[ValidationIssue] = []
for discharge_col, charge_col, quantity, code in pairs:
if not _has_column(steps_df, discharge_col) or not _has_column(
steps_df, charge_col
):
continue
dis = np.abs(np.asarray(_get_column(steps_df, discharge_col), dtype=float))
chg = np.abs(np.asarray(_get_column(steps_df, charge_col), dtype=float))
dis = np.nan_to_num(dis, nan=0.0)
chg = np.nan_to_num(chg, nan=0.0)
moved = np.maximum(dis, chg)
other = np.minimum(dis, chg)
# One column must clearly dominate; a step accumulating comparably in
# both (a bidirectional step that slipped past the sign filter, or
# phantom accumulation across unlogged gaps) is ambiguous and skipped.
considered = eligible & (moved > 0) & (moved >= dominance_ratio * other)
if not np.any(considered):
continue
# Wrong column: a discharging step (mean > 0) accumulated in the charge
# column, or vice versa.
wrong = np.where(mean > 0, chg > dis, dis > chg)
swapped = considered & wrong
total_weight = float(moved[considered].sum())
swapped_weight = float(moved[swapped].sum())
if total_weight <= 0:
continue
swapped_fraction = swapped_weight / total_weight
if swapped_fraction < swapped_fraction_threshold:
continue
n_considered = int(considered.sum())
n_swapped = int(swapped.sum())
example_steps = [
_to_py_scalar(step_counts[i]) for i in np.where(swapped)[0][:5]
]
errors.append(
ValidationIssue(
code,
f"The charge/discharge {quantity} columns appear swapped: "
f"{n_swapped} of {n_considered} clearly-signed steps "
f"({swapped_fraction * 100:.0f}% by {quantity} moved) accumulate "
f"in the column for the opposite current direction. With the "
f"platform sign convention (positive current = discharge), a "
f"discharging step must accumulate in '{discharge_col}'. Swap "
f"the '{discharge_col}' and '{charge_col}' labels (e.g. "
f"ionworksdata.transform.fix_swapped_charge_discharge_columns "
f"on the time series, then regenerate the step summary) to fix "
f"this.",
payload={
"discharge_column": discharge_col,
"charge_column": charge_col,
"swapped_fraction": swapped_fraction,
"n_steps_considered": n_considered,
"n_steps_swapped": n_swapped,
"example_steps": example_steps,
"threshold": swapped_fraction_threshold,
},
)
)
return errors
[docs]
def step_boundary_idx(step_data: np.ndarray) -> np.ndarray:
"""Per-row index of the most recent step start.
Subtracting ``series[step_boundary_idx(step_data)]`` from a running
cumulative ``series`` resets it to 0 at every change in ``step_data``.
Compute this once and pass it into multiple
:func:`running_step_reset_integral` calls when they share the same
step axis.
Parameters
----------
step_data : np.ndarray
Per-row step identifier (e.g. the ``"Step count"`` column).
Returns
-------
np.ndarray
Index of the most recent step-start row for each row, same length
as ``step_data``.
"""
groups = _get_step_group_indices(step_data)
starts = np.where(
np.concatenate([[True], np.diff(groups) != 0]),
np.arange(len(groups)),
0,
)
return np.maximum.accumulate(starts)
[docs]
def running_step_reset_integral(
signed: np.ndarray,
time: np.ndarray,
step_data: np.ndarray | None = None,
*,
boundary_idx: np.ndarray | None = None,
) -> np.ndarray:
"""Cumulative trapezoidal integral of ``signed`` that resets at each step.
Mirrors the reset semantics of the platform's cumulative capacity and
energy columns so the returned series can be compared row-by-row.
Divides by 3600 so the result is in A.h when ``signed`` is in A, or in
W.h when ``signed`` is in W.
Parameters
----------
signed : np.ndarray
Per-row signed values (e.g. ``max(I, 0)`` to integrate the
discharge half-wave only).
time : np.ndarray
Time values [s]; same length as ``signed``.
step_data : np.ndarray, optional
Per-row step identifier (e.g. ``"Step count"`` column). The
integral is reset to 0 at every change in this value. When
``None`` (and ``boundary_idx`` is also None), the integral is
not reset.
boundary_idx : np.ndarray, optional
Output of :func:`step_boundary_idx` for the same step axis.
Provide this to amortise the boundary computation across
multiple integrals that share the same ``step_data``.
Returns
-------
np.ndarray
Running per-step integral, same length as ``signed``, reset to 0
at the first row of each step.
"""
integrated = cumulative_trapezoid(signed, time, initial=0.0) / 3600.0
if boundary_idx is None:
if step_data is None:
return integrated
boundary_idx = step_boundary_idx(step_data)
return integrated - integrated[boundary_idx]
[docs]
def worst_row_relative_error(
reported: np.ndarray, integrated: np.ndarray
) -> tuple[int, float, float]:
"""Index, magnitude, and scale of the largest row-wise relative deviation.
The relative error is the largest absolute deviation between the two
series divided by the larger of their max-magnitudes (the scale). NaN
rows are ignored. Returns ``(0, inf, 0.0)`` when there is nothing to
compare (both series flat at zero, or all-NaN).
Parameters
----------
reported : np.ndarray
Reported cumulative series.
integrated : np.ndarray
Integrated series to compare against, same length as ``reported``.
Returns
-------
tuple[int, float, float]
``(worst_row_index, relative_error, max_scale)``.
"""
with warnings.catch_warnings():
warnings.simplefilter("ignore", RuntimeWarning)
max_scale = float(
np.fmax(np.nanmax(np.abs(integrated)), np.nanmax(np.abs(reported)))
)
if not np.isfinite(max_scale) or max_scale <= 0.0:
return 0, float("inf"), 0.0
diff = np.abs(reported - integrated)
if not np.any(np.isfinite(diff)):
return 0, float("inf"), max_scale
worst_idx = int(np.nanargmax(diff))
return worst_idx, float(diff[worst_idx]) / max_scale, max_scale
[docs]
def validate_capacity_energy_from_current_power(
df: DataFrame,
step_col: str = "Step count",
time_col: str = "Time [s]",
current_col: str = "Current [A]",
voltage_col: str = "Voltage [V]",
power_col: str = "Power [W]",
discharge_capacity_col: str = "Discharge capacity [A.h]",
charge_capacity_col: str = "Charge capacity [A.h]",
discharge_energy_col: str = "Discharge energy [W.h]",
charge_energy_col: str = "Charge energy [W.h]",
tolerance: float = CAPACITY_ENERGY_INTEGRAL_TOLERANCE,
) -> list[ValidationIssue]:
"""Validate cumulative capacity/energy columns match integrals of current/power.
Builds a running trapezoidal integral over the whole time series and
compares it row-by-row against the reported cumulative columns. The
reported columns reset to 0 at the start of each step, so the integral
is reset the same way and the comparison is made on the running
within-step accumulator. With the platform sign convention (positive
current = discharge):
- Discharge capacity [A.h] ≈ ∫ max(I, 0) dt / 3600
- Charge capacity [A.h] ≈ ∫ max(-I, 0) dt / 3600
- Discharge energy [W.h] ≈ ∫ max(P, 0) dt / 3600
- Charge energy [W.h] ≈ ∫ max(-P, 0) dt / 3600
The reported series and the integrated series are compared at every row.
The relative error is the maximum absolute deviation across the whole
series divided by the maximum cumulative value reached on either side,
so a local discrepancy that later cancels out still triggers the check.
A single issue is raised per cumulative column whose error exceeds
``tolerance``. Columns missing from ``df`` are silently skipped.
Parameters
----------
df : DataFrame
Time series data (pandas or polars).
step_col : str, optional
Name of the step identifier column.
time_col : str, optional
Name of the time column [s].
current_col : str, optional
Name of the signed current column [A].
voltage_col : str, optional
Name of the terminal voltage column [V]. Used to derive power as
``V * I`` when ``power_col`` is not present in ``df``.
power_col : str, optional
Name of the signed power column [W]. When absent, power is
computed from ``voltage_col * current_col``.
discharge_capacity_col, charge_capacity_col : str, optional
Names of the cumulative capacity columns [A.h].
discharge_energy_col, charge_energy_col : str, optional
Names of the cumulative energy columns [W.h].
tolerance : float, optional
Maximum allowed relative error between the integrated and reported
series, evaluated as ``max|reported - integrated| / max(reported,
integrated)`` across all rows. Defaults to ``0.10`` (10 %).
Returns
-------
list[ValidationIssue]
One issue per cumulative column whose running series deviates from
the integrated series by more than ``tolerance`` at any row. Empty
when all present columns agree within tolerance everywhere.
"""
if not _has_column(df, time_col) or not _has_column(df, step_col):
return []
time = np.asarray(_get_column(df, time_col), dtype=float)
if len(time) < 2:
return []
step_data = _get_column(df, step_col)
current = (
np.asarray(_get_column(df, current_col), dtype=float)
if _has_column(df, current_col)
else None
)
if _has_column(df, power_col):
power = np.asarray(_get_column(df, power_col), dtype=float)
power_source = power_col
elif current is not None and _has_column(df, voltage_col):
# Power isn't reported separately — derive from V * I, which is
# what the platform's conventional Power [W] column equals anyway.
power = np.asarray(_get_column(df, voltage_col), dtype=float) * current
power_source = f"{voltage_col} * {current_col}"
else:
power = None
power_source = "" # unused: the spec is filtered out when power is None
specs: list[tuple[str, np.ndarray | None, str, str, str, int, IssueCode]] = [
# (reported column, source array, source label for the message/payload,
# unit, source expression, sign-of-positive-half, issue code).
# sign=+1 picks max(x, 0), sign=-1 picks max(-x, 0).
(
discharge_capacity_col,
current,
current_col,
"A.h",
"max(I, 0)",
+1,
IssueCode.DISCHARGE_CAPACITY_INTEGRAL_MISMATCH,
),
(
charge_capacity_col,
current,
current_col,
"A.h",
"max(-I, 0)",
-1,
IssueCode.CHARGE_CAPACITY_INTEGRAL_MISMATCH,
),
(
discharge_energy_col,
power,
power_source,
"W.h",
"max(P, 0)",
+1,
IssueCode.DISCHARGE_ENERGY_INTEGRAL_MISMATCH,
),
(
charge_energy_col,
power,
power_source,
"W.h",
"max(-P, 0)",
-1,
IssueCode.CHARGE_ENERGY_INTEGRAL_MISMATCH,
),
]
active = [
spec for spec in specs if _has_column(df, spec[0]) and spec[1] is not None
]
if not active:
return []
boundary_idx = step_boundary_idx(step_data)
errors: list[ValidationIssue] = []
for (
reported_col,
source_array,
source_col,
unit,
source_expr,
sign,
code,
) in active:
signed = np.maximum(sign * source_array, 0.0)
reported = np.asarray(_get_column(df, reported_col), dtype=float)
integrated = running_step_reset_integral(
signed, time, boundary_idx=boundary_idx
)
worst_idx, relative_error, max_scale = worst_row_relative_error(
reported, integrated
)
if not np.isfinite(relative_error) or relative_error <= tolerance:
continue
errors.append(
ValidationIssue(
code,
f"Column '{reported_col}' disagrees with the running integral "
f"of '{source_expr}' over time by up to "
f"{relative_error * 100:.1f}% (exceeds {tolerance * 100:.0f}% "
f"tolerance). Worst row {worst_idx}: reported = "
f"{float(reported[worst_idx]):.4f} {unit}, integrated = "
f"{float(integrated[worst_idx]):.4f} {unit}.",
payload={
"column": reported_col,
"source_column": source_col,
"unit": unit,
"worst_row_index": worst_idx,
"reported_at_worst": float(reported[worst_idx]),
"integrated_at_worst": float(integrated[worst_idx]),
"max_scale": max_scale,
"relative_error": relative_error,
"tolerance": tolerance,
},
)
)
return errors
def _parse_dt(value: datetime | str) -> datetime | None:
"""Parse an ISO 8601 string (or pass through a datetime); None if unparseable."""
if isinstance(value, datetime):
return value
try:
return datetime.fromisoformat(value)
except (ValueError, TypeError):
return None
[docs]
def validate_measurement_timing(
start_time: datetime | str | None,
end_time: datetime | str | None,
df: DataFrame | None = None,
*,
duration_tolerance_frac: float = 0.05,
duration_tolerance_min_s: float = 60.0,
now: datetime | None = None,
) -> list[ValidationIssue]:
"""Sanity-check a measurement's wall-clock ``start_time`` / ``end_time``.
All findings are **warning** severity — timing metadata is often
approximate (clock skew, timezone sloppiness, planned runs), so these
surface a nudge without blocking an upload. Checks:
- ``start_time`` / ``end_time`` should not be in the future.
- ``end_time`` should not precede ``start_time`` (the server enforces this
as a hard error; this is an early client-side echo).
- When ``df`` is given, the wall-clock duration ``end_time - start_time``
should roughly match the elapsed span in the ``Time [s]`` column (which is
relative, starting at 0). A large mismatch suggests a mislabelled
timestamp or a paused/segmented test.
Parameters
----------
start_time, end_time : datetime | str | None
The measurement's timestamps (ISO 8601 strings or ``datetime``). A
``None`` skips the checks that need it (e.g. a still-running test has no
``end_time``).
df : DataFrame | None, optional
Time series with a ``Time [s]`` column, used for the duration check.
duration_tolerance_frac : float, optional
Allowed relative difference between wall-clock and data duration before
warning. Defaults to 0.05 (5 %).
duration_tolerance_min_s : float, optional
Absolute floor for the tolerance in seconds, so short tests aren't
flagged by rounding. Defaults to 60 s.
now : datetime | None, optional
Reference "now" for the future checks; defaults to the current UTC
time. Injectable for deterministic tests.
Returns
-------
list[ValidationIssue]
Warning-severity issues (possibly empty).
"""
issues: list[ValidationIssue] = []
ref_now = now or datetime.now(UTC)
start = _parse_dt(start_time) if start_time is not None else None
end = _parse_dt(end_time) if end_time is not None else None
def _cmp_safe(a: datetime, b: datetime) -> bool | None:
"""a < b, or None when tz-awareness differs (uncomparable)."""
if (a.tzinfo is None) != (b.tzinfo is None):
return None
return a < b
if start is not None and _cmp_safe(ref_now, start) is True:
issues.append(
ValidationIssue(
code=IssueCode.START_TIME_IN_FUTURE,
message=(
f"start_time ({start.isoformat()}) is in the future. "
f"Check the measurement's timestamp."
),
severity="warning",
payload={"start_time": start.isoformat()},
)
)
if end is not None and _cmp_safe(ref_now, end) is True:
issues.append(
ValidationIssue(
code=IssueCode.END_TIME_IN_FUTURE,
message=(
f"end_time ({end.isoformat()}) is in the future. "
f"A completed test should have finished in the past."
),
severity="warning",
payload={"end_time": end.isoformat()},
)
)
if start is not None and end is not None and _cmp_safe(end, start) is True:
issues.append(
ValidationIssue(
code=IssueCode.END_TIME_BEFORE_START_TIME,
message=(
f"end_time ({end.isoformat()}) is before start_time "
f"({start.isoformat()}). The server will reject this on upload."
),
severity="warning",
payload={"start_time": start.isoformat(), "end_time": end.isoformat()},
)
)
# Duration consistency: wall-clock (end - start) vs data span max(Time [s]).
if (
start is not None
and end is not None
and df is not None
and _has_column(df, "Time [s]")
and (start.tzinfo is None) == (end.tzinfo is None)
):
time_s = _get_column(df, "Time [s]")
if len(time_s) > 0:
data_span = float(np.nanmax(time_s) - np.nanmin(time_s))
wall_span = (end - start).total_seconds()
if wall_span >= 0 and data_span > 0:
tol = max(duration_tolerance_min_s, duration_tolerance_frac * data_span)
if abs(wall_span - data_span) > tol:
issues.append(
ValidationIssue(
code=IssueCode.DURATION_MISMATCH,
message=(
f"Wall-clock duration (end_time - start_time = "
f"{wall_span:.0f} s) differs from the time series "
f"span ({data_span:.0f} s) by more than the "
f"tolerance ({tol:.0f} s). Check the timestamps "
f"or whether the test was paused/segmented."
),
severity="warning",
payload={
"wall_clock_seconds": wall_span,
"data_span_seconds": data_span,
"tolerance_seconds": tol,
},
)
)
return issues
STRICT_CHECK_NAMES: frozenset[str] = frozenset(
{
"minimum_points_per_step",
"cycle_constant_within_step",
"time_gaps",
"voltage_continuity",
"consecutive_same_direction_full_steps",
"step_capacity_within_rated",
"charge_discharge_column_direction",
"capacity_energy_from_current_power",
"eis_zim_sign",
"eis_impedance_magnitude",
}
)
[docs]
def validate_measurement_data(
df: DataFrame,
strict: bool = False,
data_type: str | None = None,
steps_df: DataFrame | None = None,
rated_capacity: float | None = None,
voltage_window: tuple[float, float] | None = None,
skip_checks: Iterable[str] | None = None,
start_time: datetime | str | None = None,
end_time: datetime | str | None = None,
) -> None:
"""Validate measurement time series data before upload.
For standard cycler data (``data_type=None``), always runs:
1. Positive current should correspond to discharge (voltage decreases)
2. Time starts at 0
3. Time is monotonically non-decreasing
4. 'Step count' column exists, starts at 0, and increases by 1
5. Cumulative values (capacity, energy) reset at each step start and
only increase within steps
The remaining checks are strict-mode only (``strict=True``):
6. Each step has at least 2 data points
7. Cycle number does not change within a step
8. No time gap between consecutive rows exceeds 5 hours
9. When ``voltage_window`` is provided, voltage is continuous between
consecutive rows (no systematic chronological reordering)
10. When ``steps_df`` and ``rated_capacity`` are provided, two
consecutive steps delivering more than 2× rated capacity do not
share the same direction
11. When ``steps_df`` and ``rated_capacity`` are provided, no single
step exceeds 500 % of the rated capacity (soft warning)
12. Reported cumulative capacity/energy columns agree with the
trapezoidal integral of current/power within 10 %
13. When ``steps_df`` is provided, each clearly-signed step accumulates
its capacity/energy in the column matching its current direction —
catches charge/discharge column pairs whose labels are swapped
For OCP data (``data_type="ocp"``), only validates:
1. 'Voltage [V]' column exists
2. 'Step count' column exists and is sequential
For EIS data (``data_type="eis"``), always validates:
1. The 'Frequency [Hz]', 'Z_Re [Ohm]', and 'Z_Im [Ohm]' columns exist
and, in strict mode only (both skippable):
2. 'Z_Im [Ohm]' sign is not reversed (capacitive band should be negative
in the raw ``Im(Z)`` convention)
3. When ``rated_capacity`` is provided, the ohmic resistance
``R0 = min(Z_Re)`` lies within a wide (100×) plausibility band around a
capacity-derived estimate — catches order-of-magnitude unit errors
Parameters
----------
df : DataFrame
Time series data to validate (pandas or polars DataFrame).
strict : bool
If False (default), run only the always-on checks above. If True,
additionally run: minimum 2 points per step, cycle number constant
within step, time-gap check, voltage-continuity check (when
``voltage_window`` is provided), and the step-capacity checks (when
``steps_df`` and ``rated_capacity`` are provided).
data_type : str | None
The type of data being validated. Use ``"ocp"`` for open-circuit
potential data, which relaxes validation to skip current, time,
capacity, and energy checks. Use ``"eis"`` for impedance spectra,
which validates the EIS columns and (strict-only) the ``Z_Im`` sign
and impedance magnitude instead of the cycler checks. Default is
``None`` (standard cycler data).
steps_df : DataFrame | None
Optional step summary dataframe. Used only in strict mode for the
charge/discharge column-direction check and — together with
``rated_capacity`` — the consecutive same-direction and per-step
capacity checks.
rated_capacity : float | None
Rated (nominal) cell capacity in A.h. Used only in strict mode: with
``steps_df`` it enables the consecutive-full-step and per-step
capacity soft warnings; with ``data_type="eis"`` it enables the EIS
impedance-magnitude sanity check.
voltage_window : tuple[float, float] | None
Rated ``(V_min, V_max)`` voltage window of the cell. Used only in
strict mode to enable the voltage-continuity check.
start_time, end_time : datetime | str | None
The measurement's wall-clock timestamps. When either is provided, a
set of **warning-severity** timing sanity checks runs (never blocks the
upload): a future ``start_time``/``end_time``, an ``end_time`` before
``start_time`` (which the server rejects as a hard error), and — when
both are present with the time series — a wall-clock duration that
disagrees with the ``Time [s]`` span. See
:func:`validate_measurement_timing`.
skip_checks : Iterable[str] | None
Names of strict-mode checks to skip while keeping ``strict=True``
for everything else. Use this to relax a single known-problematic
check (least-privilege) instead of disabling strict mode entirely.
Recognized names are listed in :data:`STRICT_CHECK_NAMES`:
``"minimum_points_per_step"``, ``"cycle_constant_within_step"``,
``"time_gaps"``, ``"voltage_continuity"``,
``"consecutive_same_direction_full_steps"``,
``"step_capacity_within_rated"``,
``"charge_discharge_column_direction"``,
``"capacity_energy_from_current_power"``, ``"eis_zim_sign"``,
``"eis_impedance_magnitude"``. Unknown names raise
``ValueError``.
Raises
------
MeasurementValidationError
If any validation checks fail. The exception contains a list of all
errors found.
"""
skip = frozenset(skip_checks or ())
unknown = skip - STRICT_CHECK_NAMES
if unknown:
raise ValueError(
f"Unknown check name(s) in skip_checks: {sorted(unknown)}. "
f"Valid names: {sorted(STRICT_CHECK_NAMES)}"
)
all_errors: list[ValidationIssue] = []
step_col = "Step count"
if data_type == "ocp":
ocp_col_errors = validate_ocp_columns(df)
all_errors.extend(ocp_col_errors)
step_seq_errors = validate_step_count_sequential(df)
all_errors.extend(step_seq_errors)
elif data_type == "eis":
# Column presence is always-on (structural). The sign and magnitude
# checks are heuristic and strict-only + skippable, and only run once the
# required columns are present.
eis_col_errors = validate_eis_columns(df)
all_errors.extend(eis_col_errors)
if strict and not eis_col_errors:
if "eis_zim_sign" not in skip:
all_errors.extend(validate_eis_zim_sign(df))
if rated_capacity is not None and "eis_impedance_magnitude" not in skip:
all_errors.extend(validate_eis_impedance_magnitude(df, rated_capacity))
else:
all_errors.extend(validate_positive_current_is_discharge(df, step_col=step_col))
all_errors.extend(validate_time_starts_at_zero(df))
all_errors.extend(validate_time_monotonic(df))
all_errors.extend(validate_step_count_sequential(df))
if _has_column(df, step_col):
all_errors.extend(validate_cumulative_values_reset_per_step(df, step_col))
if strict:
if _has_column(df, step_col):
if "minimum_points_per_step" not in skip:
all_errors.extend(validate_minimum_points_per_step(df, step_col))
if "cycle_constant_within_step" not in skip:
all_errors.extend(validate_cycle_constant_within_step(df, step_col))
if "capacity_energy_from_current_power" not in skip:
all_errors.extend(
validate_capacity_energy_from_current_power(df, step_col)
)
if "time_gaps" not in skip:
all_errors.extend(validate_time_gaps(df))
if voltage_window is not None and "voltage_continuity" not in skip:
all_errors.extend(validate_voltage_continuity(df, voltage_window))
if steps_df is not None:
if "charge_discharge_column_direction" not in skip:
all_errors.extend(
validate_charge_discharge_column_direction(steps_df)
)
if rated_capacity is not None:
if "consecutive_same_direction_full_steps" not in skip:
all_errors.extend(
validate_consecutive_same_direction_full_steps(
steps_df, rated_capacity
)
)
if "step_capacity_within_rated" not in skip:
for issue in validate_step_capacity_within_rated(
steps_df, rated_capacity
):
warnings.warn(issue.message, stacklevel=2)
# Wall-clock timing sanity (all warning-severity: emit, never raise).
# Applies to every measurement type — the data-span check only runs when a
# "Time [s]" column is present.
if start_time is not None or end_time is not None:
for issue in validate_measurement_timing(start_time, end_time, df):
warnings.warn(issue.message, stacklevel=2)
if all_errors:
raise MeasurementValidationError(
f"Measurement data validation failed with {len(all_errors)} error(s):\n"
+ "\n".join(f" - {err}" for err in all_errors),
errors=all_errors,
)
# --- Atomic validators ------------------------------------------------------ #
[docs]
def df_to_dict_validator(v: Any) -> Any:
"""Convert DataFrame to dict with orient='list' for serialization."""
if isinstance(v, pd.DataFrame):
# Replace inf/-inf and NaN with None for JSON compatibility
return v.replace([np.inf, -np.inf, np.nan], None).to_dict(orient="list")
if isinstance(v, pl.DataFrame):
# Replace inf/-inf and NaN with None for JSON compatibility
# Process each column individually to avoid name conflicts
result = {}
for col_name in v.columns:
col = v[col_name]
if col.dtype.is_float():
# Replace inf/-inf and NaN with None
sanitized = col.to_list()
sanitized = [
None if (x is not None and (math.isinf(x) or math.isnan(x))) else x
for x in sanitized
]
result[col_name] = sanitized
else:
result[col_name] = col.to_list()
return result
return v
[docs]
def dict_to_df_validator(v: Any, return_type: str | None = None) -> Any:
"""Convert dict to DataFrame for data processing.
Parameters
----------
v : Any
Value to convert. If dict, converts to DataFrame.
return_type : str | None
Type of DataFrame to return: "polars" or "pandas".
If None, uses the global setting from set_dataframe_backend().
Returns
-------
Any
DataFrame if input was dict, otherwise unchanged.
"""
if isinstance(v, dict):
backend = return_type if return_type is not None else _dataframe_backend
# Check if all values are scalars (not lists/arrays)
all_scalars = all(
not isinstance(val, list | tuple | np.ndarray) for val in v.values()
)
if backend == "pandas":
if all_scalars:
return pd.DataFrame(v, index=[0])
return pd.DataFrame(v)
if all_scalars:
return pl.DataFrame({k: [val] for k, val in v.items()})
return pl.DataFrame(v)
return v
[docs]
def parameter_validator(v: Any) -> Any:
"""Convert pybamm.Symbol values to JSON-serializable form."""
import pybamm
if not isinstance(v, pybamm.Symbol):
return v
from pybamm.expression_tree.operations.serialise import convert_symbol_to_json
return convert_symbol_to_json(v)
[docs]
def float_sanitizer(v: Any) -> Any:
"""Sanitize float values to JSON-compatible forms.
Converts inf, -inf, and NaN to None since these are not JSON-compliant.
"""
if isinstance(v, float) and (math.isinf(v) or math.isnan(v)):
return None
if isinstance(v, np.floating) and (np.isinf(v) or np.isnan(v)):
return None
return v
[docs]
def bounds_tuple_validator(v: Any) -> Any:
"""Convert bounds 2-tuple to list for JSON serialization.
Parameters
----------
v : Any
Value to validate. If it's a tuple with 2 elements, converts to list.
Returns
-------
Any
List if input was a 2-tuple, otherwise unchanged.
"""
if isinstance(v, tuple) and len(v) == 2:
return list(v)
return v
def _raise_if_time_series_too_large(df: DataFrame) -> None:
"""Raise ``MeasurementValidationError`` if ``df`` exceeds the inline row cap.
Shared by :func:`file_scheme_validator` and
:func:`_time_series_row_count_validator` so ``file:``/``folder:`` references
inherit the exact same 1,000-row inline limit (and error message) as a bare
inline ``DataFrame``.
"""
errors = validate_time_series_row_count(df)
if errors:
raise MeasurementValidationError(errors[0].message, errors=errors)
[docs]
def file_scheme_validator(v: Any) -> Any:
"""Convert file:// and folder:// scheme paths to serialized dicts.
Handles ``file:`` prefixed paths (loads CSV as dict) and ``folder:``
prefixed paths (loads time_series and steps as dict). For ``folder:``,
parquet files are preferred over CSV when both are present.
All other values are returned unchanged.
The referenced contents are read from the local filesystem and inlined into
the submitted config, so the time series is subject to the same 1,000-row
inline limit as a bare ``DataFrame``; larger datasets must be uploaded as a
measurement and referenced with ``db:<measurement_id>``.
Raises
------
FileNotFoundError
If the file or folder path doesn't exist.
MeasurementValidationError
If the referenced time series exceeds the inline row limit.
"""
if isinstance(v, str) and v.startswith("file:"):
# removeprefix, not split(":"), so Windows drive letters (file:C:\...) survive.
path = pathlib.Path(v.removeprefix("file:")).expanduser().resolve()
if not path.exists() or not path.is_file():
raise FileNotFoundError(f"CSV file not found: {v}")
df = pd.read_csv(path)
_raise_if_time_series_too_large(df)
return df_to_dict_validator(df)
if isinstance(v, str) and v.startswith("folder:"):
path = pathlib.Path(v.removeprefix("folder:")).expanduser().resolve()
if not path.exists() or not path.is_dir():
raise FileNotFoundError(f"Folder not found: {v}")
def _read(stem: str) -> Any:
if (path / f"{stem}.parquet").exists():
df = pd.read_parquet(path / f"{stem}.parquet")
else:
df = pd.read_csv(path / f"{stem}.csv")
# Only the time series is bound by the inline row cap; the steps
# summary is small by construction.
if stem == "time_series":
_raise_if_time_series_too_large(df)
return df_to_dict_validator(df)
return {"time_series": _read("time_series"), "steps": _read("steps")}
return v
# --- Pipeline composition helpers ------------------------------------------ #
Validator = Callable[[Any], Any]
def _apply_pipeline(value: Any, validators: Iterable[Validator]) -> Any:
transformed = value
for validator in validators:
transformed = validator(transformed)
return transformed
[docs]
def pybamm_model_validator(v: Any) -> Any:
"""Convert pybamm.BaseModel instances to JSON-serializable config dicts."""
import pybamm
if not isinstance(v, pybamm.BaseModel):
return v
# Forward compat: PyBaMM PR #5411 adds to_config()
if hasattr(v, "to_config"):
return v.to_config()
class_name = v.__class__.__name__
# Built-in pybamm.lithium_ion models → lightweight config
if hasattr(pybamm.lithium_ion, class_name):
config: dict[str, Any] = {"type": class_name}
if hasattr(v, "options"):
opts = {k: val for k, val in v.options.items() if val is not None}
if opts:
config["options"] = opts
return config
# Custom / iwp models → full serialization + strip inf bounds
from pybamm.expression_tree.operations.serialise import Serialise
serialized = Serialise.serialise_custom_model(v)
stripped = _strip_inf_bounds_from_serialized_model(serialized)
return {"type": "custom", "model": stripped}
def _strip_inf_bounds_from_serialized_model(value: dict) -> dict:
"""Remove variable bounds containing inf/-inf from a serialized PyBaMM model.
PyBaMM serializes variable bounds as ``[{value: -inf}, {value: inf}]``,
which are not JSON compliant. Dropping the ``bounds`` key entirely is
safe because PyBaMM reconstructs default ``(-inf, inf)`` bounds during
deserialization when the key is absent.
"""
def _has_inf(bounds: list) -> bool:
return any(
isinstance(b, dict)
and isinstance(b.get("value"), float | np.floating)
and math.isinf(float(b["value"]))
for b in bounds
)
def _walk(obj: Any) -> Any:
if isinstance(obj, dict):
return {
k: _walk(v)
for k, v in obj.items()
if not (k == "bounds" and isinstance(v, list) and _has_inf(v))
}
if isinstance(obj, list | tuple):
return [_walk(item) for item in obj]
return obj
return _walk(value)
def _apply_recursive(value: Any, validators: Iterable[Validator]) -> Any:
if isinstance(value, dict):
# Sanitize serialized PyBaMM model dicts by stripping inf/-inf
# values that are not JSON compliant. PyBaMM reconstructs default
# bounds on deserialization so this is lossless.
if "pybamm_version" in value and "schema_version" in value:
return _strip_inf_bounds_from_serialized_model(value)
return {key: _apply_recursive(val, validators) for key, val in value.items()}
if isinstance(value, tuple):
# Apply validators to tuple first (e.g., to convert bounds tuples to lists)
transformed = _apply_pipeline(value, validators)
# If validator converted tuple to list, process recursively
if isinstance(transformed, list):
return [_apply_recursive(item, validators) for item in transformed]
# Otherwise, process tuple items recursively
return [_apply_recursive(item, validators) for item in transformed]
if isinstance(value, list):
return [_apply_recursive(item, validators) for item in value]
return _apply_pipeline(value, validators)
# --- Public pipelines ------------------------------------------------------- #
def _time_series_row_count_validator(v: Any) -> Any:
"""Block inline DataFrames that exceed the row limit.
Raises
------
MeasurementValidationError
If the DataFrame has more than 1000 rows.
"""
if isinstance(v, pd.DataFrame | pl.DataFrame):
_raise_if_time_series_too_large(v)
return v
# Order matters: file_scheme_validator must precede df_to_dict_validator so it
# can apply the inline row cap to a file:/folder: time series while it is still a
# DataFrame; _time_series_row_count_validator then catches bare inline DataFrames.
validators_outbound: list[Validator] = [
pybamm_model_validator,
float_sanitizer,
bounds_tuple_validator,
file_scheme_validator,
_time_series_row_count_validator,
df_to_dict_validator,
parameter_validator,
]
validators_inbound: list[Validator] = [
dict_to_df_validator,
]
[docs]
def run_validators_outbound(v: Any) -> Any:
"""Recursively apply outbound validators to values and nested containers."""
return _apply_recursive(v, validators_outbound)
[docs]
def run_validators_inbound(v: Any) -> Any:
"""Recursively apply inbound validators to values and nested containers."""
return _apply_recursive(v, validators_inbound)