From 4383cece5fc560a7a9c11b486a46072ee189eaa7 Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 30 Sep 2026 13:43:57 +0200 Subject: [PATCH 01/27] feat(reporting): add TimeWindowEvent for fixed-duration tumbling windows per container Introduce `TimeWindowEvent`, which splits each measurement container into consecutive fixed-duration windows (one event instance per slice), complementing the existing `ContainerEvent` (one instance per container). - Add `TumblingWindowsExpression` in the query engine to derive window boundaries from `container_metrics.start_ts` / `stop_ts` with no channel selectors; final window is clamped to `stop_ts`. - Add `TimeWindowEvent` reporting class wired into `EventType`, producing `event_instance_fact` rows and `event_dimension` metadata with `window_length` surfaced in attributes. - Update event reference docs with `TimeWindowEvent` usage, parameters, and comparison table. - Add unit tests for `TumblingWindowsExpression` and `TimeWindowEvent`, plus integration tests verifying end-to-end report behavior and aggregation joins. --- docs/impulse/docs/references/report/event.md | 70 +++- .../analyze/query/events/__init__.py | 3 + .../events/tumbling_windows_expression.py | 179 ++++++++++ src/impulse_reporting/events/event_types.py | 8 + .../events/time_window_event.py | 270 ++++++++++++++++ .../tumbling_windows_expression_test.py | 109 +++++++ .../integration/time_window_event_test.py | 305 ++++++++++++++++++ .../unit/events/time_window_event_test.py | 82 +++++ 8 files changed, 1019 insertions(+), 7 deletions(-) create mode 100644 src/impulse_query_engine/analyze/query/events/tumbling_windows_expression.py create mode 100644 src/impulse_reporting/events/time_window_event.py create mode 100644 tests/impulse_query_engine/unit/analyze/query/events/tumbling_windows_expression_test.py create mode 100644 tests/impulse_reporting/integration/time_window_event_test.py create mode 100644 tests/impulse_reporting/unit/events/time_window_event_test.py diff --git a/docs/impulse/docs/references/report/event.md b/docs/impulse/docs/references/report/event.md index 46ac02e2..4023e73c 100644 --- a/docs/impulse/docs/references/report/event.md +++ b/docs/impulse/docs/references/report/event.md @@ -195,6 +195,62 @@ The expression **must** evaluate to a `PointsInTime`; otherwise construction rai --- +## TimeWindowEvent + +A `TimeWindowEvent` divides each matching container into **consecutive fixed-duration windows** -- +one event instance per slice. Unlike `ContainerEvent` (one instance for the whole container), it +produces repeated windows (e.g. one-minute, ten-minute, hourly, or daily segments) across every +matching container. No signal expression is needed: the window boundaries are derived from the +container's `start_ts` / `stop_ts` on the `container_metrics` table. + +```python +from impulse_reporting.events.time_window_event import TimeWindowEvent + +ten_minute_windows = TimeWindowEvent( + name="ten_minute_windows", + window_length=600_000, # in the same time unit as the underlying timestamps (see note) + desc="Ten-minute segments across each measurement", +) +my_report.add_event(ten_minute_windows) +``` + +### Parameters + +| Parameter | Type | Required | Description | +|---------------------|---------------------|----------|-----------------------------------------------------------------------------------------------------------------| +| `name` | `str` | Yes | Unique event name. | +| `window_length` | `float` | Yes | Fixed window length, **in the same time unit as the underlying timestamps** (e.g. milliseconds-since-epoch). Must be strictly positive; validated at construction. | +| `desc` | `str` | No | Human-readable description. | +| `required_channels` | `list[str]` | No | Channel names required for this event. Informational; stored in the event dimension table. | +| `attributes` | `Mapping[str, str]` | No | Free-form key-value metadata. `window_length` is surfaced here automatically (without overriding a user key). | + +:::note +`window_length` follows the same convention as `SequenceOfEvents.max_overlap`: it is expressed in +the same time unit as the stored timestamps (milliseconds-since-epoch in the sample data), not +seconds or any derived unit. So 60 one-minute windows over millisecond timestamps use +`window_length=60_000`. +::: + +### How it works + +1. A tumbling-window expression reads `start_ts` and `stop_ts` from the `container_metrics` table + and tiles `[start_ts, stop_ts]` into consecutive windows of length `window_length`. +2. The **final window is clamped** to `stop_ts` when the last full window would overrun it; any + zero-length trailing slice is dropped (every instance satisfies `start_ts < end_ts`). +3. Each window becomes one **event instance** with a unique `event_instance_id`, written to the + shared `event_instance_fact` table. +4. An aggregation scoped to the event (`StatsAggregator(..., event=time_window_event)`) computes + its statistic **once per window** and joins back to those instances. + +:::note +A `TimeWindowEvent` is evaluated in the shared batch solve alongside the report's channel-based +selections, so it materializes windows for the containers covered by that solve. In practice this +is every container the report touches -- a `TimeWindowEvent` is normally paired with at least one +aggregation (its purpose), which supplies the channels. The window boundaries come from +`container_metrics`, so for a scoped aggregation to produce meaningful per-window values, those +boundaries must share the channel samples' time base (as they do in real measurement data). +::: + ## Event output schema ### event_dimension @@ -205,7 +261,7 @@ Stores event definitions (one row per event per report). |---------------------|---------------------|-----------------------------------------------------------------------------| | `event_id` | `int` | Unique event identifier (CRC32 hash of name + expression). | | `report_id` | `int` | Report identifier. | -| `event_type` | `str` | `"BASIC_EVENT"`, `"CONTAINER_EVENT"`, `"SEQUENCE_OF_EVENTS"`, or `"POINTS_IN_TIME_EVENT"`. | +| `event_type` | `str` | `"BASIC_EVENT"`, `"CONTAINER_EVENT"`, `"SEQUENCE_OF_EVENTS"`, `"POINTS_IN_TIME_EVENT"`, or `"TIME_WINDOW_EVENT"`. | | `event_name` | `str` | Event name. | | `event_description` | `str` | Event description. | | `required_channels` | `array[str]` | Required channel names (null for `ContainerEvent`). | @@ -232,9 +288,9 @@ Interval events satisfy `start_ts < end_ts`; `PointsInTimeEvent` instances are z ## Choosing between event types -| Criterion | BasicEvent | ContainerEvent | SequenceOfEvents | PointsInTimeEvent | -|----------------------------------|---------------------------------------------------------|---------------------------------------------------|---------------------------------------------------------------------------|------------------------------------------------------------| -| Requires a TSAL expression | Yes (one) | No | Yes (ordered list) | Yes (one, must evaluate to `PointsInTime`) | -| Multiple instances per container | Yes (one per matching interval) | No (always one per container) | Yes (one per joined sequence) | Yes (one per instant) | -| Instance duration | Interval (`start_ts < end_ts`) | Full container window | Interval (`start_ts < end_ts`) | Zero (`start_ts == end_ts`) | -| Use case | Signal-based conditions, operating bands, distance bins | Full-run aggregations, container-level statistics | State transitions and multi-step patterns where consecutive states overlap | Edge/instant events, e.g. `rising_edges()` / `falling_edges()` | +| Criterion | BasicEvent | ContainerEvent | SequenceOfEvents | PointsInTimeEvent | TimeWindowEvent | +|----------------------------------|---------------------------------------------------------|---------------------------------------------------|---------------------------------------------------------------------------|------------------------------------------------------------|-------------------------------------------------------------| +| Requires a TSAL expression | Yes (one) | No | Yes (ordered list) | Yes (one, must evaluate to `PointsInTime`) | No (needs a `window_length`) | +| Multiple instances per container | Yes (one per matching interval) | No (always one per container) | Yes (one per joined sequence) | Yes (one per instant) | Yes (one per fixed window) | +| Instance duration | Interval (`start_ts < end_ts`) | Full container window | Interval (`start_ts < end_ts`) | Zero (`start_ts == end_ts`) | Fixed window (last clamped to container end) | +| Use case | Signal-based conditions, operating bands, distance bins | Full-run aggregations, container-level statistics | State transitions and multi-step patterns where consecutive states overlap | Edge/instant events, e.g. `rising_edges()` / `falling_edges()` | Repeated time segments (1-min / 10-min / hourly / daily) | diff --git a/src/impulse_query_engine/analyze/query/events/__init__.py b/src/impulse_query_engine/analyze/query/events/__init__.py index 50a1ce48..4dcbc6bc 100644 --- a/src/impulse_query_engine/analyze/query/events/__init__.py +++ b/src/impulse_query_engine/analyze/query/events/__init__.py @@ -1 +1,4 @@ from .sequence_of_events_expression import SequenceOfEventsExpression +from .tumbling_windows_expression import TumblingWindowsExpression + +__all__ = ["SequenceOfEventsExpression", "TumblingWindowsExpression"] diff --git a/src/impulse_query_engine/analyze/query/events/tumbling_windows_expression.py b/src/impulse_query_engine/analyze/query/events/tumbling_windows_expression.py new file mode 100644 index 00000000..b576e94b --- /dev/null +++ b/src/impulse_query_engine/analyze/query/events/tumbling_windows_expression.py @@ -0,0 +1,179 @@ +from __future__ import annotations + +import numpy as np + +from impulse_query_engine.analyze.metadata.tag_expression import TagExpression +from impulse_query_engine.analyze.metadata.time_series_expression import ( + TimeSeriesExpression, + TimeSeriesSelector, +) +from impulse_query_engine.analyze.query.solvers.series_cache import SeriesCache +from impulse_query_engine.model.series.intervals import Intervals + +# Internal (post-``column_name_mapping``) container-metric column names for the +# measurement start/stop timestamps. These mirror ``SolverConfig.start_ts_col`` / +# ``SolverConfig.stop_ts_col`` (the same source ``ContainerEvent`` relies on) and are the +# keys under which the solve exposes them via ``SeriesCache.container_metrics``. +_START_TS_COL = "start_ts" +_STOP_TS_COL = "stop_ts" + + +class TumblingWindowsExpression(TimeSeriesExpression): + """Produce consecutive fixed-duration windows spanning a measurement container. + + The windows are derived purely from the container's ``start_ts`` / ``stop_ts`` metadata + (no channel data), so the expression declares no selectors and instead requests those + container metrics via :meth:`required_container_metrics`. Windows tile + ``[start_ts, stop_ts]`` with a fixed length ``window_length`` (expressed in the same time + unit as the underlying timestamps); the final window is clamped to ``stop_ts`` when the + last full window would overrun it. + + Visual timeline (window_length = W):: + + time ---> + container: | ------------------------------- | + windows: | --W-- | --W-- | --W-- | -rest- | + + This is the query-engine counterpart of the reporting ``TimeWindowEvent``. It evaluates + to :class:`Intervals`, so it can scope a ``StatsAggregator`` (one statistic per window). + """ + + def __init__(self, window_length: float): + """ + Initialize a TumblingWindowsExpression. + + Parameters + ---------- + window_length : float + Fixed window length, in the same time unit as the underlying timestamps + (e.g. milliseconds-since-epoch). Must be strictly positive. + + Raises + ------ + ValueError + If ``window_length`` is not strictly positive. + """ + if window_length is None or window_length <= 0: + raise ValueError( + f"TumblingWindowsExpression requires a strictly positive window_length, " + f"got {window_length!r}." + ) + self.window_length = window_length + TimeSeriesExpression.__init__(self, is_single_signal=False) + + def __str__(self) -> str: + """ + Return a string representation of the TumblingWindowsExpression. + + The ``window_length`` is included so it flows into the event's definition hash. + + Returns + ------- + str + String representation of the object. + """ + return f"TumblingWindowsExpression" + + def dtype(self): + """ + Return the Spark data type of the result. + + Returns + ------- + pyspark.sql.types.ArrayType + Same dtype as Intervals: ArrayType(ArrayType(DoubleType())). + """ + return Intervals.empty().dtype() + + def get_required_tag_exprs(self) -> set[TagExpression]: + """ + Return required tag expressions (none: windows use container metrics only). + + Returns + ------- + set of TagExpression + """ + return set() + + def required_tags(self) -> set[str]: + """ + Return required tags (none). + + Returns + ------- + set of str + """ + return set() + + def required_container_tags(self) -> set[str]: + """ + Return required container tags (none). + + Returns + ------- + set of str + """ + return set() + + def required_container_metrics(self) -> set[str]: + """ + Return the container-metric columns needed to bound the windows. + + Returns + ------- + set of str + The measurement start/stop timestamp columns. + """ + return {_START_TS_COL, _STOP_TS_COL} + + def get_selectors(self) -> list[TimeSeriesSelector]: + """ + Return channel selectors (none: windows depend on no channel data). + + Returns + ------- + list of TimeSeriesSelector + """ + return [] + + def get_selector_expr(self): + """ + Return the combined selector expression (none). + + Returns + ------- + None + """ + return None + + def build(self, cache: SeriesCache) -> Intervals: + """ + Build the tumbling windows spanning the container. + + Parameters + ---------- + cache : SeriesCache + Cache exposing the requested container metrics via ``container_metrics``. + + Returns + ------- + Intervals + Consecutive fixed-length windows over ``[start_ts, stop_ts]``, with the final + window clamped to ``stop_ts``. Empty when the container boundaries are absent + (e.g. the empty cache used for type validation) or non-positive in span. + """ + start_ts = cache.container_metrics.get(_START_TS_COL) + stop_ts = cache.container_metrics.get(_STOP_TS_COL) + + if start_ts is None or stop_ts is None or stop_ts <= start_ts: + return Intervals.empty() + + # Number of windows covering the span; the last one is clamped to stop_ts below. + window_count = int(np.ceil((stop_ts - start_ts) / self.window_length)) + if window_count <= 0: + return Intervals.empty() + + indices = np.arange(window_count) + starts = start_ts + indices * self.window_length + ends = np.minimum(start_ts + (indices + 1) * self.window_length, stop_ts) + return Intervals(starts, ends, del_last_empty=True) diff --git a/src/impulse_reporting/events/event_types.py b/src/impulse_reporting/events/event_types.py index 5b877096..fa81c71f 100644 --- a/src/impulse_reporting/events/event_types.py +++ b/src/impulse_reporting/events/event_types.py @@ -6,6 +6,7 @@ from impulse_reporting.events.container_event import ContainerEvent from impulse_reporting.events.points_in_time_event import PointsInTimeEvent from impulse_reporting.events.sequence_of_events import SequenceOfEvents +from impulse_reporting.events.time_window_event import TimeWindowEvent from impulse_reporting.persist.dimension_schema import EVENT_DIMENSION_SCHEMA from impulse_reporting.persist.fact_schema import EVENT_INSTANCE_FACT_SCHEMA @@ -25,6 +26,8 @@ class EventType(Enum): Container event type spanning the full measurement container. SEQUENCE_OF_EVENTS : SequenceOfEvents Sequence-of-events type for ordered interval sequence detection. + TIME_WINDOW_EVENT : TimeWindowEvent + Fixed-duration tumbling-window type; one instance per window across each container. """ @@ -32,6 +35,7 @@ class EventType(Enum): CONTAINER_EVENT = ContainerEvent SEQUENCE_OF_EVENTS = SequenceOfEvents POINTS_IN_TIME_EVENT = PointsInTimeEvent + TIME_WINDOW_EVENT = TimeWindowEvent def get_fact_table_name(self) -> str: """ @@ -53,6 +57,7 @@ def get_fact_table_name(self) -> str: | EventType.CONTAINER_EVENT | EventType.SEQUENCE_OF_EVENTS | EventType.POINTS_IN_TIME_EVENT + | EventType.TIME_WINDOW_EVENT ): return "event_instance_fact" case _: @@ -78,6 +83,7 @@ def get_fact_schema(self) -> StructType: | EventType.CONTAINER_EVENT | EventType.SEQUENCE_OF_EVENTS | EventType.POINTS_IN_TIME_EVENT + | EventType.TIME_WINDOW_EVENT ): return EVENT_INSTANCE_FACT_SCHEMA case _: @@ -103,6 +109,7 @@ def get_dimension_table_name(self) -> str: | EventType.CONTAINER_EVENT | EventType.SEQUENCE_OF_EVENTS | EventType.POINTS_IN_TIME_EVENT + | EventType.TIME_WINDOW_EVENT ): return "event_dimension" case _: @@ -126,6 +133,7 @@ def get_dimension_schema(self) -> StructType: | EventType.CONTAINER_EVENT | EventType.SEQUENCE_OF_EVENTS | EventType.POINTS_IN_TIME_EVENT + | EventType.TIME_WINDOW_EVENT ): return EVENT_DIMENSION_SCHEMA case _: diff --git a/src/impulse_reporting/events/time_window_event.py b/src/impulse_reporting/events/time_window_event.py new file mode 100644 index 00000000..3aadae2f --- /dev/null +++ b/src/impulse_reporting/events/time_window_event.py @@ -0,0 +1,270 @@ +"""TimeWindowEvent — splits each container into consecutive fixed-duration windows.""" + +from __future__ import annotations + +import hashlib +from collections.abc import Mapping + +import pyspark.sql.functions as f +import zlib +from pyspark.sql import DataFrame, Row, SparkSession + +from impulse_query_engine.analyze.metadata.time_series_expression import ( + TimeSeriesExpression, +) +from impulse_query_engine.analyze.query.events.tumbling_windows_expression import ( + TumblingWindowsExpression, +) +from impulse_query_engine.analyze.query.query_builder import QueryBuilder +from impulse_query_engine.analyze.query.solvers.query_solver import QuerySolver +from impulse_query_engine.model.series.intervals import Intervals +from impulse_reporting.events.event import Event +from impulse_reporting.persist.dimension_schema import EVENT_DIMENSION_SCHEMA +from impulse_reporting.persist.fact_schema import EVENT_INSTANCE_FACT_SCHEMA +from impulse_reporting.util.event_instance_util import generate_event_instance_id_column +from impulse_reporting.util.report_entity_util import ReportEntityUtil + + +class TimeWindowEvent(Event): + """Event that divides each measurement container into consecutive fixed windows. + + Unlike ``ContainerEvent`` (one instance per container), a ``TimeWindowEvent`` emits one + event instance per fixed-duration slice, tiling the container's ``start_ts`` / ``stop_ts`` + span with windows of length ``window_length``. The final slice is clamped to the + container end. Boundaries come from a :class:`TumblingWindowsExpression`, so the event + fact and any aggregation scoped to this event share the same solved windows and their + ``event_instance_id`` values match by construction. + """ + + def __init__( + self, + name: str, + window_length: float, + desc: str = None, + required_channels: list[str] = None, + attributes: Mapping[str, str] = None, + ): + """ + Initialize a TimeWindowEvent object. + + Parameters + ---------- + name : str + Name of the event. + window_length : float + Fixed window length, in the same time unit as the underlying timestamps + (e.g. milliseconds-since-epoch). Must be strictly positive. + desc : str, optional + Description of the event. + required_channels : list of str, optional + List of required channels for the event. Informational; stored in the event + dimension table. + attributes : Mapping[str, str], optional + Key-value metadata for the event. ``window_length`` is surfaced here + automatically (without overriding a user-supplied key). + + Raises + ------ + ValueError + If ``window_length`` is not strictly positive. + """ + Event.__init__(self, name) + if window_length is None or window_length <= 0: + raise ValueError( + f"TimeWindowEvent requires a strictly positive window_length, " + f"got {window_length!r}." + ) + self.window_length = window_length + self.expression = TumblingWindowsExpression(window_length).alias(name) + self.expression.require_evaluation_type( + Intervals, owner="TimeWindowEvent", example="window_length=60000" + ) + self.description = desc + self.required_channels = required_channels + normalized_attributes: dict[str, str] = {} + if attributes is not None: + normalized_attributes = {str(k): str(v) for k, v in attributes.items()} + # Surface the window length for traceability in event_dimension, without + # clobbering an explicit user-supplied attribute of the same key. + normalized_attributes.setdefault("window_length", str(window_length)) + self.attributes = normalized_attributes + + def get_id(self) -> int: + """ + Returns a unique identifier for the event. + + Returns + ------- + int + Unique positive 32-bit integer identifier for the event. + """ + hash_input = f"{self.name}" + return zlib.crc32(hash_input.encode()) & 0x7FFFFFFF # Ensures positive 32-bit int + + def get_expression(self) -> TimeSeriesExpression | None: + """ + Get the time series expression associated with the event. + + Returns + ------- + TimeSeriesExpression or None + The tumbling-windows expression for the event. + """ + return self.expression + + def get_event_type_str(self) -> str: + """Get the event type string for TimeWindowEvent. + + Returns + ------- + str + Event type string. + """ + return "TIME_WINDOW_EVENT" + + def determine_definition_hash(self) -> int: + """ + Calculate definition hash for the time-window event. + + Only includes the expression string (which encodes ``window_length``), the sole + attribute that affects the event results, so resizing the window forces a full + recompute in incremental mode. + + Excludes: name, description, required_channels, report_id + + Returns + ------- + int + Hash value representing the computation definition. + """ + hash_input = self.get_expression_str() + + # Use SHA-256 and return as int (truncated to fit LongType) + hash_bytes = hashlib.sha256(hash_input.encode()).digest() + return int.from_bytes(hash_bytes[:8], byteorder="big", signed=True) + + def as_dict(self) -> dict: + """ + Get a dictionary representation of the event. + + Returns + ------- + dict + Dictionary containing event metadata. + """ + return { + "event_id": self.get_id(), + "report_id": self.report_id, + "event_type": self.get_event_type_str(), + "event_name": self.name, + "event_description": self.description, + "required_channels": self.required_channels, + "event_expression": self.get_expression_str(), + "definition_hash": self.determine_definition_hash(), + "attributes": self.attributes, + } + + def as_spark_row(self) -> Row: + """ + Get a Spark Row representation of the event. + + Returns + ------- + Row + Spark Row containing event metadata. + """ + return Row(**self.as_dict()) + + @classmethod + def determine_events( + cls, + spark: SparkSession, + events: list[TimeWindowEvent], + *, + solved_df: DataFrame = None, + query: QueryBuilder = None, + solver: QuerySolver = None, + pre_filtered_containers_df=None, + ): + """ + Extract the event fact table for the given list of TimeWindowEvent objects. + + Each window becomes one event instance (``start_ts < end_ts``). The window intervals + are read from the centralized solve (the same column consumed by scoped aggregations), + so the resulting ``event_instance_id`` values match on both sides. + + Parameters + ---------- + spark : SparkSession + Spark session for data processing. + events : list of TimeWindowEvent + List of TimeWindowEvent objects to process. + solved_df : DataFrame, optional + Pre-solved wide DataFrame from centralized batch solve. Required. + query : QueryBuilder, optional + Query builder (unused, kept for interface compatibility). + solver : QuerySolver, optional + Query solver (unused, kept for interface compatibility). + pre_filtered_containers_df : DataFrame, optional + Pre-filtered containers for incremental processing. + + Returns + ------- + DataFrame + Spark DataFrame containing event instance facts. + """ + if solved_df is None: + raise ValueError( + "TimeWindowEvent.determine_events requires solved_df. " + "Provide a pre-solved DataFrame from the centralized batch-solve flow." + ) + + event_names = [event.get_name() for event in events] + + df = ( + solved_df.select("container_id", *event_names) + .unpivot( + f.col("container_id"), + event_names, + variableColumnName="event_name", + valueColumnName="value", + ) + .select( + "container_id", + "event_name", + f.explode(f.col("value")).alias("event_instance"), + ) + .withColumn("start_ts", f.col("event_instance").getItem(0)) + .withColumn("end_ts", f.col("event_instance").getItem(1)) + .withColumn( + "event_instance_id", + generate_event_instance_id_column(event_type=TimeWindowEvent), + ) + .withColumn( + "event_id", + ReportEntityUtil.get_event_id_column(elements=events, element_name="event_name"), + ) + .select(EVENT_INSTANCE_FACT_SCHEMA.fieldNames()) + .where(f.col("start_ts") < f.col("end_ts")) # Ensure valid time intervals + ) + return df + + @classmethod + def determine_metadata_df(cls, spark: SparkSession, events: list[TimeWindowEvent]): + """ + Create a Spark DataFrame containing event metadata. + + Parameters + ---------- + spark : SparkSession + Spark session for data processing. + events : list of TimeWindowEvent + List of TimeWindowEvent objects. + + Returns + ------- + DataFrame + Spark DataFrame containing event metadata. + """ + events = [event.as_spark_row() for event in events] + return spark.createDataFrame(events, schema=EVENT_DIMENSION_SCHEMA) diff --git a/tests/impulse_query_engine/unit/analyze/query/events/tumbling_windows_expression_test.py b/tests/impulse_query_engine/unit/analyze/query/events/tumbling_windows_expression_test.py new file mode 100644 index 00000000..5c73d99b --- /dev/null +++ b/tests/impulse_query_engine/unit/analyze/query/events/tumbling_windows_expression_test.py @@ -0,0 +1,109 @@ +from __future__ import annotations + +import numpy as np +import pytest + +from impulse_query_engine.analyze.query.events import TumblingWindowsExpression +from impulse_query_engine.analyze.query.solvers.empty_cache import EmptyTimeSeriesCache +from impulse_query_engine.model.series.intervals import Intervals + + +class _FakeCache: + """Minimal SeriesCache stand-in exposing container metrics for ``build``.""" + + def __init__(self, container_metrics: dict): + self._container_metrics = container_metrics + + @property + def container_metrics(self) -> dict: + return self._container_metrics + + @property + def container_tags(self) -> dict: + return {} + + +def _build(start_ts, stop_ts, window_length) -> Intervals: + expr = TumblingWindowsExpression(window_length) + return expr.build(_FakeCache({"start_ts": start_ts, "stop_ts": stop_ts})) + + +def test_exact_multiple_windows_last_ends_at_stop(): + """D=100, W=10 -> 10 contiguous windows, final window ends exactly at stop_ts.""" + iv = _build(0, 100, 10) + assert len(iv) == 10 + assert iv.tstarts.tolist() == [0, 10, 20, 30, 40, 50, 60, 70, 80, 90] + assert iv.tends.tolist() == [10, 20, 30, 40, 50, 60, 70, 80, 90, 100] + # Contiguity: each window's end equals the next window's start. + assert iv.tstarts[1:].tolist() == iv.tends[:-1].tolist() + assert iv.tends[-1] == 100 + + +def test_non_multiple_has_short_final_window_clamped_to_stop(): + """D=105, W=10 -> 11 windows; the last is a short slice clamped to stop_ts.""" + iv = _build(0, 105, 10) + assert len(iv) == 11 + assert iv.tstarts[-1] == 100 + assert iv.tends[-1] == 105 # clamped, not 110 + # The final window is shorter than the fixed length. + assert (iv.tends[-1] - iv.tstarts[-1]) < 10 + assert iv.tstarts[1:].tolist() == iv.tends[:-1].tolist() + + +def test_span_equal_to_window_yields_single_window(): + iv = _build(1000, 1010, 10) + assert len(iv) == 1 + assert iv.tstarts.tolist() == [1000] + assert iv.tends.tolist() == [1010] + + +def test_span_smaller_than_window_yields_single_clamped_window(): + iv = _build(1000, 1001, 10) + assert len(iv) == 1 + assert iv.tstarts.tolist() == [1000] + assert iv.tends.tolist() == [1001] + + +def test_epoch_millisecond_boundaries(): + """Realistic epoch-ms container with a 10s (10000 ms) window.""" + start, stop = 1751528502708, 1751528610253 # ~107.545 s span + iv = _build(start, stop, 10000) + assert len(iv) == int(np.ceil((stop - start) / 10000)) # 11 + assert iv.tstarts[0] == start + assert iv.tends[-1] == stop + assert iv.tstarts[1:].tolist() == iv.tends[:-1].tolist() + assert bool(np.all(iv.tstarts < iv.tends)) + + +def test_degenerate_container_yields_no_windows(): + assert len(_build(50, 50, 10)) == 0 # stop == start + assert len(_build(60, 50, 10)) == 0 # stop < start + + +def test_missing_container_metrics_yields_empty(): + expr = TumblingWindowsExpression(10) + assert len(expr.build(_FakeCache({}))) == 0 + # The empty cache used by evaluation_type() has no container metrics. + assert len(expr.build(EmptyTimeSeriesCache())) == 0 + + +def test_evaluation_type_is_intervals(): + assert TumblingWindowsExpression(10).evaluation_type() is Intervals + + +def test_no_selectors_and_requests_container_metrics(): + expr = TumblingWindowsExpression(10) + assert expr.get_selectors() == [] + assert expr.get_selector_expr() is None + assert expr.required_container_metrics() == {"start_ts", "stop_ts"} + assert expr.required_tags() == set() + + +def test_str_includes_window_length(): + assert "window_length=10" in str(TumblingWindowsExpression(10)) + + +@pytest.mark.parametrize("bad", [0, -1, -10.5, None]) +def test_non_positive_window_length_raises(bad): + with pytest.raises(ValueError, match="strictly positive"): + TumblingWindowsExpression(bad) diff --git a/tests/impulse_reporting/integration/time_window_event_test.py b/tests/impulse_reporting/integration/time_window_event_test.py new file mode 100644 index 00000000..023f5c18 --- /dev/null +++ b/tests/impulse_reporting/integration/time_window_event_test.py @@ -0,0 +1,305 @@ +"""Integration tests for TimeWindowEvent with end-to-end Report usage.""" + +from unittest.mock import create_autospec + +import pyspark.sql.functions as F +import pytest +from databricks.sdk import WorkspaceClient + +from impulse_reporting.aggregations.stats_aggregator import StatsAggregator +from impulse_reporting.config.config_parser import ( + Comparator, + ContainerFilters, + ImpulseConfig, + MetricFilter, + QueryEngine, + Solvers, + Source, + UnitySink, +) +from impulse_reporting.core.page import Page +from impulse_reporting.core.report import Report +from impulse_reporting.events.time_window_event import TimeWindowEvent +from tests.conftest import setup_basic_db, spark # noqa: F401 (pytest fixtures) + +# Container boundaries (epoch ms) for the Seat_Leon measurements in +# container_metrics.csv, as documented in container_event_test.py. +# c1: span 107545 ms, c2: 108752 ms, c3: 110083 ms +EXPECTED_CONTAINERS = { + 1: {"start_ts": 1751528502708, "stop_ts": 1751528610253}, + 2: {"start_ts": 1751528501483, "stop_ts": 1751528610235}, + 3: {"start_ts": 1751528500169, "stop_ts": 1751528610252}, +} +WINDOW_LENGTH = 10_000 # 10 seconds, in epoch-ms units + + +def _config(table_prefix: str) -> ImpulseConfig: + return ImpulseConfig( + source=Source( + container_metrics_table="spark_catalog.silver.container_metrics", + channel_metrics_table="spark_catalog.silver.channel_metrics", + channels_uri="spark_catalog.silver.channels", + ), + unity_sink=UnitySink( + catalog="spark_catalog", + schema="gold", + table_prefix=table_prefix, + ), + container_filters=ContainerFilters( + metric_filters=[ + [ + MetricFilter( + column_name="vehicle_key", comparator=Comparator.EQ, value="Seat_Leon" + ), + MetricFilter( + column_name="start_dt", + comparator=Comparator.GE, + value="2025-07-03T07:00:00.000Z", + ), + ] + ] + ), + query_engine=QueryEngine(solver=Solvers.KEY_VALUE_STORE_SOLVER), + measurement_dimensions=["container_id", "start_ts", "stop_ts"], + ) + + +def _expected_window_count(container_id: int) -> int: + span = ( + EXPECTED_CONTAINERS[container_id]["stop_ts"] + - EXPECTED_CONTAINERS[container_id]["start_ts"] + ) + return -(-span // WINDOW_LENGTH) # ceil division + + +def test_time_window_event_in_report(spark, basic_narrow_db): + """A TimeWindowEvent (co-solved with an aggregation) tiles each container into windows.""" + my_report = Report( + name="time_window_event_report", + spark=spark, + workspace_client=create_autospec(WorkspaceClient), + config=dict(_config("time_window_event_test")), + ) + + window_evt = TimeWindowEvent( + name="ten_sec", window_length=WINDOW_LENGTH, desc="Ten second windows" + ) + my_report.add_event(window_evt) + + # A channel-bearing aggregation must be present so the batched solve forms + # per-container groups the (selector-less) window expression can ride on. + query = my_report.get_db().query + page = Page(page_number=1) + my_report.add_page(page) + page.add_aggregation( + StatsAggregator( + name="rpm_stats_per_window", + input_expressions=[query.channel(channel_name="Engine RPM")], + channel_names=["Engine RPM"], + statistics=["min", "max", "mean"], + event=window_evt, + desc="Engine RPM stats per window", + ) + ) + + my_report.determine_report() + + event_dfs = my_report.event_dfs + assert "TIME_WINDOW_EVENT" in event_dfs + rows = event_dfs["TIME_WINDOW_EVENT"]["changed"].collect() + + total_expected = sum(_expected_window_count(cid) for cid in EXPECTED_CONTAINERS) + assert len(rows) == total_expected + + for container_id, expected in EXPECTED_CONTAINERS.items(): + windows = sorted( + ((r.start_ts, r.end_ts) for r in rows if r.container_id == container_id), + key=lambda w: w[0], + ) + assert len(windows) == _expected_window_count(container_id) + # First window starts at the container start. + assert windows[0][0] == expected["start_ts"] + # Windows are contiguous: each end equals the next start. + for (_, end), (nxt_start, _) in zip(windows, windows[1:], strict=False): + assert end == nxt_start + # Final window is clamped to the container stop. + assert windows[-1][1] == expected["stop_ts"] + # Every window is a valid, non-empty interval. + assert all(start < end for start, end in windows) + # Per-window instances are distinct (unlike ContainerEvent's single id). + instance_ids = [r.event_instance_id for r in rows if r.container_id == container_id] + assert len(set(instance_ids)) == len(instance_ids) + + dim_rows = my_report.event_metadata_dfs["TIME_WINDOW_EVENT"].collect() + assert len(dim_rows) == 1 + assert dim_rows[0].event_type == "TIME_WINDOW_EVENT" + assert dim_rows[0].attributes["window_length"] == str(WINDOW_LENGTH) + + +# Window length (channel-sample time unit, µs) for the aligned-boundary test. +# Per-container sample spans are ~3.9–5.4e9 µs, so 600s (=6e8 µs) yields several +# windows per container, each overlapping RPM samples. +ALIGNED_WINDOW_LENGTH = 600_000_000 +_ALIGNED_SCHEMA = "spark_catalog.silver_tw_aligned" + + +@pytest.fixture +def setup_tw_aligned_db(spark, setup_basic_db): # noqa: F811 + """Silver tables cloned from the basic db with container_metrics start_ts / stop_ts + recomputed from each container's channel-sample range, so the container boundaries + (and thus the windows) share the samples' time base.""" + spark.sql(f"CREATE SCHEMA IF NOT EXISTS {_ALIGNED_SCHEMA}") + channels = spark.read.table("spark_catalog.silver.channels") + bounds = channels.groupBy("container_id").agg( + F.min("tstart").alias("_agg_start"), + F.max("tend").alias("_agg_stop"), + ) + container_metrics = spark.read.table("spark_catalog.silver.container_metrics") + start_type = container_metrics.schema["start_ts"].dataType + stop_type = container_metrics.schema["stop_ts"].dataType + aligned_cm = ( + container_metrics.join(bounds, on="container_id", how="left") + .withColumn("start_ts", F.coalesce("_agg_start", "start_ts").cast(start_type)) + .withColumn("stop_ts", F.coalesce("_agg_stop", "stop_ts").cast(stop_type)) + .drop("_agg_start", "_agg_stop") + ) + aligned_cm.write.format("delta").mode("overwrite").option( + "overwriteSchema", "true" + ).saveAsTable(f"{_ALIGNED_SCHEMA}.container_metrics") + for table in ("channel_metrics", "channels"): + spark.read.table(f"spark_catalog.silver.{table}").write.format("delta").mode( + "overwrite" + ).saveAsTable(f"{_ALIGNED_SCHEMA}.{table}") + yield + spark.sql(f"DROP SCHEMA IF EXISTS {_ALIGNED_SCHEMA} CASCADE") + + +def test_time_window_event_aggregation_join(spark, setup_tw_aligned_db): + """Stats scoped to a TimeWindowEvent yield per-window values that join to the fact.""" + config = dict( + ImpulseConfig( + source=Source( + container_metrics_table=f"{_ALIGNED_SCHEMA}.container_metrics", + channel_metrics_table=f"{_ALIGNED_SCHEMA}.channel_metrics", + channels_uri=f"{_ALIGNED_SCHEMA}.channels", + ), + unity_sink=UnitySink( + catalog="spark_catalog", schema="gold", table_prefix="time_window_join_test" + ), + container_filters=ContainerFilters( + metric_filters=[ + [ + MetricFilter( + column_name="vehicle_key", comparator=Comparator.EQ, value="Seat_Leon" + ) + ] + ] + ), + query_engine=QueryEngine(solver=Solvers.KEY_VALUE_STORE_SOLVER), + measurement_dimensions=["container_id", "start_ts", "stop_ts"], + ) + ) + my_report = Report( + name="time_window_join_report", + spark=spark, + workspace_client=create_autospec(WorkspaceClient), + config=config, + ) + + window_evt = TimeWindowEvent(name="ten_min", window_length=ALIGNED_WINDOW_LENGTH) + my_report.add_event(window_evt) + + query = my_report.get_db().query + page = Page(page_number=1) + my_report.add_page(page) + page.add_aggregation( + StatsAggregator( + name="rpm_stats_per_window", + input_expressions=[query.channel(channel_name="Engine RPM")], + channel_names=["Engine RPM"], + statistics=["min", "max", "mean"], + event=window_evt, + desc="Engine RPM stats per window", + ) + ) + + my_report.determine_report() + my_report.persist_results() + + stats_fact = spark.read.table("spark_catalog.gold.time_window_join_test_stats_aggregator_fact") + event_instance_fact = spark.read.table( + "spark_catalog.gold.time_window_join_test_event_instance_fact" + ) + assert stats_fact.count() > 0 + assert event_instance_fact.count() > 0 + + # Real computed values: with aligned boundaries every window overlaps RPM samples, + # so each produces a positive max. + max_values = [ + r.statistic_value + for r in stats_fact.filter(F.col("aggregation_label") == "max").collect() + if r.statistic_value is not None + ] + assert len(max_values) > 0 + assert any(v > 0 for v in max_values) + + stats_event_ids = { + r.event_instance_id + for r in stats_fact.filter(F.col("event_instance_id").isNotNull()) + .select("event_instance_id") + .distinct() + .collect() + } + event_ids = { + r.event_instance_id + for r in event_instance_fact.select("event_instance_id").distinct().collect() + } + assert len(stats_event_ids) > 0 + # Every per-window stats instance must map to a materialized window instance. + assert stats_event_ids.issubset( + event_ids + ), f"stats event_instance_ids not in event_instance_fact: {stats_event_ids - event_ids}" + + +def test_multiple_time_window_events_coexist(spark, basic_narrow_db): + """Two TimeWindowEvents with different windows are allowed and both materialize.""" + my_report = Report( + name="time_window_multi_report", + spark=spark, + workspace_client=create_autospec(WorkspaceClient), + config=dict(_config("time_window_multi_test")), + ) + + evt_10s = TimeWindowEvent(name="ten_sec", window_length=WINDOW_LENGTH) + evt_30s = TimeWindowEvent(name="thirty_sec", window_length=3 * WINDOW_LENGTH) + my_report.add_event(evt_10s) + my_report.add_event(evt_30s) + + query = my_report.get_db().query + page = Page(page_number=1) + my_report.add_page(page) + page.add_aggregation( + StatsAggregator( + name="rpm_stats", + input_expressions=[query.channel(channel_name="Engine RPM")], + channel_names=["Engine RPM"], + statistics=["mean"], + event=evt_10s, + desc="Engine RPM stats per 10s window", + ) + ) + + my_report.determine_report() + + rows = my_report.event_dfs["TIME_WINDOW_EVENT"]["changed"].collect() + names = {r.event_id for r in rows} + # Two distinct events (distinct event_ids) share the shared fact table. + assert names == {evt_10s.get_id(), evt_30s.get_id()} + + # The 10s event produces strictly more windows than the 30s event. + count_10s = sum(1 for r in rows if r.event_id == evt_10s.get_id()) + count_30s = sum(1 for r in rows if r.event_id == evt_30s.get_id()) + assert count_10s > count_30s > 0 + + dim_rows = my_report.event_metadata_dfs["TIME_WINDOW_EVENT"].collect() + assert {d.event_name for d in dim_rows} == {"ten_sec", "thirty_sec"} diff --git a/tests/impulse_reporting/unit/events/time_window_event_test.py b/tests/impulse_reporting/unit/events/time_window_event_test.py new file mode 100644 index 00000000..97d6835d --- /dev/null +++ b/tests/impulse_reporting/unit/events/time_window_event_test.py @@ -0,0 +1,82 @@ +"""Unit tests for TimeWindowEvent.""" + +import pytest + +from impulse_query_engine.analyze.query.events.tumbling_windows_expression import ( + TumblingWindowsExpression, +) +from impulse_reporting.events.time_window_event import TimeWindowEvent + + +# --------------------------------------------------------------------------- +# Constructor / basic attributes +# --------------------------------------------------------------------------- +def test_init(): + event = TimeWindowEvent(name="w10", window_length=10000) + assert event.name == "w10" + assert event.window_length == 10000 + assert event.description is None + assert isinstance(event.get_expression(), TumblingWindowsExpression) + + +def test_init_surfaces_window_length_attribute(): + event = TimeWindowEvent(name="w10", window_length=10000) + assert event.attributes["window_length"] == "10000" + + +def test_init_does_not_override_user_window_length_attribute(): + event = TimeWindowEvent( + name="w10", window_length=10000, attributes={"window_length": "custom"} + ) + assert event.attributes["window_length"] == "custom" + + +@pytest.mark.parametrize("bad", [0, -1, -5.5, None]) +def test_non_positive_window_length_raises(bad): + with pytest.raises(ValueError, match="strictly positive"): + TimeWindowEvent(name="bad", window_length=bad) + + +# --------------------------------------------------------------------------- +# get_id / type string +# --------------------------------------------------------------------------- +def test_get_id_is_positive_int_and_deterministic(): + a = TimeWindowEvent(name="same", window_length=10) + b = TimeWindowEvent(name="same", window_length=99) # id keys on name only + assert isinstance(a.get_id(), int) and a.get_id() > 0 + assert a.get_id() == b.get_id() + + +def test_event_type_str(): + assert TimeWindowEvent(name="w", window_length=10).get_event_type_str() == "TIME_WINDOW_EVENT" + + +# --------------------------------------------------------------------------- +# definition hash — must move with window_length, stable otherwise +# --------------------------------------------------------------------------- +def test_definition_hash_changes_with_window_length(): + a = TimeWindowEvent(name="w", window_length=10000) + b = TimeWindowEvent(name="w", window_length=60000) + assert a.determine_definition_hash() != b.determine_definition_hash() + + +def test_definition_hash_stable_across_desc_and_attributes(): + a = TimeWindowEvent(name="w", window_length=10000, desc="a", attributes={"k": "1"}) + b = TimeWindowEvent(name="w", window_length=10000, desc="b", attributes={"k": "2"}) + assert a.determine_definition_hash() == b.determine_definition_hash() + + +# --------------------------------------------------------------------------- +# metadata dict shape +# --------------------------------------------------------------------------- +def test_as_dict_shape(): + event = TimeWindowEvent( + name="w10", window_length=10000, desc="ten second windows", required_channels=["c1"] + ) + d = event.as_dict() + assert d["event_type"] == "TIME_WINDOW_EVENT" + assert d["event_name"] == "w10" + assert d["event_description"] == "ten second windows" + assert d["required_channels"] == ["c1"] + assert d["event_expression"] != "NA" + assert d["attributes"]["window_length"] == "10000" From 4c20abf170586ea57ff98fa93d5c6c87d7e858fe Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 30 Sep 2026 15:49:34 +0200 Subject: [PATCH 02/27] refactor(query-engine): reuse SolverConfig canonical column names in TumblingWindowsExpression Replace the locally-defined `_START_TS_COL` / `_STOP_TS_COL` literals in `TumblingWindowsExpression` with `SolverConfig.start_ts_col` / `stop_ts_col` from a default `SolverConfig` instance. This keeps the internal container-metric timestamp column names consistent with the rest of the query engine and avoids duplicating config-invariant literals. --- .../events/tumbling_windows_expression.py | 19 ++++++++++--------- 1 file changed, 10 insertions(+), 9 deletions(-) diff --git a/src/impulse_query_engine/analyze/query/events/tumbling_windows_expression.py b/src/impulse_query_engine/analyze/query/events/tumbling_windows_expression.py index b576e94b..328a86da 100644 --- a/src/impulse_query_engine/analyze/query/events/tumbling_windows_expression.py +++ b/src/impulse_query_engine/analyze/query/events/tumbling_windows_expression.py @@ -8,14 +8,15 @@ TimeSeriesSelector, ) from impulse_query_engine.analyze.query.solvers.series_cache import SeriesCache +from impulse_query_engine.analyze.query.solvers.solver_config import SolverConfig from impulse_query_engine.model.series.intervals import Intervals -# Internal (post-``column_name_mapping``) container-metric column names for the -# measurement start/stop timestamps. These mirror ``SolverConfig.start_ts_col`` / -# ``SolverConfig.stop_ts_col`` (the same source ``ContainerEvent`` relies on) and are the -# keys under which the solve exposes them via ``SeriesCache.container_metrics``. -_START_TS_COL = "start_ts" -_STOP_TS_COL = "stop_ts" +# Reuse SolverConfig's canonical internal (post-``column_name_mapping``) column names for +# the measurement start/stop timestamps rather than re-declaring the literals here. These +# are the keys under which the solve exposes them via ``SeriesCache.container_metrics``, and +# they are the same names ``ContainerEvent`` relies on. A default instance suffices since +# the names are config-invariant. +_SOLVER_CONFIG = SolverConfig() class TumblingWindowsExpression(TimeSeriesExpression): @@ -124,7 +125,7 @@ def required_container_metrics(self) -> set[str]: set of str The measurement start/stop timestamp columns. """ - return {_START_TS_COL, _STOP_TS_COL} + return {_SOLVER_CONFIG.start_ts_col, _SOLVER_CONFIG.stop_ts_col} def get_selectors(self) -> list[TimeSeriesSelector]: """ @@ -162,8 +163,8 @@ def build(self, cache: SeriesCache) -> Intervals: window clamped to ``stop_ts``. Empty when the container boundaries are absent (e.g. the empty cache used for type validation) or non-positive in span. """ - start_ts = cache.container_metrics.get(_START_TS_COL) - stop_ts = cache.container_metrics.get(_STOP_TS_COL) + start_ts = cache.container_metrics.get(_SOLVER_CONFIG.start_ts_col) + stop_ts = cache.container_metrics.get(_SOLVER_CONFIG.stop_ts_col) if start_ts is None or stop_ts is None or stop_ts <= start_ts: return Intervals.empty() From d75a1d0472911ce7faa7c29a1998d1a7c076a4e7 Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 30 Sep 2026 16:04:47 +0200 Subject: [PATCH 03/27] refactor(query-engine): rename TumblingWindowsExpression to TimeWindowExpression Rename the query-engine expression class and module from `TumblingWindowsExpression` to `TimeWindowExpression` to align with the reporting `TimeWindowEvent` naming. Update all imports, exports, docstrings, and tests accordingly. --- docs/impulse/docs/references/report/event.md | 2 +- .../analyze/query/events/__init__.py | 4 ++-- ...ows_expression.py => time_window_expression.py} | 12 ++++++------ src/impulse_reporting/events/event_types.py | 2 +- src/impulse_reporting/events/time_window_event.py | 10 +++++----- ...sion_test.py => time_window_expression_test.py} | 14 +++++++------- .../unit/events/time_window_event_test.py | 6 +++--- 7 files changed, 25 insertions(+), 25 deletions(-) rename src/impulse_query_engine/analyze/query/events/{tumbling_windows_expression.py => time_window_expression.py} (93%) rename tests/impulse_query_engine/unit/analyze/query/events/{tumbling_windows_expression_test.py => time_window_expression_test.py} (89%) diff --git a/docs/impulse/docs/references/report/event.md b/docs/impulse/docs/references/report/event.md index 4023e73c..82c8fb0b 100644 --- a/docs/impulse/docs/references/report/event.md +++ b/docs/impulse/docs/references/report/event.md @@ -233,7 +233,7 @@ seconds or any derived unit. So 60 one-minute windows over millisecond timestamp ### How it works -1. A tumbling-window expression reads `start_ts` and `stop_ts` from the `container_metrics` table +1. A time-window expression reads `start_ts` and `stop_ts` from the `container_metrics` table and tiles `[start_ts, stop_ts]` into consecutive windows of length `window_length`. 2. The **final window is clamped** to `stop_ts` when the last full window would overrun it; any zero-length trailing slice is dropped (every instance satisfies `start_ts < end_ts`). diff --git a/src/impulse_query_engine/analyze/query/events/__init__.py b/src/impulse_query_engine/analyze/query/events/__init__.py index 4dcbc6bc..ad5c45a9 100644 --- a/src/impulse_query_engine/analyze/query/events/__init__.py +++ b/src/impulse_query_engine/analyze/query/events/__init__.py @@ -1,4 +1,4 @@ from .sequence_of_events_expression import SequenceOfEventsExpression -from .tumbling_windows_expression import TumblingWindowsExpression +from .time_window_expression import TimeWindowExpression -__all__ = ["SequenceOfEventsExpression", "TumblingWindowsExpression"] +__all__ = ["SequenceOfEventsExpression", "TimeWindowExpression"] diff --git a/src/impulse_query_engine/analyze/query/events/tumbling_windows_expression.py b/src/impulse_query_engine/analyze/query/events/time_window_expression.py similarity index 93% rename from src/impulse_query_engine/analyze/query/events/tumbling_windows_expression.py rename to src/impulse_query_engine/analyze/query/events/time_window_expression.py index 328a86da..94875a53 100644 --- a/src/impulse_query_engine/analyze/query/events/tumbling_windows_expression.py +++ b/src/impulse_query_engine/analyze/query/events/time_window_expression.py @@ -19,7 +19,7 @@ _SOLVER_CONFIG = SolverConfig() -class TumblingWindowsExpression(TimeSeriesExpression): +class TimeWindowExpression(TimeSeriesExpression): """Produce consecutive fixed-duration windows spanning a measurement container. The windows are derived purely from the container's ``start_ts`` / ``stop_ts`` metadata @@ -41,7 +41,7 @@ class TumblingWindowsExpression(TimeSeriesExpression): def __init__(self, window_length: float): """ - Initialize a TumblingWindowsExpression. + Initialize a TimeWindowExpression. Parameters ---------- @@ -56,7 +56,7 @@ def __init__(self, window_length: float): """ if window_length is None or window_length <= 0: raise ValueError( - f"TumblingWindowsExpression requires a strictly positive window_length, " + f"TimeWindowExpression requires a strictly positive window_length, " f"got {window_length!r}." ) self.window_length = window_length @@ -64,7 +64,7 @@ def __init__(self, window_length: float): def __str__(self) -> str: """ - Return a string representation of the TumblingWindowsExpression. + Return a string representation of the TimeWindowExpression. The ``window_length`` is included so it flows into the event's definition hash. @@ -73,7 +73,7 @@ def __str__(self) -> str: str String representation of the object. """ - return f"TumblingWindowsExpression" + return f"TimeWindowExpression" def dtype(self): """ @@ -149,7 +149,7 @@ def get_selector_expr(self): def build(self, cache: SeriesCache) -> Intervals: """ - Build the tumbling windows spanning the container. + Build the fixed-duration windows spanning the container. Parameters ---------- diff --git a/src/impulse_reporting/events/event_types.py b/src/impulse_reporting/events/event_types.py index fa81c71f..608c350e 100644 --- a/src/impulse_reporting/events/event_types.py +++ b/src/impulse_reporting/events/event_types.py @@ -27,7 +27,7 @@ class EventType(Enum): SEQUENCE_OF_EVENTS : SequenceOfEvents Sequence-of-events type for ordered interval sequence detection. TIME_WINDOW_EVENT : TimeWindowEvent - Fixed-duration tumbling-window type; one instance per window across each container. + Fixed-duration time-window type; one instance per window across each container. """ diff --git a/src/impulse_reporting/events/time_window_event.py b/src/impulse_reporting/events/time_window_event.py index 3aadae2f..e60f53a0 100644 --- a/src/impulse_reporting/events/time_window_event.py +++ b/src/impulse_reporting/events/time_window_event.py @@ -12,8 +12,8 @@ from impulse_query_engine.analyze.metadata.time_series_expression import ( TimeSeriesExpression, ) -from impulse_query_engine.analyze.query.events.tumbling_windows_expression import ( - TumblingWindowsExpression, +from impulse_query_engine.analyze.query.events.time_window_expression import ( + TimeWindowExpression, ) from impulse_query_engine.analyze.query.query_builder import QueryBuilder from impulse_query_engine.analyze.query.solvers.query_solver import QuerySolver @@ -31,7 +31,7 @@ class TimeWindowEvent(Event): Unlike ``ContainerEvent`` (one instance per container), a ``TimeWindowEvent`` emits one event instance per fixed-duration slice, tiling the container's ``start_ts`` / ``stop_ts`` span with windows of length ``window_length``. The final slice is clamped to the - container end. Boundaries come from a :class:`TumblingWindowsExpression`, so the event + container end. Boundaries come from a :class:`TimeWindowExpression`, so the event fact and any aggregation scoped to this event share the same solved windows and their ``event_instance_id`` values match by construction. """ @@ -75,7 +75,7 @@ def __init__( f"got {window_length!r}." ) self.window_length = window_length - self.expression = TumblingWindowsExpression(window_length).alias(name) + self.expression = TimeWindowExpression(window_length).alias(name) self.expression.require_evaluation_type( Intervals, owner="TimeWindowEvent", example="window_length=60000" ) @@ -108,7 +108,7 @@ def get_expression(self) -> TimeSeriesExpression | None: Returns ------- TimeSeriesExpression or None - The tumbling-windows expression for the event. + The time-window expression for the event. """ return self.expression diff --git a/tests/impulse_query_engine/unit/analyze/query/events/tumbling_windows_expression_test.py b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py similarity index 89% rename from tests/impulse_query_engine/unit/analyze/query/events/tumbling_windows_expression_test.py rename to tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py index 5c73d99b..30e45416 100644 --- a/tests/impulse_query_engine/unit/analyze/query/events/tumbling_windows_expression_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py @@ -3,7 +3,7 @@ import numpy as np import pytest -from impulse_query_engine.analyze.query.events import TumblingWindowsExpression +from impulse_query_engine.analyze.query.events import TimeWindowExpression from impulse_query_engine.analyze.query.solvers.empty_cache import EmptyTimeSeriesCache from impulse_query_engine.model.series.intervals import Intervals @@ -24,7 +24,7 @@ def container_tags(self) -> dict: def _build(start_ts, stop_ts, window_length) -> Intervals: - expr = TumblingWindowsExpression(window_length) + expr = TimeWindowExpression(window_length) return expr.build(_FakeCache({"start_ts": start_ts, "stop_ts": stop_ts})) @@ -81,18 +81,18 @@ def test_degenerate_container_yields_no_windows(): def test_missing_container_metrics_yields_empty(): - expr = TumblingWindowsExpression(10) + expr = TimeWindowExpression(10) assert len(expr.build(_FakeCache({}))) == 0 # The empty cache used by evaluation_type() has no container metrics. assert len(expr.build(EmptyTimeSeriesCache())) == 0 def test_evaluation_type_is_intervals(): - assert TumblingWindowsExpression(10).evaluation_type() is Intervals + assert TimeWindowExpression(10).evaluation_type() is Intervals def test_no_selectors_and_requests_container_metrics(): - expr = TumblingWindowsExpression(10) + expr = TimeWindowExpression(10) assert expr.get_selectors() == [] assert expr.get_selector_expr() is None assert expr.required_container_metrics() == {"start_ts", "stop_ts"} @@ -100,10 +100,10 @@ def test_no_selectors_and_requests_container_metrics(): def test_str_includes_window_length(): - assert "window_length=10" in str(TumblingWindowsExpression(10)) + assert "window_length=10" in str(TimeWindowExpression(10)) @pytest.mark.parametrize("bad", [0, -1, -10.5, None]) def test_non_positive_window_length_raises(bad): with pytest.raises(ValueError, match="strictly positive"): - TumblingWindowsExpression(bad) + TimeWindowExpression(bad) diff --git a/tests/impulse_reporting/unit/events/time_window_event_test.py b/tests/impulse_reporting/unit/events/time_window_event_test.py index 97d6835d..dd960cc3 100644 --- a/tests/impulse_reporting/unit/events/time_window_event_test.py +++ b/tests/impulse_reporting/unit/events/time_window_event_test.py @@ -2,8 +2,8 @@ import pytest -from impulse_query_engine.analyze.query.events.tumbling_windows_expression import ( - TumblingWindowsExpression, +from impulse_query_engine.analyze.query.events.time_window_expression import ( + TimeWindowExpression, ) from impulse_reporting.events.time_window_event import TimeWindowEvent @@ -16,7 +16,7 @@ def test_init(): assert event.name == "w10" assert event.window_length == 10000 assert event.description is None - assert isinstance(event.get_expression(), TumblingWindowsExpression) + assert isinstance(event.get_expression(), TimeWindowExpression) def test_init_surfaces_window_length_attribute(): From e9ce8114e55556a4a89fb1b1087015a5713043cd Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 30 Sep 2026 16:14:29 +0200 Subject: [PATCH 04/27] docs(impulse): add TimeWindowEvent API reference and update event skills Generate the `time_window_event` API reference page and register it in the pydoc loader and API sidebar. Update the events skill README and SKILL.md to include `TimeWindowEvent` alongside the other event types. --- .../events/time_window_event.md | 179 ++++++++++++++++++ docs/impulse/docs/references/api/sidebar.json | 3 +- docs/impulse/pydoc-markdown.yml | 1 + skills/README.md | 2 +- skills/impulse-events/SKILL.md | 41 +++- 5 files changed, 220 insertions(+), 6 deletions(-) create mode 100644 docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md diff --git a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md new file mode 100644 index 00000000..d5155f64 --- /dev/null +++ b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md @@ -0,0 +1,179 @@ +--- +sidebar_label: time_window_event +title: impulse_reporting.events.time_window_event +--- + +TimeWindowEvent — splits each container into consecutive fixed-duration windows. + + +## TimeWindowEvent + +```python +class TimeWindowEvent(Event) +``` + +Event that divides each measurement container into consecutive fixed windows. + +Unlike ``ContainerEvent`` (one instance per container), a ``TimeWindowEvent`` emits one +event instance per fixed-duration slice, tiling the container's ``start_ts`` / ``stop_ts`` +span with windows of length ``window_length``. The final slice is clamped to the +container end. Boundaries come from a :class:`TimeWindowExpression`, so the event +fact and any aggregation scoped to this event share the same solved windows and their +``event_instance_id`` values match by construction. + + +#### \_\_init\_\_ + +```python +def __init__(name: str, + window_length: float, + desc: str = None, + required_channels: list[str] = None, + attributes: Mapping[str, str] = None) +``` + +Initialize a TimeWindowEvent object. + +**Arguments**: + +- `name` (`str`): Name of the event. +- `window_length` (`float`): Fixed window length, in the same time unit as the underlying timestamps +(e.g. milliseconds-since-epoch). Must be strictly positive. +- `desc` (`str`): Description of the event. +- `required_channels` (`list of str`): List of required channels for the event. Informational; stored in the event +dimension table. +- `attributes` (`Mapping[str, str]`): Key-value metadata for the event. ``window_length`` is surfaced here +automatically (without overriding a user-supplied key). + +**Raises**: + +- `ValueError`: If ``window_length`` is not strictly positive. + +#### get\_id + +```python +def get_id() -> int +``` + +Returns a unique identifier for the event. + +**Returns**: + +`int`: Unique positive 32-bit integer identifier for the event. + +#### get\_expression + +```python +def get_expression() -> TimeSeriesExpression | None +``` + +Get the time series expression associated with the event. + +**Returns**: + +`TimeSeriesExpression or None`: The time-window expression for the event. + +#### get\_event\_type\_str + +```python +def get_event_type_str() -> str +``` + +Get the event type string for TimeWindowEvent. + +**Returns**: + +`str`: Event type string. + +#### determine\_definition\_hash + +```python +def determine_definition_hash() -> int +``` + +Calculate definition hash for the time-window event. + +Only includes the expression string (which encodes ``window_length``), the sole +attribute that affects the event results, so resizing the window forces a full +recompute in incremental mode. + +Excludes: name, description, required_channels, report_id + +**Returns**: + +`int`: Hash value representing the computation definition. + +#### as\_dict + +```python +def as_dict() -> dict +``` + +Get a dictionary representation of the event. + +**Returns**: + +`dict`: Dictionary containing event metadata. + +#### as\_spark\_row + +```python +def as_spark_row() -> Row +``` + +Get a Spark Row representation of the event. + +**Returns**: + +`Row`: Spark Row containing event metadata. + +#### determine\_events + +```python +def determine_events(cls, + spark: SparkSession, + events: list[TimeWindowEvent], + *, + solved_df: DataFrame = None, + query: QueryBuilder = None, + solver: QuerySolver = None, + pre_filtered_containers_df=None) +``` + +Extract the event fact table for the given list of TimeWindowEvent objects. + +Each window becomes one event instance (``start_ts < end_ts``). The window intervals +are read from the centralized solve (the same column consumed by scoped aggregations), +so the resulting ``event_instance_id`` values match on both sides. + +**Arguments**: + +- `spark` (`SparkSession`): Spark session for data processing. +- `events` (`list of TimeWindowEvent`): List of TimeWindowEvent objects to process. +- `solved_df` (`DataFrame`): Pre-solved wide DataFrame from centralized batch solve. Required. +- `query` (`QueryBuilder`): Query builder (unused, kept for interface compatibility). +- `solver` (`QuerySolver`): Query solver (unused, kept for interface compatibility). +- `pre_filtered_containers_df` (`DataFrame`): Pre-filtered containers for incremental processing. + +**Returns**: + +`DataFrame`: Spark DataFrame containing event instance facts. + +#### determine\_metadata\_df + +```python +def determine_metadata_df(cls, spark: SparkSession, + events: list[TimeWindowEvent]) +``` + +Create a Spark DataFrame containing event metadata. + +**Arguments**: + +- `spark` (`SparkSession`): Spark session for data processing. +- `events` (`list of TimeWindowEvent`): List of TimeWindowEvent objects. + +**Returns**: + +`DataFrame`: Spark DataFrame containing event metadata. + diff --git a/docs/impulse/docs/references/api/sidebar.json b/docs/impulse/docs/references/api/sidebar.json index f919f467..3f08df0f 100644 --- a/docs/impulse/docs/references/api/sidebar.json +++ b/docs/impulse/docs/references/api/sidebar.json @@ -106,7 +106,8 @@ "items": [ "references/api/impulse_reporting/events/basic_event", "references/api/impulse_reporting/events/container_event", - "references/api/impulse_reporting/events/sequence_of_events" + "references/api/impulse_reporting/events/sequence_of_events", + "references/api/impulse_reporting/events/time_window_event" ], "label": "impulse_reporting.events", "type": "category" diff --git a/docs/impulse/pydoc-markdown.yml b/docs/impulse/pydoc-markdown.yml index aa28ace6..f52dda2b 100644 --- a/docs/impulse/pydoc-markdown.yml +++ b/docs/impulse/pydoc-markdown.yml @@ -7,6 +7,7 @@ loaders: - impulse_reporting.events.basic_event - impulse_reporting.events.container_event - impulse_reporting.events.sequence_of_events + - impulse_reporting.events.time_window_event - impulse_reporting.aggregations.histogram - impulse_reporting.aggregations.histogram2d - impulse_reporting.aggregations.stats_aggregator diff --git a/skills/README.md b/skills/README.md index cb98eccf..bcae0a27 100644 --- a/skills/README.md +++ b/skills/README.md @@ -12,7 +12,7 @@ Each skill is a folder with a `SKILL.md` file that documents usage patterns. Sta | [`impulse-tsal`](./impulse-tsal/SKILL.md) | The Time Series Analytics Language DSL — selecting channels, deriving virtual signals, and the four result types (`SampleSeries`, `Intervals`, `PointsInTime`, `PointsInTimeSeries`). | | [`impulse-data-model`](./impulse-data-model/SKILL.md) | The silver-layer input tables Impulse reads, the gold-layer star schema it writes, landing your own data, and adapting an existing layout via column mappings. | | [`impulse-config`](./impulse-config/SKILL.md) | The `ImpulseConfig` schema — source tables, sink, container filters, solver options, incremental processing, and sinkless mode. | -| [`impulse-events`](./impulse-events/SKILL.md) | Defining event windows: `BasicEvent`, `ContainerEvent`, `SequenceOfEvents`, `PointsInTimeEvent`. | +| [`impulse-events`](./impulse-events/SKILL.md) | Defining event windows: `BasicEvent`, `ContainerEvent`, `SequenceOfEvents`, `PointsInTimeEvent`, `TimeWindowEvent`. | | [`impulse-aggregations`](./impulse-aggregations/SKILL.md)| Computing results over channels: 1D/2D histograms (duration/distance/custom-weight), `StatsAggregator`, `PointValueAggregator`, and pages. | | [`impulse-channels`](./impulse-channels/SKILL.md) | Calculated (derived) channels — materializing a new signal from existing channels via `CalculatedChannel` and `solve_calculated_channels`. | | [`impulse-reporting`](./impulse-reporting/SKILL.md) | The batch pipeline that persists events and aggregations to the gold-layer star schema with `Report` / `Page`, plus incremental runs. | diff --git a/skills/impulse-events/SKILL.md b/skills/impulse-events/SKILL.md index faa81740..9f63736f 100644 --- a/skills/impulse-events/SKILL.md +++ b/skills/impulse-events/SKILL.md @@ -3,9 +3,10 @@ name: impulse-events description: > Define event windows in Impulse — the time spans that scope aggregations. Use when the user wants to "define an event", segment recordings into intervals (e.g. "engine RPM between 2000 and 5000"), - aggregate over the whole recording, capture state transitions/sequences, or mark instants like rising - edges. Covers BasicEvent, ContainerEvent, SequenceOfEvents, and PointsInTimeEvent — which TSAL result - type each requires, their constructor parameters, and the event fact/dimension output. + aggregate over the whole recording, split a recording into fixed time windows (one-minute, hourly, + daily segments), capture state transitions/sequences, or mark instants like rising edges. Covers + BasicEvent, ContainerEvent, SequenceOfEvents, PointsInTimeEvent, and TimeWindowEvent — which TSAL + result type each requires, their constructor parameters, and the event fact/dimension output. --- # Impulse — events @@ -29,6 +30,7 @@ Choose the type by what you need: | `ContainerEvent` | none | exactly one | full recording | | `SequenceOfEvents` | ordered list, each `Intervals` | one per joined sequence | interval (`start < end`) | | `PointsInTimeEvent` | one, must yield `PointsInTime` | one per instant | zero (`start == end`) | +| `TimeWindowEvent` | none (needs a `window_length`) | one per fixed window | fixed window (last clamped) | The TSAL result type is validated at construction — passing the wrong type raises `ValueError`. @@ -123,12 +125,43 @@ report.add_event(rpm_rising) Parameters: `name` (required), `expr` (required, must yield `PointsInTime`), `desc`, `required_channels`, `attributes`. +## TimeWindowEvent + +Splits each recording into consecutive fixed-duration windows — one instance per slice — from the +container's `start_ts`/`stop_ts` in `container_metrics` (no expression). The final window is clamped to +the recording end. Use it for repeated segments (one-minute, ten-minute, hourly, daily) that scope +aggregations per window. + +```python +from impulse_reporting.events.time_window_event import TimeWindowEvent + +ten_minute = TimeWindowEvent( + name="ten_minute_windows", + window_length=600_000, # same time unit as the timestamps (e.g. ms-since-epoch) + desc="Ten-minute segments", +) +report.add_event(ten_minute) +``` + +| Parameter | Type | Required | Description | +|---------------------|---------------------|----------|---------------------------------------------------------------------------------| +| `name` | `str` | Yes | Unique event name. | +| `window_length` | `float` | Yes | Fixed window length, **in the same time unit as the timestamps**. Must be > 0. | +| `desc` | `str` | No | Description. | +| `required_channels` | `list[str]` | No | Informational. | +| `attributes` | `Mapping[str, str]` | No | Free-form metadata; `window_length` is added automatically. | + +Pair it with an aggregation scoped to the event (e.g. `StatsAggregator(..., event=...)`) to compute one +statistic per window. Because the windows come from `container_metrics`, those boundaries must share the +channel samples' time base for the per-window values to be meaningful. + ## Output schema All event types share two gold tables. **event_dimension** (one row per event) — key columns: `event_id`, `report_id`, -`event_type` (`"BASIC_EVENT"`, `"CONTAINER_EVENT"`, `"SEQUENCE_OF_EVENTS"`, `"POINTS_IN_TIME_EVENT"`), +`event_type` (`"BASIC_EVENT"`, `"CONTAINER_EVENT"`, `"SEQUENCE_OF_EVENTS"`, `"POINTS_IN_TIME_EVENT"`, +`"TIME_WINDOW_EVENT"`), `event_name`, `event_description`, `required_channels`, `event_expression` (TSAL string, `"NA"` for `ContainerEvent`), `definition_hash`, `attributes`. From 4dfcd41b9b2ed247fb79a0fb4b2919a9953254c6 Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 30 Sep 2026 18:44:40 +0200 Subject: [PATCH 05/27] fix(query-engine): normalize TimeWindowExpression.window_length to float for stable hashes Store `window_length` as a float in `TimeWindowExpression` so the string representation (and downstream event definition hash) is identical whether an int or float is passed. This prevents spurious full recomputes in incremental mode when the same window is described as `10` vs `10.0`. Remove the now-unreachable `window_count <= 0` guard and add unit tests covering hash stability across int/float window lengths. --- .../analyze/query/events/time_window_expression.py | 9 ++++++--- .../analyze/query/events/time_window_expression_test.py | 6 ++++++ .../unit/events/time_window_event_test.py | 8 ++++++++ 3 files changed, 20 insertions(+), 3 deletions(-) diff --git a/src/impulse_query_engine/analyze/query/events/time_window_expression.py b/src/impulse_query_engine/analyze/query/events/time_window_expression.py index 94875a53..7bba16d2 100644 --- a/src/impulse_query_engine/analyze/query/events/time_window_expression.py +++ b/src/impulse_query_engine/analyze/query/events/time_window_expression.py @@ -59,7 +59,10 @@ def __init__(self, window_length: float): f"TimeWindowExpression requires a strictly positive window_length, " f"got {window_length!r}." ) - self.window_length = window_length + # Store as float so the string form (and thus the event definition hash) is stable + # regardless of whether an int or float was passed: 10 and 10.0 are the same window + # and must not trigger a spurious full recompute in incremental mode. + self.window_length = float(window_length) TimeSeriesExpression.__init__(self, is_single_signal=False) def __str__(self) -> str: @@ -170,9 +173,9 @@ def build(self, cache: SeriesCache) -> Intervals: return Intervals.empty() # Number of windows covering the span; the last one is clamped to stop_ts below. + # The span is strictly positive (guarded above) and window_length is strictly + # positive (enforced in __init__), so window_count >= 1. window_count = int(np.ceil((stop_ts - start_ts) / self.window_length)) - if window_count <= 0: - return Intervals.empty() indices = np.arange(window_count) starts = start_ts + indices * self.window_length diff --git a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py index 30e45416..28e80472 100644 --- a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py @@ -103,6 +103,12 @@ def test_str_includes_window_length(): assert "window_length=10" in str(TimeWindowExpression(10)) +def test_str_stable_across_int_and_float_window_length(): + # int 10 and float 10.0 are the same window; the string form (which feeds the event + # definition hash) must not differ between them. + assert str(TimeWindowExpression(10)) == str(TimeWindowExpression(10.0)) + + @pytest.mark.parametrize("bad", [0, -1, -10.5, None]) def test_non_positive_window_length_raises(bad): with pytest.raises(ValueError, match="strictly positive"): diff --git a/tests/impulse_reporting/unit/events/time_window_event_test.py b/tests/impulse_reporting/unit/events/time_window_event_test.py index dd960cc3..551f3293 100644 --- a/tests/impulse_reporting/unit/events/time_window_event_test.py +++ b/tests/impulse_reporting/unit/events/time_window_event_test.py @@ -66,6 +66,14 @@ def test_definition_hash_stable_across_desc_and_attributes(): assert a.determine_definition_hash() == b.determine_definition_hash() +def test_definition_hash_stable_across_int_and_float_window_length(): + # 10000 and 10000.0 describe identical windows; the hash must not change between them + # (otherwise an int/float re-run forces a spurious full recompute in incremental mode). + a = TimeWindowEvent(name="w", window_length=10000) + b = TimeWindowEvent(name="w", window_length=10000.0) + assert a.determine_definition_hash() == b.determine_definition_hash() + + # --------------------------------------------------------------------------- # metadata dict shape # --------------------------------------------------------------------------- From 5b316f68cb0f576a6d8a6e8da677eec0efe5622b Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Thu, 1 Oct 2026 14:19:34 +0200 Subject: [PATCH 06/27] feat(reporting): compute TimeWindowEvent windows natively from container_metrics Introduce `ContainerBoundaryEvent` as a shared base for `ContainerEvent` and `TimeWindowEvent`, routing both through the solver's filter pipeline instead of the centralized channel solve. `TimeWindowEvent` now materializes windows natively in Spark from `container_metrics.start_ts` / `stop_ts` via `window_intervals_col`, so every filtered container gets windows regardless of channel coverage. Keep the query-engine `TimeWindowExpression` bit-identical to the new Spark helper by casting boundaries to double before computing window counts and clamping. Normalize `TimeWindowEvent.window_length` through the expression so int and float inputs produce identical attributes and hashes. Update docs, skills, and tests to reflect that `TimeWindowEvent` no longer requires a co-solved aggregation and that scoped aggregations still match by `event_instance_id` even with rounded double boundaries. --- .../events/container_event.md | 2 +- .../events/time_window_event.md | 40 +-- docs/impulse/docs/references/report/event.md | 24 +- skills/impulse-events/SKILL.md | 8 +- .../query/events/time_window_expression.py | 56 +++- src/impulse_reporting/core/report.py | 12 +- src/impulse_reporting/core/report_utils.py | 9 +- .../events/container_boundary_event.py | 50 +++ .../events/container_event.py | 9 +- src/impulse_reporting/events/event.py | 4 +- .../events/time_window_event.py | 80 +++-- .../events/time_window_expression_test.py | 170 ++++++++++ .../integration/time_window_event_test.py | 301 +++++++++++++++--- .../unit/events/time_window_event_test.py | 25 +- 14 files changed, 660 insertions(+), 130 deletions(-) create mode 100644 src/impulse_reporting/events/container_boundary_event.py diff --git a/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md b/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md index 463c6bbb..e4970c54 100644 --- a/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md +++ b/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md @@ -9,7 +9,7 @@ ContainerEvent — an event spanning the full measurement container. ## ContainerEvent ```python -class ContainerEvent(Event) +class ContainerEvent(ContainerBoundaryEvent) ``` Event that treats the full measurement container as a single event instance. diff --git a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md index d5155f64..e7f6606a 100644 --- a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md +++ b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md @@ -9,7 +9,7 @@ TimeWindowEvent — splits each container into consecutive fixed-duration window ## TimeWindowEvent ```python -class TimeWindowEvent(Event) +class TimeWindowEvent(ContainerBoundaryEvent) ``` Event that divides each measurement container into consecutive fixed windows. @@ -17,9 +17,9 @@ Event that divides each measurement container into consecutive fixed windows. Unlike ``ContainerEvent`` (one instance per container), a ``TimeWindowEvent`` emits one event instance per fixed-duration slice, tiling the container's ``start_ts`` / ``stop_ts`` span with windows of length ``window_length``. The final slice is clamped to the -container end. Boundaries come from a :class:`TimeWindowExpression`, so the event -fact and any aggregation scoped to this event share the same solved windows and their -``event_instance_id`` values match by construction. +container end. + +The event fact is computed natively in Spark from ``container_metrics`` (via #### \_\_init\_\_ @@ -130,29 +130,33 @@ Get a Spark Row representation of the event. #### determine\_events ```python -def determine_events(cls, - spark: SparkSession, - events: list[TimeWindowEvent], - *, - solved_df: DataFrame = None, - query: QueryBuilder = None, - solver: QuerySolver = None, - pre_filtered_containers_df=None) +def determine_events( + cls, + spark: SparkSession, + events: list[TimeWindowEvent], + *, + solved_df: DataFrame = None, + query: QueryBuilder = None, + solver: QuerySolver = None, + pre_filtered_containers_df: DataFrame = None) -> DataFrame ``` Extract the event fact table for the given list of TimeWindowEvent objects. -Each window becomes one event instance (``start_ts < end_ts``). The window intervals -are read from the centralized solve (the same column consumed by scoped aggregations), -so the resulting ``event_instance_id`` values match on both sides. +Resolves the matching containers via the solver's filter pipeline (like +``ContainerEvent``) and computes each event's windows natively from the +containers' ``start_ts`` / ``stop_ts``, so every filtered container gets windows. +Each window becomes one event instance (``start_ts < end_ts``). The windows are +bit-identical to the ones the solve computes for scoped aggregations (see +:func:`window_intervals_col`), so the ``event_instance_id`` values match. **Arguments**: - `spark` (`SparkSession`): Spark session for data processing. - `events` (`list of TimeWindowEvent`): List of TimeWindowEvent objects to process. -- `solved_df` (`DataFrame`): Pre-solved wide DataFrame from centralized batch solve. Required. -- `query` (`QueryBuilder`): Query builder (unused, kept for interface compatibility). -- `solver` (`QuerySolver`): Query solver (unused, kept for interface compatibility). +- `solved_df` (`DataFrame`): Not used by TimeWindowEvent (kept for interface compatibility). +- `query` (`QueryBuilder`): Query builder with filters applied. +- `solver` (`QuerySolver`): Solver whose filter pipeline is used for container resolution. - `pre_filtered_containers_df` (`DataFrame`): Pre-filtered containers for incremental processing. **Returns**: diff --git a/docs/impulse/docs/references/report/event.md b/docs/impulse/docs/references/report/event.md index 82c8fb0b..ae59c0bf 100644 --- a/docs/impulse/docs/references/report/event.md +++ b/docs/impulse/docs/references/report/event.md @@ -233,8 +233,9 @@ seconds or any derived unit. So 60 one-minute windows over millisecond timestamp ### How it works -1. A time-window expression reads `start_ts` and `stop_ts` from the `container_metrics` table - and tiles `[start_ts, stop_ts]` into consecutive windows of length `window_length`. +1. The event resolves the matching containers through the report's container filters (like + `ContainerEvent`), reads `start_ts` and `stop_ts` from the `container_metrics` table, and + tiles `[start_ts, stop_ts]` into consecutive windows of length `window_length`. 2. The **final window is clamped** to `stop_ts` when the last full window would overrun it; any zero-length trailing slice is dropped (every instance satisfies `start_ts < end_ts`). 3. Each window becomes one **event instance** with a unique `event_instance_id`, written to the @@ -243,12 +244,19 @@ seconds or any derived unit. So 60 one-minute windows over millisecond timestamp its statistic **once per window** and joins back to those instances. :::note -A `TimeWindowEvent` is evaluated in the shared batch solve alongside the report's channel-based -selections, so it materializes windows for the containers covered by that solve. In practice this -is every container the report touches -- a `TimeWindowEvent` is normally paired with at least one -aggregation (its purpose), which supplies the channels. The window boundaries come from -`container_metrics`, so for a scoped aggregation to produce meaningful per-window values, those -boundaries must share the channel samples' time base (as they do in real measurement data). +The windows are computed from `container_metrics` alone, so **every** container that matches the +report's filters gets windows, whether or not it has channel data and whether or not an +aggregation is scoped to the event. An aggregation scoped to the event computes the same windows +in the query engine, so its per-window rows carry the same `event_instance_id` values. For the +per-window values to be meaningful, the container boundaries must share the channel samples' time +base (as they do in real measurement data). +::: + +:::note +Window boundaries are stored as doubles (`start_ts` / `end_ts`), like every other event type. Epoch +timestamps in nanoseconds exceed the range doubles represent exactly, so their window boundaries +are rounded to about 256 ns. The rounding is the same for the event and its aggregations, so their +`event_instance_id` values still match. ::: ## Event output schema diff --git a/skills/impulse-events/SKILL.md b/skills/impulse-events/SKILL.md index 9f63736f..97e773e9 100644 --- a/skills/impulse-events/SKILL.md +++ b/skills/impulse-events/SKILL.md @@ -151,9 +151,11 @@ report.add_event(ten_minute) | `required_channels` | `list[str]` | No | Informational. | | `attributes` | `Mapping[str, str]` | No | Free-form metadata; `window_length` is added automatically. | -Pair it with an aggregation scoped to the event (e.g. `StatsAggregator(..., event=...)`) to compute one -statistic per window. Because the windows come from `container_metrics`, those boundaries must share the -channel samples' time base for the per-window values to be meaningful. +Windows are computed from `container_metrics` for every container matching the report's filters, with +or without channel data or a scoped aggregation. Pair it with an aggregation scoped to the event (e.g. +`StatsAggregator(..., event=...)`) to compute one statistic per window; those rows carry the same +`event_instance_id` values as the windows. Because the windows come from `container_metrics`, those +boundaries must share the channel samples' time base for the per-window values to be meaningful. ## Output schema diff --git a/src/impulse_query_engine/analyze/query/events/time_window_expression.py b/src/impulse_query_engine/analyze/query/events/time_window_expression.py index 7bba16d2..36632557 100644 --- a/src/impulse_query_engine/analyze/query/events/time_window_expression.py +++ b/src/impulse_query_engine/analyze/query/events/time_window_expression.py @@ -1,6 +1,8 @@ from __future__ import annotations import numpy as np +import pyspark.sql.functions as F +from pyspark.sql import Column from impulse_query_engine.analyze.metadata.tag_expression import TagExpression from impulse_query_engine.analyze.metadata.time_series_expression import ( @@ -19,6 +21,48 @@ _SOLVER_CONFIG = SolverConfig() +def window_intervals_col(start_ts: Column, stop_ts: Column, window_length: float) -> Column: + """Spark counterpart of :meth:`TimeWindowExpression.build`. + + Computes the same fixed-duration windows natively in Spark, so the reporting + ``TimeWindowEvent`` can materialize windows for every container without a solve. + The ``event_instance_id`` of a window hashes its ``start_ts`` / ``end_ts``, so the + windows computed here must be **bit-identical** to the ones ``build`` computes for + scoped aggregations. Both therefore run the same IEEE-754 operations in the same + order on the same doubles: cast the boundaries to double *before* subtracting, + ``count = ceil((stop - start) / W)``, ``start_i = start + i * W``, + ``end_i = min(start + (i + 1) * W, stop)``, and drop windows with + ``start_i >= end_i``. Keep the two implementations in sync. + + Parameters + ---------- + start_ts : pyspark.sql.Column + Container start timestamp. + stop_ts : pyspark.sql.Column + Container stop timestamp. + window_length : float + Fixed window length, in the same time unit as the timestamps. Must be strictly + positive. + + Returns + ------- + pyspark.sql.Column + ``array>`` with one ``[start, end]`` pair per window; empty when the + boundaries are null or the span is not strictly positive. + """ + start, stop = start_ts.cast("double"), stop_ts.cast("double") + w = F.lit(float(window_length)) + count = F.ceil((stop - start) / w) + windows = F.transform( + F.sequence(F.lit(0), count - F.lit(1)), + lambda i: F.array(start + i * w, F.least(start + (i + F.lit(1)) * w, stop)), + ) + windows = F.filter(windows, lambda p: p[0] < p[1]) + # Gate on a positive span: sequence(0, -1) yields [0, -1] (a descending sequence), + # not an empty array, so degenerate containers would otherwise emit bogus windows. + return F.when(stop > start, windows).otherwise(F.array().cast("array>")) + + class TimeWindowExpression(TimeSeriesExpression): """Produce consecutive fixed-duration windows spanning a measurement container. @@ -37,6 +81,8 @@ class TimeWindowExpression(TimeSeriesExpression): This is the query-engine counterpart of the reporting ``TimeWindowEvent``. It evaluates to :class:`Intervals`, so it can scope a ``StatsAggregator`` (one statistic per window). + The reporting event fact computes the same windows natively via + :func:`window_intervals_col`; the two must stay bit-identical. """ def __init__(self, window_length: float): @@ -169,7 +215,15 @@ def build(self, cache: SeriesCache) -> Intervals: start_ts = cache.container_metrics.get(_SOLVER_CONFIG.start_ts_col) stop_ts = cache.container_metrics.get(_SOLVER_CONFIG.stop_ts_col) - if start_ts is None or stop_ts is None or stop_ts <= start_ts: + if start_ts is None or stop_ts is None: + return Intervals.empty() + + # Mirror window_intervals_col exactly: convert to double *before* subtracting. A + # long column reaches pandas as int64 or float64 depending on the group (nulls + # force float64), and an exact int64 span can round differently from the double + # span for large values (e.g. ns epochs), changing the window count. + start_ts, stop_ts = float(start_ts), float(stop_ts) + if not stop_ts > start_ts: return Intervals.empty() # Number of windows covering the span; the last one is clamped to stop_ts below. diff --git a/src/impulse_reporting/core/report.py b/src/impulse_reporting/core/report.py index 131d3731..d831ceee 100644 --- a/src/impulse_reporting/core/report.py +++ b/src/impulse_reporting/core/report.py @@ -43,6 +43,7 @@ split_by_hash_change, validate_full_recalculation_scope, ) +from impulse_reporting.events.container_boundary_event import ContainerBoundaryEvent from impulse_reporting.events.container_event import ContainerEvent from impulse_reporting.events.event import Event from impulse_reporting.events.event_types import EventType @@ -1137,12 +1138,13 @@ def determine_report(self, is_incremental: bool = None): ) ) - # Collect all solvable expressions (exclude ContainerEvent) + # Collect all solvable expressions. Container-boundary events (ContainerEvent, + # TimeWindowEvent) resolve from container_metrics, not the channel solve. all_changed_expressions = collect_solvable_expressions( - changed_events_by_type, EventType, exclude_cls=ContainerEvent + changed_events_by_type, EventType, exclude_cls=ContainerBoundaryEvent ) + collect_solvable_expressions(changed_aggs_by_type, AggregationType) all_unchanged_expressions = collect_solvable_expressions( - unchanged_events_by_type, EventType, exclude_cls=ContainerEvent + unchanged_events_by_type, EventType, exclude_cls=ContainerBoundaryEvent ) + collect_solvable_expressions(unchanged_aggs_by_type, AggregationType) # Centralized solve @@ -1163,7 +1165,7 @@ def determine_report(self, is_incremental: bool = None): self.query, self.solver, changed_pre_filtered_containers_df, - ContainerEvent, + ContainerBoundaryEvent, ) unchanged_event_dfs = dispatch_events( self.spark, @@ -1173,7 +1175,7 @@ def determine_report(self, is_incremental: bool = None): self.query, self.solver, pre_filtered_containers_df, - ContainerEvent, + ContainerBoundaryEvent, ) # Merge event results into {type: {"changed": df, "unchanged": df}} and diff --git a/src/impulse_reporting/core/report_utils.py b/src/impulse_reporting/core/report_utils.py index 1a540e4b..fcbef84c 100644 --- a/src/impulse_reporting/core/report_utils.py +++ b/src/impulse_reporting/core/report_utils.py @@ -320,8 +320,8 @@ def dispatch_events( ) -> dict: """Dispatch ``determine_events`` calls per type. - Solvable event types receive ``solved_df``; ``ContainerEvent`` receives - ``query``/``solver``. + Solvable event types receive ``solved_df``; container-boundary events + (``ContainerEvent``, ``TimeWindowEvent``) receive ``query``/``solver``. Parameters ---------- @@ -333,7 +333,8 @@ def dispatch_events( solver : QuerySolver pre_filtered_containers_df : DataFrame | None container_event_cls : type - The ``ContainerEvent`` class. + Base class of the container-boundary events (``ContainerBoundaryEvent``); + subclasses are resolved via the filter pipeline instead of ``solved_df``. Returns ------- @@ -348,7 +349,7 @@ def dispatch_events( cls = type_enum[type_name].value if issubclass(cls, container_event_cls): - # ContainerEvent uses filter pipeline, not solved_df + # Container-boundary events use the filter pipeline, not solved_df event_dfs[type_name] = cls.determine_events( spark, events, diff --git a/src/impulse_reporting/events/container_boundary_event.py b/src/impulse_reporting/events/container_boundary_event.py new file mode 100644 index 00000000..bbe40928 --- /dev/null +++ b/src/impulse_reporting/events/container_boundary_event.py @@ -0,0 +1,50 @@ +"""ContainerBoundaryEvent — base for events derived from container boundaries.""" + +from __future__ import annotations + +from pyspark.sql import DataFrame, SparkSession + +from impulse_query_engine.analyze.query.query_builder import QueryBuilder +from impulse_query_engine.analyze.query.solvers.query_solver import QuerySolver +from impulse_reporting.events.event import Event + + +class ContainerBoundaryEvent(Event): + """Base class for events whose instances are derived from container boundaries. + + The instances are resolved from ``container_metrics`` (``start_ts`` / ``stop_ts``) + via the solver's filter pipeline instead of the centralized channel solve, so every + filtered container yields instances regardless of its channel data. The report + therefore excludes these event types from the solvable expressions and dispatches + them with ``query`` / ``solver`` rather than ``solved_df``. + """ + + @staticmethod + def resolve_container_metrics( + spark: SparkSession, + query: QueryBuilder, + solver: QuerySolver, + pre_filtered_containers_df: DataFrame = None, + ) -> DataFrame: + """Resolve the filtered containers' metrics via the solver filter pipeline. + + Parameters + ---------- + spark : SparkSession + Active Spark session. + query : QueryBuilder + Query builder with filters applied. + solver : QuerySolver + Solver whose filter pipeline is used for container resolution. + pre_filtered_containers_df : DataFrame, optional + Pre-filtered containers for incremental processing. + + Returns + ------- + DataFrame + Column-mapped ``container_metrics`` rows of the matching containers. + """ + container_tags_df = solver.filter_container_tags(spark, query) + return solver.filter_container_metrics( + spark, query, container_tags_df, pre_filtered_containers_df + ) diff --git a/src/impulse_reporting/events/container_event.py b/src/impulse_reporting/events/container_event.py index 7a6ea264..bc3c9a7e 100644 --- a/src/impulse_reporting/events/container_event.py +++ b/src/impulse_reporting/events/container_event.py @@ -11,7 +11,7 @@ from impulse_query_engine.analyze.query.query_builder import QueryBuilder from impulse_query_engine.analyze.query.solvers.query_solver import QuerySolver -from impulse_reporting.events.event import Event +from impulse_reporting.events.container_boundary_event import ContainerBoundaryEvent from impulse_reporting.persist.dimension_schema import EVENT_DIMENSION_SCHEMA from impulse_reporting.persist.fact_schema import EVENT_INSTANCE_FACT_SCHEMA from impulse_reporting.util.event_instance_util import generate_event_instance_id_column @@ -23,7 +23,7 @@ ) -class ContainerEvent(Event): +class ContainerEvent(ContainerBoundaryEvent): """Event that treats the full measurement container as a single event instance. Unlike ``BasicEvent``, no time-series expression is needed — the event @@ -170,9 +170,8 @@ def determine_events( Spark DataFrame matching ``EVENT_INSTANCE_FACT_SCHEMA``. """ # Resolve containers via solver filter pipeline - container_tags_df = solver.filter_container_tags(spark, query) - container_metrics_df = solver.filter_container_metrics( - spark, query, container_tags_df, pre_filtered_containers_df + container_metrics_df = cls.resolve_container_metrics( + spark, query, solver, pre_filtered_containers_df ) # Rename silver columns to gold event fact column names and cast diff --git a/src/impulse_reporting/events/event.py b/src/impulse_reporting/events/event.py index 9d590c77..a45ea243 100644 --- a/src/impulse_reporting/events/event.py +++ b/src/impulse_reporting/events/event.py @@ -171,9 +171,9 @@ def determine_events( solved_df : DataFrame, optional Pre-solved wide DataFrame from centralized batch solve. query : QueryBuilder, optional - Query builder for constructing event queries (ContainerEvent path). + Query builder for constructing event queries (container-boundary event path). solver : QuerySolver, optional - Query solver for executing queries (ContainerEvent path). + Query solver for executing queries (container-boundary event path). pre_filtered_containers_df : DataFrame, optional Pre-filtered containers for incremental processing. diff --git a/src/impulse_reporting/events/time_window_event.py b/src/impulse_reporting/events/time_window_event.py index e60f53a0..a7ab79c8 100644 --- a/src/impulse_reporting/events/time_window_event.py +++ b/src/impulse_reporting/events/time_window_event.py @@ -14,26 +14,31 @@ ) from impulse_query_engine.analyze.query.events.time_window_expression import ( TimeWindowExpression, + window_intervals_col, ) from impulse_query_engine.analyze.query.query_builder import QueryBuilder from impulse_query_engine.analyze.query.solvers.query_solver import QuerySolver from impulse_query_engine.model.series.intervals import Intervals -from impulse_reporting.events.event import Event +from impulse_reporting.events.container_boundary_event import ContainerBoundaryEvent from impulse_reporting.persist.dimension_schema import EVENT_DIMENSION_SCHEMA from impulse_reporting.persist.fact_schema import EVENT_INSTANCE_FACT_SCHEMA from impulse_reporting.util.event_instance_util import generate_event_instance_id_column from impulse_reporting.util.report_entity_util import ReportEntityUtil -class TimeWindowEvent(Event): +class TimeWindowEvent(ContainerBoundaryEvent): """Event that divides each measurement container into consecutive fixed windows. Unlike ``ContainerEvent`` (one instance per container), a ``TimeWindowEvent`` emits one event instance per fixed-duration slice, tiling the container's ``start_ts`` / ``stop_ts`` span with windows of length ``window_length``. The final slice is clamped to the - container end. Boundaries come from a :class:`TimeWindowExpression`, so the event - fact and any aggregation scoped to this event share the same solved windows and their - ``event_instance_id`` values match by construction. + container end. + + The event fact is computed natively in Spark from ``container_metrics`` (via + :func:`window_intervals_col`), so every filtered container gets windows regardless of + its channel data. Aggregations scoped to this event evaluate the + :class:`TimeWindowExpression` in the solve, which computes bit-identical windows, so + the timestamp-based ``event_instance_id`` values match on both sides. """ def __init__( @@ -68,14 +73,16 @@ def __init__( ValueError If ``window_length`` is not strictly positive. """ - Event.__init__(self, name) + ContainerBoundaryEvent.__init__(self, name) if window_length is None or window_length <= 0: raise ValueError( f"TimeWindowEvent requires a strictly positive window_length, " f"got {window_length!r}." ) - self.window_length = window_length self.expression = TimeWindowExpression(window_length).alias(name) + # Use the expression's normalized (float) length everywhere, so the event fact, + # the solve and event_dimension all see the same value for 10 and 10.0. + self.window_length = self.expression.window_length self.expression.require_evaluation_type( Intervals, owner="TimeWindowEvent", example="window_length=60000" ) @@ -86,7 +93,7 @@ def __init__( normalized_attributes = {str(k): str(v) for k, v in attributes.items()} # Surface the window length for traceability in event_dimension, without # clobbering an explicit user-supplied attribute of the same key. - normalized_attributes.setdefault("window_length", str(window_length)) + normalized_attributes.setdefault("window_length", str(self.window_length)) self.attributes = normalized_attributes def get_id(self) -> int: @@ -184,14 +191,17 @@ def determine_events( solved_df: DataFrame = None, query: QueryBuilder = None, solver: QuerySolver = None, - pre_filtered_containers_df=None, - ): + pre_filtered_containers_df: DataFrame = None, + ) -> DataFrame: """ Extract the event fact table for the given list of TimeWindowEvent objects. - Each window becomes one event instance (``start_ts < end_ts``). The window intervals - are read from the centralized solve (the same column consumed by scoped aggregations), - so the resulting ``event_instance_id`` values match on both sides. + Resolves the matching containers via the solver's filter pipeline (like + ``ContainerEvent``) and computes each event's windows natively from the + containers' ``start_ts`` / ``stop_ts``, so every filtered container gets windows. + Each window becomes one event instance (``start_ts < end_ts``). The windows are + bit-identical to the ones the solve computes for scoped aggregations (see + :func:`window_intervals_col`), so the ``event_instance_id`` values match. Parameters ---------- @@ -200,11 +210,11 @@ def determine_events( events : list of TimeWindowEvent List of TimeWindowEvent objects to process. solved_df : DataFrame, optional - Pre-solved wide DataFrame from centralized batch solve. Required. + Not used by TimeWindowEvent (kept for interface compatibility). query : QueryBuilder, optional - Query builder (unused, kept for interface compatibility). + Query builder with filters applied. solver : QuerySolver, optional - Query solver (unused, kept for interface compatibility). + Solver whose filter pipeline is used for container resolution. pre_filtered_containers_df : DataFrame, optional Pre-filtered containers for incremental processing. @@ -213,26 +223,36 @@ def determine_events( DataFrame Spark DataFrame containing event instance facts. """ - if solved_df is None: - raise ValueError( - "TimeWindowEvent.determine_events requires solved_df. " - "Provide a pre-solved DataFrame from the centralized batch-solve flow." - ) + container_metrics_df = cls.resolve_container_metrics( + spark, query, solver, pre_filtered_containers_df + ) - event_names = [event.get_name() for event in events] + # Silver-side names come from SolverConfig (column_name_mapping aware). + start_ts = f.col(solver.config.start_ts_col) + stop_ts = f.col(solver.config.stop_ts_col) + + # One (event_name, windows) struct per event, exploded in a single pass over the + # containers. start_ts / end_ts stay doubles: the event_instance_id hashes their + # string form, which must match the doubles produced by the solve. + per_event = f.array( + *[ + f.struct( + f.lit(event.get_name()).alias("event_name"), + window_intervals_col(start_ts, stop_ts, event.window_length).alias("windows"), + ) + for event in events + ] + ) df = ( - solved_df.select("container_id", *event_names) - .unpivot( - f.col("container_id"), - event_names, - variableColumnName="event_name", - valueColumnName="value", + container_metrics_df.select( + f.col(solver.config.container_id_col).alias("container_id"), + f.explode(per_event).alias("event"), ) .select( "container_id", - "event_name", - f.explode(f.col("value")).alias("event_instance"), + f.col("event.event_name").alias("event_name"), + f.explode(f.col("event.windows")).alias("event_instance"), ) .withColumn("start_ts", f.col("event_instance").getItem(0)) .withColumn("end_ts", f.col("event_instance").getItem(1)) diff --git a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py index 28e80472..d5149ad6 100644 --- a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py @@ -1,11 +1,21 @@ from __future__ import annotations +import random +from unittest.mock import MagicMock + import numpy as np +import pyspark.sql.functions as F import pytest +from impulse_query_engine.analyze.query.aggregations.stats_aggregator import StatsAggregator from impulse_query_engine.analyze.query.events import TimeWindowExpression +from impulse_query_engine.analyze.query.events.time_window_expression import ( + window_intervals_col, +) from impulse_query_engine.analyze.query.solvers.empty_cache import EmptyTimeSeriesCache from impulse_query_engine.model.series.intervals import Intervals +from impulse_query_engine.model.series.sample_series import SampleSeries +from tests.conftest import spark # noqa: F401 (pytest fixture) class _FakeCache: @@ -113,3 +123,163 @@ def test_str_stable_across_int_and_float_window_length(): def test_non_positive_window_length_raises(bad): with pytest.raises(ValueError, match="strictly positive"): TimeWindowExpression(bad) + + +def test_int64_and_float64_inputs_build_identical_windows(): + # A long container metric reaches pandas as int64 or float64 depending on the group + # (nulls force float64); both must produce the same windows. + start, stop = 1_700_000_000_000_000_123, 1_700_000_007_000_000_049 + a = _build(np.int64(start), np.int64(stop), 1_000_000_007) + b = _build(np.float64(start), np.float64(stop), 1_000_000_007) + assert a.get_data() == b.get_data() + + +def test_nan_container_metrics_yield_empty(): + # A null start/stop arrives as NaN in a float64 column. + assert len(_build(np.nan, 100.0, 10)) == 0 + assert len(_build(0.0, np.nan, 10)) == 0 + + +# --------------------------------------------------------------------------- +# window_intervals_col: the native-Spark mirror used by the reporting event fact +# --------------------------------------------------------------------------- +def _spark_windows(spark, rows, window_length, ts_type="long"): # noqa: F811 + df = spark.createDataFrame(rows, f"k int, start_ts {ts_type}, stop_ts {ts_type}") + out = df.select( + "k", window_intervals_col(F.col("start_ts"), F.col("stop_ts"), window_length).alias("w") + ) + return out, {r.k: [list(p) for p in r.w] for r in out.collect()} + + +def test_window_intervals_col_edge_cases(spark): # noqa: F811 + rows = [ + (0, 0, 100), # exact multiple -> 10 windows + (1, 0, 105), # short final window clamped to stop + (2, 1000, 1010), # span == W -> 1 window + (3, 1000, 1001), # span < W -> 1 clamped window + (4, 50, 50), # stop == start -> none (sequence(0, -1) is [0, -1], not []) + (5, 60, 50), # stop < start -> none + (6, None, 50), # null bound -> none + (7, 0, None), + ] + out, w = _spark_windows(spark, rows, 10) + + assert out.schema["w"].dataType.simpleString() == "array>" + assert w[0] == [[float(s), float(s + 10)] for s in range(0, 100, 10)] + assert w[1][-1] == [100.0, 105.0] and len(w[1]) == 11 + assert w[2] == [[1000.0, 1010.0]] + assert w[3] == [[1000.0, 1001.0]] + assert w[4] == w[5] == w[6] == w[7] == [] + assert all(s < e for windows in w.values() for s, e in windows) + + +def _as_set(windows) -> set[tuple[float, float]]: + return {(float(s), float(e)) for s, e in windows} + + +def _count_mismatch_case(window_length: float) -> tuple[int, int]: + """Find ns-epoch (start, stop) whose int64 and double spans yield different counts. + + This is exactly the case where numpy without the float() conversion (exact int64 + subtraction) and Spark (double subtraction) disagree on the number of windows. + """ + base = 1_700_000_000_000_000_000 + for start in range(base, base + 512): + for k in (1, 3, 5, 7): + for d in range(-300, 1): + stop = start + int(k * window_length) + d + exact = int(np.ceil((stop - start) / window_length)) + rounded = int(np.ceil((float(stop) - float(start)) / window_length)) + if exact != rounded: + return start, stop + raise AssertionError("no int64/double count-mismatch case found") + + +def test_window_intervals_col_bit_identical_to_build(spark): # noqa: F811 + """The id contract: the event fact (Spark) and scoped aggregations (numpy ``build``) + must produce bit-identical windows, since event_instance_id hashes start/end.""" + rnd = random.Random(7) + + # Long timestamps: ns epochs (~1.7e18, beyond 2^53) and µs epochs, with spans hugging + # multiples of W. Includes W values that are not multiples of the 256 ns double spacing. + long_cases = [] + for w in (1e9, 6e10, 333_333_333.0, 1_000_000_007.0): + for _ in range(60): + start = rnd.randint(1_600_000_000_000_000_000, 1_800_000_000_000_000_000) + stop = start + int(rnd.randint(1, 50) * w) + rnd.randint(-600, 600) + long_cases.append((start, stop, w)) + for w in (1e6, 10_000_000.0): + for _ in range(30): + start = rnd.randint(1_600_000_000_000_000, 1_800_000_000_000_000) + stop = start + int(rnd.randint(1, 50) * w) + rnd.randint(-5, 5) + long_cases.append((start, stop, w)) + edge_start, edge_stop = _count_mismatch_case(1_000_000_007.0) + long_cases.append((edge_start, edge_stop, 1_000_000_007.0)) + + # Seconds as doubles with fractional window lengths. + double_cases = [] + for w in (0.1, 0.25, 0.3, 1.7, 60.0): + for _ in range(60): + start = rnd.uniform(1.6e9, 1.8e9) + stop = start + rnd.randint(1, 50) * w + rnd.uniform(-1e-6, 1e-6) + double_cases.append((start, stop, w)) + + for cases, ts_type, np_types in ( + (long_cases, "long", (np.int64, np.float64)), + (double_cases, "double", (np.float64,)), + ): + lengths = sorted({w for _, _, w in cases}) + df = spark.createDataFrame( + [(k, s, e, w) for k, (s, e, w) in enumerate(cases)], + f"k int, start_ts {ts_type}, stop_ts {ts_type}, w double", + ) + # A single CASE WHEN column computes each row's windows for its own length only, + # so all cases run in one Spark job. + windows_col = None + for w in lengths: + branch = window_intervals_col(F.col("start_ts"), F.col("stop_ts"), w) + condition = F.col("w") == F.lit(w) + windows_col = ( + F.when(condition, branch) + if windows_col is None + else windows_col.when(condition, branch) + ) + spark_windows = { + r.k: r.windows for r in df.select("k", windows_col.alias("windows")).collect() + } + + mismatches = [] + for k, (start, stop, w) in enumerate(cases): + expected = _as_set(spark_windows[k]) + assert expected, f"case {k} produced no windows" + for np_type in np_types: + built = _as_set(_build(np_type(start), np_type(stop), w).get_data()) + if built != expected: + mismatches.append((k, np_type.__name__, start, stop, w)) + assert not mismatches, f"Spark/numpy window mismatch: {mismatches[:5]}" + + +def test_stats_aggregator_windows_equal_helper_windows(spark): # noqa: F811 + """A StatsAggregator scoped to a TimeWindowExpression emits exactly the helper's windows + (no merging of touching windows, no extra drops).""" + start, stop, w = 1_700_000_000_000_000_123, 1_700_000_007_000_000_049, 1_000_000_007.0 + _, spark_windows = _spark_windows(spark, [(0, start, stop)], w) + + # One channel sampled across the whole container, so every window has data. + ts = np.linspace(float(start), float(stop), 50) + series = SampleSeries(tstarts=ts[:-1], tends=ts[1:], values=np.arange(49, dtype=float)) + + channel = MagicMock() + channel.build.return_value = series + + agg = StatsAggregator( + input_expressions=[channel], + event_expression=TimeWindowExpression(w), + statistics=["mean"], + ) + cache = _FakeCache({"start_ts": np.float64(start), "stop_ts": np.float64(stop)}) + event_timestamps, numeric_values, _, _ = agg.build(cache) + + assert len(spark_windows[0]) == 7 + assert _as_set(event_timestamps) == _as_set(spark_windows[0]) + assert len(numeric_values[0]) == len(event_timestamps) diff --git a/tests/impulse_reporting/integration/time_window_event_test.py b/tests/impulse_reporting/integration/time_window_event_test.py index 023f5c18..cc5b3510 100644 --- a/tests/impulse_reporting/integration/time_window_event_test.py +++ b/tests/impulse_reporting/integration/time_window_event_test.py @@ -11,6 +11,7 @@ Comparator, ContainerFilters, ImpulseConfig, + IncrementalConfig, MetricFilter, QueryEngine, Solvers, @@ -73,7 +74,7 @@ def _expected_window_count(container_id: int) -> int: def test_time_window_event_in_report(spark, basic_narrow_db): - """A TimeWindowEvent (co-solved with an aggregation) tiles each container into windows.""" + """A TimeWindowEvent (alongside a scoped aggregation) tiles each container into windows.""" my_report = Report( name="time_window_event_report", spark=spark, @@ -86,8 +87,6 @@ def test_time_window_event_in_report(spark, basic_narrow_db): ) my_report.add_event(window_evt) - # A channel-bearing aggregation must be present so the batched solve forms - # per-container groups the (selector-less) window expression can ride on. query = my_report.get_db().query page = Page(page_number=1) my_report.add_page(page) @@ -133,7 +132,7 @@ def test_time_window_event_in_report(spark, basic_narrow_db): dim_rows = my_report.event_metadata_dfs["TIME_WINDOW_EVENT"].collect() assert len(dim_rows) == 1 assert dim_rows[0].event_type == "TIME_WINDOW_EVENT" - assert dim_rows[0].attributes["window_length"] == str(WINDOW_LENGTH) + assert dim_rows[0].attributes["window_length"] == str(float(WINDOW_LENGTH)) # Window length (channel-sample time unit, µs) for the aligned-boundary test. @@ -142,13 +141,25 @@ def test_time_window_event_in_report(spark, basic_narrow_db): ALIGNED_WINDOW_LENGTH = 600_000_000 _ALIGNED_SCHEMA = "spark_catalog.silver_tw_aligned" +# Customer-shaped time bases for the id-join test. Each entry: (timestamp transform applied +# to channel tstart/tend and container start_ts/stop_ts, window length in that unit). +# us: the basic db's native µs epochs (< 2^53, every boundary exactly representable). +# ns: ns epochs (~1.5e18, beyond 2^53) with a window that is NOT a multiple of the +# 256 ns double spacing there, so the boundaries round. +# sec: seconds as doubles with a fractional window, so the boundaries round. +_TIME_BASES = { + "us": (lambda c: c, ALIGNED_WINDOW_LENGTH), + "ns": (lambda c: c.cast("long") * F.lit(1000), 600_000_000_007), + "sec": (lambda c: c.cast("double") / F.lit(1e6), 600.3), +} -@pytest.fixture -def setup_tw_aligned_db(spark, setup_basic_db): # noqa: F811 - """Silver tables cloned from the basic db with container_metrics start_ts / stop_ts - recomputed from each container's channel-sample range, so the container boundaries - (and thus the windows) share the samples' time base.""" - spark.sql(f"CREATE SCHEMA IF NOT EXISTS {_ALIGNED_SCHEMA}") + +def _clone_aligned_silver(spark, schema: str, to_time_base=lambda c: c) -> None: + """Clone the basic silver tables into *schema* with container_metrics start_ts / stop_ts + recomputed from each container's channel-sample range (so the container boundaries, + and thus the windows, share the samples' time base), then map all of those timestamps + through *to_time_base*.""" + spark.sql(f"CREATE SCHEMA IF NOT EXISTS {schema}") channels = spark.read.table("spark_catalog.silver.channels") bounds = channels.groupBy("container_id").agg( F.min("tstart").alias("_agg_start"), @@ -162,29 +173,46 @@ def setup_tw_aligned_db(spark, setup_basic_db): # noqa: F811 .withColumn("start_ts", F.coalesce("_agg_start", "start_ts").cast(start_type)) .withColumn("stop_ts", F.coalesce("_agg_stop", "stop_ts").cast(stop_type)) .drop("_agg_start", "_agg_stop") + .withColumn("start_ts", to_time_base(F.col("start_ts"))) + .withColumn("stop_ts", to_time_base(F.col("stop_ts"))) ) aligned_cm.write.format("delta").mode("overwrite").option( "overwriteSchema", "true" - ).saveAsTable(f"{_ALIGNED_SCHEMA}.container_metrics") - for table in ("channel_metrics", "channels"): - spark.read.table(f"spark_catalog.silver.{table}").write.format("delta").mode( - "overwrite" - ).saveAsTable(f"{_ALIGNED_SCHEMA}.{table}") - yield - spark.sql(f"DROP SCHEMA IF EXISTS {_ALIGNED_SCHEMA} CASCADE") + ).saveAsTable(f"{schema}.container_metrics") + channels.withColumn("tstart", to_time_base(F.col("tstart"))).withColumn( + "tend", to_time_base(F.col("tend")) + ).write.format("delta").mode("overwrite").option("overwriteSchema", "true").saveAsTable( + f"{schema}.channels" + ) + spark.read.table("spark_catalog.silver.channel_metrics").write.format("delta").mode( + "overwrite" + ).saveAsTable(f"{schema}.channel_metrics") -def test_time_window_event_aggregation_join(spark, setup_tw_aligned_db): - """Stats scoped to a TimeWindowEvent yield per-window values that join to the fact.""" - config = dict( +@pytest.fixture +def setup_tw_aligned_db(spark, setup_basic_db, request): # noqa: F811 + """Aligned silver clone in the time base given by ``request.param`` (default ``us``). + + Yields ``(schema, window_length)``. + """ + time_base = getattr(request, "param", "us") + to_time_base, window_length = _TIME_BASES[time_base] + schema = f"{_ALIGNED_SCHEMA}_{time_base}" + _clone_aligned_silver(spark, schema, to_time_base) + yield schema, window_length + spark.sql(f"DROP SCHEMA IF EXISTS {schema} CASCADE") + + +def _aligned_config(schema: str, table_prefix: str, **extra) -> dict: + return dict( ImpulseConfig( source=Source( - container_metrics_table=f"{_ALIGNED_SCHEMA}.container_metrics", - channel_metrics_table=f"{_ALIGNED_SCHEMA}.channel_metrics", - channels_uri=f"{_ALIGNED_SCHEMA}.channels", + container_metrics_table=f"{schema}.container_metrics", + channel_metrics_table=f"{schema}.channel_metrics", + channels_uri=f"{schema}.channels", ), unity_sink=UnitySink( - catalog="spark_catalog", schema="gold", table_prefix="time_window_join_test" + catalog="spark_catalog", schema="gold", table_prefix=table_prefix ), container_filters=ContainerFilters( metric_filters=[ @@ -197,44 +225,35 @@ def test_time_window_event_aggregation_join(spark, setup_tw_aligned_db): ), query_engine=QueryEngine(solver=Solvers.KEY_VALUE_STORE_SOLVER), measurement_dimensions=["container_id", "start_ts", "stop_ts"], + **extra, ) ) - my_report = Report( - name="time_window_join_report", - spark=spark, - workspace_client=create_autospec(WorkspaceClient), - config=config, - ) - window_evt = TimeWindowEvent(name="ten_min", window_length=ALIGNED_WINDOW_LENGTH) - my_report.add_event(window_evt) - query = my_report.get_db().query - page = Page(page_number=1) - my_report.add_page(page) - page.add_aggregation( - StatsAggregator( - name="rpm_stats_per_window", - input_expressions=[query.channel(channel_name="Engine RPM")], - channel_names=["Engine RPM"], - statistics=["min", "max", "mean"], - event=window_evt, - desc="Engine RPM stats per window", - ) +def _rpm_stats(report: Report, event: TimeWindowEvent, statistics=("min", "max", "mean")): + query = report.get_db().query + return StatsAggregator( + name="rpm_stats_per_window", + input_expressions=[query.channel(channel_name="Engine RPM")], + channel_names=["Engine RPM"], + statistics=list(statistics), + event=event, + desc="Engine RPM stats per window", ) - my_report.determine_report() - my_report.persist_results() - stats_fact = spark.read.table("spark_catalog.gold.time_window_join_test_stats_aggregator_fact") +def _assert_ids_join(spark, table_prefix: str) -> tuple[set, set]: # noqa: F811 + """Assert every stats event_instance_id exists in event_instance_fact, with real values. + + Returns ``(stats_event_ids, event_ids)`` for further checks. + """ + stats_fact = spark.read.table(f"spark_catalog.gold.{table_prefix}_stats_aggregator_fact") event_instance_fact = spark.read.table( - "spark_catalog.gold.time_window_join_test_event_instance_fact" + f"spark_catalog.gold.{table_prefix}_event_instance_fact" ) - assert stats_fact.count() > 0 - assert event_instance_fact.count() > 0 # Real computed values: with aligned boundaries every window overlaps RPM samples, - # so each produces a positive max. + # so the windows produce a positive max. max_values = [ r.statistic_value for r in stats_fact.filter(F.col("aggregation_label") == "max").collect() @@ -259,6 +278,33 @@ def test_time_window_event_aggregation_join(spark, setup_tw_aligned_db): assert stats_event_ids.issubset( event_ids ), f"stats event_instance_ids not in event_instance_fact: {stats_event_ids - event_ids}" + return stats_event_ids, event_ids + + +@pytest.mark.parametrize("setup_tw_aligned_db", ["us", "ns", "sec"], indirect=True) +def test_time_window_event_aggregation_join(spark, setup_tw_aligned_db): + """Stats scoped to a TimeWindowEvent yield per-window values whose event_instance_id + joins to the natively computed event fact, for µs, ns and seconds-as-double time bases.""" + schema, window_length = setup_tw_aligned_db + table_prefix = f"time_window_join_test_{schema.rsplit('_', 1)[-1]}" + my_report = Report( + name="time_window_join_report", + spark=spark, + workspace_client=create_autospec(WorkspaceClient), + config=_aligned_config(schema, table_prefix), + ) + + window_evt = TimeWindowEvent(name="ten_min", window_length=window_length) + my_report.add_event(window_evt) + + page = Page(page_number=1) + my_report.add_page(page) + page.add_aggregation(_rpm_stats(my_report, window_evt)) + + my_report.determine_report() + my_report.persist_results() + + _assert_ids_join(spark, table_prefix) def test_multiple_time_window_events_coexist(spark, basic_narrow_db): @@ -303,3 +349,156 @@ def test_multiple_time_window_events_coexist(spark, basic_narrow_db): dim_rows = my_report.event_metadata_dfs["TIME_WINDOW_EVENT"].collect() assert {d.event_name for d in dim_rows} == {"ten_sec", "thirty_sec"} + + +# --------------------------------------------------------------------------- +# Coverage: windows for every filtered container, independent of channel data +# --------------------------------------------------------------------------- +_PARTIAL_SCHEMA = "spark_catalog.silver_tw_partial" +_RPM_CHANNEL_ID = 5 + + +@pytest.fixture +def setup_tw_partial_db(spark, setup_basic_db): # noqa: F811 + """Basic silver clone where container 3 has no Engine RPM channel (metrics or data).""" + spark.sql(f"CREATE SCHEMA IF NOT EXISTS {_PARTIAL_SCHEMA}") + no_rpm_on_3 = ~((F.col("container_id") == 3) & (F.col("channel_id") == _RPM_CHANNEL_ID)) + spark.read.table("spark_catalog.silver.container_metrics").write.format("delta").mode( + "overwrite" + ).saveAsTable(f"{_PARTIAL_SCHEMA}.container_metrics") + for table in ("channel_metrics", "channels"): + spark.read.table(f"spark_catalog.silver.{table}").filter(no_rpm_on_3).write.format( + "delta" + ).mode("overwrite").saveAsTable(f"{_PARTIAL_SCHEMA}.{table}") + yield + spark.sql(f"DROP SCHEMA IF EXISTS {_PARTIAL_SCHEMA} CASCADE") + + +def _assert_windows_for_all_containers(rows) -> None: + for container_id in EXPECTED_CONTAINERS: + count = sum(1 for r in rows if r.container_id == container_id) + assert count == _expected_window_count(container_id), ( + f"container {container_id}: expected {_expected_window_count(container_id)} " + f"windows, got {count}" + ) + + +def test_time_window_event_covers_containers_without_aggregated_channel( + spark, setup_tw_partial_db +): + """Windows exist for every filtered container even when the scoped aggregation's + channel is missing on some of them and the solve is split into single-channel batches. + + Previously the windows came from the channel solve, so container 3 (no Engine RPM) + got none whenever the window expression landed in the RPM batch.""" + config = _config("time_window_partial_test") + config.source = Source( + container_metrics_table=f"{_PARTIAL_SCHEMA}.container_metrics", + channel_metrics_table=f"{_PARTIAL_SCHEMA}.channel_metrics", + channels_uri=f"{_PARTIAL_SCHEMA}.channels", + ) + config.query_engine = QueryEngine( + solver=Solvers.KEY_VALUE_STORE_SOLVER, max_channels_per_batch=1 + ) + my_report = Report( + name="time_window_partial_report", + spark=spark, + workspace_client=create_autospec(WorkspaceClient), + config=dict(config), + ) + window_evt = TimeWindowEvent(name="ten_sec", window_length=WINDOW_LENGTH) + my_report.add_event(window_evt) + page = Page(page_number=1) + my_report.add_page(page) + page.add_aggregation(_rpm_stats(my_report, window_evt)) + + my_report.determine_report() + + rows = my_report.event_dfs["TIME_WINDOW_EVENT"]["changed"].collect() + _assert_windows_for_all_containers(rows) + + +def test_standalone_time_window_event_covers_all_containers(spark, basic_narrow_db): + """A TimeWindowEvent with no aggregation (nothing to solve) still materializes windows.""" + my_report = Report( + name="time_window_standalone_report", + spark=spark, + workspace_client=create_autospec(WorkspaceClient), + config=dict(_config("time_window_standalone_test")), + ) + my_report.add_event(TimeWindowEvent(name="ten_sec", window_length=WINDOW_LENGTH)) + + my_report.determine_report() + + rows = my_report.event_dfs["TIME_WINDOW_EVENT"]["changed"].collect() + _assert_windows_for_all_containers(rows) + assert all(r.start_ts < r.end_ts for r in rows) + + +# --------------------------------------------------------------------------- +# Incremental: ids still join when the aggregation and the event use different scopes +# --------------------------------------------------------------------------- +def test_time_window_event_ids_join_after_incremental_run(spark, setup_tw_aligned_db): + """Run 1 (full) on containers 1-2; run 2 (incremental) adds container 3 and changes the + aggregation's definition. The changed aggregation recomputes over all containers while + the unchanged event only computes container 3, yet every stats id must still join.""" + schema, window_length = setup_tw_aligned_db + table_prefix = "time_window_inc_test" + cm_run_1 = f"{schema}.container_metrics_run_1" + cm_run_2 = f"{schema}.container_metrics_run_2" + past = F.lit("2020-01-01 00:00:00").cast("timestamp") + cm = spark.read.table(f"{schema}.container_metrics") + cm.filter(F.col("container_id").isin([1, 2])).withColumn("timestamp", past).write.format( + "delta" + ).mode("overwrite").saveAsTable(cm_run_1) + + def _run(cm_table: str, is_incremental: bool, statistics) -> None: + config = _aligned_config( + schema, + table_prefix, + incremental=IncrementalConfig( + enabled=is_incremental, + silver_last_modified_column="timestamp", + gold_last_modified_column="_created_at", + ), + ) + config["source"].container_metrics_table = cm_table + report = Report( + name="time_window_inc_report", + spark=spark, + workspace_client=create_autospec(WorkspaceClient), + config=config, + ) + window_evt = TimeWindowEvent(name="ten_min", window_length=window_length) + report.add_event(window_evt) + page = Page(page_number=1) + report.add_page(page) + page.add_aggregation(_rpm_stats(report, window_evt, statistics)) + report.determine_report() + report.persist_results() + + _run(cm_run_1, is_incremental=False, statistics=("min", "max", "mean")) + + # Container 3 is new (recent timestamp); 1 and 2 are unchanged. + cm.withColumn( + "timestamp", F.when(F.col("container_id") == 3, F.current_timestamp()).otherwise(past) + ).write.format("delta").mode("overwrite").saveAsTable(cm_run_2) + # Adding a statistic changes the aggregation's definition hash (event unchanged). + _run(cm_run_2, is_incremental=True, statistics=("min", "max", "mean", "median")) + + stats_event_ids, _ = _assert_ids_join(spark, table_prefix) + + event_fact = spark.read.table(f"spark_catalog.gold.{table_prefix}_event_instance_fact") + stats_fact = spark.read.table(f"spark_catalog.gold.{table_prefix}_stats_aggregator_fact") + assert {r.container_id for r in event_fact.select("container_id").distinct().collect()} == { + 1, + 2, + 3, + } + # The changed aggregation was recomputed for all containers (incl. the new one). + assert {r.container_id for r in stats_fact.select("container_id").distinct().collect()} == { + 1, + 2, + 3, + } + assert stats_fact.filter(F.col("aggregation_label") == "median").count() > 0 diff --git a/tests/impulse_reporting/unit/events/time_window_event_test.py b/tests/impulse_reporting/unit/events/time_window_event_test.py index 551f3293..3c1e4910 100644 --- a/tests/impulse_reporting/unit/events/time_window_event_test.py +++ b/tests/impulse_reporting/unit/events/time_window_event_test.py @@ -5,6 +5,8 @@ from impulse_query_engine.analyze.query.events.time_window_expression import ( TimeWindowExpression, ) +from impulse_reporting.events.container_boundary_event import ContainerBoundaryEvent +from impulse_reporting.events.container_event import ContainerEvent from impulse_reporting.events.time_window_event import TimeWindowEvent @@ -21,7 +23,17 @@ def test_init(): def test_init_surfaces_window_length_attribute(): event = TimeWindowEvent(name="w10", window_length=10000) - assert event.attributes["window_length"] == "10000" + assert event.attributes["window_length"] == "10000.0" + + +def test_window_length_normalized_across_int_and_float(): + # 10000 and 10000.0 are the same windows: event_dimension must not differ between them. + a = TimeWindowEvent(name="w", window_length=10000) + b = TimeWindowEvent(name="w", window_length=10000.0) + assert isinstance(a.window_length, float) + assert a.window_length == a.get_expression().window_length + assert a.attributes == b.attributes + assert a.as_dict() == b.as_dict() def test_init_does_not_override_user_window_length_attribute(): @@ -31,6 +43,15 @@ def test_init_does_not_override_user_window_length_attribute(): assert event.attributes["window_length"] == "custom" +def test_is_container_boundary_event_but_not_container_event(): + # Routed via the filter pipeline like ContainerEvent, but a sibling (not a subclass), so + # it keeps timestamp-based instance ids and is not limited to one per report. + event = TimeWindowEvent(name="w", window_length=10) + assert isinstance(event, ContainerBoundaryEvent) + assert not isinstance(event, ContainerEvent) + assert issubclass(ContainerEvent, ContainerBoundaryEvent) + + @pytest.mark.parametrize("bad", [0, -1, -5.5, None]) def test_non_positive_window_length_raises(bad): with pytest.raises(ValueError, match="strictly positive"): @@ -87,4 +108,4 @@ def test_as_dict_shape(): assert d["event_description"] == "ten second windows" assert d["required_channels"] == ["c1"] assert d["event_expression"] != "NA" - assert d["attributes"]["window_length"] == "10000" + assert d["attributes"]["window_length"] == "10000.0" From 588df91703e41c298b5e854d7243c5b874628ff5 Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Thu, 1 Oct 2026 15:15:04 +0200 Subject: [PATCH 07/27] fix(reporting): reject non-finite window_length in TimeWindowEvent and drop redundant interval guard Validate that `TimeWindowEvent` and `TimeWindowExpression` require a finite, strictly positive `window_length`, rejecting `inf`, `-inf`, and `nan` to prevent bogus or crashing Spark window generation. Remove the now-unnecessary `start_ts < end_ts` filter from `TimeWindowEvent.determine_events` since finite positive windows guarantee valid intervals. Rename `container_event_cls` to `boundary_event_cls` in `dispatch_events` for clarity. Update docs, skills, and tests accordingly. --- .../impulse_reporting/events/time_window_event.md | 4 ++-- docs/impulse/docs/references/report/event.md | 2 +- skills/impulse-events/SKILL.md | 2 +- .../analyze/query/events/time_window_expression.py | 12 ++++++++---- src/impulse_reporting/core/report_utils.py | 6 +++--- src/impulse_reporting/events/time_window_event.py | 10 +++++----- .../query/events/time_window_expression_test.py | 2 +- .../impulse_reporting/unit/core/report_utils_test.py | 10 +++++----- .../unit/events/time_window_event_test.py | 2 +- 9 files changed, 27 insertions(+), 23 deletions(-) diff --git a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md index e7f6606a..d883b8f9 100644 --- a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md +++ b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md @@ -38,7 +38,7 @@ Initialize a TimeWindowEvent object. - `name` (`str`): Name of the event. - `window_length` (`float`): Fixed window length, in the same time unit as the underlying timestamps -(e.g. milliseconds-since-epoch). Must be strictly positive. +(e.g. milliseconds-since-epoch). Must be strictly positive and finite. - `desc` (`str`): Description of the event. - `required_channels` (`list of str`): List of required channels for the event. Informational; stored in the event dimension table. @@ -47,7 +47,7 @@ automatically (without overriding a user-supplied key). **Raises**: -- `ValueError`: If ``window_length`` is not strictly positive. +- `ValueError`: If ``window_length`` is not strictly positive and finite. #### get\_id diff --git a/docs/impulse/docs/references/report/event.md b/docs/impulse/docs/references/report/event.md index ae59c0bf..20bf4ab4 100644 --- a/docs/impulse/docs/references/report/event.md +++ b/docs/impulse/docs/references/report/event.md @@ -219,7 +219,7 @@ my_report.add_event(ten_minute_windows) | Parameter | Type | Required | Description | |---------------------|---------------------|----------|-----------------------------------------------------------------------------------------------------------------| | `name` | `str` | Yes | Unique event name. | -| `window_length` | `float` | Yes | Fixed window length, **in the same time unit as the underlying timestamps** (e.g. milliseconds-since-epoch). Must be strictly positive; validated at construction. | +| `window_length` | `float` | Yes | Fixed window length, **in the same time unit as the underlying timestamps** (e.g. milliseconds-since-epoch). Must be strictly positive and finite; validated at construction. | | `desc` | `str` | No | Human-readable description. | | `required_channels` | `list[str]` | No | Channel names required for this event. Informational; stored in the event dimension table. | | `attributes` | `Mapping[str, str]` | No | Free-form key-value metadata. `window_length` is surfaced here automatically (without overriding a user key). | diff --git a/skills/impulse-events/SKILL.md b/skills/impulse-events/SKILL.md index 97e773e9..1942d01a 100644 --- a/skills/impulse-events/SKILL.md +++ b/skills/impulse-events/SKILL.md @@ -146,7 +146,7 @@ report.add_event(ten_minute) | Parameter | Type | Required | Description | |---------------------|---------------------|----------|---------------------------------------------------------------------------------| | `name` | `str` | Yes | Unique event name. | -| `window_length` | `float` | Yes | Fixed window length, **in the same time unit as the timestamps**. Must be > 0. | +| `window_length` | `float` | Yes | Fixed window length, **in the same time unit as the timestamps**. Must be finite and > 0. | | `desc` | `str` | No | Description. | | `required_channels` | `list[str]` | No | Informational. | | `attributes` | `Mapping[str, str]` | No | Free-form metadata; `window_length` is added automatically. | diff --git a/src/impulse_query_engine/analyze/query/events/time_window_expression.py b/src/impulse_query_engine/analyze/query/events/time_window_expression.py index 36632557..ab49a044 100644 --- a/src/impulse_query_engine/analyze/query/events/time_window_expression.py +++ b/src/impulse_query_engine/analyze/query/events/time_window_expression.py @@ -1,5 +1,7 @@ from __future__ import annotations +import math + import numpy as np import pyspark.sql.functions as F from pyspark.sql import Column @@ -93,16 +95,18 @@ def __init__(self, window_length: float): ---------- window_length : float Fixed window length, in the same time unit as the underlying timestamps - (e.g. milliseconds-since-epoch). Must be strictly positive. + (e.g. milliseconds-since-epoch). Must be strictly positive and finite. Raises ------ ValueError - If ``window_length`` is not strictly positive. + If ``window_length`` is not strictly positive and finite. """ - if window_length is None or window_length <= 0: + # inf / NaN must be rejected too: inf gives a zero window count, for which Spark's + # sequence(0, -1) emits a bogus window, and NaN crashes the solve in build(). + if window_length is None or not math.isfinite(window_length) or window_length <= 0: raise ValueError( - f"TimeWindowExpression requires a strictly positive window_length, " + f"TimeWindowExpression requires a strictly positive, finite window_length, " f"got {window_length!r}." ) # Store as float so the string form (and thus the event definition hash) is stable diff --git a/src/impulse_reporting/core/report_utils.py b/src/impulse_reporting/core/report_utils.py index fcbef84c..cde1c9f5 100644 --- a/src/impulse_reporting/core/report_utils.py +++ b/src/impulse_reporting/core/report_utils.py @@ -316,7 +316,7 @@ def dispatch_events( query: QueryBuilder, solver: QuerySolver, pre_filtered_containers_df: DataFrame | None, - container_event_cls: type, + boundary_event_cls: type, ) -> dict: """Dispatch ``determine_events`` calls per type. @@ -332,7 +332,7 @@ def dispatch_events( query : QueryBuilder solver : QuerySolver pre_filtered_containers_df : DataFrame | None - container_event_cls : type + boundary_event_cls : type Base class of the container-boundary events (``ContainerBoundaryEvent``); subclasses are resolved via the filter pipeline instead of ``solved_df``. @@ -348,7 +348,7 @@ def dispatch_events( continue cls = type_enum[type_name].value - if issubclass(cls, container_event_cls): + if issubclass(cls, boundary_event_cls): # Container-boundary events use the filter pipeline, not solved_df event_dfs[type_name] = cls.determine_events( spark, diff --git a/src/impulse_reporting/events/time_window_event.py b/src/impulse_reporting/events/time_window_event.py index a7ab79c8..bf9a238b 100644 --- a/src/impulse_reporting/events/time_window_event.py +++ b/src/impulse_reporting/events/time_window_event.py @@ -3,6 +3,7 @@ from __future__ import annotations import hashlib +import math from collections.abc import Mapping import pyspark.sql.functions as f @@ -58,7 +59,7 @@ def __init__( Name of the event. window_length : float Fixed window length, in the same time unit as the underlying timestamps - (e.g. milliseconds-since-epoch). Must be strictly positive. + (e.g. milliseconds-since-epoch). Must be strictly positive and finite. desc : str, optional Description of the event. required_channels : list of str, optional @@ -71,12 +72,12 @@ def __init__( Raises ------ ValueError - If ``window_length`` is not strictly positive. + If ``window_length`` is not strictly positive and finite. """ ContainerBoundaryEvent.__init__(self, name) - if window_length is None or window_length <= 0: + if window_length is None or not math.isfinite(window_length) or window_length <= 0: raise ValueError( - f"TimeWindowEvent requires a strictly positive window_length, " + f"TimeWindowEvent requires a strictly positive, finite window_length, " f"got {window_length!r}." ) self.expression = TimeWindowExpression(window_length).alias(name) @@ -265,7 +266,6 @@ def determine_events( ReportEntityUtil.get_event_id_column(elements=events, element_name="event_name"), ) .select(EVENT_INSTANCE_FACT_SCHEMA.fieldNames()) - .where(f.col("start_ts") < f.col("end_ts")) # Ensure valid time intervals ) return df diff --git a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py index d5149ad6..fdb421d3 100644 --- a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py @@ -119,7 +119,7 @@ def test_str_stable_across_int_and_float_window_length(): assert str(TimeWindowExpression(10)) == str(TimeWindowExpression(10.0)) -@pytest.mark.parametrize("bad", [0, -1, -10.5, None]) +@pytest.mark.parametrize("bad", [0, -1, -10.5, None, float("inf"), float("-inf"), float("nan")]) def test_non_positive_window_length_raises(bad): with pytest.raises(ValueError, match="strictly positive"): TimeWindowExpression(bad) diff --git a/tests/impulse_reporting/unit/core/report_utils_test.py b/tests/impulse_reporting/unit/core/report_utils_test.py index 01551557..61faab4c 100644 --- a/tests/impulse_reporting/unit/core/report_utils_test.py +++ b/tests/impulse_reporting/unit/core/report_utils_test.py @@ -495,7 +495,7 @@ def test_empty_events_by_type_returns_empty(self): query=MagicMock(), solver=MagicMock(), pre_filtered_containers_df=None, - container_event_cls=object, + boundary_event_cls=object, ) assert event_dfs == {} @@ -532,7 +532,7 @@ def determine_metadata_df(cls, spark, events): query=MagicMock(), solver=MagicMock(), pre_filtered_containers_df=MagicMock(spec=DataFrame), - container_event_cls=FakeContainerBase, + boundary_event_cls=FakeContainerBase, ) assert received_kwargs == {"solved_df": mock_solved} @@ -573,7 +573,7 @@ def determine_metadata_df(cls, spark, events): query=mock_query, solver=mock_solver, pre_filtered_containers_df=mock_pre_filtered, - container_event_cls=FakeContainerBase, + boundary_event_cls=FakeContainerBase, ) assert received_kwargs["query"] is mock_query @@ -594,7 +594,7 @@ def test_empty_event_list_for_type_is_skipped(self): query=MagicMock(), solver=MagicMock(), pre_filtered_containers_df=None, - container_event_cls=object, + boundary_event_cls=object, ) assert "BASIC_EVENT" not in event_dfs @@ -633,7 +633,7 @@ def determine_metadata_df(cls, spark, events): query=MagicMock(), solver=MagicMock(), pre_filtered_containers_df=None, - container_event_cls=FakeContainerBase, + boundary_event_cls=FakeContainerBase, ) assert len(meta_calls) == 0 diff --git a/tests/impulse_reporting/unit/events/time_window_event_test.py b/tests/impulse_reporting/unit/events/time_window_event_test.py index 3c1e4910..78f0a25e 100644 --- a/tests/impulse_reporting/unit/events/time_window_event_test.py +++ b/tests/impulse_reporting/unit/events/time_window_event_test.py @@ -52,7 +52,7 @@ def test_is_container_boundary_event_but_not_container_event(): assert issubclass(ContainerEvent, ContainerBoundaryEvent) -@pytest.mark.parametrize("bad", [0, -1, -5.5, None]) +@pytest.mark.parametrize("bad", [0, -1, -5.5, None, float("inf"), float("-inf"), float("nan")]) def test_non_positive_window_length_raises(bad): with pytest.raises(ValueError, match="strictly positive"): TimeWindowEvent(name="bad", window_length=bad) From 7ecc300a57757bacb79f8d3767aa671e4f427c63 Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Thu, 1 Oct 2026 16:17:20 +0200 Subject: [PATCH 08/27] feat(query-engine, reporting): support TIMESTAMP container boundaries via solver_config.epoch_unit Add an opt-in `solver_config.epoch_unit` setting (`"s"`, `"ms"`, `"us"`, `"ns"`) that converts `TIMESTAMP`-typed `container_metrics.start_ts` / `stop_ts` into epoch numbers. This lets `ContainerEvent` and `TimeWindowEvent` share the same time base as the channel sample timestamps, which is required for `TimeWindowEvent` when boundaries are `TIMESTAMP` columns. - Implement `SolverConfig.normalize_container_boundaries` and `require_epoch_boundaries` for conversion and fail-fast validation. - Apply normalization in `ContainerBoundaryEvent` event resolution and in the default solver's container metadata path. - Raise a clear `TypeError` in `TimeWindowExpression` when unconverted datetime boundaries reach pandas. - Add unit and integration tests covering all epoch units, timezone independence, TIMESTAMP_NTZ/DATE rejection, and `TimeWindowEvent` aggregation joins with TIMESTAMP boundaries. - Update configuration, schema, API, and event reference docs. --- docs/impulse/docs/config/configuration.md | 14 +- .../docs/data_model/silver_layer_schema.md | 4 + .../analyze/query/solvers/solver_config.md | 68 +++++++++ docs/impulse/docs/references/report/event.md | 6 + .../query/events/time_window_expression.py | 11 ++ .../analyze/query/solvers/default_solver.py | 4 + .../analyze/query/solvers/solver_config.py | 97 +++++++++++++ .../events/container_boundary_event.py | 9 +- .../events/time_window_event.py | 3 + .../events/time_window_expression_test.py | 8 ++ .../solvers/container_boundaries_test.py | 124 +++++++++++++++++ .../default_solver_container_metadata_test.py | 57 ++++++++ .../integration/time_window_event_test.py | 131 ++++++++++++++---- 13 files changed, 507 insertions(+), 29 deletions(-) create mode 100644 tests/impulse_query_engine/unit/analyze/query/solvers/container_boundaries_test.py diff --git a/docs/impulse/docs/config/configuration.md b/docs/impulse/docs/config/configuration.md index 531fd8b2..41013167 100644 --- a/docs/impulse/docs/config/configuration.md +++ b/docs/impulse/docs/config/configuration.md @@ -170,6 +170,18 @@ Top-level fields on `SolverConfig`: the `project_id` column (after column-name mapping) of every table it reads that carries one — `container_tags` (if configured), `container_metrics`, and `channel_mapping` (if configured). Omit it if you don't need project-level scoping; the solver does not require it. +- `epoch_unit` (`"s"` | `"ms"` | `"us"` | `"ns"`, optional): Epoch unit of the channel sample + timestamps (`tstart`/`tend`). Only needed when `container_metrics.start_ts`/`stop_ts` are + `TIMESTAMP` columns **and** the report uses a `TimeWindowEvent`, whose windows must be in the + samples' time base. Such a report fails with a clear error until it is set. When set, `TIMESTAMP` + boundaries are converted to epoch numbers in that unit: + - `ContainerEvent` and `TimeWindowEvent` write `start_ts`/`end_ts` in that unit. With `"s"`, the + values are identical to the default. + - Expressions that request `start_ts`/`stop_ts` as container metrics (e.g. via + `apply(..., container_metrics=[...])`) receive epoch numbers instead of timestamps. + + `measurement_dimension` and container filters always see the original columns. When unset, + nothing is converted. `TIMESTAMP_NTZ` and `DATE` boundaries are not supported. Per-table sections (each a `TableConfig`): @@ -192,7 +204,7 @@ Internal column names that mappings can target: | `tstart`, `tend`| Sample interval start/end on the `channels` table (RLE) | | `timestamp` | Raw sample timestamp on the `channels` table (RAW mode; encoded into `tstart`/`tend`) | | `is_plausible` | Boolean plausibility flag on the `channels` table (RAW mode); consumed by `drop_implausible_data` | -| `start_ts`, `stop_ts` | Measurement start/stop epoch timestamps on the `container_metrics` table — referenced by `ContainerEvent` to derive event-fact start/end | +| `start_ts`, `stop_ts` | Measurement start/stop epoch timestamps on the `container_metrics` table — referenced by `ContainerEvent` and `TimeWindowEvent` to derive event-fact start/end. May be `TIMESTAMP` (see `epoch_unit`) | | `value` | Sample value (or attribute value on the EAV tag table) | | `key` | Attribute key on the EAV `container_tags` table | | `priority` | Tie-breaker column on the `channel_mapping` table | diff --git a/docs/impulse/docs/data_model/silver_layer_schema.md b/docs/impulse/docs/data_model/silver_layer_schema.md index d5ce7508..54f8b338 100644 --- a/docs/impulse/docs/data_model/silver_layer_schema.md +++ b/docs/impulse/docs/data_model/silver_layer_schema.md @@ -180,6 +180,10 @@ for human-readable display, `start_ts`/`stop_ts` for the gold the epoch-typed pair). Populate whichever your queries and `measurement_dimensions` config need. +`start_ts`/`stop_ts` may also be `TIMESTAMP` columns. To use them with a `TimeWindowEvent`, set +[`solver_config.epoch_unit`](../config/configuration.md#solver-column-mappings-and-filters) to the +epoch unit of the channel sample timestamps, so the window boundaries share the samples' time base. + ::: --- diff --git a/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md b/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md index e34c236e..74d57768 100644 --- a/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md +++ b/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md @@ -126,6 +126,12 @@ so that solver code can always reference the same constants. override for the channel mapping (alias) table. - `channels` (`TableConfig`): Column mappings and filters for the channel data table. - `unit_conversion` (`TableConfig`): Column mappings and filters for the unit conversion table. +- `epoch_unit` (`{"s", "ms", "us", "ns"} or None`): Epoch unit of the channel sample timestamps (``tstart`` / ``tend``). When set, +``TIMESTAMP``-typed container ``start_ts`` / ``stop_ts`` are converted to epoch +numbers in this unit for container-boundary events (``ContainerEvent``, +``TimeWindowEvent``) and for expressions that request them in the solve. Only +required for a ``TimeWindowEvent`` over ``TIMESTAMP`` boundaries; when unset, +nothing is converted. #### from\_json @@ -457,3 +463,65 @@ def col_map() -> dict[str, str] Short-key → internal-column-name mapping for UDFs and caches. +#### reject\_implausible\_channels\_filter\_in\_raw + +```python +def reject_implausible_channels_filter_in_raw(is_raw: bool) -> None +``` + +Raise if an is_plausible channels filter is set in RAW mode. + +Such a filter runs before raw encoding and bridges intervals across dropped +samples instead of splitting them; use drop_implausible_data instead. No-op +when not raw. + + +#### normalize\_container\_boundaries + +```python +def normalize_container_boundaries(df: DataFrame) -> DataFrame +``` + +Convert ``TIMESTAMP`` container start/stop columns to epoch numbers. + +Opt-in via :attr:`epoch_unit`: when it is unset, *df* is returned unchanged. +Otherwise each ``TIMESTAMP`` ``start_ts`` / ``stop_ts`` column becomes an epoch +number in that unit, computed from ``unix_micros`` (exact and independent of the +session time zone). ``"s"`` / ``"ms"`` give doubles (``"s"`` equals Spark's +``cast(timestamp as double)``), ``"us"`` / ``"ns"`` give longs. Numeric columns +are left as they are. Applying the same transform before both the event fact +and the solve keeps their window boundaries identical. + +**Arguments**: + +- `df` (`pyspark.sql.DataFrame`): Column-mapped ``container_metrics`` frame (or a projection of it). + +**Raises**: + +- `ValueError`: If :attr:`epoch_unit` is set and a boundary column is ``TIMESTAMP_NTZ`` or +``DATE`` (their epoch depends on a time zone and is not supported). + +**Returns**: + +`pyspark.sql.DataFrame`: *df* with converted boundary columns. + +#### require\_epoch\_boundaries + +```python +def require_epoch_boundaries(df: DataFrame, owner: str) -> None +``` + +Raise unless the container start/stop columns on *df* are epoch numbers. + +Call after :meth:`normalize_container_boundaries`. Checks the schema only, so it +fails fast on the driver before any Spark job runs. + +**Arguments**: + +- `df` (`pyspark.sql.DataFrame`): Normalized container_metrics frame. +- `owner` (`str`): Name of the feature that needs epoch boundaries, used in the error message. + +**Raises**: + +- `ValueError`: If ``start_ts`` / ``stop_ts`` is still a date/time type. + diff --git a/docs/impulse/docs/references/report/event.md b/docs/impulse/docs/references/report/event.md index 20bf4ab4..7d8097ea 100644 --- a/docs/impulse/docs/references/report/event.md +++ b/docs/impulse/docs/references/report/event.md @@ -229,6 +229,12 @@ my_report.add_event(ten_minute_windows) the same time unit as the stored timestamps (milliseconds-since-epoch in the sample data), not seconds or any derived unit. So 60 one-minute windows over millisecond timestamps use `window_length=60_000`. + +If `container_metrics.start_ts`/`stop_ts` are `TIMESTAMP` columns, set +[`solver_config.epoch_unit`](../../config/configuration.md#solver-column-mappings-and-filters) +to the epoch unit of the channel sample timestamps (e.g. `"s"`). The boundaries are converted to +that unit, and `window_length` is expressed in it. Without it, the report fails with an error +naming the setting. ::: ### How it works diff --git a/src/impulse_query_engine/analyze/query/events/time_window_expression.py b/src/impulse_query_engine/analyze/query/events/time_window_expression.py index ab49a044..201651c0 100644 --- a/src/impulse_query_engine/analyze/query/events/time_window_expression.py +++ b/src/impulse_query_engine/analyze/query/events/time_window_expression.py @@ -1,5 +1,6 @@ from __future__ import annotations +import datetime import math import numpy as np @@ -222,6 +223,16 @@ def build(self, cache: SeriesCache) -> Intervals: if start_ts is None or stop_ts is None: return Intervals.empty() + for name, value in (("start_ts", start_ts), ("stop_ts", stop_ts)): + # pd.Timestamp subclasses datetime.datetime; dates and numpy datetimes too. + if isinstance(value, (datetime.date, np.datetime64)): + raise TypeError( + f"TimeWindowExpression needs epoch-number container boundaries, but " + f"{name} is {type(value).__name__}. For TIMESTAMP columns, set " + "solver_config.epoch_unit to the epoch unit of the channel sample " + "timestamps so they are converted before the solve." + ) + # Mirror window_intervals_col exactly: convert to double *before* subtracting. A # long column reaches pandas as int64 or float64 depending on the group (nulls # force float64), and an exact int64 span can round differently from the double diff --git a/src/impulse_query_engine/analyze/query/solvers/default_solver.py b/src/impulse_query_engine/analyze/query/solvers/default_solver.py index d8992443..051c594e 100644 --- a/src/impulse_query_engine/analyze/query/solvers/default_solver.py +++ b/src/impulse_query_engine/analyze/query/solvers/default_solver.py @@ -1314,6 +1314,10 @@ def _build_container_metadata_df( meta_df = metrics.select(container_id_col, *metric_cols).dropDuplicates( [container_id_col] ) + # TIMESTAMP start_ts/stop_ts would reach pandas as session-local, tz-naive + # Timestamps; convert them to epoch numbers in Spark when epoch_unit is set + # (no-op otherwise), matching the container-boundary events. + meta_df = self.config.normalize_container_boundaries(meta_df) if tag_keys: if query.db.config.container_tags_table is None: diff --git a/src/impulse_query_engine/analyze/query/solvers/solver_config.py b/src/impulse_query_engine/analyze/query/solvers/solver_config.py index 813f6c44..65d51938 100644 --- a/src/impulse_query_engine/analyze/query/solvers/solver_config.py +++ b/src/impulse_query_engine/analyze/query/solvers/solver_config.py @@ -16,8 +16,12 @@ import json from enum import StrEnum +from typing import Literal +import pyspark.sql.functions as F +import pyspark.sql.types as T from pydantic import BaseModel +from pyspark.sql import Column, DataFrame class RawEncoder(StrEnum): @@ -133,9 +137,17 @@ class SolverConfig(BaseModel): Column mappings and filters for the channel data table. unit_conversion : TableConfig Column mappings and filters for the unit conversion table. + epoch_unit : {"s", "ms", "us", "ns"} or None + Epoch unit of the channel sample timestamps (``tstart`` / ``tend``). When set, + ``TIMESTAMP``-typed container ``start_ts`` / ``stop_ts`` are converted to epoch + numbers in this unit for container-boundary events (``ContainerEvent``, + ``TimeWindowEvent``) and for expressions that request them in the solve. Only + required for a ``TimeWindowEvent`` over ``TIMESTAMP`` boundaries; when unset, + nothing is converted. """ project_id: str | None = None + epoch_unit: Literal["s", "ms", "us", "ns"] | None = None container_tags: TableConfig = TableConfig() container_metrics: TableConfig = TableConfig() @@ -417,3 +429,88 @@ def reject_implausible_channels_filter_in_raw(self, is_raw: bool) -> None: "samples. Use drop_implausible_data=True instead -- it drops " "implausible points inside the encoder with correct interval boundaries." ) + + def _boundary_fields(self, df: DataFrame) -> list[T.StructField]: + """Return the container start/stop timestamp fields present on *df*.""" + names = {self.start_ts_col, self.stop_ts_col} + return [field for field in df.schema.fields if field.name in names] + + def normalize_container_boundaries(self, df: DataFrame) -> DataFrame: + """Convert ``TIMESTAMP`` container start/stop columns to epoch numbers. + + Opt-in via :attr:`epoch_unit`: when it is unset, *df* is returned unchanged. + Otherwise each ``TIMESTAMP`` ``start_ts`` / ``stop_ts`` column becomes an epoch + number in that unit, computed from ``unix_micros`` (exact and independent of the + session time zone). ``"s"`` / ``"ms"`` give doubles (``"s"`` equals Spark's + ``cast(timestamp as double)``), ``"us"`` / ``"ns"`` give longs. Numeric columns + are left as they are. Applying the same transform before both the event fact + and the solve keeps their window boundaries identical. + + Parameters + ---------- + df : pyspark.sql.DataFrame + Column-mapped ``container_metrics`` frame (or a projection of it). + + Returns + ------- + pyspark.sql.DataFrame + *df* with converted boundary columns. + + Raises + ------ + ValueError + If :attr:`epoch_unit` is set and a boundary column is ``TIMESTAMP_NTZ`` or + ``DATE`` (their epoch depends on a time zone and is not supported). + """ + if self.epoch_unit is None: + return df + for field in self._boundary_fields(df): + if isinstance(field.dataType, T.TimestampType): + df = df.withColumn(field.name, self._epoch_from_timestamp(F.col(field.name))) + elif isinstance(field.dataType, (T.TimestampNTZType, T.DateType)): + raise ValueError( + f"container_metrics column '{field.name}' has type " + f"{field.dataType.simpleString()}, which cannot be converted to an epoch " + "unambiguously (it carries no time zone). Use a TIMESTAMP or epoch-number " + "column." + ) + return df + + def _epoch_from_timestamp(self, col: Column) -> Column: + """Epoch value of a TIMESTAMP column in :attr:`epoch_unit`.""" + micros = F.unix_micros(col) + if self.epoch_unit == "s": + return micros / F.lit(1e6) + if self.epoch_unit == "ms": + return micros / F.lit(1e3) + if self.epoch_unit == "ns": + return micros * F.lit(1000) + return micros + + def require_epoch_boundaries(self, df: DataFrame, owner: str) -> None: + """Raise unless the container start/stop columns on *df* are epoch numbers. + + Call after :meth:`normalize_container_boundaries`. Checks the schema only, so it + fails fast on the driver before any Spark job runs. + + Parameters + ---------- + df : pyspark.sql.DataFrame + Normalized container_metrics frame. + owner : str + Name of the feature that needs epoch boundaries, used in the error message. + + Raises + ------ + ValueError + If ``start_ts`` / ``stop_ts`` is still a date/time type. + """ + datetime_types = (T.TimestampType, T.TimestampNTZType, T.DateType) + for field in self._boundary_fields(df): + if isinstance(field.dataType, datetime_types): + raise ValueError( + f"{owner} needs epoch-number container boundaries, but container_metrics " + f"column '{field.name}' has type {field.dataType.simpleString()}. Set " + "query_engine.solver_config.epoch_unit to the epoch unit of the channel " + "sample timestamps (one of 's', 'ms', 'us', 'ns') so it is converted." + ) diff --git a/src/impulse_reporting/events/container_boundary_event.py b/src/impulse_reporting/events/container_boundary_event.py index bbe40928..b1f9dc02 100644 --- a/src/impulse_reporting/events/container_boundary_event.py +++ b/src/impulse_reporting/events/container_boundary_event.py @@ -42,9 +42,14 @@ def resolve_container_metrics( Returns ------- DataFrame - Column-mapped ``container_metrics`` rows of the matching containers. + Column-mapped ``container_metrics`` rows of the matching containers, with + ``TIMESTAMP`` boundaries converted to epoch numbers when + ``solver.config.epoch_unit`` is set (unchanged otherwise). """ container_tags_df = solver.filter_container_tags(spark, query) - return solver.filter_container_metrics( + container_metrics_df = solver.filter_container_metrics( spark, query, container_tags_df, pre_filtered_containers_df ) + # Same transform as the solve's container metadata, so the event boundaries and + # those seen by scoped aggregations are identical. + return solver.config.normalize_container_boundaries(container_metrics_df) diff --git a/src/impulse_reporting/events/time_window_event.py b/src/impulse_reporting/events/time_window_event.py index bf9a238b..7117858f 100644 --- a/src/impulse_reporting/events/time_window_event.py +++ b/src/impulse_reporting/events/time_window_event.py @@ -227,6 +227,9 @@ def determine_events( container_metrics_df = cls.resolve_container_metrics( spark, query, solver, pre_filtered_containers_df ) + # Windows are computed in the channel samples' epoch unit, so TIMESTAMP boundaries + # need solver_config.epoch_unit (fails fast on the schema, before any Spark job). + solver.config.require_epoch_boundaries(container_metrics_df, owner="TimeWindowEvent") # Silver-side names come from SolverConfig (column_name_mapping aware). start_ts = f.col(solver.config.start_ts_col) diff --git a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py index fdb421d3..194dd462 100644 --- a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py @@ -4,6 +4,7 @@ from unittest.mock import MagicMock import numpy as np +import pandas as pd import pyspark.sql.functions as F import pytest @@ -134,6 +135,13 @@ def test_int64_and_float64_inputs_build_identical_windows(): assert a.get_data() == b.get_data() +def test_datetime_container_metrics_raise_clear_error(): + # TIMESTAMP boundaries reach pandas as pd.Timestamp unless epoch_unit converts them. + start, stop = pd.Timestamp("2025-07-03 07:41:41"), pd.Timestamp("2025-07-03 07:43:30") + with pytest.raises(TypeError, match="epoch_unit"): + _build(start, stop, 10) + + def test_nan_container_metrics_yield_empty(): # A null start/stop arrives as NaN in a float64 column. assert len(_build(np.nan, 100.0, 10)) == 0 diff --git a/tests/impulse_query_engine/unit/analyze/query/solvers/container_boundaries_test.py b/tests/impulse_query_engine/unit/analyze/query/solvers/container_boundaries_test.py new file mode 100644 index 00000000..46298ae1 --- /dev/null +++ b/tests/impulse_query_engine/unit/analyze/query/solvers/container_boundaries_test.py @@ -0,0 +1,124 @@ +# pylint: disable=missing-function-docstring, redefined-outer-name +"""Tests for SolverConfig.normalize_container_boundaries / require_epoch_boundaries. + +TIMESTAMP-typed container ``start_ts`` / ``stop_ts`` are converted to epoch numbers in +``SolverConfig.epoch_unit`` (opt-in), so container-boundary events and the solve see the +same values. With ``epoch_unit`` unset nothing changes. +""" + +import datetime as dt + +import pyspark.sql.functions as F +import pyspark.sql.types as T +import pytest +from pyspark.sql import SparkSession + +from impulse_query_engine.analyze.query.solvers.solver_config import SolverConfig +from tests.conftest import spark # noqa: F401 (pytest fixture) + +# 2025-07-03 07:41:41.483456 UTC +_EPOCH_MICROS = 1_751_528_501_483_456 + + +def _boundaries_df(spark: SparkSession): # noqa: F811 + """container_metrics-like frame with TIMESTAMP boundaries (and a null row).""" + df = spark.createDataFrame( + [(1, _EPOCH_MICROS, _EPOCH_MICROS + 60_000_000), (2, None, None)], + "container_id int, start_us long, stop_us long", + ) + return df.select( + "container_id", + F.timestamp_micros("start_us").alias("start_ts"), + F.timestamp_micros("stop_us").alias("stop_ts"), + ) + + +@pytest.mark.parametrize( + "unit, expected_type, expected_start", + [ + ("s", T.DoubleType(), _EPOCH_MICROS / 1e6), + ("ms", T.DoubleType(), _EPOCH_MICROS / 1e3), + ("us", T.LongType(), _EPOCH_MICROS), + ("ns", T.LongType(), _EPOCH_MICROS * 1000), + ], +) +@pytest.mark.parametrize("session_tz", ["UTC", "Europe/Berlin"]) +def test_timestamp_boundaries_converted_to_epoch_unit( + spark, unit, expected_type, expected_start, session_tz # noqa: F811 +): + previous_tz = spark.conf.get("spark.sql.session.timeZone") + spark.conf.set("spark.sql.session.timeZone", session_tz) + try: + out = SolverConfig(epoch_unit=unit).normalize_container_boundaries(_boundaries_df(spark)) + rows = {r.container_id: r for r in out.collect()} + finally: + spark.conf.set("spark.sql.session.timeZone", previous_tz) + + assert out.schema["start_ts"].dataType == expected_type + assert out.schema["stop_ts"].dataType == expected_type + assert rows[1].start_ts == expected_start # exact, independent of the session time zone + assert rows[2].start_ts is None and rows[2].stop_ts is None + + +def test_seconds_match_spark_cast_to_double(spark): # noqa: F811 + # "s" must equal today's ContainerEvent cast(timestamp as double), bit for bit. + df = _boundaries_df(spark) + out = SolverConfig(epoch_unit="s").normalize_container_boundaries(df) + converted = {r.container_id: (r.start_ts, r.stop_ts) for r in out.collect()} + casted = { + r.container_id: (r.s, r.e) + for r in df.select( + "container_id", + F.col("start_ts").cast("double").alias("s"), + F.col("stop_ts").cast("double").alias("e"), + ).collect() + } + assert converted == casted + + +def test_unset_epoch_unit_leaves_frame_unchanged(spark): # noqa: F811 + df = _boundaries_df(spark) + out = SolverConfig().normalize_container_boundaries(df) + assert out is df + assert isinstance(out.schema["start_ts"].dataType, T.TimestampType) + + +def test_numeric_boundaries_unchanged(spark): # noqa: F811 + df = spark.createDataFrame([(1, 100, 200)], "container_id int, start_ts long, stop_ts long") + out = SolverConfig(epoch_unit="ms").normalize_container_boundaries(df) + assert out.schema == df.schema + assert out.collect() == df.collect() + + +@pytest.mark.parametrize( + "value, ddl", + [(dt.datetime(2025, 7, 3, 7, 41, 41), "timestamp_ntz"), (dt.date(2025, 7, 3), "date")], +) +def test_zone_less_types_rejected_when_unit_set(spark, value, ddl): # noqa: F811 + df = spark.createDataFrame( + [(1, value, value)], f"container_id int, start_ts {ddl}, stop_ts {ddl}" + ) + with pytest.raises(ValueError, match="start_ts"): + SolverConfig(epoch_unit="s").normalize_container_boundaries(df) + # Opt-in only: without epoch_unit the frame passes through untouched. + assert SolverConfig().normalize_container_boundaries(df) is df + + +def test_require_epoch_boundaries(spark): # noqa: F811 + df = _boundaries_df(spark) + with pytest.raises(ValueError, match=r"TimeWindowEvent.*epoch_unit"): + SolverConfig().require_epoch_boundaries(df, owner="TimeWindowEvent") + + cfg = SolverConfig(epoch_unit="s") + cfg.require_epoch_boundaries(cfg.normalize_container_boundaries(df), owner="TimeWindowEvent") + + numeric = spark.createDataFrame( + [(1, 1.0, 2.0)], "container_id int, start_ts double, stop_ts double" + ) + SolverConfig().require_epoch_boundaries(numeric, owner="TimeWindowEvent") + + +def test_epoch_unit_validated(): + assert SolverConfig.model_validate({"epoch_unit": "ns"}).epoch_unit == "ns" + with pytest.raises(ValueError): + SolverConfig.model_validate({"epoch_unit": "minutes"}) diff --git a/tests/impulse_query_engine/unit/analyze/query/solvers/default_solver_container_metadata_test.py b/tests/impulse_query_engine/unit/analyze/query/solvers/default_solver_container_metadata_test.py index b0293fe3..3f052762 100644 --- a/tests/impulse_query_engine/unit/analyze/query/solvers/default_solver_container_metadata_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/solvers/default_solver_container_metadata_test.py @@ -14,6 +14,7 @@ """ import pandas as pd +import pyspark.sql.functions as F import pytest from pyspark.sql import SparkSession @@ -329,3 +330,59 @@ def test_cache_reads_container_meta_from_surviving_row_only(): ) assert cache.container_tags == {"brand": "BMW"} assert cache.container_metrics["num_channels"] == 11 + + +def _timestamp_boundaries_db(basic_narrow_db: MeasurementDB) -> MeasurementDB: + """Clone of basic_narrow_db with start_ts / stop_ts (epoch ms) recast to TIMESTAMP.""" + tables = dict(basic_narrow_db.config.debug_tables) + tables["container_metrics"] = ( + tables["container_metrics"] + .withColumn("start_ts", F.timestamp_millis("start_ts")) + .withColumn("stop_ts", F.timestamp_millis("stop_ts")) + ) + return MeasurementDB(MeasurementDBConfig.for_debug(tables), ws=basic_narrow_db.ws) + + +def _grab_start_ts(ts, container_metrics): + value = container_metrics["start_ts"] + if value is None: # type-inference pass on the empty cache + return 0.0 + # Encode what reached the UDF: the epoch-seconds value, or -1 for a pd.Timestamp. + return -1.0 if isinstance(value, pd.Timestamp) else float(value) + + +def test_timestamp_boundaries_converted_for_udf_when_epoch_unit_set( + spark: SparkSession, basic_narrow_db: MeasurementDB +): + """With epoch_unit set, a TIMESTAMP start_ts reaches the UDF as epoch seconds.""" + db = _timestamp_boundaries_db(basic_narrow_db) + query = db.query + result = query.select( + query.channel(channel_name="Engine RPM") + .apply(_grab_start_ts, container_metrics=["start_ts"]) + .alias("start") + ).solve(spark, solver=DefaultSolver(spark, config=SolverConfig(epoch_unit="s"))) + + expected = { + r.container_id: r.s + for r in db.container_metrics(spark) + .select("container_id", F.col("start_ts").cast("double").alias("s")) + .collect() + } + rows = {row.container_id: row.start for row in result.collect()} + assert rows and all(rows[cid] == expected[cid] for cid in rows), (rows, expected) + + +def test_timestamp_boundaries_unchanged_for_udf_without_epoch_unit( + spark: SparkSession, basic_narrow_db: MeasurementDB +): + """Backward compatibility: without epoch_unit, the UDF still gets a pd.Timestamp.""" + query = _timestamp_boundaries_db(basic_narrow_db).query + result = query.select( + query.channel(channel_name="Engine RPM") + .apply(_grab_start_ts, container_metrics=["start_ts"]) + .alias("start") + ).solve(spark, solver=DefaultSolver(spark)) + + rows = [row.start for row in result.collect()] + assert rows and all(value == -1.0 for value in rows), rows diff --git a/tests/impulse_reporting/integration/time_window_event_test.py b/tests/impulse_reporting/integration/time_window_event_test.py index cc5b3510..ebe75c8f 100644 --- a/tests/impulse_reporting/integration/time_window_event_test.py +++ b/tests/impulse_reporting/integration/time_window_event_test.py @@ -3,9 +3,11 @@ from unittest.mock import create_autospec import pyspark.sql.functions as F +import pyspark.sql.types as T import pytest from databricks.sdk import WorkspaceClient +from impulse_query_engine.analyze.query.solvers.solver_config import SolverConfig from impulse_reporting.aggregations.stats_aggregator import StatsAggregator from impulse_reporting.config.config_parser import ( Comparator, @@ -20,6 +22,7 @@ ) from impulse_reporting.core.page import Page from impulse_reporting.core.report import Report +from impulse_reporting.events.container_event import ContainerEvent from impulse_reporting.events.time_window_event import TimeWindowEvent from tests.conftest import setup_basic_db, spark # noqa: F401 (pytest fixtures) @@ -141,24 +144,41 @@ def test_time_window_event_in_report(spark, basic_narrow_db): ALIGNED_WINDOW_LENGTH = 600_000_000 _ALIGNED_SCHEMA = "spark_catalog.silver_tw_aligned" -# Customer-shaped time bases for the id-join test. Each entry: (timestamp transform applied -# to channel tstart/tend and container start_ts/stop_ts, window length in that unit). -# us: the basic db's native µs epochs (< 2^53, every boundary exactly representable). -# ns: ns epochs (~1.5e18, beyond 2^53) with a window that is NOT a multiple of the -# 256 ns double spacing there, so the boundaries round. -# sec: seconds as doubles with a fractional window, so the boundaries round. + +# Customer-shaped time bases for the id-join test, all derived from the basic db's µs epochs. +# Each entry: (transform for channel tstart/tend, transform for container start_ts/stop_ts, +# window length in the samples' unit, SolverConfig.epoch_unit). +# us: the native µs epochs (< 2^53, every boundary exactly representable). +# ns: ns epochs (~1.5e18, beyond 2^53) with a window that is NOT a multiple of the +# 256 ns double spacing there, so the boundaries round. +# sec: seconds as doubles with a fractional window, so the boundaries round. +# sec_ts: samples as seconds-as-double, container boundaries as TIMESTAMP (converted to +# epoch seconds via epoch_unit="s"). +def _to_seconds(c): + return c.cast("double") / F.lit(1e6) + + +def _to_ns(c): + return c.cast("long") * F.lit(1000) + + _TIME_BASES = { - "us": (lambda c: c, ALIGNED_WINDOW_LENGTH), - "ns": (lambda c: c.cast("long") * F.lit(1000), 600_000_000_007), - "sec": (lambda c: c.cast("double") / F.lit(1e6), 600.3), + "us": (lambda c: c, lambda c: c, ALIGNED_WINDOW_LENGTH, None), + "ns": (_to_ns, _to_ns, 600_000_000_007, None), + "sec": (_to_seconds, _to_seconds, 600.3, None), + "sec_ts": (_to_seconds, lambda c: F.timestamp_micros(c.cast("long")), 600.3, "s"), } -def _clone_aligned_silver(spark, schema: str, to_time_base=lambda c: c) -> None: +def _clone_aligned_silver( + spark, schema: str, to_time_base=lambda c: c, boundaries_to_time_base=None +) -> None: """Clone the basic silver tables into *schema* with container_metrics start_ts / stop_ts recomputed from each container's channel-sample range (so the container boundaries, - and thus the windows, share the samples' time base), then map all of those timestamps - through *to_time_base*.""" + and thus the windows, share the samples' time base). Channel timestamps are then mapped + through *to_time_base* and the container boundaries through *boundaries_to_time_base* + (default: the same transform).""" + boundaries_to_time_base = boundaries_to_time_base or to_time_base spark.sql(f"CREATE SCHEMA IF NOT EXISTS {schema}") channels = spark.read.table("spark_catalog.silver.channels") bounds = channels.groupBy("container_id").agg( @@ -173,8 +193,8 @@ def _clone_aligned_silver(spark, schema: str, to_time_base=lambda c: c) -> None: .withColumn("start_ts", F.coalesce("_agg_start", "start_ts").cast(start_type)) .withColumn("stop_ts", F.coalesce("_agg_stop", "stop_ts").cast(stop_type)) .drop("_agg_start", "_agg_stop") - .withColumn("start_ts", to_time_base(F.col("start_ts"))) - .withColumn("stop_ts", to_time_base(F.col("stop_ts"))) + .withColumn("start_ts", boundaries_to_time_base(F.col("start_ts"))) + .withColumn("stop_ts", boundaries_to_time_base(F.col("stop_ts"))) ) aligned_cm.write.format("delta").mode("overwrite").option( "overwriteSchema", "true" @@ -193,17 +213,17 @@ def _clone_aligned_silver(spark, schema: str, to_time_base=lambda c: c) -> None: def setup_tw_aligned_db(spark, setup_basic_db, request): # noqa: F811 """Aligned silver clone in the time base given by ``request.param`` (default ``us``). - Yields ``(schema, window_length)``. + Yields ``(schema, window_length, epoch_unit)``. """ time_base = getattr(request, "param", "us") - to_time_base, window_length = _TIME_BASES[time_base] + to_time_base, boundaries_to_time_base, window_length, epoch_unit = _TIME_BASES[time_base] schema = f"{_ALIGNED_SCHEMA}_{time_base}" - _clone_aligned_silver(spark, schema, to_time_base) - yield schema, window_length + _clone_aligned_silver(spark, schema, to_time_base, boundaries_to_time_base) + yield schema, window_length, epoch_unit spark.sql(f"DROP SCHEMA IF EXISTS {schema} CASCADE") -def _aligned_config(schema: str, table_prefix: str, **extra) -> dict: +def _aligned_config(schema: str, table_prefix: str, epoch_unit=None, **extra) -> dict: return dict( ImpulseConfig( source=Source( @@ -223,7 +243,10 @@ def _aligned_config(schema: str, table_prefix: str, **extra) -> dict: ] ] ), - query_engine=QueryEngine(solver=Solvers.KEY_VALUE_STORE_SOLVER), + query_engine=QueryEngine( + solver=Solvers.KEY_VALUE_STORE_SOLVER, + solver_config=SolverConfig(epoch_unit=epoch_unit) if epoch_unit else None, + ), measurement_dimensions=["container_id", "start_ts", "stop_ts"], **extra, ) @@ -281,17 +304,18 @@ def _assert_ids_join(spark, table_prefix: str) -> tuple[set, set]: # noqa: F811 return stats_event_ids, event_ids -@pytest.mark.parametrize("setup_tw_aligned_db", ["us", "ns", "sec"], indirect=True) +@pytest.mark.parametrize("setup_tw_aligned_db", ["us", "ns", "sec", "sec_ts"], indirect=True) def test_time_window_event_aggregation_join(spark, setup_tw_aligned_db): """Stats scoped to a TimeWindowEvent yield per-window values whose event_instance_id - joins to the natively computed event fact, for µs, ns and seconds-as-double time bases.""" - schema, window_length = setup_tw_aligned_db - table_prefix = f"time_window_join_test_{schema.rsplit('_', 1)[-1]}" + joins to the natively computed event fact, for µs, ns and seconds-as-double time bases, + and for TIMESTAMP container boundaries converted via epoch_unit.""" + schema, window_length, epoch_unit = setup_tw_aligned_db + table_prefix = f"time_window_join_test_{schema.removeprefix(_ALIGNED_SCHEMA + '_')}" my_report = Report( name="time_window_join_report", spark=spark, workspace_client=create_autospec(WorkspaceClient), - config=_aligned_config(schema, table_prefix), + config=_aligned_config(schema, table_prefix, epoch_unit=epoch_unit), ) window_evt = TimeWindowEvent(name="ten_min", window_length=window_length) @@ -442,7 +466,7 @@ def test_time_window_event_ids_join_after_incremental_run(spark, setup_tw_aligne """Run 1 (full) on containers 1-2; run 2 (incremental) adds container 3 and changes the aggregation's definition. The changed aggregation recomputes over all containers while the unchanged event only computes container 3, yet every stats id must still join.""" - schema, window_length = setup_tw_aligned_db + schema, window_length, _ = setup_tw_aligned_db table_prefix = "time_window_inc_test" cm_run_1 = f"{schema}.container_metrics_run_1" cm_run_2 = f"{schema}.container_metrics_run_2" @@ -502,3 +526,58 @@ def _run(cm_table: str, is_incremental: bool, statistics) -> None: 3, } assert stats_fact.filter(F.col("aggregation_label") == "median").count() > 0 + + +# --------------------------------------------------------------------------- +# TIMESTAMP container boundaries: epoch_unit is opt-in, required only by TimeWindowEvent +# --------------------------------------------------------------------------- +@pytest.mark.parametrize("setup_tw_aligned_db", ["sec_ts"], indirect=True) +def test_time_window_event_timestamp_boundaries_require_epoch_unit(spark, setup_tw_aligned_db): + """A TimeWindowEvent over TIMESTAMP boundaries without epoch_unit fails fast and clearly.""" + schema, window_length, _ = setup_tw_aligned_db + my_report = Report( + name="time_window_no_unit_report", + spark=spark, + workspace_client=create_autospec(WorkspaceClient), + config=_aligned_config(schema, "time_window_no_unit_test"), + ) + my_report.add_event(TimeWindowEvent(name="ten_min", window_length=window_length)) + + with pytest.raises(ValueError, match=r"TimeWindowEvent.*epoch_unit"): + my_report.determine_report() + + +@pytest.mark.parametrize("epoch_unit", [None, "s"]) +@pytest.mark.parametrize("setup_tw_aligned_db", ["sec_ts"], indirect=True) +def test_container_event_timestamp_boundaries(spark, setup_tw_aligned_db, epoch_unit): + """Backward compatibility: a ContainerEvent over TIMESTAMP boundaries runs without + epoch_unit (as today) and yields epoch seconds; epoch_unit="s" gives identical values. + measurement_dimension keeps the TIMESTAMP type either way.""" + schema, _, _ = setup_tw_aligned_db + table_prefix = f"container_event_ts_test_{epoch_unit or 'unset'}" + my_report = Report( + name="container_event_ts_report", + spark=spark, + workspace_client=create_autospec(WorkspaceClient), + config=_aligned_config(schema, table_prefix, epoch_unit=epoch_unit), + ) + my_report.add_event(ContainerEvent(name="full_container")) + my_report.determine_report() + my_report.persist_results() + + expected = { + r.container_id: (r.s, r.e) + for r in spark.read.table(f"{schema}.container_metrics") + .select( + "container_id", + F.col("start_ts").cast("double").alias("s"), + F.col("stop_ts").cast("double").alias("e"), + ) + .collect() + } + event_fact = spark.read.table(f"spark_catalog.gold.{table_prefix}_event_instance_fact") + actual = {r.container_id: (r.start_ts, r.end_ts) for r in event_fact.collect()} + assert actual and all(actual[cid] == expected[cid] for cid in actual), (actual, expected) + + measurement_dim = spark.read.table(f"spark_catalog.gold.{table_prefix}_measurement_dimension") + assert isinstance(measurement_dim.schema["start_ts"].dataType, T.TimestampType) From ce0fdea5b367d96ff5e0a43ea4592e3231564a7a Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 7 Oct 2026 11:33:01 +0200 Subject: [PATCH 09/27] feat(reporting, query-engine): hash TimeWindowEvent windows by position, include epoch_unit in definition hashes, and add per-container window limit Change `TimeWindowEvent` `event_instance_id` generation to hash the window's position (`container_id::event_name::window_index`) instead of its boundaries, so event facts and scoped aggregations join reliably despite double rounding of epoch timestamps. Propagate `solver_config.epoch_unit` into `ContainerBoundaryEvent` subclasses and fold it into the definition hashes of `ContainerEvent`, `TimeWindowEvent`, and aggregations scoped to a `TimeWindowEvent`, forcing a full recompute when the unit changes in incremental mode. Add `max_windows_per_container` (default 1,000,000) to `TimeWindowEvent` and `TimeWindowExpression` to fail fast when `window_length` is in the wrong unit for the boundaries, and reject non-finite container boundaries by yielding no windows. Update docs, skills, and tests. --- docs/impulse/docs/config/configuration.md | 6 + .../events/container_event.md | 4 +- .../events/time_window_event.md | 42 +++-- docs/impulse/docs/references/report/event.md | 22 +-- skills/impulse-events/SKILL.md | 7 +- .../query/events/time_window_expression.py | 145 +++++++++++++++--- .../aggregations/stats_aggregator.py | 43 ++++-- src/impulse_reporting/core/report.py | 3 + .../events/container_boundary_event.py | 19 +++ .../events/container_event.py | 6 +- .../events/time_window_event.py | 64 ++++++-- .../util/event_instance_util.py | 27 +++- .../events/time_window_expression_test.py | 93 +++++++++-- .../integration/time_window_event_test.py | 145 ++++++++++++++++++ .../unit/aggregations/definition_hash_test.py | 23 +++ .../aggregations/stats_aggregator_test.py | 55 +++++++ .../unit/events/container_event_test.py | 26 ++++ .../unit/events/time_window_event_test.py | 44 +++++- 18 files changed, 691 insertions(+), 83 deletions(-) diff --git a/docs/impulse/docs/config/configuration.md b/docs/impulse/docs/config/configuration.md index 41013167..c0a0b4a8 100644 --- a/docs/impulse/docs/config/configuration.md +++ b/docs/impulse/docs/config/configuration.md @@ -183,6 +183,12 @@ Top-level fields on `SolverConfig`: `measurement_dimension` and container filters always see the original columns. When unset, nothing is converted. `TIMESTAMP_NTZ` and `DATE` boundaries are not supported. + `epoch_unit` is part of the definition hash of `ContainerEvent`, `TimeWindowEvent` and every + aggregation scoped to a `TimeWindowEvent`, so changing it recomputes them over all containers in + incremental mode instead of mixing units in the gold tables. Expressions that read + `start_ts`/`stop_ts` through `apply(..., container_metrics=[...])` are not covered: after changing + `epoch_unit`, list them under [`full_recalculation`](#full_recalculation-optional). + Per-table sections (each a `TableConfig`): | Section | When it applies | Typical mappings | diff --git a/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md b/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md index e4970c54..5127cd57 100644 --- a/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md +++ b/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md @@ -80,7 +80,9 @@ Calculate definition hash. The hash only captures computation-relevant attributes. For a ``ContainerEvent`` the identity is fully determined by the fact that it is a container event (there is no expression to vary), -so the name of the event is hashed. +so the name of the event is hashed. When ``epoch_unit`` is set it is +hashed too, since it decides the unit of ``TIMESTAMP`` boundaries in +``start_ts`` / ``end_ts``; unset, the hash is the name alone, as before. **Returns**: diff --git a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md index d883b8f9..67fcb46c 100644 --- a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md +++ b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md @@ -29,7 +29,8 @@ def __init__(name: str, window_length: float, desc: str = None, required_channels: list[str] = None, - attributes: Mapping[str, str] = None) + attributes: Mapping[str, str] = None, + max_windows_per_container: int = MAX_WINDOWS_PER_CONTAINER) ``` Initialize a TimeWindowEvent object. @@ -44,10 +45,30 @@ Initialize a TimeWindowEvent object. dimension table. - `attributes` (`Mapping[str, str]`): Key-value metadata for the event. ``window_length`` is surfaced here automatically (without overriding a user-supplied key). +- `max_windows_per_container` (`int`): Maximum number of windows per container (default 1,000,000). A container +exceeding it fails the report with an error naming the limit, which usually +means ``window_length`` is in the wrong unit for the boundaries. Not part of +the definition hash. **Raises**: -- `ValueError`: If ``window_length`` is not strictly positive and finite. +- `ValueError`: If ``window_length`` is not strictly positive and finite, or +``max_windows_per_container`` is not a positive integer. + +#### set\_epoch\_unit + +```python +def set_epoch_unit(epoch_unit: str | None) -> None +``` + +Set the epoch unit ``TIMESTAMP`` container boundaries are converted to. + +Also recorded on the expression, whose string form feeds the definition hashes of +this event and of the aggregations scoped to it. + +**Arguments**: + +- `epoch_unit` (`str or None`): The report's ``solver_config.epoch_unit``. #### get\_id @@ -93,11 +114,13 @@ def determine_definition_hash() -> int Calculate definition hash for the time-window event. -Only includes the expression string (which encodes ``window_length``), the sole -attribute that affects the event results, so resizing the window forces a full -recompute in incremental mode. +Only includes the expression string, which encodes the attributes that affect the +event results: ``window_length`` and, when set, ``epoch_unit`` (the unit of +``TIMESTAMP`` boundaries). Resizing the window or changing the unit therefore forces +a full recompute in incremental mode. -Excludes: name, description, required_channels, report_id +Excludes: name, description, required_channels, max_windows_per_container, +report_id **Returns**: @@ -146,9 +169,10 @@ Extract the event fact table for the given list of TimeWindowEvent objects. Resolves the matching containers via the solver's filter pipeline (like ``ContainerEvent``) and computes each event's windows natively from the containers' ``start_ts`` / ``stop_ts``, so every filtered container gets windows. -Each window becomes one event instance (``start_ts < end_ts``). The windows are -bit-identical to the ones the solve computes for scoped aggregations (see -:func:`window_intervals_col`), so the ``event_instance_id`` values match. +Each window becomes one event instance (``start_ts < end_ts``) whose +``event_instance_id`` hashes its position among the container's windows. The solve +computes the same windows in the same order for scoped aggregations (see +:func:`window_intervals_col`), so the ids match. **Arguments**: diff --git a/docs/impulse/docs/references/report/event.md b/docs/impulse/docs/references/report/event.md index 7d8097ea..b67fc169 100644 --- a/docs/impulse/docs/references/report/event.md +++ b/docs/impulse/docs/references/report/event.md @@ -223,6 +223,7 @@ my_report.add_event(ten_minute_windows) | `desc` | `str` | No | Human-readable description. | | `required_channels` | `list[str]` | No | Channel names required for this event. Informational; stored in the event dimension table. | | `attributes` | `Mapping[str, str]` | No | Free-form key-value metadata. `window_length` is surfaced here automatically (without overriding a user key). | +| `max_windows_per_container` | `int` | No | Upper bound on the windows per container (default `1_000_000`). A container that would exceed it fails the report with an error naming the limit, which usually means `window_length` is in the wrong unit for the timestamps. Raise it for very long containers with short windows. Not part of the definition hash. | :::note `window_length` follows the same convention as `SequenceOfEvents.max_overlap`: it is expressed in @@ -234,7 +235,8 @@ If `container_metrics.start_ts`/`stop_ts` are `TIMESTAMP` columns, set [`solver_config.epoch_unit`](../../config/configuration.md#solver-column-mappings-and-filters) to the epoch unit of the channel sample timestamps (e.g. `"s"`). The boundaries are converted to that unit, and `window_length` is expressed in it. Without it, the report fails with an error -naming the setting. +naming the setting. `epoch_unit` is part of the event's definition (and of the aggregations +scoped to it), so changing it recomputes them over all containers in incremental mode. ::: ### How it works @@ -244,25 +246,27 @@ naming the setting. tiles `[start_ts, stop_ts]` into consecutive windows of length `window_length`. 2. The **final window is clamped** to `stop_ts` when the last full window would overrun it; any zero-length trailing slice is dropped (every instance satisfies `start_ts < end_ts`). -3. Each window becomes one **event instance** with a unique `event_instance_id`, written to the - shared `event_instance_fact` table. + Containers whose `start_ts` or `stop_ts` is null, NaN or infinite get no windows. +3. Each window becomes one **event instance**, written to the shared `event_instance_fact` table. + Its `event_instance_id` hashes the container, the event name and the window's position in + the container (0, 1, 2, ...). 4. An aggregation scoped to the event (`StatsAggregator(..., event=time_window_event)`) computes its statistic **once per window** and joins back to those instances. :::note The windows are computed from `container_metrics` alone, so **every** container that matches the report's filters gets windows, whether or not it has channel data and whether or not an -aggregation is scoped to the event. An aggregation scoped to the event computes the same windows -in the query engine, so its per-window rows carry the same `event_instance_id` values. For the -per-window values to be meaningful, the container boundaries must share the channel samples' time -base (as they do in real measurement data). +aggregation is scoped to the event. An aggregation scoped to the event computes the same windows, +in the same order, in the query engine, so its per-window rows carry the same `event_instance_id` +values. For the per-window values to be meaningful, the container boundaries must share the channel +samples' time base (as they do in real measurement data). ::: :::note Window boundaries are stored as doubles (`start_ts` / `end_ts`), like every other event type. Epoch timestamps in nanoseconds exceed the range doubles represent exactly, so their window boundaries -are rounded to about 256 ns. The rounding is the same for the event and its aggregations, so their -`event_instance_id` values still match. +are rounded to about 256 ns. The `event_instance_id` depends on the window's position, not on its +boundaries, so the rounding does not affect how aggregations join to the windows. ::: ## Event output schema diff --git a/skills/impulse-events/SKILL.md b/skills/impulse-events/SKILL.md index 1942d01a..2d21cade 100644 --- a/skills/impulse-events/SKILL.md +++ b/skills/impulse-events/SKILL.md @@ -150,12 +150,15 @@ report.add_event(ten_minute) | `desc` | `str` | No | Description. | | `required_channels` | `list[str]` | No | Informational. | | `attributes` | `Mapping[str, str]` | No | Free-form metadata; `window_length` is added automatically. | +| `max_windows_per_container` | `int` | No | Windows-per-container limit (default 1,000,000); exceeding it fails the report, usually a `window_length` unit mismatch. | Windows are computed from `container_metrics` for every container matching the report's filters, with or without channel data or a scoped aggregation. Pair it with an aggregation scoped to the event (e.g. `StatsAggregator(..., event=...)`) to compute one statistic per window; those rows carry the same -`event_instance_id` values as the windows. Because the windows come from `container_metrics`, those -boundaries must share the channel samples' time base for the per-window values to be meaningful. +`event_instance_id` values as the windows (the id hashes container, event name and window position). +Because the windows come from `container_metrics`, those boundaries must share the channel samples' +time base for the per-window values to be meaningful. Containers with null, NaN or infinite +boundaries get no windows. ## Output schema diff --git a/src/impulse_query_engine/analyze/query/events/time_window_expression.py b/src/impulse_query_engine/analyze/query/events/time_window_expression.py index 201651c0..200ef43e 100644 --- a/src/impulse_query_engine/analyze/query/events/time_window_expression.py +++ b/src/impulse_query_engine/analyze/query/events/time_window_expression.py @@ -2,6 +2,7 @@ import datetime import math +import numbers import numpy as np import pyspark.sql.functions as F @@ -23,16 +24,65 @@ # the names are config-invariant. _SOLVER_CONFIG = SolverConfig() +# Default upper bound on the windows per container. A window_length in the wrong unit for the +# boundaries (e.g. 60 meant as seconds over ns epochs) would otherwise yield billions of +# windows: Spark's sequence fails with an opaque COLLECTION_SIZE_LIMIT_EXCEEDED and numpy +# allocates arrays of that size. +MAX_WINDOWS_PER_CONTAINER = 1_000_000 -def window_intervals_col(start_ts: Column, stop_ts: Column, window_length: float) -> Column: +_WINDOW_LIMIT_HINT = ( + "Check that window_length is in the epoch unit of the container boundaries " + "(solver_config.epoch_unit), or raise max_windows_per_container." +) + + +def _validate_max_windows(max_windows: int) -> int: + """Return *max_windows* as an int, raising unless it is a positive integer. + + Parameters + ---------- + max_windows : int + Maximum number of windows per container. + + Returns + ------- + int + The validated limit. + + Raises + ------ + ValueError + If *max_windows* is not a positive integer. + """ + if ( + isinstance(max_windows, bool) + or not isinstance(max_windows, numbers.Integral) + or max_windows <= 0 + ): + raise ValueError(f"max_windows must be a positive integer, got {max_windows!r}.") + return int(max_windows) + + +def _is_finite(col: Column) -> Column: + """True for finite doubles; false for NaN / +-inf; null for null.""" + return ~F.isnan(col) & (F.abs(col) != F.lit(float("inf"))) + + +def window_intervals_col( + start_ts: Column, + stop_ts: Column, + window_length: float, + max_windows: int = MAX_WINDOWS_PER_CONTAINER, +) -> Column: """Spark counterpart of :meth:`TimeWindowExpression.build`. Computes the same fixed-duration windows natively in Spark, so the reporting ``TimeWindowEvent`` can materialize windows for every container without a solve. - The ``event_instance_id`` of a window hashes its ``start_ts`` / ``end_ts``, so the - windows computed here must be **bit-identical** to the ones ``build`` computes for - scoped aggregations. Both therefore run the same IEEE-754 operations in the same - order on the same doubles: cast the boundaries to double *before* subtracting, + The ``event_instance_id`` of a window hashes its position in the returned array, so + the windows computed here must match the ones ``build`` computes for scoped + aggregations in **count and order**. Both therefore run the same IEEE-754 operations + in the same order on the same doubles (which also keeps the stored boundaries + identical): cast the boundaries to double *before* subtracting, ``count = ceil((stop - start) / W)``, ``start_i = start + i * W``, ``end_i = min(start + (i + 1) * W, stop)``, and drop windows with ``start_i >= end_i``. Keep the two implementations in sync. @@ -46,13 +96,18 @@ def window_intervals_col(start_ts: Column, stop_ts: Column, window_length: float window_length : float Fixed window length, in the same time unit as the timestamps. Must be strictly positive. + max_windows : int, optional + Maximum number of windows per container (default + :data:`MAX_WINDOWS_PER_CONTAINER`). A container exceeding it fails the query with + an error naming the limit. Returns ------- pyspark.sql.Column - ``array>`` with one ``[start, end]`` pair per window; empty when the - boundaries are null or the span is not strictly positive. + ``array>`` with one ``[start, end]`` pair per window; empty when a + boundary is null, NaN or infinite, or the span is not strictly positive. """ + max_windows = _validate_max_windows(max_windows) start, stop = start_ts.cast("double"), stop_ts.cast("double") w = F.lit(float(window_length)) count = F.ceil((stop - start) / w) @@ -61,9 +116,24 @@ def window_intervals_col(start_ts: Column, stop_ts: Column, window_length: float lambda i: F.array(start + i * w, F.least(start + (i + F.lit(1)) * w, stop)), ) windows = F.filter(windows, lambda p: p[0] < p[1]) - # Gate on a positive span: sequence(0, -1) yields [0, -1] (a descending sequence), - # not an empty array, so degenerate containers would otherwise emit bogus windows. - return F.when(stop > start, windows).otherwise(F.array().cast("array>")) + too_many = F.raise_error( + F.concat( + F.lit("TimeWindowExpression: "), + count.cast("string"), + F.lit(f" windows of length {float(window_length)} over a container span of "), + (stop - start).cast("string"), + F.lit(f" exceed max_windows={max_windows}. {_WINDOW_LIMIT_HINT}"), + ) + ) + # Gate on finite boundaries and a positive span before anything reaches sequence: + # sequence(0, -1) yields [0, -1] (a descending sequence), not an empty array, and Spark + # orders NaN above every number, so a NaN stop_ts would pass ``stop > start`` alone. + valid = _is_finite(start) & _is_finite(stop) & (stop > start) + return ( + F.when(valid & (count > F.lit(max_windows)), too_many) + .when(valid, windows) + .otherwise(F.array().cast("array>")) + ) class TimeWindowExpression(TimeSeriesExpression): @@ -85,10 +155,19 @@ class TimeWindowExpression(TimeSeriesExpression): This is the query-engine counterpart of the reporting ``TimeWindowEvent``. It evaluates to :class:`Intervals`, so it can scope a ``StatsAggregator`` (one statistic per window). The reporting event fact computes the same windows natively via - :func:`window_intervals_col`; the two must stay bit-identical. + :func:`window_intervals_col`; the two must produce the same windows in the same order. + + Attributes + ---------- + epoch_unit : str or None + Epoch unit the solver converts ``TIMESTAMP`` boundaries to + (``solver_config.epoch_unit``), set by the reporting ``TimeWindowEvent``. + Descriptive only: :meth:`build` does not convert (the solver does). It is part of + the string form, so the definition hashes of the event and of every aggregation + scoped to it change with the unit. """ - def __init__(self, window_length: float): + def __init__(self, window_length: float, max_windows: int = MAX_WINDOWS_PER_CONTAINER): """ Initialize a TimeWindowExpression. @@ -97,11 +176,16 @@ def __init__(self, window_length: float): window_length : float Fixed window length, in the same time unit as the underlying timestamps (e.g. milliseconds-since-epoch). Must be strictly positive and finite. + max_windows : int, optional + Maximum number of windows per container (default + :data:`MAX_WINDOWS_PER_CONTAINER`); :meth:`build` raises beyond it. Not part + of the string form, since it only decides between an error and a result. Raises ------ ValueError - If ``window_length`` is not strictly positive and finite. + If ``window_length`` is not strictly positive and finite, or ``max_windows`` + is not a positive integer. """ # inf / NaN must be rejected too: inf gives a zero window count, for which Spark's # sequence(0, -1) emits a bogus window, and NaN crashes the solve in build(). @@ -114,20 +198,25 @@ def __init__(self, window_length: float): # regardless of whether an int or float was passed: 10 and 10.0 are the same window # and must not trigger a spurious full recompute in incremental mode. self.window_length = float(window_length) + self.max_windows = _validate_max_windows(max_windows) + self.epoch_unit: str | None = None TimeSeriesExpression.__init__(self, is_single_signal=False) def __str__(self) -> str: """ Return a string representation of the TimeWindowExpression. - The ``window_length`` is included so it flows into the event's definition hash. + The ``window_length`` (and ``epoch_unit``, when set) is included so it flows into + the definition hashes of the event and of the aggregations scoped to it. An unset + ``epoch_unit`` is omitted, keeping the string identical to the unit-less form. Returns ------- str String representation of the object. """ - return f"TimeWindowExpression" + unit = f", epoch_unit={self.epoch_unit}" if self.epoch_unit is not None else "" + return f"TimeWindowExpression" def dtype(self): """ @@ -215,7 +304,15 @@ def build(self, cache: SeriesCache) -> Intervals: Intervals Consecutive fixed-length windows over ``[start_ts, stop_ts]``, with the final window clamped to ``stop_ts``. Empty when the container boundaries are absent - (e.g. the empty cache used for type validation) or non-positive in span. + (e.g. the empty cache used for type validation), NaN or infinite, or + non-positive in span. + + Raises + ------ + TypeError + If a boundary is a date/time value rather than an epoch number. + ValueError + If the container would produce more than ``max_windows`` windows. """ start_ts = cache.container_metrics.get(_SOLVER_CONFIG.start_ts_col) stop_ts = cache.container_metrics.get(_SOLVER_CONFIG.stop_ts_col) @@ -238,13 +335,23 @@ def build(self, cache: SeriesCache) -> Intervals: # force float64), and an exact int64 span can round differently from the double # span for large values (e.g. ns epochs), changing the window count. start_ts, stop_ts = float(start_ts), float(stop_ts) - if not stop_ts > start_ts: + # Same gate as window_intervals_col: NaN / infinite boundaries (e.g. an unfinished + # recording) yield no windows, like nulls. + if not (math.isfinite(start_ts) and math.isfinite(stop_ts) and stop_ts > start_ts): return Intervals.empty() # Number of windows covering the span; the last one is clamped to stop_ts below. # The span is strictly positive (guarded above) and window_length is strictly - # positive (enforced in __init__), so window_count >= 1. - window_count = int(np.ceil((stop_ts - start_ts) / self.window_length)) + # positive (enforced in __init__), so window_count >= 1. Compared before the int + # conversion, since an overflowing span gives an infinite count. + window_count = np.ceil((stop_ts - start_ts) / self.window_length) + if window_count > self.max_windows: + raise ValueError( + f"TimeWindowExpression: {window_count:.0f} windows of length {self.window_length} " + f"over a container span of {stop_ts - start_ts} exceed " + f"max_windows={self.max_windows}. {_WINDOW_LIMIT_HINT}" + ) + window_count = int(window_count) indices = np.arange(window_count) starts = start_ts + indices * self.window_length diff --git a/src/impulse_reporting/aggregations/stats_aggregator.py b/src/impulse_reporting/aggregations/stats_aggregator.py index e5a6eb7f..31c5df36 100644 --- a/src/impulse_reporting/aggregations/stats_aggregator.py +++ b/src/impulse_reporting/aggregations/stats_aggregator.py @@ -481,7 +481,9 @@ def _explode_stats_values(df: DataFrame) -> DataFrame: Returns ------- pyspark.sql.DataFrame - DataFrame with exploded statistics for each signal and interval. + DataFrame with exploded statistics for each signal and interval, carrying the + interval's position as ``interval_index`` (the window index of a + ``TimeWindowEvent``). """ # Step 1: Explode by signal index to get one row per signal. # @@ -526,6 +528,7 @@ def _explode_stats_values(df: DataFrame) -> DataFrame: "event_id", "event_name", "signal_index", + "interval_index", f.col("zipped.event_timestamps").getItem(0).alias("start_ts"), f.col("zipped.event_timestamps").getItem(1).alias("end_ts"), f.col("zipped.signal_stats_per_interval").alias("statistics"), @@ -538,6 +541,7 @@ def _explode_stats_values(df: DataFrame) -> DataFrame: "event_name", "event_id", "signal_index", + "interval_index", "start_ts", "end_ts", f.explode(f.col("statistics")).alias("aggregation_label", "statistic_value"), @@ -663,8 +667,9 @@ def _add_event_instance_id_column( Add an event_instance_id column, matching ``event_instance_fact``. The id comes from ``generate_event_instance_id_column``: a ``ContainerEvent`` - gets ``xxhash64(container_id)`` (one id per container), all other event types get - the timestamp-based hash. The container-event case is applied per row (keyed on + gets ``xxhash64(container_id)`` (one id per container), a ``TimeWindowEvent`` the + window-index hash over ``interval_index``, all other event types the + timestamp-based hash. The event-type cases are applied per row (keyed on ``stats_name``) since a frame may mix event types. Parameters @@ -678,22 +683,34 @@ def _add_event_instance_id_column( Function that adds the event_instance_id column to a DataFrame. """ from impulse_reporting.events.container_event import ContainerEvent + from impulse_reporting.events.time_window_event import TimeWindowEvent def _(df: DataFrame) -> DataFrame: - container_event_stats_names = [ - agg.get_name() - for agg in aggregations - if agg and isinstance(agg.get_event(), ContainerEvent) - ] - - timestamp_based_id = generate_event_instance_id_column() + def stats_names_scoped_to(event_cls: type) -> list[str]: + return [ + agg.get_name() + for agg in aggregations + if agg and isinstance(agg.get_event(), event_cls) + ] + + container_event_stats_names = stats_names_scoped_to(ContainerEvent) + time_window_stats_names = stats_names_scoped_to(TimeWindowEvent) + + event_instance_id_column = generate_event_instance_id_column() + # Only reference interval_index when a TimeWindowEvent is in play: frames of + # other aggregation types (e.g. PointValueAggregator) do not carry it. + if time_window_stats_names: + event_instance_id_column = f.when( + f.col("stats_name").isin(time_window_stats_names), + generate_event_instance_id_column( + event_type=TimeWindowEvent, window_index_col="interval_index" + ), + ).otherwise(event_instance_id_column) if container_event_stats_names: event_instance_id_column = f.when( f.col("stats_name").isin(container_event_stats_names), generate_event_instance_id_column(event_type=ContainerEvent), - ).otherwise(timestamp_based_id) - else: - event_instance_id_column = timestamp_based_id + ).otherwise(event_instance_id_column) return df.withColumn("event_instance_id", event_instance_id_column) diff --git a/src/impulse_reporting/core/report.py b/src/impulse_reporting/core/report.py index d831ceee..9a680881 100644 --- a/src/impulse_reporting/core/report.py +++ b/src/impulse_reporting/core/report.py @@ -415,6 +415,9 @@ def add_event(self, event: Event): ) self.events.append(event) event.set_report_id(self.report_id) + if isinstance(event, ContainerBoundaryEvent): + # The unit of TIMESTAMP boundaries is part of these events' definitions. + event.set_epoch_unit(self.solver.config.epoch_unit) def get_events(self) -> list[Event]: """ diff --git a/src/impulse_reporting/events/container_boundary_event.py b/src/impulse_reporting/events/container_boundary_event.py index b1f9dc02..3172679a 100644 --- a/src/impulse_reporting/events/container_boundary_event.py +++ b/src/impulse_reporting/events/container_boundary_event.py @@ -17,8 +17,27 @@ class ContainerBoundaryEvent(Event): filtered container yields instances regardless of its channel data. The report therefore excludes these event types from the solvable expressions and dispatches them with ``query`` / ``solver`` rather than ``solved_df``. + + Attributes + ---------- + epoch_unit : str or None + ``solver_config.epoch_unit`` of the report the event belongs to, set by + ``Report.add_event``. It decides the unit of ``TIMESTAMP`` boundaries, so + subclasses fold it into their definition hash. """ + epoch_unit: str | None = None + + def set_epoch_unit(self, epoch_unit: str | None) -> None: + """Set the epoch unit ``TIMESTAMP`` container boundaries are converted to. + + Parameters + ---------- + epoch_unit : str or None + The report's ``solver_config.epoch_unit``. + """ + self.epoch_unit = epoch_unit + @staticmethod def resolve_container_metrics( spark: SparkSession, diff --git a/src/impulse_reporting/events/container_event.py b/src/impulse_reporting/events/container_event.py index bc3c9a7e..caa961f7 100644 --- a/src/impulse_reporting/events/container_event.py +++ b/src/impulse_reporting/events/container_event.py @@ -90,7 +90,9 @@ def determine_definition_hash(self) -> int: The hash only captures computation-relevant attributes. For a ``ContainerEvent`` the identity is fully determined by the fact that it is a container event (there is no expression to vary), - so the name of the event is hashed. + so the name of the event is hashed. When ``epoch_unit`` is set it is + hashed too, since it decides the unit of ``TIMESTAMP`` boundaries in + ``start_ts`` / ``end_ts``; unset, the hash is the name alone, as before. Returns ------- @@ -98,6 +100,8 @@ def determine_definition_hash(self) -> int: Hash value representing the computation definition. """ hash_input = self.name + if self.epoch_unit is not None: + hash_input = f"{self.name}::epoch_unit={self.epoch_unit}" hash_bytes = hashlib.sha256(hash_input.encode()).digest() return int.from_bytes(hash_bytes[:8], byteorder="big", signed=True) diff --git a/src/impulse_reporting/events/time_window_event.py b/src/impulse_reporting/events/time_window_event.py index 7117858f..8b967b33 100644 --- a/src/impulse_reporting/events/time_window_event.py +++ b/src/impulse_reporting/events/time_window_event.py @@ -14,6 +14,7 @@ TimeSeriesExpression, ) from impulse_query_engine.analyze.query.events.time_window_expression import ( + MAX_WINDOWS_PER_CONTAINER, TimeWindowExpression, window_intervals_col, ) @@ -38,8 +39,9 @@ class TimeWindowEvent(ContainerBoundaryEvent): The event fact is computed natively in Spark from ``container_metrics`` (via :func:`window_intervals_col`), so every filtered container gets windows regardless of its channel data. Aggregations scoped to this event evaluate the - :class:`TimeWindowExpression` in the solve, which computes bit-identical windows, so - the timestamp-based ``event_instance_id`` values match on both sides. + :class:`TimeWindowExpression` in the solve, which computes the same windows in the + same order. ``event_instance_id`` hashes the window's position rather than its + boundaries, so both sides match without relying on bit-identical doubles. """ def __init__( @@ -49,6 +51,7 @@ def __init__( desc: str = None, required_channels: list[str] = None, attributes: Mapping[str, str] = None, + max_windows_per_container: int = MAX_WINDOWS_PER_CONTAINER, ): """ Initialize a TimeWindowEvent object. @@ -68,11 +71,17 @@ def __init__( attributes : Mapping[str, str], optional Key-value metadata for the event. ``window_length`` is surfaced here automatically (without overriding a user-supplied key). + max_windows_per_container : int, optional + Maximum number of windows per container (default 1,000,000). A container + exceeding it fails the report with an error naming the limit, which usually + means ``window_length`` is in the wrong unit for the boundaries. Not part of + the definition hash. Raises ------ ValueError - If ``window_length`` is not strictly positive and finite. + If ``window_length`` is not strictly positive and finite, or + ``max_windows_per_container`` is not a positive integer. """ ContainerBoundaryEvent.__init__(self, name) if window_length is None or not math.isfinite(window_length) or window_length <= 0: @@ -80,10 +89,13 @@ def __init__( f"TimeWindowEvent requires a strictly positive, finite window_length, " f"got {window_length!r}." ) - self.expression = TimeWindowExpression(window_length).alias(name) + self.expression = TimeWindowExpression( + window_length, max_windows=max_windows_per_container + ).alias(name) # Use the expression's normalized (float) length everywhere, so the event fact, # the solve and event_dimension all see the same value for 10 and 10.0. self.window_length = self.expression.window_length + self.max_windows_per_container = self.expression.max_windows self.expression.require_evaluation_type( Intervals, owner="TimeWindowEvent", example="window_length=60000" ) @@ -97,6 +109,20 @@ def __init__( normalized_attributes.setdefault("window_length", str(self.window_length)) self.attributes = normalized_attributes + def set_epoch_unit(self, epoch_unit: str | None) -> None: + """Set the epoch unit ``TIMESTAMP`` container boundaries are converted to. + + Also recorded on the expression, whose string form feeds the definition hashes of + this event and of the aggregations scoped to it. + + Parameters + ---------- + epoch_unit : str or None + The report's ``solver_config.epoch_unit``. + """ + ContainerBoundaryEvent.set_epoch_unit(self, epoch_unit) + self.expression.epoch_unit = epoch_unit + def get_id(self) -> int: """ Returns a unique identifier for the event. @@ -134,11 +160,13 @@ def determine_definition_hash(self) -> int: """ Calculate definition hash for the time-window event. - Only includes the expression string (which encodes ``window_length``), the sole - attribute that affects the event results, so resizing the window forces a full - recompute in incremental mode. + Only includes the expression string, which encodes the attributes that affect the + event results: ``window_length`` and, when set, ``epoch_unit`` (the unit of + ``TIMESTAMP`` boundaries). Resizing the window or changing the unit therefore forces + a full recompute in incremental mode. - Excludes: name, description, required_channels, report_id + Excludes: name, description, required_channels, max_windows_per_container, + report_id Returns ------- @@ -200,9 +228,10 @@ def determine_events( Resolves the matching containers via the solver's filter pipeline (like ``ContainerEvent``) and computes each event's windows natively from the containers' ``start_ts`` / ``stop_ts``, so every filtered container gets windows. - Each window becomes one event instance (``start_ts < end_ts``). The windows are - bit-identical to the ones the solve computes for scoped aggregations (see - :func:`window_intervals_col`), so the ``event_instance_id`` values match. + Each window becomes one event instance (``start_ts < end_ts``) whose + ``event_instance_id`` hashes its position among the container's windows. The solve + computes the same windows in the same order for scoped aggregations (see + :func:`window_intervals_col`), so the ids match. Parameters ---------- @@ -236,13 +265,18 @@ def determine_events( stop_ts = f.col(solver.config.stop_ts_col) # One (event_name, windows) struct per event, exploded in a single pass over the - # containers. start_ts / end_ts stay doubles: the event_instance_id hashes their - # string form, which must match the doubles produced by the solve. + # containers. posexplode yields each window's position, which the + # event_instance_id hashes (scoped aggregations use the same position). per_event = f.array( *[ f.struct( f.lit(event.get_name()).alias("event_name"), - window_intervals_col(start_ts, stop_ts, event.window_length).alias("windows"), + window_intervals_col( + start_ts, + stop_ts, + event.window_length, + max_windows=event.max_windows_per_container, + ).alias("windows"), ) for event in events ] @@ -256,7 +290,7 @@ def determine_events( .select( "container_id", f.col("event.event_name").alias("event_name"), - f.explode(f.col("event.windows")).alias("event_instance"), + f.posexplode(f.col("event.windows")).alias("window_index", "event_instance"), ) .withColumn("start_ts", f.col("event_instance").getItem(0)) .withColumn("end_ts", f.col("event_instance").getItem(1)) diff --git a/src/impulse_reporting/util/event_instance_util.py b/src/impulse_reporting/util/event_instance_util.py index 64763986..f104b294 100644 --- a/src/impulse_reporting/util/event_instance_util.py +++ b/src/impulse_reporting/util/event_instance_util.py @@ -17,21 +17,26 @@ def generate_event_instance_id_column( event_name_col: str = "event_name", start_ts_col: str = "start_ts", end_ts_col: str = "end_ts", + window_index_col: str = "window_index", ) -> Column: """ Generate an event_instance_id column. The id is an xxHash64 of ``container_id::event_name::start_ts::end_ts``. For ``ContainerEvent`` only ``container_id`` is hashed, since a container - event produces exactly one instance per container. The result is a signed - 64-bit long (may be negative), wide enough to keep this merge/join key - collision-free at scale. + event produces exactly one instance per container. For ``TimeWindowEvent`` + the window's position replaces the timestamps + (``container_id::event_name::window_index``), so the event fact and the + aggregations scoped to it agree without bit-identical window boundaries. + The result is a signed 64-bit long (may be negative), wide enough to keep + this merge/join key collision-free at scale. Parameters ---------- event_type : type[Event] or None, optional The event class. When the class is ``ContainerEvent``, the - ``container_id`` column is hashed. For any other value (including + ``container_id`` column is hashed; for ``TimeWindowEvent`` the + window-index hash is returned. For any other value (including ``None`` for backward-compatibility) the timestamp-based hash column is returned. container_id_col : str, optional @@ -42,6 +47,9 @@ def generate_event_instance_id_column( Name of the start timestamp column, defaults to "start_ts". end_ts_col : str, optional Name of the end timestamp column, defaults to "end_ts". + window_index_col : str, optional + Name of the window position column (``TimeWindowEvent`` only), defaults + to "window_index". Returns ------- @@ -49,10 +57,21 @@ def generate_event_instance_id_column( A column expression for the event_instance_id. """ from impulse_reporting.events.container_event import ContainerEvent + from impulse_reporting.events.time_window_event import TimeWindowEvent if event_type is ContainerEvent: return f.xxhash64(f.col(container_id_col).cast("string")) + if event_type is TimeWindowEvent: + return f.xxhash64( + f.concat_ws( + "::", + f.col(container_id_col), + f.col(event_name_col), + f.col(window_index_col), + ) + ) + return f.xxhash64( f.concat_ws( "::", diff --git a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py index 194dd462..e60ee1bd 100644 --- a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py @@ -11,6 +11,7 @@ from impulse_query_engine.analyze.query.aggregations.stats_aggregator import StatsAggregator from impulse_query_engine.analyze.query.events import TimeWindowExpression from impulse_query_engine.analyze.query.events.time_window_expression import ( + MAX_WINDOWS_PER_CONTAINER, window_intervals_col, ) from impulse_query_engine.analyze.query.solvers.empty_cache import EmptyTimeSeriesCache @@ -34,8 +35,8 @@ def container_tags(self) -> dict: return {} -def _build(start_ts, stop_ts, window_length) -> Intervals: - expr = TimeWindowExpression(window_length) +def _build(start_ts, stop_ts, window_length, **kwargs) -> Intervals: + expr = TimeWindowExpression(window_length, **kwargs) return expr.build(_FakeCache({"start_ts": start_ts, "stop_ts": stop_ts})) @@ -120,6 +121,42 @@ def test_str_stable_across_int_and_float_window_length(): assert str(TimeWindowExpression(10)) == str(TimeWindowExpression(10.0)) +def test_str_includes_epoch_unit_only_when_set(): + # The string feeds the definition hashes of the event and its scoped aggregations, so + # the unit must move them, while an unset unit keeps the unit-less form. + expr = TimeWindowExpression(10) + assert str(expr) == "TimeWindowExpression" + expr.epoch_unit = "ms" + assert str(expr) == "TimeWindowExpression" + + +def test_max_windows_not_part_of_str(): + # The cap only decides between an error and a result, so it must not force a recompute. + assert MAX_WINDOWS_PER_CONTAINER == 1_000_000 + assert TimeWindowExpression(10).max_windows == MAX_WINDOWS_PER_CONTAINER + assert str(TimeWindowExpression(10, max_windows=5)) == str(TimeWindowExpression(10)) + + +@pytest.mark.parametrize("bad", [0, -1, 1.5, True, None, "10"]) +def test_invalid_max_windows_raises(bad): + with pytest.raises(ValueError, match="max_windows must be a positive integer"): + TimeWindowExpression(10, max_windows=bad) + + +def test_build_raises_beyond_max_windows(): + # 10 windows are fine at max_windows=10, not at 9. + assert len(_build(0, 100, 10, max_windows=10)) == 10 + with pytest.raises(ValueError, match="10 windows of length 10.0 .* exceed max_windows=9"): + _build(0, 100, 10, max_windows=9) + + +def test_build_unit_mismatch_hits_default_cap(): + # window_length=60 meant as seconds over a 1 h ns-epoch span: 6e10 windows. + start = 1_700_000_000_000_000_000 + with pytest.raises(ValueError, match="epoch unit of the container boundaries"): + _build(np.int64(start), np.int64(start + 3_600_000_000_000), 60) + + @pytest.mark.parametrize("bad", [0, -1, -10.5, None, float("inf"), float("-inf"), float("nan")]) def test_non_positive_window_length_raises(bad): with pytest.raises(ValueError, match="strictly positive"): @@ -148,6 +185,12 @@ def test_nan_container_metrics_yield_empty(): assert len(_build(0.0, np.nan, 10)) == 0 +def test_infinite_container_metrics_yield_empty(): + # Used to overflow in int(np.ceil(inf)). + assert len(_build(0.0, np.inf, 10)) == 0 + assert len(_build(-np.inf, 100.0, 10)) == 0 + + # --------------------------------------------------------------------------- # window_intervals_col: the native-Spark mirror used by the reporting event fact # --------------------------------------------------------------------------- @@ -181,8 +224,38 @@ def test_window_intervals_col_edge_cases(spark): # noqa: F811 assert all(s < e for windows in w.values() for s, e in windows) -def _as_set(windows) -> set[tuple[float, float]]: - return {(float(s), float(e)) for s, e in windows} +def test_window_intervals_col_non_finite_bounds_yield_no_windows(spark): # noqa: F811 + """NaN / infinite boundaries give no windows on both sides. Spark orders NaN above + every number, so a NaN stop_ts used to pass ``stop > start`` and emit [[0, 4], [-4, 0]].""" + nan, inf = float("nan"), float("inf") + rows = [(0, 0.0, nan), (1, nan, 10.0), (2, nan, nan), (3, 0.0, inf), (4, -inf, 10.0)] + _, w = _spark_windows(spark, rows, 4, ts_type="double") + + for k, start, stop in rows: + assert w[k] == [], f"row {k} ({start}, {stop}) produced {w[k]}" + assert _build(start, stop, 4).get_data() == [] + + +def test_window_intervals_col_raises_beyond_max_windows(spark): # noqa: F811 + df = spark.createDataFrame([(0, 0, 100)], "k int, start_ts long, stop_ts long") + ok = df.select(window_intervals_col(F.col("start_ts"), F.col("stop_ts"), 10, max_windows=10)) + assert len(ok.collect()[0][0]) == 10 + + too_many = df.select( + window_intervals_col(F.col("start_ts"), F.col("stop_ts"), 10, max_windows=9) + ) + with pytest.raises(Exception, match="10 windows of length 10.0 .* exceed max_windows=9"): + too_many.collect() + + +def test_window_intervals_col_invalid_max_windows_raises(): + with pytest.raises(ValueError, match="max_windows must be a positive integer"): + window_intervals_col(F.col("start_ts"), F.col("stop_ts"), 10, max_windows=0) + + +def _as_list(windows) -> list[tuple[float, float]]: + """Windows as ordered (start, end) pairs: the order is the window index the ids hash.""" + return [(float(s), float(e)) for s, e in windows] def _count_mismatch_case(window_length: float) -> tuple[int, int]: @@ -204,8 +277,9 @@ def _count_mismatch_case(window_length: float) -> tuple[int, int]: def test_window_intervals_col_bit_identical_to_build(spark): # noqa: F811 - """The id contract: the event fact (Spark) and scoped aggregations (numpy ``build``) - must produce bit-identical windows, since event_instance_id hashes start/end.""" + """The event fact (Spark) and scoped aggregations (numpy ``build``) produce the same + windows in the same order: event_instance_id hashes the window's position, and the + stored boundaries must describe the window the statistics were computed over.""" rnd = random.Random(7) # Long timestamps: ns epochs (~1.7e18, beyond 2^53) and µs epochs, with spans hugging @@ -258,10 +332,10 @@ def test_window_intervals_col_bit_identical_to_build(spark): # noqa: F811 mismatches = [] for k, (start, stop, w) in enumerate(cases): - expected = _as_set(spark_windows[k]) + expected = _as_list(spark_windows[k]) assert expected, f"case {k} produced no windows" for np_type in np_types: - built = _as_set(_build(np_type(start), np_type(stop), w).get_data()) + built = _as_list(_build(np_type(start), np_type(stop), w).get_data()) if built != expected: mismatches.append((k, np_type.__name__, start, stop, w)) assert not mismatches, f"Spark/numpy window mismatch: {mismatches[:5]}" @@ -289,5 +363,6 @@ def test_stats_aggregator_windows_equal_helper_windows(spark): # noqa: F811 event_timestamps, numeric_values, _, _ = agg.build(cache) assert len(spark_windows[0]) == 7 - assert _as_set(event_timestamps) == _as_set(spark_windows[0]) + # Same windows in the same order: event_timestamps' position is the window index. + assert _as_list(event_timestamps) == _as_list(spark_windows[0]) assert len(numeric_values[0]) == len(event_timestamps) diff --git a/tests/impulse_reporting/integration/time_window_event_test.py b/tests/impulse_reporting/integration/time_window_event_test.py index ebe75c8f..c2dd3c3b 100644 --- a/tests/impulse_reporting/integration/time_window_event_test.py +++ b/tests/impulse_reporting/integration/time_window_event_test.py @@ -1,5 +1,6 @@ """Integration tests for TimeWindowEvent with end-to-end Report usage.""" +import math from unittest.mock import create_autospec import pyspark.sql.functions as F @@ -304,6 +305,62 @@ def _assert_ids_join(spark, table_prefix: str) -> tuple[set, set]: # noqa: F811 return stats_event_ids, event_ids +def _assert_window_stats_match_samples(spark, schema: str, table_prefix: str): # noqa: F811 + """Each window's RPM min / max equal those of the silver samples overlapping that window, + and windows without RPM samples carry no value (RPM only covers each container's first + minute, so most windows are empty). + + The ids hash the window's position, so a position that drifted between the event fact + and the solve would attach a neighbouring window's values; this pins every stats row to + the window whose boundaries event_instance_fact stores. + """ + rpm_channels = ( + spark.read.table(f"{schema}.channel_metrics") + .filter(F.col("channel_name") == "Engine RPM") + .select("container_id", "channel_id") + ) + samples = ( + spark.read.table(f"{schema}.channels") + .join(rpm_channels, ["container_id", "channel_id"]) + .select( + "container_id", + F.col("tstart").cast("double").alias("tstart"), + F.col("tend").cast("double").alias("tend"), + F.col("value").cast("double").alias("value"), + ) + ) + windows = spark.read.table(f"spark_catalog.gold.{table_prefix}_event_instance_fact") + expected = ( + windows.join(samples, "container_id") + .filter((F.col("tstart") < F.col("end_ts")) & (F.col("tend") > F.col("start_ts"))) + .groupBy("container_id", "event_instance_id") + .agg(F.min("value").alias("expected_min"), F.max("value").alias("expected_max")) + ) + actual = ( + spark.read.table(f"spark_catalog.gold.{table_prefix}_stats_aggregator_fact") + .groupBy("container_id", "event_instance_id") + .pivot("aggregation_label", ["min", "max"]) + .agg(F.first("statistic_value")) + ) + rows = actual.join(expected, ["container_id", "event_instance_id"], "left").collect() + + def _is_missing(value) -> bool: + return value is None or math.isnan(value) + + with_samples = [r for r in rows if r.expected_max is not None] + assert with_samples and len(with_samples) < len(rows) + mismatches = [ + r + for r in rows + if ( + (r["min"], r["max"]) != (r.expected_min, r.expected_max) + if r.expected_max is not None + else not (_is_missing(r["min"]) and _is_missing(r["max"])) + ) + ] + assert not mismatches, mismatches[:5] + + @pytest.mark.parametrize("setup_tw_aligned_db", ["us", "ns", "sec", "sec_ts"], indirect=True) def test_time_window_event_aggregation_join(spark, setup_tw_aligned_db): """Stats scoped to a TimeWindowEvent yield per-window values whose event_instance_id @@ -329,6 +386,7 @@ def test_time_window_event_aggregation_join(spark, setup_tw_aligned_db): my_report.persist_results() _assert_ids_join(spark, table_prefix) + _assert_window_stats_match_samples(spark, schema, table_prefix) def test_multiple_time_window_events_coexist(spark, basic_narrow_db): @@ -581,3 +639,90 @@ def test_container_event_timestamp_boundaries(spark, setup_tw_aligned_db, epoch_ measurement_dim = spark.read.table(f"spark_catalog.gold.{table_prefix}_measurement_dimension") assert isinstance(measurement_dim.schema["start_ts"].dataType, T.TimestampType) + + +@pytest.mark.parametrize("setup_tw_aligned_db", ["sec_ts"], indirect=True) +def test_epoch_unit_change_recomputes_boundary_events(spark, setup_tw_aligned_db): + """Changing epoch_unit between incremental runs moves the definition hashes of the + TimeWindowEvent, the ContainerEvent and the aggregation scoped to the windows. They + recompute over all containers, so the gold tables never mix units.""" + schema, window_length, epoch_unit = setup_tw_aligned_db + assert epoch_unit == "s" + table_prefix = "time_window_epoch_unit_test" + cm_run_1 = f"{schema}.container_metrics_run_1" + cm_run_2 = f"{schema}.container_metrics_run_2" + past = F.lit("2020-01-01 00:00:00").cast("timestamp") + cm = spark.read.table(f"{schema}.container_metrics") + cm.filter(F.col("container_id").isin([1, 2])).withColumn("timestamp", past).write.format( + "delta" + ).mode("overwrite").saveAsTable(cm_run_1) + # Container 3 is new in run 2 (recent timestamp); 1 and 2 are unchanged. + cm.withColumn( + "timestamp", F.when(F.col("container_id") == 3, F.current_timestamp()).otherwise(past) + ).write.format("delta").mode("overwrite").saveAsTable(cm_run_2) + + def _run(cm_table: str, unit: str, is_incremental: bool): + config = _aligned_config( + schema, + table_prefix, + epoch_unit=unit, + incremental=IncrementalConfig( + enabled=is_incremental, + silver_last_modified_column="timestamp", + gold_last_modified_column="_created_at", + ), + ) + config["source"].container_metrics_table = cm_table + report = Report( + name="time_window_epoch_unit_report", + spark=spark, + workspace_client=create_autospec(WorkspaceClient), + config=config, + ) + window_evt = TimeWindowEvent(name="ten_min", window_length=window_length) + container_evt = ContainerEvent(name="full_container") + report.add_event(window_evt) + report.add_event(container_evt) + page = Page(page_number=1) + report.add_page(page) + stats = _rpm_stats(report, window_evt) + page.add_aggregation(stats) + report.determine_report() + report.persist_results() + return report, window_evt, container_evt, stats + + def _event_rows(event_id: int): + return ( + spark.read.table(f"spark_catalog.gold.{table_prefix}_event_instance_fact") + .filter(F.col("event_id") == event_id) + .collect() + ) + + _, _, container_evt, _ = _run(cm_run_1, "s", is_incremental=False) + starts_in_s = {r.container_id: r.start_ts for r in _event_rows(container_evt.get_id())} + assert set(starts_in_s) == {1, 2} + + report, window_evt, container_evt, stats = _run(cm_run_2, "ms", is_incremental=True) + + changed_events = {i for ids in report._changed_event_ids.values() for i in ids} + changed_aggs = {i for ids in report._changed_aggregation_ids.values() for i in ids} + assert {window_evt.get_id(), container_evt.get_id()} <= changed_events + assert stats.get_id() in changed_aggs + + # The unchanged containers 1 and 2 were rewritten in ms, not left in seconds. + container_rows = _event_rows(container_evt.get_id()) + assert {r.container_id for r in container_rows} == {1, 2, 3} + for r in container_rows: + if r.container_id in starts_in_s: + assert r.start_ts == pytest.approx(starts_in_s[r.container_id] * 1000) + starts_in_ms = {r.container_id: r.start_ts for r in container_rows} + + # Every window tiles the container's ms span: none is left over from the seconds run. + window_rows = _event_rows(window_evt.get_id()) + assert {r.container_id for r in window_rows} == {1, 2, 3} + first_window = {} + for r in window_rows: + first_window[r.container_id] = min( + first_window.get(r.container_id, r.start_ts), r.start_ts + ) + assert first_window == starts_in_ms diff --git a/tests/impulse_reporting/unit/aggregations/definition_hash_test.py b/tests/impulse_reporting/unit/aggregations/definition_hash_test.py index c901f3b0..3dca9c68 100644 --- a/tests/impulse_reporting/unit/aggregations/definition_hash_test.py +++ b/tests/impulse_reporting/unit/aggregations/definition_hash_test.py @@ -17,6 +17,7 @@ ) from impulse_reporting.aggregations.stats_aggregator import StatsAggregator from impulse_reporting.events.basic_event import BasicEvent +from impulse_reporting.events.time_window_event import TimeWindowEvent class TestHistogramDefinitionHash: @@ -354,6 +355,28 @@ def test_hash_without_custom_stats_matches_formula(self): assert stats_agg.determine_definition_hash() == expected + def test_time_window_epoch_unit_changes_hash(self): + """Statistics scoped to a TimeWindowEvent are computed per window, and the windows + tile TIMESTAMP boundaries in the report's epoch_unit, so the unit must move the + aggregation's hash too (it does so through the event expression string).""" + event = TimeWindowEvent(name="windows", window_length=10_000) + stats_agg = StatsAggregator( + name="stats", + input_expressions=[TimeSeriesSelector(None)], + channel_names=["ch_a"], + statistics=["min", "max"], + event=event, + ) + hist = HistogramDuration( + name="hist", base_expr=TimeSeriesSelector(None), bins=[0.0, 1.0], event=event + ) + before = (stats_agg.determine_definition_hash(), hist.determine_definition_hash()) + + event.set_epoch_unit("ms") + + assert stats_agg.determine_definition_hash() != before[0] + assert hist.determine_definition_hash() != before[1] + def test_renaming_channel_names_changes_hash(self): """channel_names is the fact-table merge key, so a rename must force recompute.""" agg1 = self._make(channel_names=["ch_a", "ch_b"]) diff --git a/tests/impulse_reporting/unit/aggregations/stats_aggregator_test.py b/tests/impulse_reporting/unit/aggregations/stats_aggregator_test.py index 646d350b..d36d0e77 100644 --- a/tests/impulse_reporting/unit/aggregations/stats_aggregator_test.py +++ b/tests/impulse_reporting/unit/aggregations/stats_aggregator_test.py @@ -18,6 +18,7 @@ from impulse_reporting.events.basic_event import BasicEvent from impulse_reporting.events.container_event import ContainerEvent from impulse_reporting.events.points_in_time_event import PointsInTimeEvent +from impulse_reporting.events.time_window_event import TimeWindowEvent from impulse_reporting.persist.dimension_schema import STATS_AGGREGATOR_DIMENSION_SCHEMA @@ -530,6 +531,60 @@ def test_determine_aggregations_container_event_instance_id(spark, basic_narrow_ assert all(row.event_instance_id != row.expected_container_id for row in basic_rows) +def test_determine_aggregations_time_window_event_instance_id(spark, basic_narrow_db): + """Time-window stats rows use the window-index id that ``TimeWindowEvent.determine_events`` + writes to ``event_instance_fact``: one id per window, all of them materialized. Basic-event + stats in the same frame keep the timestamp-based id.""" + eng_rpm = basic_narrow_db.query.channel(channel_name="Engine RPM") + + window_event = TimeWindowEvent(name="ten_s", window_length=10_000) + basic_event = BasicEvent(name="rpm_event", expr=eng_rpm > 500) + window_stats = StatsAggregator( + name="window_stats", + input_expressions=[eng_rpm], + channel_names=["Engine RPM"], + statistics=["min", "max"], + event=window_event, + ) + basic_stats = StatsAggregator( + name="basic_stats", + input_expressions=[eng_rpm], + channel_names=["Engine RPM"], + statistics=["min", "max"], + event=basic_event, + ) + + solver = DefaultSolver(spark) + solved_df = basic_narrow_db.query.select( + window_stats.get_expression(), basic_stats.get_expression() + ).solve(spark, solver) + df = StatsAggregator.determine_aggregations( + spark=spark, aggregations=[window_stats, basic_stats], solved_df=solved_df + ) + windows = TimeWindowEvent.determine_events( + spark, [window_event], query=basic_narrow_db.query, solver=solver + ) + + window_ids = {r.event_instance_id for r in windows.collect()} + window_count = { + r.container_id: r.n + for r in windows.groupBy("container_id").count().withColumnRenamed("count", "n").collect() + } + window_rows = df.filter(f.col("visual_id") == window_stats.get_id()) + assert window_rows.count() > 0 + assert {r.event_instance_id for r in window_rows.collect()} <= window_ids + # Every window of every solved container gets its own id (no collapsed indices). + per_container = window_rows.groupBy("container_id").agg( + f.countDistinct("event_instance_id").alias("n") + ) + for row in per_container.collect(): + assert row.n == window_count[row.container_id], row + + basic_rows = df.filter(f.col("visual_id") == basic_stats.get_id()).collect() + assert len(basic_rows) > 0 + assert not {r.event_instance_id for r in basic_rows} & window_ids + + def test_determine_metadata_df(spark, basic_narrow_db): """Test that determine_metadata_df returns a DataFrame with expected columns.""" eng_rpm = basic_narrow_db.query.channel(channel_name="Engine RPM") diff --git a/tests/impulse_reporting/unit/events/container_event_test.py b/tests/impulse_reporting/unit/events/container_event_test.py index 9a9914da..3fc6d0e3 100644 --- a/tests/impulse_reporting/unit/events/container_event_test.py +++ b/tests/impulse_reporting/unit/events/container_event_test.py @@ -1,5 +1,7 @@ """Unit tests for ContainerEvent.""" +import hashlib + import pyspark.sql.functions as f from impulse_query_engine.analyze.query.solvers.default_solver import DefaultSolver @@ -141,6 +143,30 @@ def test_definition_hash_ignores_description(): assert ev1.determine_definition_hash() == ev2.determine_definition_hash() +def _sha256_long(text: str) -> int: + return int.from_bytes(hashlib.sha256(text.encode()).digest()[:8], "big", signed=True) + + +def test_definition_hash_without_epoch_unit_is_name_only(): + """Unset epoch_unit keeps the pre-existing name-only hash, so gold tables written before + epoch_unit existed are not recomputed.""" + event = ContainerEvent(name="ev") + assert event.determine_definition_hash() == _sha256_long("ev") + event.set_epoch_unit(None) + assert event.determine_definition_hash() == _sha256_long("ev") + + +def test_definition_hash_changes_with_epoch_unit(): + """epoch_unit decides the unit of TIMESTAMP boundaries in start_ts / end_ts, so changing + it must force a recompute instead of mixing units in event_instance_fact.""" + unset, ms, us = ContainerEvent(name="ev"), ContainerEvent(name="ev"), ContainerEvent(name="ev") + ms.set_epoch_unit("ms") + us.set_epoch_unit("us") + hashes = {e.determine_definition_hash() for e in (unset, ms, us)} + assert len(hashes) == 3 + assert ms.as_dict()["definition_hash"] == ms.determine_definition_hash() + + # --------------------------------------------------------------------------- # determine_events (integration-ish, needs Spark) # --------------------------------------------------------------------------- diff --git a/tests/impulse_reporting/unit/events/time_window_event_test.py b/tests/impulse_reporting/unit/events/time_window_event_test.py index 78f0a25e..e33e775e 100644 --- a/tests/impulse_reporting/unit/events/time_window_event_test.py +++ b/tests/impulse_reporting/unit/events/time_window_event_test.py @@ -3,6 +3,7 @@ import pytest from impulse_query_engine.analyze.query.events.time_window_expression import ( + MAX_WINDOWS_PER_CONTAINER, TimeWindowExpression, ) from impulse_reporting.events.container_boundary_event import ContainerBoundaryEvent @@ -45,7 +46,7 @@ def test_init_does_not_override_user_window_length_attribute(): def test_is_container_boundary_event_but_not_container_event(): # Routed via the filter pipeline like ContainerEvent, but a sibling (not a subclass), so - # it keeps timestamp-based instance ids and is not limited to one per report. + # it gets its own (window-index) instance ids and is not limited to one per report. event = TimeWindowEvent(name="w", window_length=10) assert isinstance(event, ContainerBoundaryEvent) assert not isinstance(event, ContainerEvent) @@ -95,6 +96,47 @@ def test_definition_hash_stable_across_int_and_float_window_length(): assert a.determine_definition_hash() == b.determine_definition_hash() +def test_definition_hash_changes_with_epoch_unit(): + # epoch_unit decides the unit TIMESTAMP boundaries are tiled in, so flipping it must + # force a full recompute. It reaches the hash through the expression string. + unset = TimeWindowEvent(name="w", window_length=10000) + cleared = TimeWindowEvent(name="w", window_length=10000) + cleared.set_epoch_unit(None) + s, ms = TimeWindowEvent(name="w", window_length=10000), TimeWindowEvent( + name="w", window_length=10000 + ) + s.set_epoch_unit("s") + ms.set_epoch_unit("ms") + + assert ms.epoch_unit == ms.get_expression().epoch_unit == "ms" + assert "epoch_unit=ms" in ms.as_dict()["event_expression"] + assert unset.determine_definition_hash() == cleared.determine_definition_hash() + assert len({e.determine_definition_hash() for e in (unset, s, ms)}) == 3 + + +# --------------------------------------------------------------------------- +# max_windows_per_container — guard rail, not part of the definition +# --------------------------------------------------------------------------- +def test_max_windows_per_container_default_and_override(): + assert TimeWindowEvent(name="w", window_length=10).max_windows_per_container == ( + MAX_WINDOWS_PER_CONTAINER + ) + event = TimeWindowEvent(name="w", window_length=10, max_windows_per_container=5) + assert event.max_windows_per_container == event.get_expression().max_windows == 5 + + +def test_max_windows_per_container_excluded_from_hash(): + a = TimeWindowEvent(name="w", window_length=10) + b = TimeWindowEvent(name="w", window_length=10, max_windows_per_container=5) + assert a.determine_definition_hash() == b.determine_definition_hash() + + +@pytest.mark.parametrize("bad", [0, -1, 2.5, True, None]) +def test_invalid_max_windows_per_container_raises(bad): + with pytest.raises(ValueError, match="max_windows must be a positive integer"): + TimeWindowEvent(name="w", window_length=10, max_windows_per_container=bad) + + # --------------------------------------------------------------------------- # metadata dict shape # --------------------------------------------------------------------------- From 98477444d6208d5dd9284653f1b54685437809b2 Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 7 Oct 2026 11:53:25 +0200 Subject: [PATCH 10/27] feat(reporting, query-engine): support RAW channel data with TimeWindowEvent and clarify epoch_unit semantics Extend `TimeWindowEvent` integration tests to cover `data_type=RAW` with both `Rle` and `Interval` raw encoders, including TIMESTAMP container boundaries converted via `epoch_unit="us"`. Clarify in docs and code that `epoch_unit` describes the existing unit of channel sample timestamps (`tstart`/`tend` or `timestamp` for RAW data) and that channel timestamps are never converted themselves. --- docs/impulse/docs/config/configuration.md | 10 +- .../docs/data_model/silver_layer_schema.md | 3 +- .../analyze/query/solvers/solver_config.md | 14 +-- docs/impulse/docs/references/report/event.md | 3 +- .../analyze/query/solvers/solver_config.py | 14 +-- .../integration/time_window_event_test.py | 100 +++++++++++++++--- 6 files changed, 113 insertions(+), 31 deletions(-) diff --git a/docs/impulse/docs/config/configuration.md b/docs/impulse/docs/config/configuration.md index c0a0b4a8..c648c11a 100644 --- a/docs/impulse/docs/config/configuration.md +++ b/docs/impulse/docs/config/configuration.md @@ -171,10 +171,12 @@ Top-level fields on `SolverConfig`: `container_tags` (if configured), `container_metrics`, and `channel_mapping` (if configured). Omit it if you don't need project-level scoping; the solver does not require it. - `epoch_unit` (`"s"` | `"ms"` | `"us"` | `"ns"`, optional): Epoch unit of the channel sample - timestamps (`tstart`/`tend`). Only needed when `container_metrics.start_ts`/`stop_ts` are - `TIMESTAMP` columns **and** the report uses a `TimeWindowEvent`, whose windows must be in the - samples' time base. Such a report fails with a clear error until it is set. When set, `TIMESTAMP` - boundaries are converted to epoch numbers in that unit: + timestamps (`tstart`/`tend`, or `timestamp` when `data_type = "RAW"`). It only describes them: + channel timestamps are never converted and must already be epoch numbers. Only needed when + `container_metrics.start_ts`/`stop_ts` are `TIMESTAMP` columns **and** the report uses a + `TimeWindowEvent`, whose windows must be in the samples' time base. Such a report fails with a + clear error until it is set. When set, `TIMESTAMP` boundaries are converted to epoch numbers in + that unit: - `ContainerEvent` and `TimeWindowEvent` write `start_ts`/`end_ts` in that unit. With `"s"`, the values are identical to the default. - Expressions that request `start_ts`/`stop_ts` as container metrics (e.g. via diff --git a/docs/impulse/docs/data_model/silver_layer_schema.md b/docs/impulse/docs/data_model/silver_layer_schema.md index 54f8b338..ac98f911 100644 --- a/docs/impulse/docs/data_model/silver_layer_schema.md +++ b/docs/impulse/docs/data_model/silver_layer_schema.md @@ -182,7 +182,8 @@ the epoch-typed pair). Populate whichever your queries and `start_ts`/`stop_ts` may also be `TIMESTAMP` columns. To use them with a `TimeWindowEvent`, set [`solver_config.epoch_unit`](../config/configuration.md#solver-column-mappings-and-filters) to the -epoch unit of the channel sample timestamps, so the window boundaries share the samples' time base. +epoch unit of the channel sample timestamps (`tstart`/`tend`, or `timestamp` in the raw format), so +the window boundaries share the samples' time base. ::: diff --git a/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md b/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md index 74d57768..4aa010be 100644 --- a/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md +++ b/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md @@ -126,12 +126,14 @@ so that solver code can always reference the same constants. override for the channel mapping (alias) table. - `channels` (`TableConfig`): Column mappings and filters for the channel data table. - `unit_conversion` (`TableConfig`): Column mappings and filters for the unit conversion table. -- `epoch_unit` (`{"s", "ms", "us", "ns"} or None`): Epoch unit of the channel sample timestamps (``tstart`` / ``tend``). When set, -``TIMESTAMP``-typed container ``start_ts`` / ``stop_ts`` are converted to epoch -numbers in this unit for container-boundary events (``ContainerEvent``, -``TimeWindowEvent``) and for expressions that request them in the solve. Only -required for a ``TimeWindowEvent`` over ``TIMESTAMP`` boundaries; when unset, -nothing is converted. +- `epoch_unit` (`{"s", "ms", "us", "ns"} or None`): Epoch unit of the channel sample timestamps (``tstart`` / ``tend``, or +``timestamp`` for RAW data). When set, ``TIMESTAMP``-typed container +``start_ts`` / ``stop_ts`` are converted to epoch numbers in this unit for +container-boundary events (``ContainerEvent``, ``TimeWindowEvent``) and for +expressions that request them in the solve. Channel timestamps are never +converted; they must already be epoch numbers. Only required for a +``TimeWindowEvent`` over ``TIMESTAMP`` boundaries; when unset, nothing is +converted. #### from\_json diff --git a/docs/impulse/docs/references/report/event.md b/docs/impulse/docs/references/report/event.md index b67fc169..66f34e55 100644 --- a/docs/impulse/docs/references/report/event.md +++ b/docs/impulse/docs/references/report/event.md @@ -233,7 +233,8 @@ seconds or any derived unit. So 60 one-minute windows over millisecond timestamp If `container_metrics.start_ts`/`stop_ts` are `TIMESTAMP` columns, set [`solver_config.epoch_unit`](../../config/configuration.md#solver-column-mappings-and-filters) -to the epoch unit of the channel sample timestamps (e.g. `"s"`). The boundaries are converted to +to the epoch unit of the channel sample timestamps (`tstart`/`tend`, or `timestamp` for RAW +data; e.g. `"s"`). The boundaries are converted to that unit, and `window_length` is expressed in it. Without it, the report fails with an error naming the setting. `epoch_unit` is part of the event's definition (and of the aggregations scoped to it), so changing it recomputes them over all containers in incremental mode. diff --git a/src/impulse_query_engine/analyze/query/solvers/solver_config.py b/src/impulse_query_engine/analyze/query/solvers/solver_config.py index 65d51938..d704ca14 100644 --- a/src/impulse_query_engine/analyze/query/solvers/solver_config.py +++ b/src/impulse_query_engine/analyze/query/solvers/solver_config.py @@ -138,12 +138,14 @@ class SolverConfig(BaseModel): unit_conversion : TableConfig Column mappings and filters for the unit conversion table. epoch_unit : {"s", "ms", "us", "ns"} or None - Epoch unit of the channel sample timestamps (``tstart`` / ``tend``). When set, - ``TIMESTAMP``-typed container ``start_ts`` / ``stop_ts`` are converted to epoch - numbers in this unit for container-boundary events (``ContainerEvent``, - ``TimeWindowEvent``) and for expressions that request them in the solve. Only - required for a ``TimeWindowEvent`` over ``TIMESTAMP`` boundaries; when unset, - nothing is converted. + Epoch unit of the channel sample timestamps (``tstart`` / ``tend``, or + ``timestamp`` for RAW data). When set, ``TIMESTAMP``-typed container + ``start_ts`` / ``stop_ts`` are converted to epoch numbers in this unit for + container-boundary events (``ContainerEvent``, ``TimeWindowEvent``) and for + expressions that request them in the solve. Channel timestamps are never + converted; they must already be epoch numbers. Only required for a + ``TimeWindowEvent`` over ``TIMESTAMP`` boundaries; when unset, nothing is + converted. """ project_id: str | None = None diff --git a/tests/impulse_reporting/integration/time_window_event_test.py b/tests/impulse_reporting/integration/time_window_event_test.py index c2dd3c3b..b87d2a73 100644 --- a/tests/impulse_reporting/integration/time_window_event_test.py +++ b/tests/impulse_reporting/integration/time_window_event_test.py @@ -7,12 +7,14 @@ import pyspark.sql.types as T import pytest from databricks.sdk import WorkspaceClient +from pyspark.sql import Window -from impulse_query_engine.analyze.query.solvers.solver_config import SolverConfig +from impulse_query_engine.analyze.query.solvers.solver_config import RawEncoder, SolverConfig from impulse_reporting.aggregations.stats_aggregator import StatsAggregator from impulse_reporting.config.config_parser import ( Comparator, ContainerFilters, + DataType, ImpulseConfig, IncrementalConfig, MetricFilter, @@ -155,6 +157,8 @@ def test_time_window_event_in_report(spark, basic_narrow_db): # sec: seconds as doubles with a fractional window, so the boundaries round. # sec_ts: samples as seconds-as-double, container boundaries as TIMESTAMP (converted to # epoch seconds via epoch_unit="s"). +# us_ts: native µs samples, container boundaries as TIMESTAMP (converted to epoch µs via +# epoch_unit="us", the long path of the conversion). def _to_seconds(c): return c.cast("double") / F.lit(1e6) @@ -168,6 +172,12 @@ def _to_ns(c): "ns": (_to_ns, _to_ns, 600_000_000_007, None), "sec": (_to_seconds, _to_seconds, 600.3, None), "sec_ts": (_to_seconds, lambda c: F.timestamp_micros(c.cast("long")), 600.3, "s"), + "us_ts": ( + lambda c: c, + lambda c: F.timestamp_micros(c.cast("long")), + ALIGNED_WINDOW_LENGTH, + "us", + ), } @@ -224,13 +234,21 @@ def setup_tw_aligned_db(spark, setup_basic_db, request): # noqa: F811 spark.sql(f"DROP SCHEMA IF EXISTS {schema} CASCADE") -def _aligned_config(schema: str, table_prefix: str, epoch_unit=None, **extra) -> dict: +def _aligned_config( + schema: str, + table_prefix: str, + epoch_unit=None, + raw_encoder: RawEncoder | None = None, + channels_table: str = "channels", + **extra, +) -> dict: + """Report config over the aligned clone; a *raw_encoder* switches to ``data_type=RAW``.""" return dict( ImpulseConfig( source=Source( container_metrics_table=f"{schema}.container_metrics", channel_metrics_table=f"{schema}.channel_metrics", - channels_uri=f"{schema}.channels", + channels_uri=f"{schema}.{channels_table}", ), unity_sink=UnitySink( catalog="spark_catalog", schema="gold", table_prefix=table_prefix @@ -247,6 +265,8 @@ def _aligned_config(schema: str, table_prefix: str, epoch_unit=None, **extra) -> query_engine=QueryEngine( solver=Solvers.KEY_VALUE_STORE_SOLVER, solver_config=SolverConfig(epoch_unit=epoch_unit) if epoch_unit else None, + data_type=DataType.RAW if raw_encoder else DataType.RLE, + raw_encoder=raw_encoder, ), measurement_dimensions=["container_id", "start_ts", "stop_ts"], **extra, @@ -305,7 +325,9 @@ def _assert_ids_join(spark, table_prefix: str) -> tuple[set, set]: # noqa: F811 return stats_event_ids, event_ids -def _assert_window_stats_match_samples(spark, schema: str, table_prefix: str): # noqa: F811 +def _assert_window_stats_match_samples( + spark, schema: str, table_prefix: str, channels_table: str = "channels" # noqa: F811 +): """Each window's RPM min / max equal those of the silver samples overlapping that window, and windows without RPM samples carry no value (RPM only covers each container's first minute, so most windows are empty). @@ -319,15 +341,19 @@ def _assert_window_stats_match_samples(spark, schema: str, table_prefix: str): .filter(F.col("channel_name") == "Engine RPM") .select("container_id", "channel_id") ) - samples = ( - spark.read.table(f"{schema}.channels") - .join(rpm_channels, ["container_id", "channel_id"]) - .select( - "container_id", - F.col("tstart").cast("double").alias("tstart"), - F.col("tend").cast("double").alias("tend"), - F.col("value").cast("double").alias("value"), + channels = spark.read.table(f"{schema}.{channels_table}") + if "tend" not in channels.columns: + # RAW points: each sample is valid until the next one, the last one only at its own + # timestamp (the documented raw->interval rule of both encoders). + by_time = Window.partitionBy("container_id", "channel_id").orderBy("tstart") + channels = channels.withColumnRenamed("timestamp", "tstart").withColumn( + "tend", F.coalesce(F.lead("tstart").over(by_time), F.col("tstart")) ) + samples = channels.join(rpm_channels, ["container_id", "channel_id"]).select( + "container_id", + F.col("tstart").cast("double").alias("tstart"), + F.col("tend").cast("double").alias("tend"), + F.col("value").cast("double").alias("value"), ) windows = spark.read.table(f"spark_catalog.gold.{table_prefix}_event_instance_fact") expected = ( @@ -361,7 +387,9 @@ def _is_missing(value) -> bool: assert not mismatches, mismatches[:5] -@pytest.mark.parametrize("setup_tw_aligned_db", ["us", "ns", "sec", "sec_ts"], indirect=True) +@pytest.mark.parametrize( + "setup_tw_aligned_db", ["us", "ns", "sec", "sec_ts", "us_ts"], indirect=True +) def test_time_window_event_aggregation_join(spark, setup_tw_aligned_db): """Stats scoped to a TimeWindowEvent yield per-window values whose event_instance_id joins to the natively computed event fact, for µs, ns and seconds-as-double time bases, @@ -389,6 +417,52 @@ def test_time_window_event_aggregation_join(spark, setup_tw_aligned_db): _assert_window_stats_match_samples(spark, schema, table_prefix) +def _write_raw_channels(spark, schema: str) -> str: # noqa: F811 + """Write the aligned channels in the raw format (one ``timestamp`` per sample, no + ``tend``) next to the RLE table, and return the new table's name.""" + table = "channels_raw" + spark.read.table(f"{schema}.channels").select( + "container_id", "channel_id", F.col("tstart").alias("timestamp"), "value" + ).write.format("delta").mode("overwrite").saveAsTable(f"{schema}.{table}") + return table + + +@pytest.mark.parametrize("raw_encoder", [RawEncoder.RLE, RawEncoder.INTERVAL]) +@pytest.mark.parametrize("setup_tw_aligned_db", ["us", "us_ts"], indirect=True) +def test_time_window_event_aggregation_join_raw(spark, setup_tw_aligned_db, raw_encoder): + """With data_type=RAW both encoders derive [tstart, tend) from the raw ``timestamp`` + column without changing its unit, so windows over numeric or TIMESTAMP (epoch_unit="us") + container boundaries line up with the samples exactly as for RLE silver data.""" + schema, window_length, epoch_unit = setup_tw_aligned_db + channels_table = _write_raw_channels(spark, schema) + time_base = schema.removeprefix(_ALIGNED_SCHEMA + "_") + table_prefix = f"time_window_raw_test_{time_base}_{raw_encoder.value.lower()}" + my_report = Report( + name="time_window_raw_report", + spark=spark, + workspace_client=create_autospec(WorkspaceClient), + config=_aligned_config( + schema, + table_prefix, + epoch_unit=epoch_unit, + raw_encoder=raw_encoder, + channels_table=channels_table, + ), + ) + + window_evt = TimeWindowEvent(name="ten_min", window_length=window_length) + my_report.add_event(window_evt) + page = Page(page_number=1) + my_report.add_page(page) + page.add_aggregation(_rpm_stats(my_report, window_evt)) + + my_report.determine_report() + my_report.persist_results() + + _assert_ids_join(spark, table_prefix) + _assert_window_stats_match_samples(spark, schema, table_prefix, channels_table) + + def test_multiple_time_window_events_coexist(spark, basic_narrow_db): """Two TimeWindowEvents with different windows are allowed and both materialize.""" my_report = Report( From 7f02bba872a1215f311eb18f58c94b215b5b2ab2 Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 7 Oct 2026 12:15:23 +0200 Subject: [PATCH 11/27] feat(reporting, query-engine): validate TimeWindowEvent.max_windows_per_container and strengthen multi-event tests Expose `validate_max_windows` from `TimeWindowExpression` with a configurable parameter name so `TimeWindowEvent` can validate `max_windows_per_container` with an error message that names the caller's parameter. Harden the `test_multiple_time_window_events_coexist` integration test to verify that two events with different window lengths each tile every container, that scoped stats join only to their own event's windows, and that stats values match the underlying samples. Reuse the tiling assertion across other TimeWindowEvent tests. --- .../query/events/time_window_expression.py | 14 ++- .../events/time_window_event.py | 5 + .../integration/time_window_event_test.py | 117 ++++++++++++------ .../unit/events/time_window_event_test.py | 2 +- 4 files changed, 97 insertions(+), 41 deletions(-) diff --git a/src/impulse_query_engine/analyze/query/events/time_window_expression.py b/src/impulse_query_engine/analyze/query/events/time_window_expression.py index 200ef43e..40f569cf 100644 --- a/src/impulse_query_engine/analyze/query/events/time_window_expression.py +++ b/src/impulse_query_engine/analyze/query/events/time_window_expression.py @@ -32,17 +32,21 @@ _WINDOW_LIMIT_HINT = ( "Check that window_length is in the epoch unit of the container boundaries " - "(solver_config.epoch_unit), or raise max_windows_per_container." + "(solver_config.epoch_unit), or raise the limit (TimeWindowEvent " + "max_windows_per_container, TimeWindowExpression max_windows)." ) -def _validate_max_windows(max_windows: int) -> int: +def validate_max_windows(max_windows: int, param_name: str = "max_windows") -> int: """Return *max_windows* as an int, raising unless it is a positive integer. Parameters ---------- max_windows : int Maximum number of windows per container. + param_name : str, optional + Name of the caller's parameter, used in the error message (default + ``"max_windows"``). Returns ------- @@ -59,7 +63,7 @@ def _validate_max_windows(max_windows: int) -> int: or not isinstance(max_windows, numbers.Integral) or max_windows <= 0 ): - raise ValueError(f"max_windows must be a positive integer, got {max_windows!r}.") + raise ValueError(f"{param_name} must be a positive integer, got {max_windows!r}.") return int(max_windows) @@ -107,7 +111,7 @@ def window_intervals_col( ``array>`` with one ``[start, end]`` pair per window; empty when a boundary is null, NaN or infinite, or the span is not strictly positive. """ - max_windows = _validate_max_windows(max_windows) + max_windows = validate_max_windows(max_windows) start, stop = start_ts.cast("double"), stop_ts.cast("double") w = F.lit(float(window_length)) count = F.ceil((stop - start) / w) @@ -198,7 +202,7 @@ def __init__(self, window_length: float, max_windows: int = MAX_WINDOWS_PER_CONT # regardless of whether an int or float was passed: 10 and 10.0 are the same window # and must not trigger a spurious full recompute in incremental mode. self.window_length = float(window_length) - self.max_windows = _validate_max_windows(max_windows) + self.max_windows = validate_max_windows(max_windows) self.epoch_unit: str | None = None TimeSeriesExpression.__init__(self, is_single_signal=False) diff --git a/src/impulse_reporting/events/time_window_event.py b/src/impulse_reporting/events/time_window_event.py index 8b967b33..aa70e6be 100644 --- a/src/impulse_reporting/events/time_window_event.py +++ b/src/impulse_reporting/events/time_window_event.py @@ -16,6 +16,7 @@ from impulse_query_engine.analyze.query.events.time_window_expression import ( MAX_WINDOWS_PER_CONTAINER, TimeWindowExpression, + validate_max_windows, window_intervals_col, ) from impulse_query_engine.analyze.query.query_builder import QueryBuilder @@ -89,6 +90,10 @@ def __init__( f"TimeWindowEvent requires a strictly positive, finite window_length, " f"got {window_length!r}." ) + # Validated here so the error names this event's parameter, not the expression's. + max_windows_per_container = validate_max_windows( + max_windows_per_container, param_name="max_windows_per_container" + ) self.expression = TimeWindowExpression( window_length, max_windows=max_windows_per_container ).alias(name) diff --git a/tests/impulse_reporting/integration/time_window_event_test.py b/tests/impulse_reporting/integration/time_window_event_test.py index b87d2a73..83205288 100644 --- a/tests/impulse_reporting/integration/time_window_event_test.py +++ b/tests/impulse_reporting/integration/time_window_event_test.py @@ -274,10 +274,15 @@ def _aligned_config( ) -def _rpm_stats(report: Report, event: TimeWindowEvent, statistics=("min", "max", "mean")): +def _rpm_stats( + report: Report, + event: TimeWindowEvent, + statistics=("min", "max", "mean"), + name: str = "rpm_stats_per_window", +): query = report.get_db().query return StatsAggregator( - name="rpm_stats_per_window", + name=name, input_expressions=[query.channel(channel_name="Engine RPM")], channel_names=["Engine RPM"], statistics=list(statistics), @@ -463,48 +468,81 @@ def test_time_window_event_aggregation_join_raw(spark, setup_tw_aligned_db, raw_ _assert_window_stats_match_samples(spark, schema, table_prefix, channels_table) -def test_multiple_time_window_events_coexist(spark, basic_narrow_db): - """Two TimeWindowEvents with different windows are allowed and both materialize.""" +def _container_boundaries(spark, schema: str) -> dict: # noqa: F811 + """``{container_id: (start_ts, stop_ts)}`` of the clone's container_metrics, as doubles.""" + return { + r.container_id: (float(r.start_ts), float(r.stop_ts)) + for r in spark.read.table(f"{schema}.container_metrics") + .select("container_id", "start_ts", "stop_ts") + .collect() + } + + +def _assert_windows_tile_containers(rows, boundaries: dict, window_length: float) -> None: + """Each container's windows tile its ``[start_ts, stop_ts]`` exactly: they start at + ``start_ts``, are contiguous, all but the last are ``window_length`` long, and the last + one is clamped to ``stop_ts``.""" + for container_id, (start, stop) in boundaries.items(): + windows = sorted((r.start_ts, r.end_ts) for r in rows if r.container_id == container_id) + assert len(windows) == math.ceil((stop - start) / window_length), container_id + assert windows[0][0] == start and windows[-1][1] == stop, (container_id, windows) + assert all(prev[1] == nxt[0] for prev, nxt in zip(windows, windows[1:])), container_id + assert all(e - s == window_length for s, e in windows[:-1]), container_id + assert 0 < windows[-1][1] - windows[-1][0] <= window_length, container_id + + +def test_multiple_time_window_events_coexist(spark, setup_tw_aligned_db): + """Two TimeWindowEvents with different window lengths coexist in one report: each tiles + every container with its own windows, and the statistics scoped to each event carry + the values of that event's windows. The ids hash the event name, so window k of one + event never joins window k of the other.""" + schema, window_length, _ = setup_tw_aligned_db + table_prefix = "time_window_multi_test" my_report = Report( name="time_window_multi_report", spark=spark, workspace_client=create_autospec(WorkspaceClient), - config=dict(_config("time_window_multi_test")), + config=_aligned_config(schema, table_prefix), ) - evt_10s = TimeWindowEvent(name="ten_sec", window_length=WINDOW_LENGTH) - evt_30s = TimeWindowEvent(name="thirty_sec", window_length=3 * WINDOW_LENGTH) - my_report.add_event(evt_10s) - my_report.add_event(evt_30s) + evt_short = TimeWindowEvent(name="ten_min", window_length=window_length) + evt_long = TimeWindowEvent(name="thirty_min", window_length=3 * window_length) + my_report.add_event(evt_short) + my_report.add_event(evt_long) - query = my_report.get_db().query page = Page(page_number=1) my_report.add_page(page) - page.add_aggregation( - StatsAggregator( - name="rpm_stats", - input_expressions=[query.channel(channel_name="Engine RPM")], - channel_names=["Engine RPM"], - statistics=["mean"], - event=evt_10s, - desc="Engine RPM stats per 10s window", - ) - ) + page.add_aggregation(_rpm_stats(my_report, evt_short, name="rpm_stats_ten_min")) + page.add_aggregation(_rpm_stats(my_report, evt_long, name="rpm_stats_thirty_min")) my_report.determine_report() + my_report.persist_results() - rows = my_report.event_dfs["TIME_WINDOW_EVENT"]["changed"].collect() - names = {r.event_id for r in rows} - # Two distinct events (distinct event_ids) share the shared fact table. - assert names == {evt_10s.get_id(), evt_30s.get_id()} + event_fact = spark.read.table(f"spark_catalog.gold.{table_prefix}_event_instance_fact") + event_rows = event_fact.collect() + boundaries = _container_boundaries(spark, schema) + for event, length in ((evt_short, window_length), (evt_long, 3 * window_length)): + _assert_windows_tile_containers( + [r for r in event_rows if r.event_id == event.get_id()], boundaries, length + ) - # The 10s event produces strictly more windows than the 30s event. - count_10s = sum(1 for r in rows if r.event_id == evt_10s.get_id()) - count_30s = sum(1 for r in rows if r.event_id == evt_30s.get_id()) - assert count_10s > count_30s > 0 + # Every stats row joins a window of its own event, with its window's sample values. + _assert_ids_join(spark, table_prefix) + stats_fact = spark.read.table(f"spark_catalog.gold.{table_prefix}_stats_aggregator_fact") + cross_event = ( + stats_fact.select("event_instance_id", F.col("event_id").alias("stats_event_id")) + .join(event_fact.select("event_instance_id", "event_id"), "event_instance_id") + .filter(F.col("stats_event_id") != F.col("event_id")) + ) + assert cross_event.count() == 0 + assert {r.event_id for r in stats_fact.select("event_id").distinct().collect()} == { + evt_short.get_id(), + evt_long.get_id(), + } + _assert_window_stats_match_samples(spark, schema, table_prefix) dim_rows = my_report.event_metadata_dfs["TIME_WINDOW_EVENT"].collect() - assert {d.event_name for d in dim_rows} == {"ten_sec", "thirty_sec"} + assert {d.event_name for d in dim_rows} == {"ten_min", "thirty_min"} # --------------------------------------------------------------------------- @@ -531,12 +569,11 @@ def setup_tw_partial_db(spark, setup_basic_db): # noqa: F811 def _assert_windows_for_all_containers(rows) -> None: - for container_id in EXPECTED_CONTAINERS: - count = sum(1 for r in rows if r.container_id == container_id) - assert count == _expected_window_count(container_id), ( - f"container {container_id}: expected {_expected_window_count(container_id)} " - f"windows, got {count}" - ) + """Every filtered container is tiled into WINDOW_LENGTH windows over its boundaries.""" + boundaries = { + cid: (float(b["start_ts"]), float(b["stop_ts"])) for cid, b in EXPECTED_CONTAINERS.items() + } + _assert_windows_tile_containers(rows, boundaries, WINDOW_LENGTH) def test_time_window_event_covers_containers_without_aggregated_channel( @@ -573,6 +610,16 @@ def test_time_window_event_covers_containers_without_aggregated_channel( rows = my_report.event_dfs["TIME_WINDOW_EVENT"]["changed"].collect() _assert_windows_for_all_containers(rows) + # The stats cover every window of the containers that have Engine RPM, and none of + # container 3, whose windows exist regardless. + window_ids = {r.event_instance_id for r in rows} + stats_rows = my_report.aggregation_dfs["STATS_AGGREGATOR"]["changed"].collect() + assert {r.container_id for r in stats_rows} == {1, 2} + assert {r.event_instance_id for r in stats_rows} == { + r.event_instance_id for r in rows if r.container_id in (1, 2) + } + assert {r.event_instance_id for r in stats_rows} <= window_ids + def test_standalone_time_window_event_covers_all_containers(spark, basic_narrow_db): """A TimeWindowEvent with no aggregation (nothing to solve) still materializes windows.""" diff --git a/tests/impulse_reporting/unit/events/time_window_event_test.py b/tests/impulse_reporting/unit/events/time_window_event_test.py index e33e775e..9f0ff2ef 100644 --- a/tests/impulse_reporting/unit/events/time_window_event_test.py +++ b/tests/impulse_reporting/unit/events/time_window_event_test.py @@ -133,7 +133,7 @@ def test_max_windows_per_container_excluded_from_hash(): @pytest.mark.parametrize("bad", [0, -1, 2.5, True, None]) def test_invalid_max_windows_per_container_raises(bad): - with pytest.raises(ValueError, match="max_windows must be a positive integer"): + with pytest.raises(ValueError, match="^max_windows_per_container must be a positive integer"): TimeWindowEvent(name="w", window_length=10, max_windows_per_container=bad) From e9a1b66e367cae859cad24eb93d2d6da828380dd Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 7 Oct 2026 12:45:48 +0200 Subject: [PATCH 12/27] refactor(tests): use itertools.pairwise for adjacent window assertions --- tests/impulse_reporting/integration/time_window_event_test.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tests/impulse_reporting/integration/time_window_event_test.py b/tests/impulse_reporting/integration/time_window_event_test.py index 83205288..96101526 100644 --- a/tests/impulse_reporting/integration/time_window_event_test.py +++ b/tests/impulse_reporting/integration/time_window_event_test.py @@ -1,6 +1,7 @@ """Integration tests for TimeWindowEvent with end-to-end Report usage.""" import math +from itertools import pairwise from unittest.mock import create_autospec import pyspark.sql.functions as F @@ -486,7 +487,7 @@ def _assert_windows_tile_containers(rows, boundaries: dict, window_length: float windows = sorted((r.start_ts, r.end_ts) for r in rows if r.container_id == container_id) assert len(windows) == math.ceil((stop - start) / window_length), container_id assert windows[0][0] == start and windows[-1][1] == stop, (container_id, windows) - assert all(prev[1] == nxt[0] for prev, nxt in zip(windows, windows[1:])), container_id + assert all(prev[1] == nxt[0] for prev, nxt in pairwise(windows)), container_id assert all(e - s == window_length for s, e in windows[:-1]), container_id assert 0 < windows[-1][1] - windows[-1][0] <= window_length, container_id From 54a0ed6d04c35983c1cb5da88e4e14f8fbc502b7 Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 7 Oct 2026 13:43:20 +0200 Subject: [PATCH 13/27] refactor(reporting): move shared event metadata helpers to ContainerBoundaryEvent Move `get_id`, `as_spark_row`, and `determine_metadata_df` from `ContainerEvent` and `TimeWindowEvent` into their common base class `ContainerBoundaryEvent` to eliminate duplication. Remove redundant `window_length` and `Intervals` validation from `TimeWindowEvent` now handled by `TimeWindowExpression`. Update API docs to reflect the removed methods on subclasses. --- .../events/container_event.md | 42 ------------- .../events/time_window_event.md | 42 ------------- .../events/container_boundary_event.py | 45 +++++++++++++- .../events/container_event.py | 42 +------------ .../events/time_window_event.py | 60 +------------------ 5 files changed, 48 insertions(+), 183 deletions(-) diff --git a/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md b/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md index 5127cd57..5c03dcc7 100644 --- a/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md +++ b/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md @@ -33,18 +33,6 @@ Initialise a ContainerEvent. - `desc` (`str`): Human-readable description. - `attributes` (`dict`): Key-value metadata for the event. -#### get\_id - -```python -def get_id() -> int -``` - -Return a unique identifier derived from the event name. - -**Returns**: - -`int`: Positive 32-bit integer identifier. - #### get\_expression ```python @@ -100,18 +88,6 @@ Return a dictionary representation of the event. `dict`: -#### as\_spark\_row - -```python -def as_spark_row() -> Row -``` - -Return a Spark ``Row`` representation. - -**Returns**: - -`Row`: - #### determine\_events ```python @@ -144,21 +120,3 @@ produces one event instance per container. `DataFrame`: Spark DataFrame matching ``EVENT_INSTANCE_FACT_SCHEMA``. -#### determine\_metadata\_df - -```python -def determine_metadata_df(cls, spark: SparkSession, - events: list[ContainerEvent]) -> DataFrame -``` - -Create a Spark DataFrame containing event metadata. - -**Arguments**: - -- `spark` (`SparkSession`): Active Spark session. -- `events` (`list of ContainerEvent`): List of ContainerEvent objects. - -**Returns**: - -`DataFrame`: Spark DataFrame matching ``EVENT_DIMENSION_SCHEMA``. - diff --git a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md index 67fcb46c..7b107ed1 100644 --- a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md +++ b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md @@ -70,18 +70,6 @@ this event and of the aggregations scoped to it. - `epoch_unit` (`str or None`): The report's ``solver_config.epoch_unit``. -#### get\_id - -```python -def get_id() -> int -``` - -Returns a unique identifier for the event. - -**Returns**: - -`int`: Unique positive 32-bit integer identifier for the event. - #### get\_expression ```python @@ -138,18 +126,6 @@ Get a dictionary representation of the event. `dict`: Dictionary containing event metadata. -#### as\_spark\_row - -```python -def as_spark_row() -> Row -``` - -Get a Spark Row representation of the event. - -**Returns**: - -`Row`: Spark Row containing event metadata. - #### determine\_events ```python @@ -187,21 +163,3 @@ computes the same windows in the same order for scoped aggregations (see `DataFrame`: Spark DataFrame containing event instance facts. -#### determine\_metadata\_df - -```python -def determine_metadata_df(cls, spark: SparkSession, - events: list[TimeWindowEvent]) -``` - -Create a Spark DataFrame containing event metadata. - -**Arguments**: - -- `spark` (`SparkSession`): Spark session for data processing. -- `events` (`list of TimeWindowEvent`): List of TimeWindowEvent objects. - -**Returns**: - -`DataFrame`: Spark DataFrame containing event metadata. - diff --git a/src/impulse_reporting/events/container_boundary_event.py b/src/impulse_reporting/events/container_boundary_event.py index 3172679a..60ce4c02 100644 --- a/src/impulse_reporting/events/container_boundary_event.py +++ b/src/impulse_reporting/events/container_boundary_event.py @@ -2,11 +2,14 @@ from __future__ import annotations -from pyspark.sql import DataFrame, SparkSession +import zlib + +from pyspark.sql import DataFrame, Row, SparkSession from impulse_query_engine.analyze.query.query_builder import QueryBuilder from impulse_query_engine.analyze.query.solvers.query_solver import QuerySolver from impulse_reporting.events.event import Event +from impulse_reporting.persist.dimension_schema import EVENT_DIMENSION_SCHEMA class ContainerBoundaryEvent(Event): @@ -38,6 +41,46 @@ def set_epoch_unit(self, epoch_unit: str | None) -> None: """ self.epoch_unit = epoch_unit + def get_id(self) -> int: + """Return a unique identifier derived from the event name. + + Returns + ------- + int + Positive 32-bit integer identifier. + """ + return zlib.crc32(self.name.encode()) & 0x7FFFFFFF + + def as_spark_row(self) -> Row: + """Return a Spark ``Row`` representation of :meth:`as_dict`. + + Returns + ------- + Row + """ + return Row(**self.as_dict()) + + @classmethod + def determine_metadata_df( + cls, spark: SparkSession, events: list[ContainerBoundaryEvent] + ) -> DataFrame: + """Create a Spark DataFrame containing event metadata. + + Parameters + ---------- + spark : SparkSession + Active Spark session. + events : list of ContainerBoundaryEvent + Events of one container-boundary type. + + Returns + ------- + DataFrame + Spark DataFrame matching ``EVENT_DIMENSION_SCHEMA``. + """ + rows = [event.as_spark_row() for event in events] + return spark.createDataFrame(rows, schema=EVENT_DIMENSION_SCHEMA) + @staticmethod def resolve_container_metrics( spark: SparkSession, diff --git a/src/impulse_reporting/events/container_event.py b/src/impulse_reporting/events/container_event.py index caa961f7..b1d5f429 100644 --- a/src/impulse_reporting/events/container_event.py +++ b/src/impulse_reporting/events/container_event.py @@ -6,13 +6,11 @@ from typing import TYPE_CHECKING import pyspark.sql.functions as f -import zlib -from pyspark.sql import DataFrame, Row, SparkSession +from pyspark.sql import DataFrame, SparkSession from impulse_query_engine.analyze.query.query_builder import QueryBuilder from impulse_query_engine.analyze.query.solvers.query_solver import QuerySolver from impulse_reporting.events.container_boundary_event import ContainerBoundaryEvent -from impulse_reporting.persist.dimension_schema import EVENT_DIMENSION_SCHEMA from impulse_reporting.persist.fact_schema import EVENT_INSTANCE_FACT_SCHEMA from impulse_reporting.util.event_instance_util import generate_event_instance_id_column from impulse_reporting.util.report_entity_util import ReportEntityUtil @@ -55,16 +53,6 @@ def __init__(self, name: str, desc: str = None, attributes: dict[str, str] = Non # Instance methods # ------------------------------------------------------------------ - def get_id(self) -> int: - """Return a unique identifier derived from the event name. - - Returns - ------- - int - Positive 32-bit integer identifier. - """ - return zlib.crc32(self.name.encode()) & 0x7FFFFFFF - def get_expression(self) -> TimeSeriesExpression | None: """ContainerEvent has no time-series expression. @@ -124,15 +112,6 @@ def as_dict(self) -> dict: "attributes": self.attributes, } - def as_spark_row(self) -> Row: - """Return a Spark ``Row`` representation. - - Returns - ------- - Row - """ - return Row(**self.as_dict()) - # ------------------------------------------------------------------ # Class methods # ------------------------------------------------------------------ @@ -209,22 +188,3 @@ def determine_events( # Select only the columns defined in the fact schema return df.select(EVENT_INSTANCE_FACT_SCHEMA.fieldNames()) - - @classmethod - def determine_metadata_df(cls, spark: SparkSession, events: list[ContainerEvent]) -> DataFrame: - """Create a Spark DataFrame containing event metadata. - - Parameters - ---------- - spark : SparkSession - Active Spark session. - events : list of ContainerEvent - List of ContainerEvent objects. - - Returns - ------- - DataFrame - Spark DataFrame matching ``EVENT_DIMENSION_SCHEMA``. - """ - rows = [event.as_spark_row() for event in events] - return spark.createDataFrame(rows, schema=EVENT_DIMENSION_SCHEMA) diff --git a/src/impulse_reporting/events/time_window_event.py b/src/impulse_reporting/events/time_window_event.py index aa70e6be..8579c286 100644 --- a/src/impulse_reporting/events/time_window_event.py +++ b/src/impulse_reporting/events/time_window_event.py @@ -3,12 +3,10 @@ from __future__ import annotations import hashlib -import math from collections.abc import Mapping import pyspark.sql.functions as f -import zlib -from pyspark.sql import DataFrame, Row, SparkSession +from pyspark.sql import DataFrame, SparkSession from impulse_query_engine.analyze.metadata.time_series_expression import ( TimeSeriesExpression, @@ -21,9 +19,7 @@ ) from impulse_query_engine.analyze.query.query_builder import QueryBuilder from impulse_query_engine.analyze.query.solvers.query_solver import QuerySolver -from impulse_query_engine.model.series.intervals import Intervals from impulse_reporting.events.container_boundary_event import ContainerBoundaryEvent -from impulse_reporting.persist.dimension_schema import EVENT_DIMENSION_SCHEMA from impulse_reporting.persist.fact_schema import EVENT_INSTANCE_FACT_SCHEMA from impulse_reporting.util.event_instance_util import generate_event_instance_id_column from impulse_reporting.util.report_entity_util import ReportEntityUtil @@ -85,12 +81,8 @@ def __init__( ``max_windows_per_container`` is not a positive integer. """ ContainerBoundaryEvent.__init__(self, name) - if window_length is None or not math.isfinite(window_length) or window_length <= 0: - raise ValueError( - f"TimeWindowEvent requires a strictly positive, finite window_length, " - f"got {window_length!r}." - ) - # Validated here so the error names this event's parameter, not the expression's. + # window_length is validated by TimeWindowExpression. max_windows_per_container is + # validated here so the error names this event's parameter, not the expression's. max_windows_per_container = validate_max_windows( max_windows_per_container, param_name="max_windows_per_container" ) @@ -101,9 +93,6 @@ def __init__( # the solve and event_dimension all see the same value for 10 and 10.0. self.window_length = self.expression.window_length self.max_windows_per_container = self.expression.max_windows - self.expression.require_evaluation_type( - Intervals, owner="TimeWindowEvent", example="window_length=60000" - ) self.description = desc self.required_channels = required_channels normalized_attributes: dict[str, str] = {} @@ -128,18 +117,6 @@ def set_epoch_unit(self, epoch_unit: str | None) -> None: ContainerBoundaryEvent.set_epoch_unit(self, epoch_unit) self.expression.epoch_unit = epoch_unit - def get_id(self) -> int: - """ - Returns a unique identifier for the event. - - Returns - ------- - int - Unique positive 32-bit integer identifier for the event. - """ - hash_input = f"{self.name}" - return zlib.crc32(hash_input.encode()) & 0x7FFFFFFF # Ensures positive 32-bit int - def get_expression(self) -> TimeSeriesExpression | None: """ Get the time series expression associated with the event. @@ -205,17 +182,6 @@ def as_dict(self) -> dict: "attributes": self.attributes, } - def as_spark_row(self) -> Row: - """ - Get a Spark Row representation of the event. - - Returns - ------- - Row - Spark Row containing event metadata. - """ - return Row(**self.as_dict()) - @classmethod def determine_events( cls, @@ -310,23 +276,3 @@ def determine_events( .select(EVENT_INSTANCE_FACT_SCHEMA.fieldNames()) ) return df - - @classmethod - def determine_metadata_df(cls, spark: SparkSession, events: list[TimeWindowEvent]): - """ - Create a Spark DataFrame containing event metadata. - - Parameters - ---------- - spark : SparkSession - Spark session for data processing. - events : list of TimeWindowEvent - List of TimeWindowEvent objects. - - Returns - ------- - DataFrame - Spark DataFrame containing event metadata. - """ - events = [event.as_spark_row() for event in events] - return spark.createDataFrame(events, schema=EVENT_DIMENSION_SCHEMA) From 4f9aebe0d9490213c5a607fee2029f7ea9aaba31 Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 7 Oct 2026 14:03:43 +0200 Subject: [PATCH 14/27] test(reporting): harden TimeWindowEvent integration assertions and scope boundary helper Refactor `test_time_window_event_in_report` to reuse the shared `_assert_windows_for_all_containers` helper. Update `_container_boundaries` to filter on the report's `vehicle_key` scope so boundary expectations match the containers actually processed. Strengthen `_assert_windows_tile_containers` to compute exact expected windows using the same double arithmetic as the event, covering edge cases like zero-span containers and fractional windows, and drop the `itertools.pairwise` dependency. Remove a redundant `event_instance_id` subset check in the aggregation coverage test. --- .../integration/time_window_event_test.py | 47 ++++++++----------- 1 file changed, 20 insertions(+), 27 deletions(-) diff --git a/tests/impulse_reporting/integration/time_window_event_test.py b/tests/impulse_reporting/integration/time_window_event_test.py index 96101526..cabfd152 100644 --- a/tests/impulse_reporting/integration/time_window_event_test.py +++ b/tests/impulse_reporting/integration/time_window_event_test.py @@ -1,7 +1,6 @@ """Integration tests for TimeWindowEvent with end-to-end Report usage.""" import math -from itertools import pairwise from unittest.mock import create_autospec import pyspark.sql.functions as F @@ -117,21 +116,8 @@ def test_time_window_event_in_report(spark, basic_narrow_db): total_expected = sum(_expected_window_count(cid) for cid in EXPECTED_CONTAINERS) assert len(rows) == total_expected - for container_id, expected in EXPECTED_CONTAINERS.items(): - windows = sorted( - ((r.start_ts, r.end_ts) for r in rows if r.container_id == container_id), - key=lambda w: w[0], - ) - assert len(windows) == _expected_window_count(container_id) - # First window starts at the container start. - assert windows[0][0] == expected["start_ts"] - # Windows are contiguous: each end equals the next start. - for (_, end), (nxt_start, _) in zip(windows, windows[1:], strict=False): - assert end == nxt_start - # Final window is clamped to the container stop. - assert windows[-1][1] == expected["stop_ts"] - # Every window is a valid, non-empty interval. - assert all(start < end for start, end in windows) + _assert_windows_for_all_containers(rows) + for container_id in EXPECTED_CONTAINERS: # Per-window instances are distinct (unlike ContainerEvent's single id). instance_ids = [r.event_instance_id for r in rows if r.container_id == container_id] assert len(set(instance_ids)) == len(instance_ids) @@ -470,26 +456,35 @@ def test_time_window_event_aggregation_join_raw(spark, setup_tw_aligned_db, raw_ def _container_boundaries(spark, schema: str) -> dict: # noqa: F811 - """``{container_id: (start_ts, stop_ts)}`` of the clone's container_metrics, as doubles.""" + """``{container_id: (start_ts, stop_ts)}`` as doubles, for the containers in the report's + scope (``_aligned_config`` filters on ``vehicle_key == "Seat_Leon"``).""" return { r.container_id: (float(r.start_ts), float(r.stop_ts)) for r in spark.read.table(f"{schema}.container_metrics") + .filter(F.col("vehicle_key") == "Seat_Leon") .select("container_id", "start_ts", "stop_ts") .collect() } def _assert_windows_tile_containers(rows, boundaries: dict, window_length: float) -> None: - """Each container's windows tile its ``[start_ts, stop_ts]`` exactly: they start at - ``start_ts``, are contiguous, all but the last are ``window_length`` long, and the last - one is clamped to ``stop_ts``.""" + """Each container's windows tile its ``[start_ts, stop_ts]`` exactly: window ``i`` spans + ``[start + i * W, min(start + (i + 1) * W, stop)]``, so the windows start at + ``start_ts``, are contiguous and the last one is clamped to ``stop_ts``. + + The expected boundaries use the same double arithmetic as the event, so the comparison + is exact for every time base (ns epochs, fractional windows). A container without a + positive span expects no windows. + """ for container_id, (start, stop) in boundaries.items(): windows = sorted((r.start_ts, r.end_ts) for r in rows if r.container_id == container_id) - assert len(windows) == math.ceil((stop - start) / window_length), container_id - assert windows[0][0] == start and windows[-1][1] == stop, (container_id, windows) - assert all(prev[1] == nxt[0] for prev, nxt in pairwise(windows)), container_id - assert all(e - s == window_length for s, e in windows[:-1]), container_id - assert 0 < windows[-1][1] - windows[-1][0] <= window_length, container_id + count = math.ceil((stop - start) / window_length) if stop > start else 0 + expected = [ + (start + i * window_length, min(start + (i + 1) * window_length, stop)) + for i in range(count) + ] + expected = [(s, e) for s, e in expected if s < e] + assert windows == expected, (container_id, windows[:3], expected[:3]) def test_multiple_time_window_events_coexist(spark, setup_tw_aligned_db): @@ -613,13 +608,11 @@ def test_time_window_event_covers_containers_without_aggregated_channel( # The stats cover every window of the containers that have Engine RPM, and none of # container 3, whose windows exist regardless. - window_ids = {r.event_instance_id for r in rows} stats_rows = my_report.aggregation_dfs["STATS_AGGREGATOR"]["changed"].collect() assert {r.container_id for r in stats_rows} == {1, 2} assert {r.event_instance_id for r in stats_rows} == { r.event_instance_id for r in rows if r.container_id in (1, 2) } - assert {r.event_instance_id for r in stats_rows} <= window_ids def test_standalone_time_window_event_covers_all_containers(spark, basic_narrow_db): From 4a92625292ce2b8ff870c9ed0f017e53f543ce2d Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 7 Oct 2026 14:18:13 +0200 Subject: [PATCH 15/27] docs(query-engine, skills): clarify epoch_unit only converts container_metrics start_ts/stop_ts Clarify across configuration docs, API reference, skills, and docstrings that `solver_config.epoch_unit` is the epoch unit of the `channels` table timestamps (which are never converted) and that only `TIMESTAMP`-typed `container_metrics.start_ts`/`stop_ts` are converted to epoch numbers. Update the impulse-config skill example to use `start_ts`/`stop_ts` column names and add the `epoch_unit` description. Tighten the TimeWindowExpression error message to name the converted columns explicitly. --- docs/impulse/docs/config/configuration.md | 4 ++-- .../analyze/query/solvers/solver_config.md | 15 +++++++-------- skills/impulse-config/SKILL.md | 11 ++++++++++- skills/impulse-events/SKILL.md | 7 +++++-- .../query/events/time_window_expression.py | 2 +- .../analyze/query/solvers/solver_config.py | 18 +++++++++--------- 6 files changed, 34 insertions(+), 23 deletions(-) diff --git a/docs/impulse/docs/config/configuration.md b/docs/impulse/docs/config/configuration.md index c648c11a..af67f69c 100644 --- a/docs/impulse/docs/config/configuration.md +++ b/docs/impulse/docs/config/configuration.md @@ -175,8 +175,8 @@ Top-level fields on `SolverConfig`: channel timestamps are never converted and must already be epoch numbers. Only needed when `container_metrics.start_ts`/`stop_ts` are `TIMESTAMP` columns **and** the report uses a `TimeWindowEvent`, whose windows must be in the samples' time base. Such a report fails with a - clear error until it is set. When set, `TIMESTAMP` boundaries are converted to epoch numbers in - that unit: + clear error until it is set. When set, `TIMESTAMP`-typed `container_metrics.start_ts`/`stop_ts` + are converted to epoch numbers in that unit: - `ContainerEvent` and `TimeWindowEvent` write `start_ts`/`end_ts` in that unit. With `"s"`, the values are identical to the default. - Expressions that request `start_ts`/`stop_ts` as container metrics (e.g. via diff --git a/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md b/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md index 4aa010be..7610f845 100644 --- a/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md +++ b/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md @@ -126,14 +126,13 @@ so that solver code can always reference the same constants. override for the channel mapping (alias) table. - `channels` (`TableConfig`): Column mappings and filters for the channel data table. - `unit_conversion` (`TableConfig`): Column mappings and filters for the unit conversion table. -- `epoch_unit` (`{"s", "ms", "us", "ns"} or None`): Epoch unit of the channel sample timestamps (``tstart`` / ``tend``, or -``timestamp`` for RAW data). When set, ``TIMESTAMP``-typed container -``start_ts`` / ``stop_ts`` are converted to epoch numbers in this unit for -container-boundary events (``ContainerEvent``, ``TimeWindowEvent``) and for -expressions that request them in the solve. Channel timestamps are never -converted; they must already be epoch numbers. Only required for a -``TimeWindowEvent`` over ``TIMESTAMP`` boundaries; when unset, nothing is -converted. +- `epoch_unit` (`{"s", "ms", "us", "ns"} or None`): Epoch unit of the timestamps in the ``channels`` table (``tstart`` / ``tend``, +or ``timestamp`` for RAW data); these are never converted. Only +``TIMESTAMP``-typed ``start_ts`` / ``stop_ts`` of the ``container_metrics`` +table are converted, into epoch numbers in this unit so they match the channel +timestamps. ``ContainerEvent``, ``TimeWindowEvent`` and expressions that read +these columns see the converted values. Only needed for a ``TimeWindowEvent`` +over ``TIMESTAMP`` container boundaries; unset means nothing is converted. #### from\_json diff --git a/skills/impulse-config/SKILL.md b/skills/impulse-config/SKILL.md index 6e2c9eed..1994715f 100644 --- a/skills/impulse-config/SKILL.md +++ b/skills/impulse-config/SKILL.md @@ -124,6 +124,13 @@ rejected (use `drop_implausible_data`), any other channels filter warns. Top-level `project_id` (str, optional) applies an equality filter on the `project_id` column of every table that has one (`container_tags`, `container_metrics`, `channel_mapping`). Omit if not needed. +Top-level `epoch_unit` (`"s"` | `"ms"` | `"us"` | `"ns"`, optional) is the epoch unit of the `channels` +timestamps (`tstart`/`tend`, or `timestamp` with `data_type="RAW"`); those are never converted. Only +`TIMESTAMP`-typed `container_metrics.start_ts`/`stop_ts` are converted, into epoch numbers in that +unit, so `ContainerEvent` / `TimeWindowEvent` boundaries match the channel timestamps. Needed only +for a `TimeWindowEvent` over `TIMESTAMP` boundaries; with `"ns"`, boundaries must lie between +1677-09-21 and 2262-04-11 (the int64 nanosecond range). + ```python "query_engine": { "solver": "DefaultSolver", @@ -133,7 +140,9 @@ table that has one (`container_tags`, `container_metrics`, `channel_mapping`). O "column_name_mapping": {"entity_id": "container_id"}, "filters": {"parent_id": "my_parent_id"} }, - "container_metrics": {"column_name_mapping": {"start_dt": "tstart", "stop_dt": "tend"}}, + "container_metrics": { + "column_name_mapping": {"measurement_start": "start_ts", "measurement_end": "stop_ts"} + }, "channel_mapping": {"filters": {"toolbox_id": "my_toolbox"}} } } diff --git a/skills/impulse-events/SKILL.md b/skills/impulse-events/SKILL.md index 2d21cade..5eeaea24 100644 --- a/skills/impulse-events/SKILL.md +++ b/skills/impulse-events/SKILL.md @@ -157,8 +157,11 @@ or without channel data or a scoped aggregation. Pair it with an aggregation sco `StatsAggregator(..., event=...)`) to compute one statistic per window; those rows carry the same `event_instance_id` values as the windows (the id hashes container, event name and window position). Because the windows come from `container_metrics`, those boundaries must share the channel samples' -time base for the per-window values to be meaningful. Containers with null, NaN or infinite -boundaries get no windows. +time base for the per-window values to be meaningful. If `container_metrics.start_ts`/`stop_ts` are +`TIMESTAMP` columns, set `query_engine.solver_config.epoch_unit` to the unit of the channel +timestamps (`tstart`/`tend`, or `timestamp` for RAW); only those two container columns are +converted, never the channel timestamps. Containers with null, NaN or infinite boundaries get no +windows. ## Output schema diff --git a/src/impulse_query_engine/analyze/query/events/time_window_expression.py b/src/impulse_query_engine/analyze/query/events/time_window_expression.py index 40f569cf..aacad549 100644 --- a/src/impulse_query_engine/analyze/query/events/time_window_expression.py +++ b/src/impulse_query_engine/analyze/query/events/time_window_expression.py @@ -331,7 +331,7 @@ def build(self, cache: SeriesCache) -> Intervals: f"TimeWindowExpression needs epoch-number container boundaries, but " f"{name} is {type(value).__name__}. For TIMESTAMP columns, set " "solver_config.epoch_unit to the epoch unit of the channel sample " - "timestamps so they are converted before the solve." + "timestamps so start_ts / stop_ts are converted before the solve." ) # Mirror window_intervals_col exactly: convert to double *before* subtracting. A diff --git a/src/impulse_query_engine/analyze/query/solvers/solver_config.py b/src/impulse_query_engine/analyze/query/solvers/solver_config.py index d704ca14..3c2eb3de 100644 --- a/src/impulse_query_engine/analyze/query/solvers/solver_config.py +++ b/src/impulse_query_engine/analyze/query/solvers/solver_config.py @@ -138,14 +138,13 @@ class SolverConfig(BaseModel): unit_conversion : TableConfig Column mappings and filters for the unit conversion table. epoch_unit : {"s", "ms", "us", "ns"} or None - Epoch unit of the channel sample timestamps (``tstart`` / ``tend``, or - ``timestamp`` for RAW data). When set, ``TIMESTAMP``-typed container - ``start_ts`` / ``stop_ts`` are converted to epoch numbers in this unit for - container-boundary events (``ContainerEvent``, ``TimeWindowEvent``) and for - expressions that request them in the solve. Channel timestamps are never - converted; they must already be epoch numbers. Only required for a - ``TimeWindowEvent`` over ``TIMESTAMP`` boundaries; when unset, nothing is - converted. + Epoch unit of the timestamps in the ``channels`` table (``tstart`` / ``tend``, + or ``timestamp`` for RAW data); these are never converted. Only + ``TIMESTAMP``-typed ``start_ts`` / ``stop_ts`` of the ``container_metrics`` + table are converted, into epoch numbers in this unit so they match the channel + timestamps. ``ContainerEvent``, ``TimeWindowEvent`` and expressions that read + these columns see the converted values. Only needed for a ``TimeWindowEvent`` + over ``TIMESTAMP`` container boundaries; unset means nothing is converted. """ project_id: str | None = None @@ -514,5 +513,6 @@ def require_epoch_boundaries(self, df: DataFrame, owner: str) -> None: f"{owner} needs epoch-number container boundaries, but container_metrics " f"column '{field.name}' has type {field.dataType.simpleString()}. Set " "query_engine.solver_config.epoch_unit to the epoch unit of the channel " - "sample timestamps (one of 's', 'ms', 'us', 'ns') so it is converted." + "sample timestamps (one of 's', 'ms', 'us', 'ns') so that column is " + "converted to epoch numbers in that unit." ) From 0a238e388daf0f5bcea26f891a728988c6449eab Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 7 Oct 2026 15:39:11 +0200 Subject: [PATCH 16/27] feat(query-engine, reporting): replace epoch_unit with channel_time_unit/channel_time_origin for TimeWindowEvent Rename `solver_config.epoch_unit` to `channel_time_unit` and add `channel_time_origin` (`"epoch"` default, `"container_start"`). The new settings describe the time frame of channel sample timestamps and are used only to compute `TimeWindowEvent` windows from `container_metrics.start_ts`/`stop_ts`; the raw boundary columns are no longer converted in place and remain visible unchanged to `ContainerEvent`, `measurement_dimension`, and UDFs. - Add `SolverConfig.with_window_bounds` to derive prefixed `__window_start`/`__window_stop` columns in the channel time frame, supporting both absolute epoch and container-start-relative origins. - Remove boundary conversion from `ContainerBoundaryEvent` and `ContainerEvent`; drop `epoch_unit` from their definition hashes. - Update `TimeWindowExpression` and `TimeWindowEvent` to read the derived window-bound columns and include the channel time frame in definition hashes. - Update configuration docs, API references, skills, and tests. --- docs/impulse/docs/config/configuration.md | 46 ++-- .../docs/data_model/silver_layer_schema.md | 7 +- .../analyze/query/solvers/solver_config.md | 100 ++++--- .../events/container_event.md | 4 +- .../events/time_window_event.md | 23 +- docs/impulse/docs/references/report/event.md | 25 +- skills/impulse-config/SKILL.md | 14 +- skills/impulse-events/SKILL.md | 9 +- .../query/events/time_window_expression.py | 85 +++--- .../analyze/query/solvers/default_solver.py | 10 +- .../analyze/query/solvers/solver_config.py | 175 ++++++++----- src/impulse_reporting/core/report.py | 10 +- .../events/container_boundary_event.py | 29 +- .../events/container_event.py | 6 +- .../events/time_window_event.py | 41 +-- .../events/time_window_expression_test.py | 35 +-- .../solvers/container_boundaries_test.py | 122 ++++++--- .../default_solver_container_metadata_test.py | 89 +++++-- .../integration/time_window_event_test.py | 247 +++++++++++++----- .../unit/aggregations/definition_hash_test.py | 8 +- .../unit/events/container_event_test.py | 20 +- .../unit/events/time_window_event_test.py | 32 +-- 22 files changed, 690 insertions(+), 447 deletions(-) diff --git a/docs/impulse/docs/config/configuration.md b/docs/impulse/docs/config/configuration.md index af67f69c..50f18868 100644 --- a/docs/impulse/docs/config/configuration.md +++ b/docs/impulse/docs/config/configuration.md @@ -170,26 +170,30 @@ Top-level fields on `SolverConfig`: the `project_id` column (after column-name mapping) of every table it reads that carries one — `container_tags` (if configured), `container_metrics`, and `channel_mapping` (if configured). Omit it if you don't need project-level scoping; the solver does not require it. -- `epoch_unit` (`"s"` | `"ms"` | `"us"` | `"ns"`, optional): Epoch unit of the channel sample - timestamps (`tstart`/`tend`, or `timestamp` when `data_type = "RAW"`). It only describes them: - channel timestamps are never converted and must already be epoch numbers. Only needed when - `container_metrics.start_ts`/`stop_ts` are `TIMESTAMP` columns **and** the report uses a - `TimeWindowEvent`, whose windows must be in the samples' time base. Such a report fails with a - clear error until it is set. When set, `TIMESTAMP`-typed `container_metrics.start_ts`/`stop_ts` - are converted to epoch numbers in that unit: - - `ContainerEvent` and `TimeWindowEvent` write `start_ts`/`end_ts` in that unit. With `"s"`, the - values are identical to the default. - - Expressions that request `start_ts`/`stop_ts` as container metrics (e.g. via - `apply(..., container_metrics=[...])`) receive epoch numbers instead of timestamps. - - `measurement_dimension` and container filters always see the original columns. When unset, - nothing is converted. `TIMESTAMP_NTZ` and `DATE` boundaries are not supported. - - `epoch_unit` is part of the definition hash of `ContainerEvent`, `TimeWindowEvent` and every - aggregation scoped to a `TimeWindowEvent`, so changing it recomputes them over all containers in - incremental mode instead of mixing units in the gold tables. Expressions that read - `start_ts`/`stop_ts` through `apply(..., container_metrics=[...])` are not covered: after changing - `epoch_unit`, list them under [`full_recalculation`](#full_recalculation-optional). +- `channel_time_unit` (`"s"` | `"ms"` | `"us"` | `"ns"`, optional) and `channel_time_origin` + (`"epoch"` (default) | `"container_start"`): the time frame of the channel timestamps + (`tstart`/`tend`, or `timestamp` when `data_type = "RAW"`). `"epoch"` means absolute epoch + numbers; `"container_start"` means time relative to the container's `start_ts` (e.g. seconds + since the recording started). Channel timestamps are never converted; they must already be + numbers in that frame. + + Both settings are only used by `TimeWindowEvent`, whose windows must lie in the channel time + frame. It derives its window bounds from `container_metrics.start_ts`/`stop_ts`: + - origin `"epoch"`: numeric boundaries as they are; `TIMESTAMP` boundaries as epoch numbers in + `channel_time_unit`; + - origin `"container_start"`: `0` to `stop_ts - start_ts`, in `channel_time_unit` for + `TIMESTAMP` boundaries; numeric boundaries are only shifted (they must already be in the + channels' unit). + + `channel_time_unit` is required when the boundaries are `TIMESTAMP` columns; a report with a + `TimeWindowEvent` fails with a clear error until it is set. `TIMESTAMP_NTZ` and `DATE` boundaries + are not supported. Everything else sees the original `start_ts`/`stop_ts`: `ContainerEvent`, + `measurement_dimension`, container filters, and UDFs that request them via + `apply(..., container_metrics=[...])` (a `TIMESTAMP` arrives there as a `pd.Timestamp`). + + The channel time frame is part of the definition hash of every `TimeWindowEvent` and of the + aggregations scoped to it, so changing it recomputes them over all containers in incremental + mode instead of mixing time frames in the gold tables. Per-table sections (each a `TableConfig`): @@ -212,7 +216,7 @@ Internal column names that mappings can target: | `tstart`, `tend`| Sample interval start/end on the `channels` table (RLE) | | `timestamp` | Raw sample timestamp on the `channels` table (RAW mode; encoded into `tstart`/`tend`) | | `is_plausible` | Boolean plausibility flag on the `channels` table (RAW mode); consumed by `drop_implausible_data` | -| `start_ts`, `stop_ts` | Measurement start/stop epoch timestamps on the `container_metrics` table — referenced by `ContainerEvent` and `TimeWindowEvent` to derive event-fact start/end. May be `TIMESTAMP` (see `epoch_unit`) | +| `start_ts`, `stop_ts` | Measurement start/stop epoch timestamps on the `container_metrics` table — referenced by `ContainerEvent` and `TimeWindowEvent` to derive event-fact start/end. May be `TIMESTAMP` (see `channel_time_unit`) | | `value` | Sample value (or attribute value on the EAV tag table) | | `key` | Attribute key on the EAV `container_tags` table | | `priority` | Tie-breaker column on the `channel_mapping` table | diff --git a/docs/impulse/docs/data_model/silver_layer_schema.md b/docs/impulse/docs/data_model/silver_layer_schema.md index ac98f911..5522a975 100644 --- a/docs/impulse/docs/data_model/silver_layer_schema.md +++ b/docs/impulse/docs/data_model/silver_layer_schema.md @@ -181,9 +181,10 @@ the epoch-typed pair). Populate whichever your queries and `measurement_dimensions` config need. `start_ts`/`stop_ts` may also be `TIMESTAMP` columns. To use them with a `TimeWindowEvent`, set -[`solver_config.epoch_unit`](../config/configuration.md#solver-column-mappings-and-filters) to the -epoch unit of the channel sample timestamps (`tstart`/`tend`, or `timestamp` in the raw format), so -the window boundaries share the samples' time base. +[`solver_config.channel_time_unit`](../config/configuration.md#solver-column-mappings-and-filters) +to the unit of the channel sample timestamps (`tstart`/`tend`, or `timestamp` in the raw format), +plus `channel_time_origin="container_start"` if those are relative to the container start, so +the window boundaries share the samples' time base. The columns themselves are not converted. ::: diff --git a/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md b/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md index 7610f845..7874f692 100644 --- a/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md +++ b/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md @@ -126,13 +126,15 @@ so that solver code can always reference the same constants. override for the channel mapping (alias) table. - `channels` (`TableConfig`): Column mappings and filters for the channel data table. - `unit_conversion` (`TableConfig`): Column mappings and filters for the unit conversion table. -- `epoch_unit` (`{"s", "ms", "us", "ns"} or None`): Epoch unit of the timestamps in the ``channels`` table (``tstart`` / ``tend``, -or ``timestamp`` for RAW data); these are never converted. Only -``TIMESTAMP``-typed ``start_ts`` / ``stop_ts`` of the ``container_metrics`` -table are converted, into epoch numbers in this unit so they match the channel -timestamps. ``ContainerEvent``, ``TimeWindowEvent`` and expressions that read -these columns see the converted values. Only needed for a ``TimeWindowEvent`` -over ``TIMESTAMP`` container boundaries; unset means nothing is converted. +- `channel_time_unit` (`{"s", "ms", "us", "ns"} or None`): Time unit of the timestamps in the ``channels`` table (``tstart`` / ``tend``, or +``timestamp`` for RAW data). Only used to compute ``TimeWindowEvent`` windows in +that unit (see :meth:`with_window_bounds`); required when ``container_metrics`` +``start_ts`` / ``stop_ts`` are ``TIMESTAMP`` columns. Nothing else is converted: +channel timestamps, and the ``start_ts`` / ``stop_ts`` seen by UDFs, +``ContainerEvent`` and ``measurement_dimension``, keep their original values. +- `channel_time_origin` (`{"epoch", "container_start"}`): Origin of the channel timestamps: absolute epoch (default), or relative to the +container's ``start_ts``. Like :attr:`channel_time_unit`, only used for the +``TimeWindowEvent`` windows. #### from\_json @@ -222,6 +224,30 @@ def start_ts_col() -> str Internal column name for the measurement-start epoch timestamp on container_metrics. +#### window\_start\_col + +```python +def window_start_col() -> str +``` + +Internal column name for the container start in the channel time frame. + +Added by :meth:`with_window_bounds`; prefixed so it cannot clash with a customer +column. + + +#### window\_stop\_col + +```python +def window_stop_col() -> str +``` + +Internal column name for the container stop in the channel time frame. + +Added by :meth:`with_window_bounds`; prefixed so it cannot clash with a customer +column. + + #### stop\_ts\_col ```python @@ -477,52 +503,44 @@ samples instead of splitting them; use drop_implausible_data instead. No-op when not raw. -#### normalize\_container\_boundaries +#### with\_window\_bounds ```python -def normalize_container_boundaries(df: DataFrame) -> DataFrame +def with_window_bounds(df: DataFrame) -> DataFrame ``` -Convert ``TIMESTAMP`` container start/stop columns to epoch numbers. +Add the container start/stop in the channel time frame, for ``TimeWindowEvent``. -Opt-in via :attr:`epoch_unit`: when it is unset, *df* is returned unchanged. -Otherwise each ``TIMESTAMP`` ``start_ts`` / ``stop_ts`` column becomes an epoch -number in that unit, computed from ``unix_micros`` (exact and independent of the -session time zone). ``"s"`` / ``"ms"`` give doubles (``"s"`` equals Spark's -``cast(timestamp as double)``), ``"us"`` / ``"ns"`` give longs. Numeric columns -are left as they are. Applying the same transform before both the event fact -and the solve keeps their window boundaries identical. +A ``TimeWindowEvent`` tiles each container into windows that must be in the same +time frame as the channel timestamps (:attr:`channel_time_unit`, +:attr:`channel_time_origin`). This adds :attr:`window_start_col` / +:attr:`window_stop_col`, derived from the raw ``start_ts`` / ``stop_ts``, which stay +unchanged for UDFs, ``ContainerEvent`` and ``measurement_dimension``: + +- origin ``"epoch"``: numeric boundaries as they are; ``TIMESTAMP`` boundaries as + epoch numbers in :attr:`channel_time_unit`; +- origin ``"container_start"``: ``0`` and ``stop_ts - start_ts``, for ``TIMESTAMP`` + boundaries in :attr:`channel_time_unit`, for numeric ones as they are (they must + already be in the channels' unit). + +``TIMESTAMP`` values are converted via ``unix_micros``, which is exact and +independent of the session time zone; ``"s"`` / ``"ms"`` give doubles, ``"us"`` / +``"ns"`` longs. The event fact and the solve both call this, so their windows use +the same bounds. The types are checked on the schema, so a missing setting fails +before any Spark job runs. **Arguments**: -- `df` (`pyspark.sql.DataFrame`): Column-mapped ``container_metrics`` frame (or a projection of it). +- `df` (`pyspark.sql.DataFrame`): Column-mapped ``container_metrics`` frame (or a projection of it) with +``start_ts`` and ``stop_ts``. **Raises**: -- `ValueError`: If :attr:`epoch_unit` is set and a boundary column is ``TIMESTAMP_NTZ`` or -``DATE`` (their epoch depends on a time zone and is not supported). +- `ValueError`: If ``start_ts`` / ``stop_ts`` are missing, are ``TIMESTAMP_NTZ`` or ``DATE``, +mix ``TIMESTAMP`` and numeric types, or are ``TIMESTAMP`` while +:attr:`channel_time_unit` is unset. **Returns**: -`pyspark.sql.DataFrame`: *df* with converted boundary columns. - -#### require\_epoch\_boundaries - -```python -def require_epoch_boundaries(df: DataFrame, owner: str) -> None -``` - -Raise unless the container start/stop columns on *df* are epoch numbers. - -Call after :meth:`normalize_container_boundaries`. Checks the schema only, so it -fails fast on the driver before any Spark job runs. - -**Arguments**: - -- `df` (`pyspark.sql.DataFrame`): Normalized container_metrics frame. -- `owner` (`str`): Name of the feature that needs epoch boundaries, used in the error message. - -**Raises**: - -- `ValueError`: If ``start_ts`` / ``stop_ts`` is still a date/time type. +`pyspark.sql.DataFrame`: *df* with the two window-bound columns added. diff --git a/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md b/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md index 5c03dcc7..5e710ba1 100644 --- a/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md +++ b/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md @@ -68,9 +68,7 @@ Calculate definition hash. The hash only captures computation-relevant attributes. For a ``ContainerEvent`` the identity is fully determined by the fact that it is a container event (there is no expression to vary), -so the name of the event is hashed. When ``epoch_unit`` is set it is -hashed too, since it decides the unit of ``TIMESTAMP`` boundaries in -``start_ts`` / ``end_ts``; unset, the hash is the name alone, as before. +so the name of the event is hashed. **Returns**: diff --git a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md index 7b107ed1..9bb40d35 100644 --- a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md +++ b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md @@ -55,20 +55,22 @@ the definition hash. - `ValueError`: If ``window_length`` is not strictly positive and finite, or ``max_windows_per_container`` is not a positive integer. -#### set\_epoch\_unit +#### set\_channel\_time ```python -def set_epoch_unit(epoch_unit: str | None) -> None +def set_channel_time(unit: str | None, origin: str = "epoch") -> None ``` -Set the epoch unit ``TIMESTAMP`` container boundaries are converted to. +Record the channel time frame the windows are computed in. -Also recorded on the expression, whose string form feeds the definition hashes of -this event and of the aggregations scoped to it. +Set by ``Report.add_event`` from the report's ``solver_config``. Stored on the +expression, whose string form feeds the definition hashes of this event and of the +aggregations scoped to it. **Arguments**: -- `epoch_unit` (`str or None`): The report's ``solver_config.epoch_unit``. +- `unit` (`str or None`): The report's ``solver_config.channel_time_unit``. +- `origin` (`str`): The report's ``solver_config.channel_time_origin`` (default ``"epoch"``). #### get\_expression @@ -103,9 +105,9 @@ def determine_definition_hash() -> int Calculate definition hash for the time-window event. Only includes the expression string, which encodes the attributes that affect the -event results: ``window_length`` and, when set, ``epoch_unit`` (the unit of -``TIMESTAMP`` boundaries). Resizing the window or changing the unit therefore forces -a full recompute in incremental mode. +event results: ``window_length`` and the channel time frame (``channel_time_unit``, +``channel_time_origin``; omitted while unset / default). Resizing the window or +changing the time frame therefore forces a full recompute in incremental mode. Excludes: name, description, required_channels, max_windows_per_container, report_id @@ -144,7 +146,8 @@ Extract the event fact table for the given list of TimeWindowEvent objects. Resolves the matching containers via the solver's filter pipeline (like ``ContainerEvent``) and computes each event's windows natively from the -containers' ``start_ts`` / ``stop_ts``, so every filtered container gets windows. +containers' ``start_ts`` / ``stop_ts`` in the channel time frame +(``SolverConfig.with_window_bounds``), so every filtered container gets windows. Each window becomes one event instance (``start_ts < end_ts``) whose ``event_instance_id`` hashes its position among the container's windows. The solve computes the same windows in the same order for scoped aggregations (see diff --git a/docs/impulse/docs/references/report/event.md b/docs/impulse/docs/references/report/event.md index 66f34e55..c73bcd5b 100644 --- a/docs/impulse/docs/references/report/event.md +++ b/docs/impulse/docs/references/report/event.md @@ -231,20 +231,29 @@ the same time unit as the stored timestamps (milliseconds-since-epoch in the sam seconds or any derived unit. So 60 one-minute windows over millisecond timestamps use `window_length=60_000`. -If `container_metrics.start_ts`/`stop_ts` are `TIMESTAMP` columns, set -[`solver_config.epoch_unit`](../../config/configuration.md#solver-column-mappings-and-filters) -to the epoch unit of the channel sample timestamps (`tstart`/`tend`, or `timestamp` for RAW -data; e.g. `"s"`). The boundaries are converted to -that unit, and `window_length` is expressed in it. Without it, the report fails with an error -naming the setting. `epoch_unit` is part of the event's definition (and of the aggregations -scoped to it), so changing it recomputes them over all containers in incremental mode. +The windows are computed in the time frame of the channel timestamps, set by +[`solver_config.channel_time_unit` and `channel_time_origin`](../../config/configuration.md#solver-column-mappings-and-filters): + +- If `container_metrics.start_ts`/`stop_ts` are `TIMESTAMP` columns, set `channel_time_unit` to the + unit of the channel timestamps (`tstart`/`tend`, or `timestamp` for RAW data; e.g. `"s"`), and + `window_length` is expressed in it. Without it, the report fails with an error naming the + setting. +- If the channel timestamps are relative to the container start (e.g. seconds since the + recording started), also set `channel_time_origin="container_start"`. The windows then run from + `0` to `stop_ts - start_ts`. + +Only the windows use these settings: `ContainerEvent`, `measurement_dimension` and UDFs that read +`start_ts`/`stop_ts` keep seeing the original values. The channel time frame is part of the +event's definition (and of the aggregations scoped to it), so changing it recomputes them over all +containers in incremental mode. ::: ### How it works 1. The event resolves the matching containers through the report's container filters (like `ContainerEvent`), reads `start_ts` and `stop_ts` from the `container_metrics` table, and - tiles `[start_ts, stop_ts]` into consecutive windows of length `window_length`. + tiles `[start_ts, stop_ts]`, in the channel time frame, into consecutive windows of length + `window_length`. The window instances in `event_instance_fact` are in that frame too. 2. The **final window is clamped** to `stop_ts` when the last full window would overrun it; any zero-length trailing slice is dropped (every instance satisfies `start_ts < end_ts`). Containers whose `start_ts` or `stop_ts` is null, NaN or infinite get no windows. diff --git a/skills/impulse-config/SKILL.md b/skills/impulse-config/SKILL.md index 1994715f..97df3016 100644 --- a/skills/impulse-config/SKILL.md +++ b/skills/impulse-config/SKILL.md @@ -124,12 +124,14 @@ rejected (use `drop_implausible_data`), any other channels filter warns. Top-level `project_id` (str, optional) applies an equality filter on the `project_id` column of every table that has one (`container_tags`, `container_metrics`, `channel_mapping`). Omit if not needed. -Top-level `epoch_unit` (`"s"` | `"ms"` | `"us"` | `"ns"`, optional) is the epoch unit of the `channels` -timestamps (`tstart`/`tend`, or `timestamp` with `data_type="RAW"`); those are never converted. Only -`TIMESTAMP`-typed `container_metrics.start_ts`/`stop_ts` are converted, into epoch numbers in that -unit, so `ContainerEvent` / `TimeWindowEvent` boundaries match the channel timestamps. Needed only -for a `TimeWindowEvent` over `TIMESTAMP` boundaries; with `"ns"`, boundaries must lie between -1677-09-21 and 2262-04-11 (the int64 nanosecond range). +Top-level `channel_time_unit` (`"s"` | `"ms"` | `"us"` | `"ns"`, optional) and `channel_time_origin` +(`"epoch"` default | `"container_start"`) describe the time frame of the `channels` timestamps +(`tstart`/`tend`, or `timestamp` with `data_type="RAW"`): absolute epoch numbers, or time relative +to the container's `start_ts`. Only `TimeWindowEvent` uses them, to compute its windows in that +frame from `container_metrics.start_ts`/`stop_ts` (origin `"container_start"`: from `0` to +`stop_ts - start_ts`). `channel_time_unit` is required when those are `TIMESTAMP` columns. Nothing +is converted in place: the channel timestamps, and the `start_ts`/`stop_ts` seen by +`ContainerEvent`, `measurement_dimension` and UDFs (a `pd.Timestamp`), keep their original values. ```python "query_engine": { diff --git a/skills/impulse-events/SKILL.md b/skills/impulse-events/SKILL.md index 5eeaea24..39e31adf 100644 --- a/skills/impulse-events/SKILL.md +++ b/skills/impulse-events/SKILL.md @@ -158,10 +158,11 @@ or without channel data or a scoped aggregation. Pair it with an aggregation sco `event_instance_id` values as the windows (the id hashes container, event name and window position). Because the windows come from `container_metrics`, those boundaries must share the channel samples' time base for the per-window values to be meaningful. If `container_metrics.start_ts`/`stop_ts` are -`TIMESTAMP` columns, set `query_engine.solver_config.epoch_unit` to the unit of the channel -timestamps (`tstart`/`tend`, or `timestamp` for RAW); only those two container columns are -converted, never the channel timestamps. Containers with null, NaN or infinite boundaries get no -windows. +`TIMESTAMP` columns, set `query_engine.solver_config.channel_time_unit` to the unit of the channel +timestamps (`tstart`/`tend`, or `timestamp` for RAW), and `channel_time_origin="container_start"` if +they are relative to the container start (windows then run from `0`). Only the windows use these +settings; `start_ts`/`stop_ts` themselves keep their original values for `ContainerEvent` and UDFs. +Containers with null, NaN or infinite boundaries get no windows. ## Output schema diff --git a/src/impulse_query_engine/analyze/query/events/time_window_expression.py b/src/impulse_query_engine/analyze/query/events/time_window_expression.py index aacad549..3cbf8ad9 100644 --- a/src/impulse_query_engine/analyze/query/events/time_window_expression.py +++ b/src/impulse_query_engine/analyze/query/events/time_window_expression.py @@ -1,6 +1,5 @@ from __future__ import annotations -import datetime import math import numbers @@ -17,11 +16,10 @@ from impulse_query_engine.analyze.query.solvers.solver_config import SolverConfig from impulse_query_engine.model.series.intervals import Intervals -# Reuse SolverConfig's canonical internal (post-``column_name_mapping``) column names for -# the measurement start/stop timestamps rather than re-declaring the literals here. These -# are the keys under which the solve exposes them via ``SeriesCache.container_metrics``, and -# they are the same names ``ContainerEvent`` relies on. A default instance suffices since -# the names are config-invariant. +# Reuse SolverConfig's internal column names for the container bounds in the channel time +# frame (see SolverConfig.with_window_bounds) rather than re-declaring the literals here. +# These are the keys under which the solve exposes them via ``SeriesCache.container_metrics``. +# A default instance suffices since the names are config-invariant. _SOLVER_CONFIG = SolverConfig() # Default upper bound on the windows per container. A window_length in the wrong unit for the @@ -31,8 +29,8 @@ MAX_WINDOWS_PER_CONTAINER = 1_000_000 _WINDOW_LIMIT_HINT = ( - "Check that window_length is in the epoch unit of the container boundaries " - "(solver_config.epoch_unit), or raise the limit (TimeWindowEvent " + "Check that window_length is in the unit of the channel timestamps " + "(solver_config.channel_time_unit), or raise the limit (TimeWindowEvent " "max_windows_per_container, TimeWindowExpression max_windows)." ) @@ -144,11 +142,12 @@ class TimeWindowExpression(TimeSeriesExpression): """Produce consecutive fixed-duration windows spanning a measurement container. The windows are derived purely from the container's ``start_ts`` / ``stop_ts`` metadata - (no channel data), so the expression declares no selectors and instead requests those - container metrics via :meth:`required_container_metrics`. Windows tile - ``[start_ts, stop_ts]`` with a fixed length ``window_length`` (expressed in the same time - unit as the underlying timestamps); the final window is clamped to ``stop_ts`` when the - last full window would overrun it. + (no channel data), so the expression declares no selectors and instead requests the + container bounds in the channel time frame via :meth:`required_container_metrics` + (computed by ``SolverConfig.with_window_bounds``). Windows tile those bounds with a + fixed length ``window_length`` (expressed in the same time unit as the channel + timestamps); the final window is clamped to the stop bound when the last full window + would overrun it. Visual timeline (window_length = W):: @@ -163,12 +162,14 @@ class TimeWindowExpression(TimeSeriesExpression): Attributes ---------- - epoch_unit : str or None - Epoch unit the solver converts ``TIMESTAMP`` boundaries to - (``solver_config.epoch_unit``), set by the reporting ``TimeWindowEvent``. - Descriptive only: :meth:`build` does not convert (the solver does). It is part of - the string form, so the definition hashes of the event and of every aggregation - scoped to it change with the unit. + channel_time_unit : str or None + ``solver_config.channel_time_unit``, set by the reporting ``TimeWindowEvent``. + channel_time_origin : str + ``solver_config.channel_time_origin`` (default ``"epoch"``), set the same way. + + Both are descriptive only: :meth:`build` does not convert (the solver computes the + bounds). They are part of the string form, so the definition hashes of the event and of + every aggregation scoped to it change with the channel time frame. """ def __init__(self, window_length: float, max_windows: int = MAX_WINDOWS_PER_CONTAINER): @@ -203,24 +204,30 @@ def __init__(self, window_length: float, max_windows: int = MAX_WINDOWS_PER_CONT # and must not trigger a spurious full recompute in incremental mode. self.window_length = float(window_length) self.max_windows = validate_max_windows(max_windows) - self.epoch_unit: str | None = None + self.channel_time_unit: str | None = None + self.channel_time_origin: str = "epoch" TimeSeriesExpression.__init__(self, is_single_signal=False) def __str__(self) -> str: """ Return a string representation of the TimeWindowExpression. - The ``window_length`` (and ``epoch_unit``, when set) is included so it flows into - the definition hashes of the event and of the aggregations scoped to it. An unset - ``epoch_unit`` is omitted, keeping the string identical to the unit-less form. + The ``window_length`` and the channel time frame are included so they flow into the + definition hashes of the event and of the aggregations scoped to it. An unset + ``channel_time_unit`` and the default ``"epoch"`` origin are omitted, keeping the + default string unchanged. Returns ------- str String representation of the object. """ - unit = f", epoch_unit={self.epoch_unit}" if self.epoch_unit is not None else "" - return f"TimeWindowExpression" + frame = "" + if self.channel_time_unit is not None: + frame += f", channel_time_unit={self.channel_time_unit}" + if self.channel_time_origin != "epoch": + frame += f", channel_time_origin={self.channel_time_origin}" + return f"TimeWindowExpression" def dtype(self): """ @@ -270,9 +277,10 @@ def required_container_metrics(self) -> set[str]: Returns ------- set of str - The measurement start/stop timestamp columns. + The container start/stop in the channel time frame, which the solver derives + from ``start_ts`` / ``stop_ts`` (``SolverConfig.with_window_bounds``). """ - return {_SOLVER_CONFIG.start_ts_col, _SOLVER_CONFIG.stop_ts_col} + return {_SOLVER_CONFIG.window_start_col, _SOLVER_CONFIG.window_stop_col} def get_selectors(self) -> list[TimeSeriesSelector]: """ @@ -306,34 +314,21 @@ def build(self, cache: SeriesCache) -> Intervals: Returns ------- Intervals - Consecutive fixed-length windows over ``[start_ts, stop_ts]``, with the final - window clamped to ``stop_ts``. Empty when the container boundaries are absent - (e.g. the empty cache used for type validation), NaN or infinite, or - non-positive in span. + Consecutive fixed-length windows over the container bounds, with the final + window clamped to the stop bound. Empty when the bounds are absent (e.g. the + empty cache used for type validation), NaN or infinite, or non-positive in span. Raises ------ - TypeError - If a boundary is a date/time value rather than an epoch number. ValueError If the container would produce more than ``max_windows`` windows. """ - start_ts = cache.container_metrics.get(_SOLVER_CONFIG.start_ts_col) - stop_ts = cache.container_metrics.get(_SOLVER_CONFIG.stop_ts_col) + start_ts = cache.container_metrics.get(_SOLVER_CONFIG.window_start_col) + stop_ts = cache.container_metrics.get(_SOLVER_CONFIG.window_stop_col) if start_ts is None or stop_ts is None: return Intervals.empty() - for name, value in (("start_ts", start_ts), ("stop_ts", stop_ts)): - # pd.Timestamp subclasses datetime.datetime; dates and numpy datetimes too. - if isinstance(value, (datetime.date, np.datetime64)): - raise TypeError( - f"TimeWindowExpression needs epoch-number container boundaries, but " - f"{name} is {type(value).__name__}. For TIMESTAMP columns, set " - "solver_config.epoch_unit to the epoch unit of the channel sample " - "timestamps so start_ts / stop_ts are converted before the solve." - ) - # Mirror window_intervals_col exactly: convert to double *before* subtracting. A # long column reaches pandas as int64 or float64 depending on the group (nulls # force float64), and an exact int64 span can round differently from the double diff --git a/src/impulse_query_engine/analyze/query/solvers/default_solver.py b/src/impulse_query_engine/analyze/query/solvers/default_solver.py index 051c594e..5e0b9272 100644 --- a/src/impulse_query_engine/analyze/query/solvers/default_solver.py +++ b/src/impulse_query_engine/analyze/query/solvers/default_solver.py @@ -1305,6 +1305,12 @@ def _build_container_metadata_df( if metric_cols: metrics = self.scoped_container_metrics(self.spark, query, pre_filtered_containers_df) + window_bounds = {self.config.window_start_col, self.config.window_stop_col} + if window_bounds & set(metric_cols): + # TimeWindowExpression reads the container bounds in the channel time + # frame, computed exactly like the TimeWindowEvent fact does. The raw + # start_ts/stop_ts stay unchanged for any other expression (e.g. UDFs). + metrics = self.config.with_window_bounds(metrics) missing = [c for c in metric_cols if c not in metrics.columns] if missing: raise ValueError( @@ -1314,10 +1320,6 @@ def _build_container_metadata_df( meta_df = metrics.select(container_id_col, *metric_cols).dropDuplicates( [container_id_col] ) - # TIMESTAMP start_ts/stop_ts would reach pandas as session-local, tz-naive - # Timestamps; convert them to epoch numbers in Spark when epoch_unit is set - # (no-op otherwise), matching the container-boundary events. - meta_df = self.config.normalize_container_boundaries(meta_df) if tag_keys: if query.db.config.container_tags_table is None: diff --git a/src/impulse_query_engine/analyze/query/solvers/solver_config.py b/src/impulse_query_engine/analyze/query/solvers/solver_config.py index 3c2eb3de..c901a8d3 100644 --- a/src/impulse_query_engine/analyze/query/solvers/solver_config.py +++ b/src/impulse_query_engine/analyze/query/solvers/solver_config.py @@ -137,18 +137,22 @@ class SolverConfig(BaseModel): Column mappings and filters for the channel data table. unit_conversion : TableConfig Column mappings and filters for the unit conversion table. - epoch_unit : {"s", "ms", "us", "ns"} or None - Epoch unit of the timestamps in the ``channels`` table (``tstart`` / ``tend``, - or ``timestamp`` for RAW data); these are never converted. Only - ``TIMESTAMP``-typed ``start_ts`` / ``stop_ts`` of the ``container_metrics`` - table are converted, into epoch numbers in this unit so they match the channel - timestamps. ``ContainerEvent``, ``TimeWindowEvent`` and expressions that read - these columns see the converted values. Only needed for a ``TimeWindowEvent`` - over ``TIMESTAMP`` container boundaries; unset means nothing is converted. + channel_time_unit : {"s", "ms", "us", "ns"} or None + Time unit of the timestamps in the ``channels`` table (``tstart`` / ``tend``, or + ``timestamp`` for RAW data). Only used to compute ``TimeWindowEvent`` windows in + that unit (see :meth:`with_window_bounds`); required when ``container_metrics`` + ``start_ts`` / ``stop_ts`` are ``TIMESTAMP`` columns. Nothing else is converted: + channel timestamps, and the ``start_ts`` / ``stop_ts`` seen by UDFs, + ``ContainerEvent`` and ``measurement_dimension``, keep their original values. + channel_time_origin : {"epoch", "container_start"} + Origin of the channel timestamps: absolute epoch (default), or relative to the + container's ``start_ts``. Like :attr:`channel_time_unit`, only used for the + ``TimeWindowEvent`` windows. """ project_id: str | None = None - epoch_unit: Literal["s", "ms", "us", "ns"] | None = None + channel_time_unit: Literal["s", "ms", "us", "ns"] | None = None + channel_time_origin: Literal["epoch", "container_start"] = "epoch" container_tags: TableConfig = TableConfig() container_metrics: TableConfig = TableConfig() @@ -235,6 +239,24 @@ def start_ts_col(self) -> str: """Internal column name for the measurement-start epoch timestamp on container_metrics.""" return "start_ts" + @property + def window_start_col(self) -> str: + """Internal column name for the container start in the channel time frame. + + Added by :meth:`with_window_bounds`; prefixed so it cannot clash with a customer + column. + """ + return "__window_start" + + @property + def window_stop_col(self) -> str: + """Internal column name for the container stop in the channel time frame. + + Added by :meth:`with_window_bounds`; prefixed so it cannot clash with a customer + column. + """ + return "__window_stop" + @property def stop_ts_col(self) -> str: """Internal column name for the measurement-stop epoch timestamp on container_metrics.""" @@ -436,83 +458,100 @@ def _boundary_fields(self, df: DataFrame) -> list[T.StructField]: names = {self.start_ts_col, self.stop_ts_col} return [field for field in df.schema.fields if field.name in names] - def normalize_container_boundaries(self, df: DataFrame) -> DataFrame: - """Convert ``TIMESTAMP`` container start/stop columns to epoch numbers. + def with_window_bounds(self, df: DataFrame) -> DataFrame: + """Add the container start/stop in the channel time frame, for ``TimeWindowEvent``. - Opt-in via :attr:`epoch_unit`: when it is unset, *df* is returned unchanged. - Otherwise each ``TIMESTAMP`` ``start_ts`` / ``stop_ts`` column becomes an epoch - number in that unit, computed from ``unix_micros`` (exact and independent of the - session time zone). ``"s"`` / ``"ms"`` give doubles (``"s"`` equals Spark's - ``cast(timestamp as double)``), ``"us"`` / ``"ns"`` give longs. Numeric columns - are left as they are. Applying the same transform before both the event fact - and the solve keeps their window boundaries identical. + A ``TimeWindowEvent`` tiles each container into windows that must be in the same + time frame as the channel timestamps (:attr:`channel_time_unit`, + :attr:`channel_time_origin`). This adds :attr:`window_start_col` / + :attr:`window_stop_col`, derived from the raw ``start_ts`` / ``stop_ts``, which stay + unchanged for UDFs, ``ContainerEvent`` and ``measurement_dimension``: + + - origin ``"epoch"``: numeric boundaries as they are; ``TIMESTAMP`` boundaries as + epoch numbers in :attr:`channel_time_unit`; + - origin ``"container_start"``: ``0`` and ``stop_ts - start_ts``, for ``TIMESTAMP`` + boundaries in :attr:`channel_time_unit`, for numeric ones as they are (they must + already be in the channels' unit). + + ``TIMESTAMP`` values are converted via ``unix_micros``, which is exact and + independent of the session time zone; ``"s"`` / ``"ms"`` give doubles, ``"us"`` / + ``"ns"`` longs. The event fact and the solve both call this, so their windows use + the same bounds. The types are checked on the schema, so a missing setting fails + before any Spark job runs. Parameters ---------- df : pyspark.sql.DataFrame - Column-mapped ``container_metrics`` frame (or a projection of it). + Column-mapped ``container_metrics`` frame (or a projection of it) with + ``start_ts`` and ``stop_ts``. Returns ------- pyspark.sql.DataFrame - *df* with converted boundary columns. + *df* with the two window-bound columns added. Raises ------ ValueError - If :attr:`epoch_unit` is set and a boundary column is ``TIMESTAMP_NTZ`` or - ``DATE`` (their epoch depends on a time zone and is not supported). + If ``start_ts`` / ``stop_ts`` are missing, are ``TIMESTAMP_NTZ`` or ``DATE``, + mix ``TIMESTAMP`` and numeric types, or are ``TIMESTAMP`` while + :attr:`channel_time_unit` is unset. """ - if self.epoch_unit is None: - return df - for field in self._boundary_fields(df): - if isinstance(field.dataType, T.TimestampType): - df = df.withColumn(field.name, self._epoch_from_timestamp(F.col(field.name))) - elif isinstance(field.dataType, (T.TimestampNTZType, T.DateType)): + types = {field.name: field.dataType for field in self._boundary_fields(df)} + missing = [c for c in (self.start_ts_col, self.stop_ts_col) if c not in types] + if missing: + raise ValueError( + f"TimeWindowEvent needs the container_metrics columns {missing} to compute " + f"its windows. Available columns: {df.columns}" + ) + for name, dtype in types.items(): + if isinstance(dtype, (T.TimestampNTZType, T.DateType)): raise ValueError( - f"container_metrics column '{field.name}' has type " - f"{field.dataType.simpleString()}, which cannot be converted to an epoch " - "unambiguously (it carries no time zone). Use a TIMESTAMP or epoch-number " - "column." + f"container_metrics column '{name}' has type {dtype.simpleString()}, " + "which cannot be converted to an epoch unambiguously (it carries no time " + "zone). Use a TIMESTAMP or epoch-number column." ) - return df + is_timestamp = {isinstance(dtype, T.TimestampType) for dtype in types.values()} + if len(is_timestamp) > 1: + raise ValueError( + f"container_metrics columns '{self.start_ts_col}' and '{self.stop_ts_col}' " + "must both be TIMESTAMP or both be numeric to compute TimeWindowEvent windows." + ) + timestamps = is_timestamp.pop() + if timestamps and self.channel_time_unit is None: + raise ValueError( + f"TimeWindowEvent needs its windows in the channel time frame, but " + f"container_metrics '{self.start_ts_col}' / '{self.stop_ts_col}' are " + "TIMESTAMP columns. Set query_engine.solver_config.channel_time_unit to the " + "unit of the channel timestamps (one of 's', 'ms', 'us', 'ns'), and " + "channel_time_origin to 'container_start' if they are relative to the " + "container start." + ) - def _epoch_from_timestamp(self, col: Column) -> Column: - """Epoch value of a TIMESTAMP column in :attr:`epoch_unit`.""" - micros = F.unix_micros(col) - if self.epoch_unit == "s": + start, stop = F.col(self.start_ts_col), F.col(self.stop_ts_col) + if self.channel_time_origin == "container_start": + window_start = F.lit(0) + # Subtract exactly in microseconds before scaling to the channel unit. + window_stop = ( + self._micros_in_unit(F.unix_micros(stop) - F.unix_micros(start)) + if timestamps + else stop - start + ) + elif timestamps: + window_start = self._micros_in_unit(F.unix_micros(start)) + window_stop = self._micros_in_unit(F.unix_micros(stop)) + else: + window_start, window_stop = start, stop + return df.withColumn(self.window_start_col, window_start).withColumn( + self.window_stop_col, window_stop + ) + + def _micros_in_unit(self, micros: Column) -> Column: + """Microseconds converted to :attr:`channel_time_unit`.""" + if self.channel_time_unit == "s": return micros / F.lit(1e6) - if self.epoch_unit == "ms": + if self.channel_time_unit == "ms": return micros / F.lit(1e3) - if self.epoch_unit == "ns": + if self.channel_time_unit == "ns": return micros * F.lit(1000) return micros - - def require_epoch_boundaries(self, df: DataFrame, owner: str) -> None: - """Raise unless the container start/stop columns on *df* are epoch numbers. - - Call after :meth:`normalize_container_boundaries`. Checks the schema only, so it - fails fast on the driver before any Spark job runs. - - Parameters - ---------- - df : pyspark.sql.DataFrame - Normalized container_metrics frame. - owner : str - Name of the feature that needs epoch boundaries, used in the error message. - - Raises - ------ - ValueError - If ``start_ts`` / ``stop_ts`` is still a date/time type. - """ - datetime_types = (T.TimestampType, T.TimestampNTZType, T.DateType) - for field in self._boundary_fields(df): - if isinstance(field.dataType, datetime_types): - raise ValueError( - f"{owner} needs epoch-number container boundaries, but container_metrics " - f"column '{field.name}' has type {field.dataType.simpleString()}. Set " - "query_engine.solver_config.epoch_unit to the epoch unit of the channel " - "sample timestamps (one of 's', 'ms', 'us', 'ns') so that column is " - "converted to epoch numbers in that unit." - ) diff --git a/src/impulse_reporting/core/report.py b/src/impulse_reporting/core/report.py index 9a680881..64ed3cc5 100644 --- a/src/impulse_reporting/core/report.py +++ b/src/impulse_reporting/core/report.py @@ -47,6 +47,7 @@ from impulse_reporting.events.container_event import ContainerEvent from impulse_reporting.events.event import Event from impulse_reporting.events.event_types import EventType +from impulse_reporting.events.time_window_event import TimeWindowEvent from impulse_reporting.incremental.container_detector import ContainerUpsertDetector from impulse_reporting.incremental.definition_hash_comparator import ( DefinitionHashComparator, @@ -415,9 +416,12 @@ def add_event(self, event: Event): ) self.events.append(event) event.set_report_id(self.report_id) - if isinstance(event, ContainerBoundaryEvent): - # The unit of TIMESTAMP boundaries is part of these events' definitions. - event.set_epoch_unit(self.solver.config.epoch_unit) + if isinstance(event, TimeWindowEvent): + # The windows are computed in the channel time frame, so it is part of the + # event's (and its scoped aggregations') definition. + event.set_channel_time( + self.solver.config.channel_time_unit, self.solver.config.channel_time_origin + ) def get_events(self) -> list[Event]: """ diff --git a/src/impulse_reporting/events/container_boundary_event.py b/src/impulse_reporting/events/container_boundary_event.py index 60ce4c02..21d8c58c 100644 --- a/src/impulse_reporting/events/container_boundary_event.py +++ b/src/impulse_reporting/events/container_boundary_event.py @@ -20,27 +20,8 @@ class ContainerBoundaryEvent(Event): filtered container yields instances regardless of its channel data. The report therefore excludes these event types from the solvable expressions and dispatches them with ``query`` / ``solver`` rather than ``solved_df``. - - Attributes - ---------- - epoch_unit : str or None - ``solver_config.epoch_unit`` of the report the event belongs to, set by - ``Report.add_event``. It decides the unit of ``TIMESTAMP`` boundaries, so - subclasses fold it into their definition hash. """ - epoch_unit: str | None = None - - def set_epoch_unit(self, epoch_unit: str | None) -> None: - """Set the epoch unit ``TIMESTAMP`` container boundaries are converted to. - - Parameters - ---------- - epoch_unit : str or None - The report's ``solver_config.epoch_unit``. - """ - self.epoch_unit = epoch_unit - def get_id(self) -> int: """Return a unique identifier derived from the event name. @@ -104,14 +85,10 @@ def resolve_container_metrics( Returns ------- DataFrame - Column-mapped ``container_metrics`` rows of the matching containers, with - ``TIMESTAMP`` boundaries converted to epoch numbers when - ``solver.config.epoch_unit`` is set (unchanged otherwise). + Column-mapped ``container_metrics`` rows of the matching containers, with the + original ``start_ts`` / ``stop_ts``. """ container_tags_df = solver.filter_container_tags(spark, query) - container_metrics_df = solver.filter_container_metrics( + return solver.filter_container_metrics( spark, query, container_tags_df, pre_filtered_containers_df ) - # Same transform as the solve's container metadata, so the event boundaries and - # those seen by scoped aggregations are identical. - return solver.config.normalize_container_boundaries(container_metrics_df) diff --git a/src/impulse_reporting/events/container_event.py b/src/impulse_reporting/events/container_event.py index b1d5f429..74a509d8 100644 --- a/src/impulse_reporting/events/container_event.py +++ b/src/impulse_reporting/events/container_event.py @@ -78,9 +78,7 @@ def determine_definition_hash(self) -> int: The hash only captures computation-relevant attributes. For a ``ContainerEvent`` the identity is fully determined by the fact that it is a container event (there is no expression to vary), - so the name of the event is hashed. When ``epoch_unit`` is set it is - hashed too, since it decides the unit of ``TIMESTAMP`` boundaries in - ``start_ts`` / ``end_ts``; unset, the hash is the name alone, as before. + so the name of the event is hashed. Returns ------- @@ -88,8 +86,6 @@ def determine_definition_hash(self) -> int: Hash value representing the computation definition. """ hash_input = self.name - if self.epoch_unit is not None: - hash_input = f"{self.name}::epoch_unit={self.epoch_unit}" hash_bytes = hashlib.sha256(hash_input.encode()).digest() return int.from_bytes(hash_bytes[:8], byteorder="big", signed=True) diff --git a/src/impulse_reporting/events/time_window_event.py b/src/impulse_reporting/events/time_window_event.py index 8579c286..180ecf89 100644 --- a/src/impulse_reporting/events/time_window_event.py +++ b/src/impulse_reporting/events/time_window_event.py @@ -103,19 +103,22 @@ def __init__( normalized_attributes.setdefault("window_length", str(self.window_length)) self.attributes = normalized_attributes - def set_epoch_unit(self, epoch_unit: str | None) -> None: - """Set the epoch unit ``TIMESTAMP`` container boundaries are converted to. + def set_channel_time(self, unit: str | None, origin: str = "epoch") -> None: + """Record the channel time frame the windows are computed in. - Also recorded on the expression, whose string form feeds the definition hashes of - this event and of the aggregations scoped to it. + Set by ``Report.add_event`` from the report's ``solver_config``. Stored on the + expression, whose string form feeds the definition hashes of this event and of the + aggregations scoped to it. Parameters ---------- - epoch_unit : str or None - The report's ``solver_config.epoch_unit``. + unit : str or None + The report's ``solver_config.channel_time_unit``. + origin : str, optional + The report's ``solver_config.channel_time_origin`` (default ``"epoch"``). """ - ContainerBoundaryEvent.set_epoch_unit(self, epoch_unit) - self.expression.epoch_unit = epoch_unit + self.expression.channel_time_unit = unit + self.expression.channel_time_origin = origin def get_expression(self) -> TimeSeriesExpression | None: """ @@ -143,9 +146,9 @@ def determine_definition_hash(self) -> int: Calculate definition hash for the time-window event. Only includes the expression string, which encodes the attributes that affect the - event results: ``window_length`` and, when set, ``epoch_unit`` (the unit of - ``TIMESTAMP`` boundaries). Resizing the window or changing the unit therefore forces - a full recompute in incremental mode. + event results: ``window_length`` and the channel time frame (``channel_time_unit``, + ``channel_time_origin``; omitted while unset / default). Resizing the window or + changing the time frame therefore forces a full recompute in incremental mode. Excludes: name, description, required_channels, max_windows_per_container, report_id @@ -198,7 +201,8 @@ def determine_events( Resolves the matching containers via the solver's filter pipeline (like ``ContainerEvent``) and computes each event's windows natively from the - containers' ``start_ts`` / ``stop_ts``, so every filtered container gets windows. + containers' ``start_ts`` / ``stop_ts`` in the channel time frame + (``SolverConfig.with_window_bounds``), so every filtered container gets windows. Each window becomes one event instance (``start_ts < end_ts``) whose ``event_instance_id`` hashes its position among the container's windows. The solve computes the same windows in the same order for scoped aggregations (see @@ -227,13 +231,12 @@ def determine_events( container_metrics_df = cls.resolve_container_metrics( spark, query, solver, pre_filtered_containers_df ) - # Windows are computed in the channel samples' epoch unit, so TIMESTAMP boundaries - # need solver_config.epoch_unit (fails fast on the schema, before any Spark job). - solver.config.require_epoch_boundaries(container_metrics_df, owner="TimeWindowEvent") - - # Silver-side names come from SolverConfig (column_name_mapping aware). - start_ts = f.col(solver.config.start_ts_col) - stop_ts = f.col(solver.config.stop_ts_col) + # The windows are computed in the channel time frame, from the same bounds the solve + # uses for scoped aggregations (fails fast on the schema, e.g. when TIMESTAMP + # boundaries lack solver_config.channel_time_unit). + container_metrics_df = solver.config.with_window_bounds(container_metrics_df) + start_ts = f.col(solver.config.window_start_col) + stop_ts = f.col(solver.config.window_stop_col) # One (event_name, windows) struct per event, exploded in a single pass over the # containers. posexplode yields each window's position, which the diff --git a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py index e60ee1bd..4ef267f7 100644 --- a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py @@ -4,7 +4,6 @@ from unittest.mock import MagicMock import numpy as np -import pandas as pd import pyspark.sql.functions as F import pytest @@ -35,9 +34,13 @@ def container_tags(self) -> dict: return {} +# The container bounds in the channel time frame, as SolverConfig.with_window_bounds adds them. +_WINDOW_START, _WINDOW_STOP = "__window_start", "__window_stop" + + def _build(start_ts, stop_ts, window_length, **kwargs) -> Intervals: expr = TimeWindowExpression(window_length, **kwargs) - return expr.build(_FakeCache({"start_ts": start_ts, "stop_ts": stop_ts})) + return expr.build(_FakeCache({_WINDOW_START: start_ts, _WINDOW_STOP: stop_ts})) def test_exact_multiple_windows_last_ends_at_stop(): @@ -107,7 +110,9 @@ def test_no_selectors_and_requests_container_metrics(): expr = TimeWindowExpression(10) assert expr.get_selectors() == [] assert expr.get_selector_expr() is None - assert expr.required_container_metrics() == {"start_ts", "stop_ts"} + # The bounds in the channel time frame, not the raw start_ts / stop_ts (which UDFs + # keep reading unconverted). + assert expr.required_container_metrics() == {_WINDOW_START, _WINDOW_STOP} assert expr.required_tags() == set() @@ -121,13 +126,18 @@ def test_str_stable_across_int_and_float_window_length(): assert str(TimeWindowExpression(10)) == str(TimeWindowExpression(10.0)) -def test_str_includes_epoch_unit_only_when_set(): +def test_str_includes_channel_time_frame_only_when_set(): # The string feeds the definition hashes of the event and its scoped aggregations, so - # the unit must move them, while an unset unit keeps the unit-less form. + # the channel time frame must move them, while the defaults keep the plain form. expr = TimeWindowExpression(10) assert str(expr) == "TimeWindowExpression" - expr.epoch_unit = "ms" - assert str(expr) == "TimeWindowExpression" + expr.channel_time_unit = "ms" + assert str(expr) == "TimeWindowExpression" + expr.channel_time_origin = "container_start" + assert str(expr) == ( + "TimeWindowExpression" + ) def test_max_windows_not_part_of_str(): @@ -153,7 +163,7 @@ def test_build_raises_beyond_max_windows(): def test_build_unit_mismatch_hits_default_cap(): # window_length=60 meant as seconds over a 1 h ns-epoch span: 6e10 windows. start = 1_700_000_000_000_000_000 - with pytest.raises(ValueError, match="epoch unit of the container boundaries"): + with pytest.raises(ValueError, match="unit of the channel timestamps"): _build(np.int64(start), np.int64(start + 3_600_000_000_000), 60) @@ -172,13 +182,6 @@ def test_int64_and_float64_inputs_build_identical_windows(): assert a.get_data() == b.get_data() -def test_datetime_container_metrics_raise_clear_error(): - # TIMESTAMP boundaries reach pandas as pd.Timestamp unless epoch_unit converts them. - start, stop = pd.Timestamp("2025-07-03 07:41:41"), pd.Timestamp("2025-07-03 07:43:30") - with pytest.raises(TypeError, match="epoch_unit"): - _build(start, stop, 10) - - def test_nan_container_metrics_yield_empty(): # A null start/stop arrives as NaN in a float64 column. assert len(_build(np.nan, 100.0, 10)) == 0 @@ -359,7 +362,7 @@ def test_stats_aggregator_windows_equal_helper_windows(spark): # noqa: F811 event_expression=TimeWindowExpression(w), statistics=["mean"], ) - cache = _FakeCache({"start_ts": np.float64(start), "stop_ts": np.float64(stop)}) + cache = _FakeCache({_WINDOW_START: np.float64(start), _WINDOW_STOP: np.float64(stop)}) event_timestamps, numeric_values, _, _ = agg.build(cache) assert len(spark_windows[0]) == 7 diff --git a/tests/impulse_query_engine/unit/analyze/query/solvers/container_boundaries_test.py b/tests/impulse_query_engine/unit/analyze/query/solvers/container_boundaries_test.py index 46298ae1..1c5d8b37 100644 --- a/tests/impulse_query_engine/unit/analyze/query/solvers/container_boundaries_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/solvers/container_boundaries_test.py @@ -1,9 +1,10 @@ # pylint: disable=missing-function-docstring, redefined-outer-name -"""Tests for SolverConfig.normalize_container_boundaries / require_epoch_boundaries. +"""Tests for SolverConfig.with_window_bounds. -TIMESTAMP-typed container ``start_ts`` / ``stop_ts`` are converted to epoch numbers in -``SolverConfig.epoch_unit`` (opt-in), so container-boundary events and the solve see the -same values. With ``epoch_unit`` unset nothing changes. +``TimeWindowEvent`` windows are computed in the channel time frame +(``channel_time_unit`` / ``channel_time_origin``). ``with_window_bounds`` derives the container +start/stop in that frame as two extra columns and leaves the raw ``start_ts`` / ``stop_ts`` +untouched for everyone else (UDFs, ``ContainerEvent``, ``measurement_dimension``). """ import datetime as dt @@ -18,12 +19,15 @@ # 2025-07-03 07:41:41.483456 UTC _EPOCH_MICROS = 1_751_528_501_483_456 +# One hour and half a second later. +_SPAN_MICROS = 3_600_500_000 +_START, _STOP = "__window_start", "__window_stop" def _boundaries_df(spark: SparkSession): # noqa: F811 """container_metrics-like frame with TIMESTAMP boundaries (and a null row).""" df = spark.createDataFrame( - [(1, _EPOCH_MICROS, _EPOCH_MICROS + 60_000_000), (2, None, None)], + [(1, _EPOCH_MICROS, _EPOCH_MICROS + _SPAN_MICROS), (2, None, None)], "container_id int, start_us long, stop_us long", ) return df.select( @@ -33,6 +37,16 @@ def _boundaries_df(spark: SparkSession): # noqa: F811 ) +def _bounds(cfg: SolverConfig, df) -> dict: + out = cfg.with_window_bounds(df) + return {r.container_id: (r[_START], r[_STOP]) for r in out.collect()} + + +def test_window_bound_column_names(): + cfg = SolverConfig() + assert (cfg.window_start_col, cfg.window_stop_col) == (_START, _STOP) + + @pytest.mark.parametrize( "unit, expected_type, expected_start", [ @@ -43,28 +57,25 @@ def _boundaries_df(spark: SparkSession): # noqa: F811 ], ) @pytest.mark.parametrize("session_tz", ["UTC", "Europe/Berlin"]) -def test_timestamp_boundaries_converted_to_epoch_unit( +def test_epoch_origin_converts_timestamps_to_unit( spark, unit, expected_type, expected_start, session_tz # noqa: F811 ): previous_tz = spark.conf.get("spark.sql.session.timeZone") spark.conf.set("spark.sql.session.timeZone", session_tz) try: - out = SolverConfig(epoch_unit=unit).normalize_container_boundaries(_boundaries_df(spark)) + out = SolverConfig(channel_time_unit=unit).with_window_bounds(_boundaries_df(spark)) rows = {r.container_id: r for r in out.collect()} finally: spark.conf.set("spark.sql.session.timeZone", previous_tz) - assert out.schema["start_ts"].dataType == expected_type - assert out.schema["stop_ts"].dataType == expected_type - assert rows[1].start_ts == expected_start # exact, independent of the session time zone - assert rows[2].start_ts is None and rows[2].stop_ts is None + assert out.schema[_START].dataType == expected_type + assert rows[1][_START] == expected_start # exact, independent of the session time zone + assert rows[2][_START] is None and rows[2][_STOP] is None -def test_seconds_match_spark_cast_to_double(spark): # noqa: F811 - # "s" must equal today's ContainerEvent cast(timestamp as double), bit for bit. +def test_epoch_seconds_match_spark_cast_to_double(spark): # noqa: F811 + # "s" equals Spark's cast(timestamp as double), bit for bit. df = _boundaries_df(spark) - out = SolverConfig(epoch_unit="s").normalize_container_boundaries(df) - converted = {r.container_id: (r.start_ts, r.stop_ts) for r in out.collect()} casted = { r.container_id: (r.s, r.e) for r in df.select( @@ -73,52 +84,79 @@ def test_seconds_match_spark_cast_to_double(spark): # noqa: F811 F.col("stop_ts").cast("double").alias("e"), ).collect() } - assert converted == casted + assert _bounds(SolverConfig(channel_time_unit="s"), df) == casted + + +@pytest.mark.parametrize( + "unit, expected_stop", [("s", 3600.5), ("ms", 3_600_500.0), ("us", _SPAN_MICROS)] +) +@pytest.mark.parametrize("session_tz", ["UTC", "Europe/Berlin"]) +def test_container_start_origin_gives_relative_bounds( + spark, unit, expected_stop, session_tz # noqa: F811 +): + previous_tz = spark.conf.get("spark.sql.session.timeZone") + spark.conf.set("spark.sql.session.timeZone", session_tz) + try: + cfg = SolverConfig(channel_time_unit=unit, channel_time_origin="container_start") + bounds = _bounds(cfg, _boundaries_df(spark)) + finally: + spark.conf.set("spark.sql.session.timeZone", previous_tz) + + assert bounds[1] == (0, expected_stop) + # A null boundary leaves a null stop bound, so the container gets no windows. + assert bounds[2][1] is None + + +def test_numeric_boundaries_epoch_as_is_and_container_start_shifted(spark): # noqa: F811 + df = spark.createDataFrame( + [(1, 1000.5, 4601.0)], "container_id int, start_ts double, stop_ts double" + ) + assert _bounds(SolverConfig(), df) == {1: (1000.5, 4601.0)} + # Shift only: numeric boundaries are already in the channels' unit, so no unit is needed. + assert _bounds(SolverConfig(channel_time_origin="container_start"), df) == {1: (0, 3600.5)} -def test_unset_epoch_unit_leaves_frame_unchanged(spark): # noqa: F811 +def test_raw_boundaries_stay_unchanged(spark): # noqa: F811 df = _boundaries_df(spark) - out = SolverConfig().normalize_container_boundaries(df) - assert out is df + cfg = SolverConfig(channel_time_unit="s", channel_time_origin="container_start") + out = cfg.with_window_bounds(df) assert isinstance(out.schema["start_ts"].dataType, T.TimestampType) + assert out.select("container_id", "start_ts", "stop_ts").collect() == df.collect() -def test_numeric_boundaries_unchanged(spark): # noqa: F811 - df = spark.createDataFrame([(1, 100, 200)], "container_id int, start_ts long, stop_ts long") - out = SolverConfig(epoch_unit="ms").normalize_container_boundaries(df) - assert out.schema == df.schema - assert out.collect() == df.collect() +def test_timestamp_boundaries_without_unit_rejected(spark): # noqa: F811 + for origin in ("epoch", "container_start"): + with pytest.raises(ValueError, match=r"TimeWindowEvent.*channel_time_unit"): + SolverConfig(channel_time_origin=origin).with_window_bounds(_boundaries_df(spark)) @pytest.mark.parametrize( "value, ddl", [(dt.datetime(2025, 7, 3, 7, 41, 41), "timestamp_ntz"), (dt.date(2025, 7, 3), "date")], ) -def test_zone_less_types_rejected_when_unit_set(spark, value, ddl): # noqa: F811 +def test_zone_less_types_rejected(spark, value, ddl): # noqa: F811 df = spark.createDataFrame( [(1, value, value)], f"container_id int, start_ts {ddl}, stop_ts {ddl}" ) with pytest.raises(ValueError, match="start_ts"): - SolverConfig(epoch_unit="s").normalize_container_boundaries(df) - # Opt-in only: without epoch_unit the frame passes through untouched. - assert SolverConfig().normalize_container_boundaries(df) is df + SolverConfig(channel_time_unit="s").with_window_bounds(df) -def test_require_epoch_boundaries(spark): # noqa: F811 - df = _boundaries_df(spark) - with pytest.raises(ValueError, match=r"TimeWindowEvent.*epoch_unit"): - SolverConfig().require_epoch_boundaries(df, owner="TimeWindowEvent") +def test_mixed_and_missing_boundaries_rejected(spark): # noqa: F811 + mixed = _boundaries_df(spark).withColumn("stop_ts", F.lit(1.0)) + with pytest.raises(ValueError, match="both be TIMESTAMP or both be numeric"): + SolverConfig(channel_time_unit="s").with_window_bounds(mixed) + with pytest.raises(ValueError, match="stop_ts"): + SolverConfig().with_window_bounds(_boundaries_df(spark).drop("stop_ts")) - cfg = SolverConfig(epoch_unit="s") - cfg.require_epoch_boundaries(cfg.normalize_container_boundaries(df), owner="TimeWindowEvent") - numeric = spark.createDataFrame( - [(1, 1.0, 2.0)], "container_id int, start_ts double, stop_ts double" +def test_channel_time_settings_validated(): + cfg = SolverConfig.model_validate( + {"channel_time_unit": "ns", "channel_time_origin": "container_start"} ) - SolverConfig().require_epoch_boundaries(numeric, owner="TimeWindowEvent") - - -def test_epoch_unit_validated(): - assert SolverConfig.model_validate({"epoch_unit": "ns"}).epoch_unit == "ns" + assert (cfg.channel_time_unit, cfg.channel_time_origin) == ("ns", "container_start") + assert SolverConfig().channel_time_origin == "epoch" + with pytest.raises(ValueError): + SolverConfig.model_validate({"channel_time_unit": "minutes"}) with pytest.raises(ValueError): - SolverConfig.model_validate({"epoch_unit": "minutes"}) + SolverConfig.model_validate({"channel_time_origin": "recording_start"}) diff --git a/tests/impulse_query_engine/unit/analyze/query/solvers/default_solver_container_metadata_test.py b/tests/impulse_query_engine/unit/analyze/query/solvers/default_solver_container_metadata_test.py index 3f052762..c87b52b8 100644 --- a/tests/impulse_query_engine/unit/analyze/query/solvers/default_solver_container_metadata_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/solvers/default_solver_container_metadata_test.py @@ -19,7 +19,9 @@ from pyspark.sql import SparkSession import impulse_query_engine.schema as S +from impulse_query_engine.analyze.query.aggregations.stats_aggregator import StatsAggregator from impulse_query_engine.analyze.query.channels.calculated_channel import CalculatedChannel +from impulse_query_engine.analyze.query.events import TimeWindowExpression from impulse_query_engine.analyze.query.solvers.default_solver import ( DefaultSolver, TimeSeriesCache, @@ -347,42 +349,79 @@ def _grab_start_ts(ts, container_metrics): value = container_metrics["start_ts"] if value is None: # type-inference pass on the empty cache return 0.0 - # Encode what reached the UDF: the epoch-seconds value, or -1 for a pd.Timestamp. - return -1.0 if isinstance(value, pd.Timestamp) else float(value) - - -def test_timestamp_boundaries_converted_for_udf_when_epoch_unit_set( - spark: SparkSession, basic_narrow_db: MeasurementDB + # Like a customer UDF rebuilding absolute time: a pd.Timestamp (naive, in the session + # time zone, here UTC) becomes epoch microseconds (exact); anything else is flagged -1. + if not isinstance(value, pd.Timestamp): + return -1.0 + return float((value - pd.Timestamp("1970-01-01")) // pd.Timedelta(microseconds=1)) + + +@pytest.mark.parametrize( + "config", + [ + SolverConfig(), + SolverConfig(channel_time_unit="ms"), + SolverConfig(channel_time_unit="ms", channel_time_origin="container_start"), + ], +) +def test_udf_gets_raw_timestamp_start_ts_regardless_of_channel_time( + spark: SparkSession, basic_narrow_db: MeasurementDB, config: SolverConfig ): - """With epoch_unit set, a TIMESTAMP start_ts reaches the UDF as epoch seconds.""" + """A UDF reading a TIMESTAMP start_ts always gets the absolute pd.Timestamp: the channel + time settings only shape TimeWindowEvent windows, never the raw container metrics.""" db = _timestamp_boundaries_db(basic_narrow_db) query = db.query - result = query.select( - query.channel(channel_name="Engine RPM") - .apply(_grab_start_ts, container_metrics=["start_ts"]) - .alias("start") - ).solve(spark, solver=DefaultSolver(spark, config=SolverConfig(epoch_unit="s"))) + previous_tz = spark.conf.get("spark.sql.session.timeZone") + spark.conf.set("spark.sql.session.timeZone", "UTC") + try: + result = query.select( + query.channel(channel_name="Engine RPM") + .apply(_grab_start_ts, container_metrics=["start_ts"]) + .alias("start") + ).solve(spark, solver=DefaultSolver(spark, config=config)) + rows = {row.container_id: row.start for row in result.collect()} + finally: + spark.conf.set("spark.sql.session.timeZone", previous_tz) expected = { - r.container_id: r.s + r.container_id: float(r.us) for r in db.container_metrics(spark) - .select("container_id", F.col("start_ts").cast("double").alias("s")) + .select("container_id", F.unix_micros("start_ts").alias("us")) .collect() } - rows = {row.container_id: row.start for row in result.collect()} assert rows and all(rows[cid] == expected[cid] for cid in rows), (rows, expected) -def test_timestamp_boundaries_unchanged_for_udf_without_epoch_unit( +def test_time_window_expression_and_udf_share_a_solve( spark: SparkSession, basic_narrow_db: MeasurementDB ): - """Backward compatibility: without epoch_unit, the UDF still gets a pd.Timestamp.""" - query = _timestamp_boundaries_db(basic_narrow_db).query + """In one solve, a TimeWindowExpression tiles the container in the channel time frame + (here relative ms) while a UDF still reads the absolute TIMESTAMP start_ts.""" + db = _timestamp_boundaries_db(basic_narrow_db) + query = db.query + rpm = query.channel(channel_name="Engine RPM") + windows = StatsAggregator( + input_expressions=[rpm], + event_expression=TimeWindowExpression(10_000), + statistics=["mean"], + ).alias("windows") + config = SolverConfig(channel_time_unit="ms", channel_time_origin="container_start") result = query.select( - query.channel(channel_name="Engine RPM") - .apply(_grab_start_ts, container_metrics=["start_ts"]) - .alias("start") - ).solve(spark, solver=DefaultSolver(spark)) - - rows = [row.start for row in result.collect()] - assert rows and all(value == -1.0 for value in rows), rows + windows, rpm.apply(_grab_start_ts, container_metrics=["start_ts"]).alias("start") + ).solve(spark, solver=DefaultSolver(spark, config=config)) + rows = {row.container_id: row for row in result.collect()} + + durations = { + r.container_id: float(r.d) + for r in basic_narrow_db.container_metrics(spark) + .select("container_id", (F.col("stop_ts") - F.col("start_ts")).alias("d")) + .collect() + } + assert rows + for container_id, row in rows.items(): + event_timestamps = row.windows.event_timestamps + # Relative windows: from 0 to the container's duration in ms, 10 s apart. + assert event_timestamps[0] == [0.0, 10_000.0] + assert event_timestamps[-1][1] == durations[container_id] + assert len(event_timestamps) == -(-durations[container_id] // 10_000) + assert row.start > 0 # absolute epoch microseconds, not -1 (non-timestamp) diff --git a/tests/impulse_reporting/integration/time_window_event_test.py b/tests/impulse_reporting/integration/time_window_event_test.py index cabfd152..ff4d6029 100644 --- a/tests/impulse_reporting/integration/time_window_event_test.py +++ b/tests/impulse_reporting/integration/time_window_event_test.py @@ -3,6 +3,7 @@ import math from unittest.mock import create_autospec +import pandas as pd import pyspark.sql.functions as F import pyspark.sql.types as T import pytest @@ -137,15 +138,17 @@ def test_time_window_event_in_report(spark, basic_narrow_db): # Customer-shaped time bases for the id-join test, all derived from the basic db's µs epochs. # Each entry: (transform for channel tstart/tend, transform for container start_ts/stop_ts, -# window length in the samples' unit, SolverConfig.epoch_unit). +# window length in the samples' unit, SolverConfig channel time settings). # us: the native µs epochs (< 2^53, every boundary exactly representable). # ns: ns epochs (~1.5e18, beyond 2^53) with a window that is NOT a multiple of the # 256 ns double spacing there, so the boundaries round. # sec: seconds as doubles with a fractional window, so the boundaries round. -# sec_ts: samples as seconds-as-double, container boundaries as TIMESTAMP (converted to -# epoch seconds via epoch_unit="s"). -# us_ts: native µs samples, container boundaries as TIMESTAMP (converted to epoch µs via -# epoch_unit="us", the long path of the conversion). +# sec_ts: samples as seconds-as-double, container boundaries as TIMESTAMP (windows in +# epoch seconds via channel_time_unit="s"). +# us_ts: native µs samples, container boundaries as TIMESTAMP (windows in epoch µs via +# channel_time_unit="us", the long path of the conversion). +# rel_sec: samples as seconds since the container start (double), container boundaries as +# TIMESTAMP (windows from 0 via channel_time_origin="container_start"). def _to_seconds(c): return c.cast("double") / F.lit(1e6) @@ -154,28 +157,35 @@ def _to_ns(c): return c.cast("long") * F.lit(1000) +def _to_timestamp(c): + return F.timestamp_micros(c.cast("long")) + + +_RELATIVE_SECONDS = {"channel_time_unit": "s", "channel_time_origin": "container_start"} + _TIME_BASES = { - "us": (lambda c: c, lambda c: c, ALIGNED_WINDOW_LENGTH, None), - "ns": (_to_ns, _to_ns, 600_000_000_007, None), - "sec": (_to_seconds, _to_seconds, 600.3, None), - "sec_ts": (_to_seconds, lambda c: F.timestamp_micros(c.cast("long")), 600.3, "s"), - "us_ts": ( - lambda c: c, - lambda c: F.timestamp_micros(c.cast("long")), - ALIGNED_WINDOW_LENGTH, - "us", - ), + "us": (lambda c: c, lambda c: c, ALIGNED_WINDOW_LENGTH, {}), + "ns": (_to_ns, _to_ns, 600_000_000_007, {}), + "sec": (_to_seconds, _to_seconds, 600.3, {}), + "sec_ts": (_to_seconds, _to_timestamp, 600.3, {"channel_time_unit": "s"}), + "us_ts": (lambda c: c, _to_timestamp, ALIGNED_WINDOW_LENGTH, {"channel_time_unit": "us"}), + "rel_sec": (_to_seconds, _to_timestamp, 600.3, _RELATIVE_SECONDS), } def _clone_aligned_silver( - spark, schema: str, to_time_base=lambda c: c, boundaries_to_time_base=None + spark, + schema: str, + to_time_base=lambda c: c, + boundaries_to_time_base=None, + relative_channels: bool = False, ) -> None: """Clone the basic silver tables into *schema* with container_metrics start_ts / stop_ts recomputed from each container's channel-sample range (so the container boundaries, and thus the windows, share the samples' time base). Channel timestamps are then mapped - through *to_time_base* and the container boundaries through *boundaries_to_time_base* - (default: the same transform).""" + through *to_time_base* (after subtracting the container start when *relative_channels*) + and the container boundaries through *boundaries_to_time_base* (default: the same + transform).""" boundaries_to_time_base = boundaries_to_time_base or to_time_base spark.sql(f"CREATE SCHEMA IF NOT EXISTS {schema}") channels = spark.read.table("spark_catalog.silver.channels") @@ -197,6 +207,13 @@ def _clone_aligned_silver( aligned_cm.write.format("delta").mode("overwrite").option( "overwriteSchema", "true" ).saveAsTable(f"{schema}.container_metrics") + if relative_channels: + channels = ( + channels.join(bounds, on="container_id") + .withColumn("tstart", F.col("tstart") - F.col("_agg_start")) + .withColumn("tend", F.col("tend") - F.col("_agg_start")) + .drop("_agg_start", "_agg_stop") + ) channels.withColumn("tstart", to_time_base(F.col("tstart"))).withColumn( "tend", to_time_base(F.col("tend")) ).write.format("delta").mode("overwrite").option("overwriteSchema", "true").saveAsTable( @@ -211,20 +228,27 @@ def _clone_aligned_silver( def setup_tw_aligned_db(spark, setup_basic_db, request): # noqa: F811 """Aligned silver clone in the time base given by ``request.param`` (default ``us``). - Yields ``(schema, window_length, epoch_unit)``. + Yields ``(schema, window_length, channel_time)``, where *channel_time* holds the + SolverConfig channel time settings for that time base. """ time_base = getattr(request, "param", "us") - to_time_base, boundaries_to_time_base, window_length, epoch_unit = _TIME_BASES[time_base] + to_time_base, boundaries_to_time_base, window_length, channel_time = _TIME_BASES[time_base] schema = f"{_ALIGNED_SCHEMA}_{time_base}" - _clone_aligned_silver(spark, schema, to_time_base, boundaries_to_time_base) - yield schema, window_length, epoch_unit + _clone_aligned_silver( + spark, + schema, + to_time_base, + boundaries_to_time_base, + relative_channels=channel_time.get("channel_time_origin") == "container_start", + ) + yield schema, window_length, channel_time spark.sql(f"DROP SCHEMA IF EXISTS {schema} CASCADE") def _aligned_config( schema: str, table_prefix: str, - epoch_unit=None, + channel_time: dict | None = None, raw_encoder: RawEncoder | None = None, channels_table: str = "channels", **extra, @@ -251,7 +275,7 @@ def _aligned_config( ), query_engine=QueryEngine( solver=Solvers.KEY_VALUE_STORE_SOLVER, - solver_config=SolverConfig(epoch_unit=epoch_unit) if epoch_unit else None, + solver_config=SolverConfig(**channel_time) if channel_time else None, data_type=DataType.RAW if raw_encoder else DataType.RLE, raw_encoder=raw_encoder, ), @@ -380,19 +404,20 @@ def _is_missing(value) -> bool: @pytest.mark.parametrize( - "setup_tw_aligned_db", ["us", "ns", "sec", "sec_ts", "us_ts"], indirect=True + "setup_tw_aligned_db", ["us", "ns", "sec", "sec_ts", "us_ts", "rel_sec"], indirect=True ) def test_time_window_event_aggregation_join(spark, setup_tw_aligned_db): """Stats scoped to a TimeWindowEvent yield per-window values whose event_instance_id joins to the natively computed event fact, for µs, ns and seconds-as-double time bases, - and for TIMESTAMP container boundaries converted via epoch_unit.""" - schema, window_length, epoch_unit = setup_tw_aligned_db + for TIMESTAMP container boundaries (channel_time_unit), and for channel timestamps + relative to the container start (channel_time_origin="container_start").""" + schema, window_length, channel_time = setup_tw_aligned_db table_prefix = f"time_window_join_test_{schema.removeprefix(_ALIGNED_SCHEMA + '_')}" my_report = Report( name="time_window_join_report", spark=spark, workspace_client=create_autospec(WorkspaceClient), - config=_aligned_config(schema, table_prefix, epoch_unit=epoch_unit), + config=_aligned_config(schema, table_prefix, channel_time=channel_time), ) window_evt = TimeWindowEvent(name="ten_min", window_length=window_length) @@ -407,6 +432,19 @@ def test_time_window_event_aggregation_join(spark, setup_tw_aligned_db): _assert_ids_join(spark, table_prefix) _assert_window_stats_match_samples(spark, schema, table_prefix) + if channel_time.get("channel_time_origin") == "container_start": + _assert_windows_start_at_zero(spark, table_prefix) + + +def _assert_windows_start_at_zero(spark, table_prefix: str) -> None: # noqa: F811 + """Relative channel time: every container's first window starts at 0.""" + first = ( + spark.read.table(f"spark_catalog.gold.{table_prefix}_event_instance_fact") + .groupBy("container_id") + .agg(F.min("start_ts").alias("first")) + .collect() + ) + assert first and all(r.first == 0.0 for r in first), first def _write_raw_channels(spark, schema: str) -> str: # noqa: F811 @@ -423,9 +461,10 @@ def _write_raw_channels(spark, schema: str) -> str: # noqa: F811 @pytest.mark.parametrize("setup_tw_aligned_db", ["us", "us_ts"], indirect=True) def test_time_window_event_aggregation_join_raw(spark, setup_tw_aligned_db, raw_encoder): """With data_type=RAW both encoders derive [tstart, tend) from the raw ``timestamp`` - column without changing its unit, so windows over numeric or TIMESTAMP (epoch_unit="us") - container boundaries line up with the samples exactly as for RLE silver data.""" - schema, window_length, epoch_unit = setup_tw_aligned_db + column without changing its unit, so windows over numeric or TIMESTAMP + (channel_time_unit="us") container boundaries line up with the samples exactly as for + RLE silver data.""" + schema, window_length, channel_time = setup_tw_aligned_db channels_table = _write_raw_channels(spark, schema) time_base = schema.removeprefix(_ALIGNED_SCHEMA + "_") table_prefix = f"time_window_raw_test_{time_base}_{raw_encoder.value.lower()}" @@ -436,7 +475,7 @@ def test_time_window_event_aggregation_join_raw(spark, setup_tw_aligned_db, raw_ config=_aligned_config( schema, table_prefix, - epoch_unit=epoch_unit, + channel_time=channel_time, raw_encoder=raw_encoder, channels_table=channels_table, ), @@ -702,11 +741,14 @@ def _run(cm_table: str, is_incremental: bool, statistics) -> None: # --------------------------------------------------------------------------- -# TIMESTAMP container boundaries: epoch_unit is opt-in, required only by TimeWindowEvent +# TIMESTAMP container boundaries: channel_time_unit is required only by TimeWindowEvent # --------------------------------------------------------------------------- @pytest.mark.parametrize("setup_tw_aligned_db", ["sec_ts"], indirect=True) -def test_time_window_event_timestamp_boundaries_require_epoch_unit(spark, setup_tw_aligned_db): - """A TimeWindowEvent over TIMESTAMP boundaries without epoch_unit fails fast and clearly.""" +def test_time_window_event_timestamp_boundaries_require_channel_time_unit( + spark, setup_tw_aligned_db +): + """A TimeWindowEvent over TIMESTAMP boundaries without channel_time_unit fails fast and + clearly.""" schema, window_length, _ = setup_tw_aligned_db my_report = Report( name="time_window_no_unit_report", @@ -716,23 +758,29 @@ def test_time_window_event_timestamp_boundaries_require_epoch_unit(spark, setup_ ) my_report.add_event(TimeWindowEvent(name="ten_min", window_length=window_length)) - with pytest.raises(ValueError, match=r"TimeWindowEvent.*epoch_unit"): + with pytest.raises(ValueError, match=r"TimeWindowEvent.*channel_time_unit"): my_report.determine_report() -@pytest.mark.parametrize("epoch_unit", [None, "s"]) +@pytest.mark.parametrize( + "channel_time", + [None, {"channel_time_unit": "ms"}, _RELATIVE_SECONDS], + ids=["unset", "ms", "relative_s"], +) @pytest.mark.parametrize("setup_tw_aligned_db", ["sec_ts"], indirect=True) -def test_container_event_timestamp_boundaries(spark, setup_tw_aligned_db, epoch_unit): - """Backward compatibility: a ContainerEvent over TIMESTAMP boundaries runs without - epoch_unit (as today) and yields epoch seconds; epoch_unit="s" gives identical values. - measurement_dimension keeps the TIMESTAMP type either way.""" +def test_container_event_timestamp_boundaries(spark, setup_tw_aligned_db, channel_time): + """A ContainerEvent ignores the channel time settings: over TIMESTAMP boundaries it + always writes the raw boundaries as epoch seconds (as on main). measurement_dimension + keeps the TIMESTAMP type.""" schema, _, _ = setup_tw_aligned_db - table_prefix = f"container_event_ts_test_{epoch_unit or 'unset'}" + unit = (channel_time or {}).get("channel_time_unit", "unset") + origin = (channel_time or {}).get("channel_time_origin", "epoch") + table_prefix = f"container_event_ts_test_{unit}_{origin}" my_report = Report( name="container_event_ts_report", spark=spark, workspace_client=create_autospec(WorkspaceClient), - config=_aligned_config(schema, table_prefix, epoch_unit=epoch_unit), + config=_aligned_config(schema, table_prefix, channel_time=channel_time), ) my_report.add_event(ContainerEvent(name="full_container")) my_report.determine_report() @@ -757,13 +805,14 @@ def test_container_event_timestamp_boundaries(spark, setup_tw_aligned_db, epoch_ @pytest.mark.parametrize("setup_tw_aligned_db", ["sec_ts"], indirect=True) -def test_epoch_unit_change_recomputes_boundary_events(spark, setup_tw_aligned_db): - """Changing epoch_unit between incremental runs moves the definition hashes of the - TimeWindowEvent, the ContainerEvent and the aggregation scoped to the windows. They - recompute over all containers, so the gold tables never mix units.""" - schema, window_length, epoch_unit = setup_tw_aligned_db - assert epoch_unit == "s" - table_prefix = "time_window_epoch_unit_test" +def test_channel_time_change_recomputes_time_window_event_only(spark, setup_tw_aligned_db): + """Changing channel_time_unit between incremental runs moves the definition hashes of the + TimeWindowEvent and the aggregation scoped to its windows, so they recompute over all + containers and the windows never mix units. The ContainerEvent writes the raw boundaries, + so it keeps its hash and its rows.""" + schema, window_length, channel_time = setup_tw_aligned_db + assert channel_time == {"channel_time_unit": "s"} + table_prefix = "time_window_channel_time_test" cm_run_1 = f"{schema}.container_metrics_run_1" cm_run_2 = f"{schema}.container_metrics_run_2" past = F.lit("2020-01-01 00:00:00").cast("timestamp") @@ -780,7 +829,7 @@ def _run(cm_table: str, unit: str, is_incremental: bool): config = _aligned_config( schema, table_prefix, - epoch_unit=unit, + channel_time={"channel_time_unit": unit}, incremental=IncrementalConfig( enabled=is_incremental, silver_last_modified_column="timestamp", @@ -789,7 +838,7 @@ def _run(cm_table: str, unit: str, is_incremental: bool): ) config["source"].container_metrics_table = cm_table report = Report( - name="time_window_epoch_unit_report", + name="time_window_channel_time_report", spark=spark, workspace_client=create_autospec(WorkspaceClient), config=config, @@ -813,24 +862,27 @@ def _event_rows(event_id: int): .collect() ) - _, _, container_evt, _ = _run(cm_run_1, "s", is_incremental=False) - starts_in_s = {r.container_id: r.start_ts for r in _event_rows(container_evt.get_id())} - assert set(starts_in_s) == {1, 2} - + _run(cm_run_1, "s", is_incremental=False) report, window_evt, container_evt, stats = _run(cm_run_2, "ms", is_incremental=True) changed_events = {i for ids in report._changed_event_ids.values() for i in ids} changed_aggs = {i for ids in report._changed_aggregation_ids.values() for i in ids} - assert {window_evt.get_id(), container_evt.get_id()} <= changed_events + assert window_evt.get_id() in changed_events assert stats.get_id() in changed_aggs + assert container_evt.get_id() not in changed_events - # The unchanged containers 1 and 2 were rewritten in ms, not left in seconds. + boundaries = { + r.container_id: r + for r in cm.select( + "container_id", + F.col("start_ts").cast("double").alias("start_s"), + (F.unix_micros("start_ts") / F.lit(1e3)).alias("start_ms"), + ).collect() + } + # ContainerEvent: the raw boundaries as epoch seconds, for the old and the new containers. container_rows = _event_rows(container_evt.get_id()) assert {r.container_id for r in container_rows} == {1, 2, 3} - for r in container_rows: - if r.container_id in starts_in_s: - assert r.start_ts == pytest.approx(starts_in_s[r.container_id] * 1000) - starts_in_ms = {r.container_id: r.start_ts for r in container_rows} + assert all(r.start_ts == boundaries[r.container_id].start_s for r in container_rows) # Every window tiles the container's ms span: none is left over from the seconds run. window_rows = _event_rows(window_evt.get_id()) @@ -840,4 +892,73 @@ def _event_rows(event_id: int): first_window[r.container_id] = min( first_window.get(r.container_id, r.start_ts), r.start_ts ) - assert first_window == starts_in_ms + assert first_window == {cid: b.start_ms for cid, b in boundaries.items()} + + +def _absolute_start_micros(ts, container_metrics): + """Customer-style UDF: rebuilding absolute time needs the TIMESTAMP start_ts, which + arrives as a pd.Timestamp (naive, in the session time zone). Returns it as a constant + series of epoch microseconds.""" + value = container_metrics["start_ts"] + if value is None: # type-inference pass on the empty cache + return ts * 0 + micros = (value - pd.Timestamp("1970-01-01")) // pd.Timedelta(microseconds=1) + return ts * 0 + float(micros) + + +@pytest.mark.parametrize("setup_tw_aligned_db", ["rel_sec"], indirect=True) +def test_udf_reads_absolute_start_ts_next_to_relative_time_window_event( + spark, setup_tw_aligned_db +): + """Customer pattern: channels hold seconds since the container start, start_ts / stop_ts + are TIMESTAMP. A TimeWindowEvent tiles each container in that relative frame, while a UDF + in the same report still reads the absolute start_ts as a pd.Timestamp.""" + schema, window_length, channel_time = setup_tw_aligned_db + table_prefix = "time_window_udf_absolute_test" + previous_tz = spark.conf.get("spark.sql.session.timeZone") + spark.conf.set("spark.sql.session.timeZone", "UTC") + try: + report = Report( + name="time_window_udf_absolute_report", + spark=spark, + workspace_client=create_autospec(WorkspaceClient), + config=_aligned_config(schema, table_prefix, channel_time=channel_time), + ) + window_evt = TimeWindowEvent(name="ten_min", window_length=window_length) + report.add_event(window_evt) + page = Page(page_number=1) + report.add_page(page) + page.add_aggregation(_rpm_stats(report, window_evt)) + rpm = report.get_db().query.channel(channel_name="Engine RPM") + absolute_start = StatsAggregator( + name="absolute_start_per_window", + input_expressions=[rpm.apply(_absolute_start_micros, container_metrics=["start_ts"])], + channel_names=["absolute_start_us"], + statistics=["max"], + event=window_evt, + ) + page.add_aggregation(absolute_start) + report.determine_report() + report.persist_results() + finally: + spark.conf.set("spark.sql.session.timeZone", previous_tz) + + _assert_ids_join(spark, table_prefix) + _assert_windows_start_at_zero(spark, table_prefix) + + # The UDF saw the absolute start: in every window with samples, its value equals + # unix_micros(start_ts) of that container. + expected = { + r.container_id: float(r.us) + for r in spark.read.table(f"{schema}.container_metrics") + .select("container_id", F.unix_micros("start_ts").alias("us")) + .collect() + } + values = ( + spark.read.table(f"spark_catalog.gold.{table_prefix}_stats_aggregator_fact") + .filter(F.col("visual_id") == absolute_start.get_id()) + .filter(F.col("statistic_value").isNotNull() & ~F.isnan("statistic_value")) + .collect() + ) + assert values + assert all(r.statistic_value == expected[r.container_id] for r in values), values[:3] diff --git a/tests/impulse_reporting/unit/aggregations/definition_hash_test.py b/tests/impulse_reporting/unit/aggregations/definition_hash_test.py index 3dca9c68..b68fe493 100644 --- a/tests/impulse_reporting/unit/aggregations/definition_hash_test.py +++ b/tests/impulse_reporting/unit/aggregations/definition_hash_test.py @@ -355,10 +355,10 @@ def test_hash_without_custom_stats_matches_formula(self): assert stats_agg.determine_definition_hash() == expected - def test_time_window_epoch_unit_changes_hash(self): + def test_time_window_channel_time_changes_hash(self): """Statistics scoped to a TimeWindowEvent are computed per window, and the windows - tile TIMESTAMP boundaries in the report's epoch_unit, so the unit must move the - aggregation's hash too (it does so through the event expression string).""" + lie in the report's channel time frame, so changing it must move the aggregation's + hash too (it does so through the event expression string).""" event = TimeWindowEvent(name="windows", window_length=10_000) stats_agg = StatsAggregator( name="stats", @@ -372,7 +372,7 @@ def test_time_window_epoch_unit_changes_hash(self): ) before = (stats_agg.determine_definition_hash(), hist.determine_definition_hash()) - event.set_epoch_unit("ms") + event.set_channel_time("ms", "container_start") assert stats_agg.determine_definition_hash() != before[0] assert hist.determine_definition_hash() != before[1] diff --git a/tests/impulse_reporting/unit/events/container_event_test.py b/tests/impulse_reporting/unit/events/container_event_test.py index 3fc6d0e3..ad85a4a3 100644 --- a/tests/impulse_reporting/unit/events/container_event_test.py +++ b/tests/impulse_reporting/unit/events/container_event_test.py @@ -147,24 +147,12 @@ def _sha256_long(text: str) -> int: return int.from_bytes(hashlib.sha256(text.encode()).digest()[:8], "big", signed=True) -def test_definition_hash_without_epoch_unit_is_name_only(): - """Unset epoch_unit keeps the pre-existing name-only hash, so gold tables written before - epoch_unit existed are not recomputed.""" +def test_definition_hash_is_name_only(): + """The channel time settings only shape TimeWindowEvent windows; a ContainerEvent writes + the raw boundaries, so its hash stays the name alone and never forces a recompute.""" event = ContainerEvent(name="ev") assert event.determine_definition_hash() == _sha256_long("ev") - event.set_epoch_unit(None) - assert event.determine_definition_hash() == _sha256_long("ev") - - -def test_definition_hash_changes_with_epoch_unit(): - """epoch_unit decides the unit of TIMESTAMP boundaries in start_ts / end_ts, so changing - it must force a recompute instead of mixing units in event_instance_fact.""" - unset, ms, us = ContainerEvent(name="ev"), ContainerEvent(name="ev"), ContainerEvent(name="ev") - ms.set_epoch_unit("ms") - us.set_epoch_unit("us") - hashes = {e.determine_definition_hash() for e in (unset, ms, us)} - assert len(hashes) == 3 - assert ms.as_dict()["definition_hash"] == ms.determine_definition_hash() + assert event.as_dict()["definition_hash"] == _sha256_long("ev") # --------------------------------------------------------------------------- diff --git a/tests/impulse_reporting/unit/events/time_window_event_test.py b/tests/impulse_reporting/unit/events/time_window_event_test.py index 9f0ff2ef..d3bc63fe 100644 --- a/tests/impulse_reporting/unit/events/time_window_event_test.py +++ b/tests/impulse_reporting/unit/events/time_window_event_test.py @@ -96,22 +96,24 @@ def test_definition_hash_stable_across_int_and_float_window_length(): assert a.determine_definition_hash() == b.determine_definition_hash() -def test_definition_hash_changes_with_epoch_unit(): - # epoch_unit decides the unit TIMESTAMP boundaries are tiled in, so flipping it must - # force a full recompute. It reaches the hash through the expression string. - unset = TimeWindowEvent(name="w", window_length=10000) - cleared = TimeWindowEvent(name="w", window_length=10000) - cleared.set_epoch_unit(None) - s, ms = TimeWindowEvent(name="w", window_length=10000), TimeWindowEvent( - name="w", window_length=10000 - ) - s.set_epoch_unit("s") - ms.set_epoch_unit("ms") +def test_definition_hash_changes_with_channel_time_frame(): + # The channel time frame decides where the windows lie, so changing the unit or the + # origin must force a full recompute. It reaches the hash through the expression string. + def event(unit=None, origin="epoch") -> TimeWindowEvent: + e = TimeWindowEvent(name="w", window_length=10000) + e.set_channel_time(unit, origin) + return e - assert ms.epoch_unit == ms.get_expression().epoch_unit == "ms" - assert "epoch_unit=ms" in ms.as_dict()["event_expression"] - assert unset.determine_definition_hash() == cleared.determine_definition_hash() - assert len({e.determine_definition_hash() for e in (unset, s, ms)}) == 3 + unset = TimeWindowEvent(name="w", window_length=10000) + s, ms, ms_relative = event("s"), event("ms"), event("ms", "container_start") + + assert ms_relative.get_expression().channel_time_unit == "ms" + assert ms_relative.get_expression().channel_time_origin == "container_start" + assert "channel_time_origin=container_start" in ms_relative.as_dict()["event_expression"] + # The defaults keep today's hash. + assert unset.determine_definition_hash() == event().determine_definition_hash() + hashes = {e.determine_definition_hash() for e in (unset, s, ms, ms_relative)} + assert len(hashes) == 4 # --------------------------------------------------------------------------- From b73437ebe381bbf164959ea3e670c938bd2fc64d Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 7 Oct 2026 16:36:37 +0200 Subject: [PATCH 17/27] feat(query-engine, reporting): add solver_config.container_time_unit for numeric boundaries MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add `solver_config.container_time_unit` to convert numeric `container_metrics.start_ts`/`stop_ts` into the channel time unit used by `TimeWindowEvent` windows. This supports cases like epoch-ms boundaries with µs channel samples. The setting requires `channel_time_unit`, is rejected for `TIMESTAMP` boundaries, and is included in `TimeWindowEvent` definition hashes. Update docs, API references, skills, and tests. --- docs/impulse/docs/config/configuration.md | 18 ++++-- .../docs/data_model/silver_layer_schema.md | 4 +- .../analyze/query/solvers/solver_config.md | 30 ++++++--- .../events/time_window_event.md | 7 ++- docs/impulse/docs/references/report/event.md | 7 ++- skills/impulse-config/SKILL.md | 6 +- skills/impulse-events/SKILL.md | 6 +- .../query/events/time_window_expression.py | 16 +++-- .../analyze/query/solvers/solver_config.py | 62 +++++++++++++++--- src/impulse_reporting/core/report.py | 5 +- .../events/time_window_event.py | 9 ++- .../events/time_window_expression_test.py | 2 + .../solvers/container_boundaries_test.py | 63 +++++++++++++++++++ .../integration/time_window_event_test.py | 21 ++++++- .../unit/events/time_window_event_test.py | 10 +-- 15 files changed, 219 insertions(+), 47 deletions(-) diff --git a/docs/impulse/docs/config/configuration.md b/docs/impulse/docs/config/configuration.md index 50f18868..85eb3934 100644 --- a/docs/impulse/docs/config/configuration.md +++ b/docs/impulse/docs/config/configuration.md @@ -177,13 +177,19 @@ Top-level fields on `SolverConfig`: since the recording started). Channel timestamps are never converted; they must already be numbers in that frame. - Both settings are only used by `TimeWindowEvent`, whose windows must lie in the channel time +- `container_time_unit` (`"s"` | `"ms"` | `"us"` | `"ns"`, optional): the unit of **numeric** + `container_metrics.start_ts`/`stop_ts`, when it differs from the channels' unit. For example, + boundaries in epoch ms and channel samples in epoch µs need `container_time_unit = "ms"` and + `channel_time_unit = "us"`. Requires `channel_time_unit`; not allowed for `TIMESTAMP` + boundaries, which carry their own unit. Unset means the numeric boundaries are already in the + channels' unit. + + These settings are only used by `TimeWindowEvent`, whose windows must lie in the channel time frame. It derives its window bounds from `container_metrics.start_ts`/`stop_ts`: - - origin `"epoch"`: numeric boundaries as they are; `TIMESTAMP` boundaries as epoch numbers in - `channel_time_unit`; - - origin `"container_start"`: `0` to `stop_ts - start_ts`, in `channel_time_unit` for - `TIMESTAMP` boundaries; numeric boundaries are only shifted (they must already be in the - channels' unit). + - origin `"epoch"`: `TIMESTAMP` boundaries as epoch numbers in `channel_time_unit`; numeric + boundaries converted from `container_time_unit` to `channel_time_unit` (as they are when + `container_time_unit` is unset); + - origin `"container_start"`: `0` to `stop_ts - start_ts`, converted the same way. `channel_time_unit` is required when the boundaries are `TIMESTAMP` columns; a report with a `TimeWindowEvent` fails with a clear error until it is set. `TIMESTAMP_NTZ` and `DATE` boundaries diff --git a/docs/impulse/docs/data_model/silver_layer_schema.md b/docs/impulse/docs/data_model/silver_layer_schema.md index 5522a975..36491670 100644 --- a/docs/impulse/docs/data_model/silver_layer_schema.md +++ b/docs/impulse/docs/data_model/silver_layer_schema.md @@ -184,7 +184,9 @@ the epoch-typed pair). Populate whichever your queries and [`solver_config.channel_time_unit`](../config/configuration.md#solver-column-mappings-and-filters) to the unit of the channel sample timestamps (`tstart`/`tend`, or `timestamp` in the raw format), plus `channel_time_origin="container_start"` if those are relative to the container start, so -the window boundaries share the samples' time base. The columns themselves are not converted. +the window boundaries share the samples' time base. Numeric `start_ts`/`stop_ts` may use a different +epoch unit than the channel samples (e.g. ms boundaries, µs samples); then also set +`solver_config.container_time_unit`. The columns themselves are not converted. ::: diff --git a/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md b/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md index 7874f692..a8db9e9d 100644 --- a/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md +++ b/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md @@ -135,6 +135,12 @@ channel timestamps, and the ``start_ts`` / ``stop_ts`` seen by UDFs, - `channel_time_origin` (`{"epoch", "container_start"}`): Origin of the channel timestamps: absolute epoch (default), or relative to the container's ``start_ts``. Like :attr:`channel_time_unit`, only used for the ``TimeWindowEvent`` windows. +- `container_time_unit` (`{"s", "ms", "us", "ns"} or None`): Unit of **numeric** ``container_metrics`` ``start_ts`` / ``stop_ts``, when it differs +from :attr:`channel_time_unit` (e.g. boundaries in epoch ms, channels in µs). Only +used to convert them into :attr:`channel_time_unit` for the ``TimeWindowEvent`` +windows; requires :attr:`channel_time_unit`. Unset means the numeric boundaries are +already in the channels' unit. Not allowed for ``TIMESTAMP`` boundaries, which carry +their own unit. #### from\_json @@ -517,15 +523,16 @@ time frame as the channel timestamps (:attr:`channel_time_unit`, :attr:`window_stop_col`, derived from the raw ``start_ts`` / ``stop_ts``, which stay unchanged for UDFs, ``ContainerEvent`` and ``measurement_dimension``: -- origin ``"epoch"``: numeric boundaries as they are; ``TIMESTAMP`` boundaries as - epoch numbers in :attr:`channel_time_unit`; -- origin ``"container_start"``: ``0`` and ``stop_ts - start_ts``, for ``TIMESTAMP`` - boundaries in :attr:`channel_time_unit`, for numeric ones as they are (they must - already be in the channels' unit). +- origin ``"epoch"``: ``TIMESTAMP`` boundaries as epoch numbers in + :attr:`channel_time_unit`; numeric boundaries converted from + :attr:`container_time_unit` to :attr:`channel_time_unit` (as they are when unset); +- origin ``"container_start"``: ``0`` and ``stop_ts - start_ts``, converted the same + way (the difference is taken first, in the boundaries' own unit). ``TIMESTAMP`` values are converted via ``unix_micros``, which is exact and independent of the session time zone; ``"s"`` / ``"ms"`` give doubles, ``"us"`` / -``"ns"`` longs. The event fact and the solve both call this, so their windows use +``"ns"`` longs. Numeric boundaries converted to a finer unit are multiplied by an +integer (exact, keeping longs), to a coarser unit divided (doubles). The event fact and the solve both call this, so their windows use the same bounds. The types are checked on the schema, so a missing setting fails before any Spark job runs. @@ -538,9 +545,18 @@ before any Spark job runs. - `ValueError`: If ``start_ts`` / ``stop_ts`` are missing, are ``TIMESTAMP_NTZ`` or ``DATE``, mix ``TIMESTAMP`` and numeric types, or are ``TIMESTAMP`` while -:attr:`channel_time_unit` is unset. +:attr:`channel_time_unit` is unset or :attr:`container_time_unit` is set. **Returns**: `pyspark.sql.DataFrame`: *df* with the two window-bound columns added. +#### validate\_container\_time\_unit\_requires\_channel\_time\_unit + +```python +def validate_container_time_unit_requires_channel_time_unit() +``` + +``container_time_unit`` converts into ``channel_time_unit``, so it needs one. + + diff --git a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md index 9bb40d35..5545ca12 100644 --- a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md +++ b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md @@ -58,7 +58,9 @@ the definition hash. #### set\_channel\_time ```python -def set_channel_time(unit: str | None, origin: str = "epoch") -> None +def set_channel_time(unit: str | None, + origin: str = "epoch", + container_unit: str | None = None) -> None ``` Record the channel time frame the windows are computed in. @@ -71,6 +73,7 @@ aggregations scoped to it. - `unit` (`str or None`): The report's ``solver_config.channel_time_unit``. - `origin` (`str`): The report's ``solver_config.channel_time_origin`` (default ``"epoch"``). +- `container_unit` (`str or None`): The report's ``solver_config.container_time_unit``. #### get\_expression @@ -106,7 +109,7 @@ Calculate definition hash for the time-window event. Only includes the expression string, which encodes the attributes that affect the event results: ``window_length`` and the channel time frame (``channel_time_unit``, -``channel_time_origin``; omitted while unset / default). Resizing the window or +``channel_time_origin``, ``container_time_unit``; omitted while unset / default). Resizing the window or changing the time frame therefore forces a full recompute in incremental mode. Excludes: name, description, required_channels, max_windows_per_container, diff --git a/docs/impulse/docs/references/report/event.md b/docs/impulse/docs/references/report/event.md index c73bcd5b..9fefaa30 100644 --- a/docs/impulse/docs/references/report/event.md +++ b/docs/impulse/docs/references/report/event.md @@ -227,12 +227,12 @@ my_report.add_event(ten_minute_windows) :::note `window_length` follows the same convention as `SequenceOfEvents.max_overlap`: it is expressed in -the same time unit as the stored timestamps (milliseconds-since-epoch in the sample data), not +the same time unit as the channel timestamps (microseconds-since-epoch in the sample data), not seconds or any derived unit. So 60 one-minute windows over millisecond timestamps use `window_length=60_000`. The windows are computed in the time frame of the channel timestamps, set by -[`solver_config.channel_time_unit` and `channel_time_origin`](../../config/configuration.md#solver-column-mappings-and-filters): +[`solver_config.channel_time_unit`, `channel_time_origin` and `container_time_unit`](../../config/configuration.md#solver-column-mappings-and-filters): - If `container_metrics.start_ts`/`stop_ts` are `TIMESTAMP` columns, set `channel_time_unit` to the unit of the channel timestamps (`tstart`/`tend`, or `timestamp` for RAW data; e.g. `"s"`), and @@ -241,6 +241,9 @@ The windows are computed in the time frame of the channel timestamps, set by - If the channel timestamps are relative to the container start (e.g. seconds since the recording started), also set `channel_time_origin="container_start"`. The windows then run from `0` to `stop_ts - start_ts`. +- If numeric `start_ts`/`stop_ts` are in another unit than the channel timestamps (e.g. epoch ms + boundaries, µs samples), set `container_time_unit` to their unit (e.g. `"ms"`) and + `channel_time_unit` to the channels' (e.g. `"us"`). Otherwise no window overlaps the samples. Only the windows use these settings: `ContainerEvent`, `measurement_dimension` and UDFs that read `start_ts`/`stop_ts` keep seeing the original values. The channel time frame is part of the diff --git a/skills/impulse-config/SKILL.md b/skills/impulse-config/SKILL.md index 97df3016..30f06463 100644 --- a/skills/impulse-config/SKILL.md +++ b/skills/impulse-config/SKILL.md @@ -129,8 +129,10 @@ Top-level `channel_time_unit` (`"s"` | `"ms"` | `"us"` | `"ns"`, optional) and ` (`tstart`/`tend`, or `timestamp` with `data_type="RAW"`): absolute epoch numbers, or time relative to the container's `start_ts`. Only `TimeWindowEvent` uses them, to compute its windows in that frame from `container_metrics.start_ts`/`stop_ts` (origin `"container_start"`: from `0` to -`stop_ts - start_ts`). `channel_time_unit` is required when those are `TIMESTAMP` columns. Nothing -is converted in place: the channel timestamps, and the `start_ts`/`stop_ts` seen by +`stop_ts - start_ts`). `channel_time_unit` is required when those are `TIMESTAMP` columns. If they +are numeric but in another unit than the channels (e.g. epoch ms boundaries, µs samples), also set +`container_time_unit` (e.g. `"ms"`; requires `channel_time_unit`, not allowed for `TIMESTAMP`). +Nothing is converted in place: the channel timestamps, and the `start_ts`/`stop_ts` seen by `ContainerEvent`, `measurement_dimension` and UDFs (a `pd.Timestamp`), keep their original values. ```python diff --git a/skills/impulse-events/SKILL.md b/skills/impulse-events/SKILL.md index 39e31adf..0efd32c0 100644 --- a/skills/impulse-events/SKILL.md +++ b/skills/impulse-events/SKILL.md @@ -160,8 +160,10 @@ Because the windows come from `container_metrics`, those boundaries must share t time base for the per-window values to be meaningful. If `container_metrics.start_ts`/`stop_ts` are `TIMESTAMP` columns, set `query_engine.solver_config.channel_time_unit` to the unit of the channel timestamps (`tstart`/`tend`, or `timestamp` for RAW), and `channel_time_origin="container_start"` if -they are relative to the container start (windows then run from `0`). Only the windows use these -settings; `start_ts`/`stop_ts` themselves keep their original values for `ContainerEvent` and UDFs. +they are relative to the container start (windows then run from `0`). Numeric boundaries in another +unit than the channels (e.g. epoch ms vs. µs samples) need `container_time_unit` as well. Only the +windows use these settings; `start_ts`/`stop_ts` keep their original values for `ContainerEvent` and +UDFs. Containers with null, NaN or infinite boundaries get no windows. ## Output schema diff --git a/src/impulse_query_engine/analyze/query/events/time_window_expression.py b/src/impulse_query_engine/analyze/query/events/time_window_expression.py index 3cbf8ad9..54c0b093 100644 --- a/src/impulse_query_engine/analyze/query/events/time_window_expression.py +++ b/src/impulse_query_engine/analyze/query/events/time_window_expression.py @@ -30,8 +30,9 @@ _WINDOW_LIMIT_HINT = ( "Check that window_length is in the unit of the channel timestamps " - "(solver_config.channel_time_unit), or raise the limit (TimeWindowEvent " - "max_windows_per_container, TimeWindowExpression max_windows)." + "(solver_config.channel_time_unit) and, for numeric container boundaries in another " + "unit, that solver_config.container_time_unit is set, or raise the limit " + "(TimeWindowEvent max_windows_per_container, TimeWindowExpression max_windows)." ) @@ -166,6 +167,8 @@ class TimeWindowExpression(TimeSeriesExpression): ``solver_config.channel_time_unit``, set by the reporting ``TimeWindowEvent``. channel_time_origin : str ``solver_config.channel_time_origin`` (default ``"epoch"``), set the same way. + container_time_unit : str or None + ``solver_config.container_time_unit``, set the same way. Both are descriptive only: :meth:`build` does not convert (the solver computes the bounds). They are part of the string form, so the definition hashes of the event and of @@ -206,6 +209,7 @@ def __init__(self, window_length: float, max_windows: int = MAX_WINDOWS_PER_CONT self.max_windows = validate_max_windows(max_windows) self.channel_time_unit: str | None = None self.channel_time_origin: str = "epoch" + self.container_time_unit: str | None = None TimeSeriesExpression.__init__(self, is_single_signal=False) def __str__(self) -> str: @@ -213,9 +217,9 @@ def __str__(self) -> str: Return a string representation of the TimeWindowExpression. The ``window_length`` and the channel time frame are included so they flow into the - definition hashes of the event and of the aggregations scoped to it. An unset - ``channel_time_unit`` and the default ``"epoch"`` origin are omitted, keeping the - default string unchanged. + definition hashes of the event and of the aggregations scoped to it. Unset units + and the default ``"epoch"`` origin are omitted, keeping the default string + unchanged. Returns ------- @@ -227,6 +231,8 @@ def __str__(self) -> str: frame += f", channel_time_unit={self.channel_time_unit}" if self.channel_time_origin != "epoch": frame += f", channel_time_origin={self.channel_time_origin}" + if self.container_time_unit is not None: + frame += f", container_time_unit={self.container_time_unit}" return f"TimeWindowExpression" def dtype(self): diff --git a/src/impulse_query_engine/analyze/query/solvers/solver_config.py b/src/impulse_query_engine/analyze/query/solvers/solver_config.py index c901a8d3..7a419ea2 100644 --- a/src/impulse_query_engine/analyze/query/solvers/solver_config.py +++ b/src/impulse_query_engine/analyze/query/solvers/solver_config.py @@ -20,9 +20,12 @@ import pyspark.sql.functions as F import pyspark.sql.types as T -from pydantic import BaseModel +from pydantic import BaseModel, model_validator from pyspark.sql import Column, DataFrame +# Nanoseconds per time unit, for converting container boundaries into the channel unit. +_NANOS_PER_UNIT = {"s": 10**9, "ms": 10**6, "us": 10**3, "ns": 1} + class RawEncoder(StrEnum): """Encoder used to convert RAW point data into intervals for solving. @@ -148,11 +151,19 @@ class SolverConfig(BaseModel): Origin of the channel timestamps: absolute epoch (default), or relative to the container's ``start_ts``. Like :attr:`channel_time_unit`, only used for the ``TimeWindowEvent`` windows. + container_time_unit : {"s", "ms", "us", "ns"} or None + Unit of **numeric** ``container_metrics`` ``start_ts`` / ``stop_ts``, when it differs + from :attr:`channel_time_unit` (e.g. boundaries in epoch ms, channels in µs). Only + used to convert them into :attr:`channel_time_unit` for the ``TimeWindowEvent`` + windows; requires :attr:`channel_time_unit`. Unset means the numeric boundaries are + already in the channels' unit. Not allowed for ``TIMESTAMP`` boundaries, which carry + their own unit. """ project_id: str | None = None channel_time_unit: Literal["s", "ms", "us", "ns"] | None = None channel_time_origin: Literal["epoch", "container_start"] = "epoch" + container_time_unit: Literal["s", "ms", "us", "ns"] | None = None container_tags: TableConfig = TableConfig() container_metrics: TableConfig = TableConfig() @@ -467,15 +478,16 @@ def with_window_bounds(self, df: DataFrame) -> DataFrame: :attr:`window_stop_col`, derived from the raw ``start_ts`` / ``stop_ts``, which stay unchanged for UDFs, ``ContainerEvent`` and ``measurement_dimension``: - - origin ``"epoch"``: numeric boundaries as they are; ``TIMESTAMP`` boundaries as - epoch numbers in :attr:`channel_time_unit`; - - origin ``"container_start"``: ``0`` and ``stop_ts - start_ts``, for ``TIMESTAMP`` - boundaries in :attr:`channel_time_unit`, for numeric ones as they are (they must - already be in the channels' unit). + - origin ``"epoch"``: ``TIMESTAMP`` boundaries as epoch numbers in + :attr:`channel_time_unit`; numeric boundaries converted from + :attr:`container_time_unit` to :attr:`channel_time_unit` (as they are when unset); + - origin ``"container_start"``: ``0`` and ``stop_ts - start_ts``, converted the same + way (the difference is taken first, in the boundaries' own unit). ``TIMESTAMP`` values are converted via ``unix_micros``, which is exact and independent of the session time zone; ``"s"`` / ``"ms"`` give doubles, ``"us"`` / - ``"ns"`` longs. The event fact and the solve both call this, so their windows use + ``"ns"`` longs. Numeric boundaries converted to a finer unit are multiplied by an + integer (exact, keeping longs), to a coarser unit divided (doubles). The event fact and the solve both call this, so their windows use the same bounds. The types are checked on the schema, so a missing setting fails before any Spark job runs. @@ -495,7 +507,7 @@ def with_window_bounds(self, df: DataFrame) -> DataFrame: ValueError If ``start_ts`` / ``stop_ts`` are missing, are ``TIMESTAMP_NTZ`` or ``DATE``, mix ``TIMESTAMP`` and numeric types, or are ``TIMESTAMP`` while - :attr:`channel_time_unit` is unset. + :attr:`channel_time_unit` is unset or :attr:`container_time_unit` is set. """ types = {field.name: field.dataType for field in self._boundary_fields(df)} missing = [c for c in (self.start_ts_col, self.stop_ts_col) if c not in types] @@ -527,6 +539,12 @@ def with_window_bounds(self, df: DataFrame) -> DataFrame: "channel_time_origin to 'container_start' if they are relative to the " "container start." ) + if timestamps and self.container_time_unit is not None: + raise ValueError( + f"container_time_unit only applies to numeric container_metrics " + f"'{self.start_ts_col}' / '{self.stop_ts_col}', but they are TIMESTAMP " + "columns, which carry their own unit. Remove container_time_unit." + ) start, stop = F.col(self.start_ts_col), F.col(self.stop_ts_col) if self.channel_time_origin == "container_start": @@ -535,17 +553,41 @@ def with_window_bounds(self, df: DataFrame) -> DataFrame: window_stop = ( self._micros_in_unit(F.unix_micros(stop) - F.unix_micros(start)) if timestamps - else stop - start + else self._container_to_channel_unit(stop - start) ) elif timestamps: window_start = self._micros_in_unit(F.unix_micros(start)) window_stop = self._micros_in_unit(F.unix_micros(stop)) else: - window_start, window_stop = start, stop + window_start = self._container_to_channel_unit(start) + window_stop = self._container_to_channel_unit(stop) return df.withColumn(self.window_start_col, window_start).withColumn( self.window_stop_col, window_stop ) + def _container_to_channel_unit(self, col: Column) -> Column: + """Numeric boundary *col* converted from :attr:`container_time_unit` to + :attr:`channel_time_unit` (unchanged when unset or equal).""" + if self.container_time_unit is None or self.container_time_unit == self.channel_time_unit: + return col + source = _NANOS_PER_UNIT[self.container_time_unit] + target = _NANOS_PER_UNIT[self.channel_time_unit] + if source > target: + # Finer target unit: an integer factor keeps long boundaries exact. + return col * F.lit(source // target) + return col / F.lit(float(target // source)) + + @model_validator(mode="after") + def validate_container_time_unit_requires_channel_time_unit(self): + """``container_time_unit`` converts into ``channel_time_unit``, so it needs one.""" + if self.container_time_unit is not None and self.channel_time_unit is None: + raise ValueError( + "container_time_unit requires channel_time_unit: numeric container boundaries " + "are converted from container_time_unit into the unit of the channel " + "timestamps." + ) + return self + def _micros_in_unit(self, micros: Column) -> Column: """Microseconds converted to :attr:`channel_time_unit`.""" if self.channel_time_unit == "s": diff --git a/src/impulse_reporting/core/report.py b/src/impulse_reporting/core/report.py index 64ed3cc5..82955e8a 100644 --- a/src/impulse_reporting/core/report.py +++ b/src/impulse_reporting/core/report.py @@ -419,8 +419,11 @@ def add_event(self, event: Event): if isinstance(event, TimeWindowEvent): # The windows are computed in the channel time frame, so it is part of the # event's (and its scoped aggregations') definition. + solver_config = self.solver.config event.set_channel_time( - self.solver.config.channel_time_unit, self.solver.config.channel_time_origin + solver_config.channel_time_unit, + solver_config.channel_time_origin, + solver_config.container_time_unit, ) def get_events(self) -> list[Event]: diff --git a/src/impulse_reporting/events/time_window_event.py b/src/impulse_reporting/events/time_window_event.py index 180ecf89..854e2826 100644 --- a/src/impulse_reporting/events/time_window_event.py +++ b/src/impulse_reporting/events/time_window_event.py @@ -103,7 +103,9 @@ def __init__( normalized_attributes.setdefault("window_length", str(self.window_length)) self.attributes = normalized_attributes - def set_channel_time(self, unit: str | None, origin: str = "epoch") -> None: + def set_channel_time( + self, unit: str | None, origin: str = "epoch", container_unit: str | None = None + ) -> None: """Record the channel time frame the windows are computed in. Set by ``Report.add_event`` from the report's ``solver_config``. Stored on the @@ -116,9 +118,12 @@ def set_channel_time(self, unit: str | None, origin: str = "epoch") -> None: The report's ``solver_config.channel_time_unit``. origin : str, optional The report's ``solver_config.channel_time_origin`` (default ``"epoch"``). + container_unit : str or None, optional + The report's ``solver_config.container_time_unit``. """ self.expression.channel_time_unit = unit self.expression.channel_time_origin = origin + self.expression.container_time_unit = container_unit def get_expression(self) -> TimeSeriesExpression | None: """ @@ -147,7 +152,7 @@ def determine_definition_hash(self) -> int: Only includes the expression string, which encodes the attributes that affect the event results: ``window_length`` and the channel time frame (``channel_time_unit``, - ``channel_time_origin``; omitted while unset / default). Resizing the window or + ``channel_time_origin``, ``container_time_unit``; omitted while unset / default). Resizing the window or changing the time frame therefore forces a full recompute in incremental mode. Excludes: name, description, required_channels, max_windows_per_container, diff --git a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py index 4ef267f7..f130f251 100644 --- a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py @@ -138,6 +138,8 @@ def test_str_includes_channel_time_frame_only_when_set(): "TimeWindowExpression" ) + expr.container_time_unit = "s" + assert str(expr).endswith(", container_time_unit=s>") def test_max_windows_not_part_of_str(): diff --git a/tests/impulse_query_engine/unit/analyze/query/solvers/container_boundaries_test.py b/tests/impulse_query_engine/unit/analyze/query/solvers/container_boundaries_test.py index 1c5d8b37..7b6e7fbc 100644 --- a/tests/impulse_query_engine/unit/analyze/query/solvers/container_boundaries_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/solvers/container_boundaries_test.py @@ -116,6 +116,69 @@ def test_numeric_boundaries_epoch_as_is_and_container_start_shifted(spark): # n assert _bounds(SolverConfig(channel_time_origin="container_start"), df) == {1: (0, 3600.5)} +def _ms_boundaries_df(spark: SparkSession): # noqa: F811 + """The TIMESTAMP boundaries of _boundaries_df as epoch-ms longs (container 1 only).""" + return ( + _boundaries_df(spark) + .filter(F.col("container_id") == 1) + .select( + "container_id", + F.unix_millis("start_ts").alias("start_ts"), + F.unix_millis("stop_ts").alias("stop_ts"), + ) + ) + + +def test_numeric_ms_boundaries_converted_to_finer_channel_unit_exactly(spark): # noqa: F811 + # Boundaries in epoch ms, channels in µs: an integer factor keeps the longs exact. + cfg = SolverConfig(channel_time_unit="us", container_time_unit="ms") + out = cfg.with_window_bounds(_ms_boundaries_df(spark)) + start_ms = _EPOCH_MICROS // 1000 + stop_ms = (_EPOCH_MICROS + _SPAN_MICROS) // 1000 + assert out.schema[_START].dataType == T.LongType() + assert _bounds(cfg, _ms_boundaries_df(spark)) == {1: (start_ms * 1000, stop_ms * 1000)} + + relative = SolverConfig( + channel_time_unit="us", channel_time_origin="container_start", container_time_unit="ms" + ) + # The difference is taken in ms first, then converted. + assert _bounds(relative, _ms_boundaries_df(spark)) == {1: (0, (stop_ms - start_ms) * 1000)} + + +def test_numeric_boundaries_converted_to_coarser_channel_unit(spark): # noqa: F811 + # Boundaries in epoch µs, channels in ms: a division, giving doubles. + df = spark.createDataFrame( + [(1, _EPOCH_MICROS, _EPOCH_MICROS + _SPAN_MICROS)], + "container_id int, start_ts long, stop_ts long", + ) + cfg = SolverConfig(channel_time_unit="ms", container_time_unit="us") + out = cfg.with_window_bounds(df) + assert out.schema[_START].dataType == T.DoubleType() + assert _bounds(cfg, df) == { + 1: (_EPOCH_MICROS / 1000.0, (_EPOCH_MICROS + _SPAN_MICROS) / 1000.0) + } + + +def test_numeric_boundaries_unchanged_for_equal_or_unset_container_unit(spark): # noqa: F811 + df = _ms_boundaries_df(spark) + raw = {r.container_id: (r.start_ts, r.stop_ts) for r in df.collect()} + assert _bounds(SolverConfig(channel_time_unit="ms", container_time_unit="ms"), df) == raw + assert _bounds(SolverConfig(channel_time_unit="us"), df) == raw + + +def test_container_time_unit_rejected_for_timestamp_boundaries(spark): # noqa: F811 + cfg = SolverConfig(channel_time_unit="s", container_time_unit="ms") + with pytest.raises(ValueError, match="container_time_unit only applies to numeric"): + cfg.with_window_bounds(_boundaries_df(spark)) + + +def test_container_time_unit_requires_channel_time_unit(): + with pytest.raises(ValueError, match="container_time_unit requires channel_time_unit"): + SolverConfig.model_validate({"container_time_unit": "ms"}) + cfg = SolverConfig.model_validate({"channel_time_unit": "us", "container_time_unit": "ms"}) + assert (cfg.channel_time_unit, cfg.container_time_unit) == ("us", "ms") + + def test_raw_boundaries_stay_unchanged(spark): # noqa: F811 df = _boundaries_df(spark) cfg = SolverConfig(channel_time_unit="s", channel_time_origin="container_start") diff --git a/tests/impulse_reporting/integration/time_window_event_test.py b/tests/impulse_reporting/integration/time_window_event_test.py index ff4d6029..65470e39 100644 --- a/tests/impulse_reporting/integration/time_window_event_test.py +++ b/tests/impulse_reporting/integration/time_window_event_test.py @@ -149,6 +149,8 @@ def test_time_window_event_in_report(spark, basic_narrow_db): # channel_time_unit="us", the long path of the conversion). # rel_sec: samples as seconds since the container start (double), container boundaries as # TIMESTAMP (windows from 0 via channel_time_origin="container_start"). +# ms_bounds: native µs samples, container boundaries as epoch-ms longs (windows in µs via +# container_time_unit="ms"); without the conversion no window overlaps a sample. def _to_seconds(c): return c.cast("double") / F.lit(1e6) @@ -161,6 +163,10 @@ def _to_timestamp(c): return F.timestamp_micros(c.cast("long")) +def _to_ms(c): + return F.floor(c.cast("long") / F.lit(1000)).cast("long") + + _RELATIVE_SECONDS = {"channel_time_unit": "s", "channel_time_origin": "container_start"} _TIME_BASES = { @@ -170,6 +176,12 @@ def _to_timestamp(c): "sec_ts": (_to_seconds, _to_timestamp, 600.3, {"channel_time_unit": "s"}), "us_ts": (lambda c: c, _to_timestamp, ALIGNED_WINDOW_LENGTH, {"channel_time_unit": "us"}), "rel_sec": (_to_seconds, _to_timestamp, 600.3, _RELATIVE_SECONDS), + "ms_bounds": ( + lambda c: c, + _to_ms, + ALIGNED_WINDOW_LENGTH, + {"channel_time_unit": "us", "container_time_unit": "ms"}, + ), } @@ -404,13 +416,16 @@ def _is_missing(value) -> bool: @pytest.mark.parametrize( - "setup_tw_aligned_db", ["us", "ns", "sec", "sec_ts", "us_ts", "rel_sec"], indirect=True + "setup_tw_aligned_db", + ["us", "ns", "sec", "sec_ts", "us_ts", "rel_sec", "ms_bounds"], + indirect=True, ) def test_time_window_event_aggregation_join(spark, setup_tw_aligned_db): """Stats scoped to a TimeWindowEvent yield per-window values whose event_instance_id joins to the natively computed event fact, for µs, ns and seconds-as-double time bases, - for TIMESTAMP container boundaries (channel_time_unit), and for channel timestamps - relative to the container start (channel_time_origin="container_start").""" + for TIMESTAMP container boundaries (channel_time_unit), for channel timestamps relative + to the container start (channel_time_origin="container_start"), and for numeric + boundaries in another unit than the channels (container_time_unit).""" schema, window_length, channel_time = setup_tw_aligned_db table_prefix = f"time_window_join_test_{schema.removeprefix(_ALIGNED_SCHEMA + '_')}" my_report = Report( diff --git a/tests/impulse_reporting/unit/events/time_window_event_test.py b/tests/impulse_reporting/unit/events/time_window_event_test.py index d3bc63fe..849b97d1 100644 --- a/tests/impulse_reporting/unit/events/time_window_event_test.py +++ b/tests/impulse_reporting/unit/events/time_window_event_test.py @@ -99,21 +99,23 @@ def test_definition_hash_stable_across_int_and_float_window_length(): def test_definition_hash_changes_with_channel_time_frame(): # The channel time frame decides where the windows lie, so changing the unit or the # origin must force a full recompute. It reaches the hash through the expression string. - def event(unit=None, origin="epoch") -> TimeWindowEvent: + def event(unit=None, origin="epoch", container_unit=None) -> TimeWindowEvent: e = TimeWindowEvent(name="w", window_length=10000) - e.set_channel_time(unit, origin) + e.set_channel_time(unit, origin, container_unit) return e unset = TimeWindowEvent(name="w", window_length=10000) s, ms, ms_relative = event("s"), event("ms"), event("ms", "container_start") + ms_from_s = event("ms", container_unit="s") assert ms_relative.get_expression().channel_time_unit == "ms" assert ms_relative.get_expression().channel_time_origin == "container_start" assert "channel_time_origin=container_start" in ms_relative.as_dict()["event_expression"] # The defaults keep today's hash. assert unset.determine_definition_hash() == event().determine_definition_hash() - hashes = {e.determine_definition_hash() for e in (unset, s, ms, ms_relative)} - assert len(hashes) == 4 + assert ms_from_s.get_expression().container_time_unit == "s" + hashes = {e.determine_definition_hash() for e in (unset, s, ms, ms_relative, ms_from_s)} + assert len(hashes) == 5 # --------------------------------------------------------------------------- From 070281f4908e85887e5ed9abf81875a795e69428 Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 7 Oct 2026 17:17:39 +0200 Subject: [PATCH 18/27] feat(query-engine, reporting): unify TimeWindowEvent window computation in a single pandas UDF-backed function Replace the native-Spark `window_intervals_col` with a scalar pandas UDF (`window_intervals_udf`) that delegates to a shared `tile_windows` implementation. This ensures the reporting event fact and the solve-side `TimeWindowExpression` produce identical windows from the same `container_metrics` bounds, keeping `event_instance_id` values consistent. Update docstrings, API references, and tests to reflect the new UDF and shared tiling logic. --- .../events/time_window_event.md | 6 +- docs/impulse/docs/references/report/event.md | 5 +- .../query/events/time_window_expression.py | 186 +++++++++--------- .../events/time_window_event.py | 28 ++- .../events/time_window_expression_test.py | 174 ++++++++-------- 5 files changed, 191 insertions(+), 208 deletions(-) diff --git a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md index 5545ca12..df8c3f9e 100644 --- a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md +++ b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md @@ -19,7 +19,7 @@ event instance per fixed-duration slice, tiling the container's ``start_ts`` / ` span with windows of length ``window_length``. The final slice is clamped to the container end. -The event fact is computed natively in Spark from ``container_metrics`` (via +The event fact is computed from ``container_metrics`` alone (via #### \_\_init\_\_ @@ -153,8 +153,8 @@ containers' ``start_ts`` / ``stop_ts`` in the channel time frame (``SolverConfig.with_window_bounds``), so every filtered container gets windows. Each window becomes one event instance (``start_ts < end_ts``) whose ``event_instance_id`` hashes its position among the container's windows. The solve -computes the same windows in the same order for scoped aggregations (see -:func:`window_intervals_col`), so the ids match. +uses the same window function for scoped aggregations (see +:func:`window_intervals_udf`), so the ids match. **Arguments**: diff --git a/docs/impulse/docs/references/report/event.md b/docs/impulse/docs/references/report/event.md index 9fefaa30..7645a359 100644 --- a/docs/impulse/docs/references/report/event.md +++ b/docs/impulse/docs/references/report/event.md @@ -269,9 +269,8 @@ containers in incremental mode. :::note The windows are computed from `container_metrics` alone, so **every** container that matches the report's filters gets windows, whether or not it has channel data and whether or not an -aggregation is scoped to the event. An aggregation scoped to the event computes the same windows, -in the same order, in the query engine, so its per-window rows carry the same `event_instance_id` -values. For the per-window values to be meaningful, the container boundaries must share the channel +aggregation is scoped to the event. An aggregation scoped to the event uses the same window +function in the query engine, so its per-window rows carry the same `event_instance_id` values. For the per-window values to be meaningful, the container boundaries must share the channel samples' time base (as they do in real measurement data). ::: diff --git a/src/impulse_query_engine/analyze/query/events/time_window_expression.py b/src/impulse_query_engine/analyze/query/events/time_window_expression.py index 54c0b093..1fa905c4 100644 --- a/src/impulse_query_engine/analyze/query/events/time_window_expression.py +++ b/src/impulse_query_engine/analyze/query/events/time_window_expression.py @@ -4,8 +4,8 @@ import numbers import numpy as np +import pandas as pd import pyspark.sql.functions as F -from pyspark.sql import Column from impulse_query_engine.analyze.metadata.tag_expression import TagExpression from impulse_query_engine.analyze.metadata.time_series_expression import ( @@ -24,8 +24,7 @@ # Default upper bound on the windows per container. A window_length in the wrong unit for the # boundaries (e.g. 60 meant as seconds over ns epochs) would otherwise yield billions of -# windows: Spark's sequence fails with an opaque COLLECTION_SIZE_LIMIT_EXCEEDED and numpy -# allocates arrays of that size. +# windows: numpy would allocate arrays of that size and the event fact explode as many rows. MAX_WINDOWS_PER_CONTAINER = 1_000_000 _WINDOW_LIMIT_HINT = ( @@ -66,39 +65,79 @@ def validate_max_windows(max_windows: int, param_name: str = "max_windows") -> i return int(max_windows) -def _is_finite(col: Column) -> Column: - """True for finite doubles; false for NaN / +-inf; null for null.""" - return ~F.isnan(col) & (F.abs(col) != F.lit(float("inf"))) +def tile_windows( + start, stop, window_length: float, max_windows: int = MAX_WINDOWS_PER_CONTAINER +) -> tuple[np.ndarray, np.ndarray]: + """Tile ``[start, stop]`` into consecutive windows of length *window_length*. + The one window implementation behind ``TimeWindowEvent``: the solve calls it through + :meth:`TimeWindowExpression.build` (scoped aggregations), the event fact through + :func:`window_intervals_udf`. ``event_instance_id`` hashes a window's position, so both + sides must produce the same windows in the same order, which a single function + guarantees as long as both pass in the same values. Both read the same Spark-computed + bounds (``SolverConfig.with_window_bounds``), but pandas hands them over as ``int64`` or + ``float64`` (nulls force ``float64``), or as ``None`` / ``NaN``. The bounds are + therefore converted to ``float`` first: ``int64`` -> ``float64`` rounds to the nearest + double on either path, so the arithmetic below runs on identical doubles. -def window_intervals_col( - start_ts: Column, - stop_ts: Column, - window_length: float, - max_windows: int = MAX_WINDOWS_PER_CONTAINER, -) -> Column: - """Spark counterpart of :meth:`TimeWindowExpression.build`. + Window ``i`` spans ``[start + i * W, min(start + (i + 1) * W, stop)]``, so the last one + is clamped to *stop*; windows with ``start_i >= end_i`` (possible only through rounding) + are dropped. - Computes the same fixed-duration windows natively in Spark, so the reporting - ``TimeWindowEvent`` can materialize windows for every container without a solve. - The ``event_instance_id`` of a window hashes its position in the returned array, so - the windows computed here must match the ones ``build`` computes for scoped - aggregations in **count and order**. Both therefore run the same IEEE-754 operations - in the same order on the same doubles (which also keeps the stored boundaries - identical): cast the boundaries to double *before* subtracting, - ``count = ceil((stop - start) / W)``, ``start_i = start + i * W``, - ``end_i = min(start + (i + 1) * W, stop)``, and drop windows with - ``start_i >= end_i``. Keep the two implementations in sync. + Parameters + ---------- + start, stop : float, int, None + Container bounds in the channel time frame. + window_length : float + Fixed window length, in the same unit as the bounds. Strictly positive. + max_windows : int, optional + Maximum number of windows (default :data:`MAX_WINDOWS_PER_CONTAINER`). + + Returns + ------- + tuple of numpy.ndarray + ``(starts, ends)`` as float64 arrays; empty when a bound is null, NaN or infinite, + or the span is not strictly positive. + + Raises + ------ + ValueError + If the span would produce more than *max_windows* windows. + """ + empty = (np.empty(0), np.empty(0)) + if pd.isna(start) or pd.isna(stop): + return empty + start, stop = float(start), float(stop) + # NaN / infinite bounds (e.g. an unfinished recording) yield no windows, like nulls. + if not (math.isfinite(start) and math.isfinite(stop) and stop > start): + return empty + + # Compared before the int conversion, since an overflowing span gives an infinite count. + window_count = np.ceil((stop - start) / window_length) + if window_count > max_windows: + raise ValueError( + f"TimeWindowExpression: {window_count:.0f} windows of length {window_length} " + f"over a container span of {stop - start} exceed " + f"max_windows={max_windows}. {_WINDOW_LIMIT_HINT}" + ) + indices = np.arange(int(window_count)) + starts = start + indices * window_length + ends = np.minimum(start + (indices + 1) * window_length, stop) + keep = starts < ends + return starts[keep], ends[keep] + + +def window_intervals_udf(window_length: float, max_windows: int = MAX_WINDOWS_PER_CONTAINER): + """Scalar pandas UDF giving each container's windows via :func:`tile_windows`. + + Used by the reporting ``TimeWindowEvent`` for its event fact, on one row per container + (its plan reads only ``container_metrics`` / ``container_tags``, never the channels + table), so the event fact and the solve share one window implementation. Parameters ---------- - start_ts : pyspark.sql.Column - Container start timestamp. - stop_ts : pyspark.sql.Column - Container stop timestamp. window_length : float - Fixed window length, in the same time unit as the timestamps. Must be strictly - positive. + Fixed window length, in the same unit as the bounds. Strictly positive. max_windows : int, optional Maximum number of windows per container (default :data:`MAX_WINDOWS_PER_CONTAINER`). A container exceeding it fails the query with @@ -106,37 +145,24 @@ def window_intervals_col( Returns ------- - pyspark.sql.Column - ``array>`` with one ``[start, end]`` pair per window; empty when a - boundary is null, NaN or infinite, or the span is not strictly positive. + callable + A pandas UDF ``(start, stop) -> array>`` with one ``[start, end]`` pair + per window, in order; empty when a bound is null, NaN or infinite, or the span is not + strictly positive. """ + window_length = float(window_length) max_windows = validate_max_windows(max_windows) - start, stop = start_ts.cast("double"), stop_ts.cast("double") - w = F.lit(float(window_length)) - count = F.ceil((stop - start) / w) - windows = F.transform( - F.sequence(F.lit(0), count - F.lit(1)), - lambda i: F.array(start + i * w, F.least(start + (i + F.lit(1)) * w, stop)), - ) - windows = F.filter(windows, lambda p: p[0] < p[1]) - too_many = F.raise_error( - F.concat( - F.lit("TimeWindowExpression: "), - count.cast("string"), - F.lit(f" windows of length {float(window_length)} over a container span of "), - (stop - start).cast("string"), - F.lit(f" exceed max_windows={max_windows}. {_WINDOW_LIMIT_HINT}"), + + @F.pandas_udf("array>") + def windows(start: pd.Series, stop: pd.Series) -> pd.Series: + return pd.Series( + [ + np.column_stack(tile_windows(s, e, window_length, max_windows)).tolist() + for s, e in zip(start, stop, strict=True) + ] ) - ) - # Gate on finite boundaries and a positive span before anything reaches sequence: - # sequence(0, -1) yields [0, -1] (a descending sequence), not an empty array, and Spark - # orders NaN above every number, so a NaN stop_ts would pass ``stop > start`` alone. - valid = _is_finite(start) & _is_finite(stop) & (stop > start) - return ( - F.when(valid & (count > F.lit(max_windows)), too_many) - .when(valid, windows) - .otherwise(F.array().cast("array>")) - ) + + return windows class TimeWindowExpression(TimeSeriesExpression): @@ -158,8 +184,8 @@ class TimeWindowExpression(TimeSeriesExpression): This is the query-engine counterpart of the reporting ``TimeWindowEvent``. It evaluates to :class:`Intervals`, so it can scope a ``StatsAggregator`` (one statistic per window). - The reporting event fact computes the same windows natively via - :func:`window_intervals_col`; the two must produce the same windows in the same order. + The windows come from :func:`tile_windows`, which the reporting event fact also uses (via + :func:`window_intervals_udf`), so both produce the same windows in the same order. Attributes ---------- @@ -195,8 +221,8 @@ def __init__(self, window_length: float, max_windows: int = MAX_WINDOWS_PER_CONT If ``window_length`` is not strictly positive and finite, or ``max_windows`` is not a positive integer. """ - # inf / NaN must be rejected too: inf gives a zero window count, for which Spark's - # sequence(0, -1) emits a bogus window, and NaN crashes the solve in build(). + # inf / NaN must be rejected too: inf gives a zero window count and NaN an undefined + # one in tile_windows. if window_length is None or not math.isfinite(window_length) or window_length <= 0: raise ValueError( f"TimeWindowExpression requires a strictly positive, finite window_length, " @@ -329,36 +355,12 @@ def build(self, cache: SeriesCache) -> Intervals: ValueError If the container would produce more than ``max_windows`` windows. """ - start_ts = cache.container_metrics.get(_SOLVER_CONFIG.window_start_col) - stop_ts = cache.container_metrics.get(_SOLVER_CONFIG.window_stop_col) - - if start_ts is None or stop_ts is None: - return Intervals.empty() - - # Mirror window_intervals_col exactly: convert to double *before* subtracting. A - # long column reaches pandas as int64 or float64 depending on the group (nulls - # force float64), and an exact int64 span can round differently from the double - # span for large values (e.g. ns epochs), changing the window count. - start_ts, stop_ts = float(start_ts), float(stop_ts) - # Same gate as window_intervals_col: NaN / infinite boundaries (e.g. an unfinished - # recording) yield no windows, like nulls. - if not (math.isfinite(start_ts) and math.isfinite(stop_ts) and stop_ts > start_ts): + starts, ends = tile_windows( + cache.container_metrics.get(_SOLVER_CONFIG.window_start_col), + cache.container_metrics.get(_SOLVER_CONFIG.window_stop_col), + self.window_length, + self.max_windows, + ) + if len(starts) == 0: return Intervals.empty() - - # Number of windows covering the span; the last one is clamped to stop_ts below. - # The span is strictly positive (guarded above) and window_length is strictly - # positive (enforced in __init__), so window_count >= 1. Compared before the int - # conversion, since an overflowing span gives an infinite count. - window_count = np.ceil((stop_ts - start_ts) / self.window_length) - if window_count > self.max_windows: - raise ValueError( - f"TimeWindowExpression: {window_count:.0f} windows of length {self.window_length} " - f"over a container span of {stop_ts - start_ts} exceed " - f"max_windows={self.max_windows}. {_WINDOW_LIMIT_HINT}" - ) - window_count = int(window_count) - - indices = np.arange(window_count) - starts = start_ts + indices * self.window_length - ends = np.minimum(start_ts + (indices + 1) * self.window_length, stop_ts) return Intervals(starts, ends, del_last_empty=True) diff --git a/src/impulse_reporting/events/time_window_event.py b/src/impulse_reporting/events/time_window_event.py index 854e2826..9883b52d 100644 --- a/src/impulse_reporting/events/time_window_event.py +++ b/src/impulse_reporting/events/time_window_event.py @@ -15,7 +15,7 @@ MAX_WINDOWS_PER_CONTAINER, TimeWindowExpression, validate_max_windows, - window_intervals_col, + window_intervals_udf, ) from impulse_query_engine.analyze.query.query_builder import QueryBuilder from impulse_query_engine.analyze.query.solvers.query_solver import QuerySolver @@ -33,12 +33,12 @@ class TimeWindowEvent(ContainerBoundaryEvent): span with windows of length ``window_length``. The final slice is clamped to the container end. - The event fact is computed natively in Spark from ``container_metrics`` (via - :func:`window_intervals_col`), so every filtered container gets windows regardless of + The event fact is computed from ``container_metrics`` alone (via + :func:`window_intervals_udf`), so every filtered container gets windows regardless of its channel data. Aggregations scoped to this event evaluate the - :class:`TimeWindowExpression` in the solve, which computes the same windows in the - same order. ``event_instance_id`` hashes the window's position rather than its - boundaries, so both sides match without relying on bit-identical doubles. + :class:`TimeWindowExpression` in the solve. Both use the same window function + (``tile_windows``), and ``event_instance_id`` hashes the window's position, so the + ids match on both sides. """ def __init__( @@ -210,8 +210,8 @@ def determine_events( (``SolverConfig.with_window_bounds``), so every filtered container gets windows. Each window becomes one event instance (``start_ts < end_ts``) whose ``event_instance_id`` hashes its position among the container's windows. The solve - computes the same windows in the same order for scoped aggregations (see - :func:`window_intervals_col`), so the ids match. + uses the same window function for scoped aggregations (see + :func:`window_intervals_udf`), so the ids match. Parameters ---------- @@ -245,17 +245,15 @@ def determine_events( # One (event_name, windows) struct per event, exploded in a single pass over the # containers. posexplode yields each window's position, which the - # event_instance_id hashes (scoped aggregations use the same position). + # event_instance_id hashes (scoped aggregations use the same position). The windows + # UDF only sees this container-level plan, never the channels table. per_event = f.array( *[ f.struct( f.lit(event.get_name()).alias("event_name"), - window_intervals_col( - start_ts, - stop_ts, - event.window_length, - max_windows=event.max_windows_per_container, - ).alias("windows"), + window_intervals_udf( + event.window_length, max_windows=event.max_windows_per_container + )(start_ts, stop_ts).alias("windows"), ) for event in events ] diff --git a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py index f130f251..1b275744 100644 --- a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py @@ -4,6 +4,7 @@ from unittest.mock import MagicMock import numpy as np +import pandas as pd import pyspark.sql.functions as F import pytest @@ -11,7 +12,8 @@ from impulse_query_engine.analyze.query.events import TimeWindowExpression from impulse_query_engine.analyze.query.events.time_window_expression import ( MAX_WINDOWS_PER_CONTAINER, - window_intervals_col, + tile_windows, + window_intervals_udf, ) from impulse_query_engine.analyze.query.solvers.empty_cache import EmptyTimeSeriesCache from impulse_query_engine.model.series.intervals import Intervals @@ -196,29 +198,34 @@ def test_infinite_container_metrics_yield_empty(): assert len(_build(-np.inf, 100.0, 10)) == 0 +def test_tile_windows_none_and_na_yield_empty(): + for start, stop in ((None, 100), (0, None), (pd.NA, 100), (0, np.nan)): + starts, ends = tile_windows(start, stop, 10) + assert len(starts) == len(ends) == 0 + + # --------------------------------------------------------------------------- -# window_intervals_col: the native-Spark mirror used by the reporting event fact +# window_intervals_udf: tile_windows on the event fact side (one row per container) # --------------------------------------------------------------------------- -def _spark_windows(spark, rows, window_length, ts_type="long"): # noqa: F811 +def _udf_windows(spark, rows, window_length, ts_type="long", **kwargs): # noqa: F811 df = spark.createDataFrame(rows, f"k int, start_ts {ts_type}, stop_ts {ts_type}") - out = df.select( - "k", window_intervals_col(F.col("start_ts"), F.col("stop_ts"), window_length).alias("w") - ) + windows = window_intervals_udf(window_length, **kwargs) + out = df.select("k", windows(F.col("start_ts"), F.col("stop_ts")).alias("w")) return out, {r.k: [list(p) for p in r.w] for r in out.collect()} -def test_window_intervals_col_edge_cases(spark): # noqa: F811 +def test_window_intervals_udf_edge_cases(spark): # noqa: F811 rows = [ (0, 0, 100), # exact multiple -> 10 windows (1, 0, 105), # short final window clamped to stop (2, 1000, 1010), # span == W -> 1 window (3, 1000, 1001), # span < W -> 1 clamped window - (4, 50, 50), # stop == start -> none (sequence(0, -1) is [0, -1], not []) + (4, 50, 50), # stop == start -> none (5, 60, 50), # stop < start -> none (6, None, 50), # null bound -> none (7, 0, None), ] - out, w = _spark_windows(spark, rows, 10) + out, w = _udf_windows(spark, rows, 10) assert out.schema["w"].dataType.simpleString() == "array>" assert w[0] == [[float(s), float(s + 10)] for s in range(0, 100, 10)] @@ -226,36 +233,25 @@ def test_window_intervals_col_edge_cases(spark): # noqa: F811 assert w[2] == [[1000.0, 1010.0]] assert w[3] == [[1000.0, 1001.0]] assert w[4] == w[5] == w[6] == w[7] == [] - assert all(s < e for windows in w.values() for s, e in windows) -def test_window_intervals_col_non_finite_bounds_yield_no_windows(spark): # noqa: F811 - """NaN / infinite boundaries give no windows on both sides. Spark orders NaN above - every number, so a NaN stop_ts used to pass ``stop > start`` and emit [[0, 4], [-4, 0]].""" +def test_window_intervals_udf_non_finite_bounds_yield_no_windows(spark): # noqa: F811 nan, inf = float("nan"), float("inf") rows = [(0, 0.0, nan), (1, nan, 10.0), (2, nan, nan), (3, 0.0, inf), (4, -inf, 10.0)] - _, w = _spark_windows(spark, rows, 4, ts_type="double") + _, w = _udf_windows(spark, rows, 4, ts_type="double") + assert all(w[k] == [] for k, _, _ in rows), w - for k, start, stop in rows: - assert w[k] == [], f"row {k} ({start}, {stop}) produced {w[k]}" - assert _build(start, stop, 4).get_data() == [] - -def test_window_intervals_col_raises_beyond_max_windows(spark): # noqa: F811 - df = spark.createDataFrame([(0, 0, 100)], "k int, start_ts long, stop_ts long") - ok = df.select(window_intervals_col(F.col("start_ts"), F.col("stop_ts"), 10, max_windows=10)) - assert len(ok.collect()[0][0]) == 10 - - too_many = df.select( - window_intervals_col(F.col("start_ts"), F.col("stop_ts"), 10, max_windows=9) - ) +def test_window_intervals_udf_raises_beyond_max_windows(spark): # noqa: F811 + _, ok = _udf_windows(spark, [(0, 0, 100)], 10, max_windows=10) + assert len(ok[0]) == 10 with pytest.raises(Exception, match="10 windows of length 10.0 .* exceed max_windows=9"): - too_many.collect() + _udf_windows(spark, [(0, 0, 100)], 10, max_windows=9) -def test_window_intervals_col_invalid_max_windows_raises(): +def test_window_intervals_udf_invalid_max_windows_raises(): with pytest.raises(ValueError, match="max_windows must be a positive integer"): - window_intervals_col(F.col("start_ts"), F.col("stop_ts"), 10, max_windows=0) + window_intervals_udf(10, max_windows=0) def _as_list(windows) -> list[tuple[float, float]]: @@ -266,8 +262,8 @@ def _as_list(windows) -> list[tuple[float, float]]: def _count_mismatch_case(window_length: float) -> tuple[int, int]: """Find ns-epoch (start, stop) whose int64 and double spans yield different counts. - This is exactly the case where numpy without the float() conversion (exact int64 - subtraction) and Spark (double subtraction) disagree on the number of windows. + This is exactly the case where exact int64 subtraction and double subtraction disagree + on the number of windows, so tile_windows must convert to float first on every path. """ base = 1_700_000_000_000_000_000 for start in range(base, base + 512): @@ -281,76 +277,64 @@ def _count_mismatch_case(window_length: float) -> tuple[int, int]: raise AssertionError("no int64/double count-mismatch case found") -def test_window_intervals_col_bit_identical_to_build(spark): # noqa: F811 - """The event fact (Spark) and scoped aggregations (numpy ``build``) produce the same - windows in the same order: event_instance_id hashes the window's position, and the - stored boundaries must describe the window the statistics were computed over.""" - rnd = random.Random(7) +def _batch_dtype_udf(): + """Pandas UDF reporting the dtype the bounds arrive in, per row of the batch (created + lazily: defining a pandas UDF needs an active Spark session).""" + + @F.pandas_udf("string") + def batch_dtype(start: pd.Series) -> pd.Series: + return pd.Series([str(start.dtype)] * len(start)) + + return batch_dtype - # Long timestamps: ns epochs (~1.7e18, beyond 2^53) and µs epochs, with spans hugging - # multiples of W. Includes W values that are not multiples of the 256 ns double spacing. - long_cases = [] - for w in (1e9, 6e10, 333_333_333.0, 1_000_000_007.0): - for _ in range(60): - start = rnd.randint(1_600_000_000_000_000_000, 1_800_000_000_000_000_000) - stop = start + int(rnd.randint(1, 50) * w) + rnd.randint(-600, 600) - long_cases.append((start, stop, w)) - for w in (1e6, 10_000_000.0): - for _ in range(30): - start = rnd.randint(1_600_000_000_000_000, 1_800_000_000_000_000) - stop = start + int(rnd.randint(1, 50) * w) + rnd.randint(-5, 5) - long_cases.append((start, stop, w)) - edge_start, edge_stop = _count_mismatch_case(1_000_000_007.0) - long_cases.append((edge_start, edge_stop, 1_000_000_007.0)) - - # Seconds as doubles with fractional window lengths. - double_cases = [] - for w in (0.1, 0.25, 0.3, 1.7, 60.0): - for _ in range(60): - start = rnd.uniform(1.6e9, 1.8e9) - stop = start + rnd.randint(1, 50) * w + rnd.uniform(-1e-6, 1e-6) - double_cases.append((start, stop, w)) - - for cases, ts_type, np_types in ( - (long_cases, "long", (np.int64, np.float64)), - (double_cases, "double", (np.float64,)), - ): - lengths = sorted({w for _, _, w in cases}) - df = spark.createDataFrame( - [(k, s, e, w) for k, (s, e, w) in enumerate(cases)], - f"k int, start_ts {ts_type}, stop_ts {ts_type}, w double", + +def test_event_fact_and_solve_windows_identical_across_input_dtypes(spark): # noqa: F811 + """Both sides call tile_windows, but pandas hands the bounds over differently: the event + fact UDF gets int64 for a batch without nulls and float64 once a null is in the batch, + the solve gets float64 (its container metrics are nulled on most rows). For ns epochs + beyond 2^53, including a span where int64 and double subtraction disagree on the count, + all paths must produce the same windows in the same order.""" + rnd = random.Random(7) + window_length = 1_000_000_007.0 + cases = [] + for _ in range(60): + start = rnd.randint(1_600_000_000_000_000_000, 1_800_000_000_000_000_000) + cases.append( + (start, start + int(rnd.randint(1, 50) * window_length) + rnd.randint(-600, 600)) ) - # A single CASE WHEN column computes each row's windows for its own length only, - # so all cases run in one Spark job. - windows_col = None - for w in lengths: - branch = window_intervals_col(F.col("start_ts"), F.col("stop_ts"), w) - condition = F.col("w") == F.lit(w) - windows_col = ( - F.when(condition, branch) - if windows_col is None - else windows_col.when(condition, branch) - ) - spark_windows = { - r.k: r.windows for r in df.select("k", windows_col.alias("windows")).collect() - } - - mismatches = [] - for k, (start, stop, w) in enumerate(cases): - expected = _as_list(spark_windows[k]) - assert expected, f"case {k} produced no windows" - for np_type in np_types: - built = _as_list(_build(np_type(start), np_type(stop), w).get_data()) - if built != expected: - mismatches.append((k, np_type.__name__, start, stop, w)) - assert not mismatches, f"Spark/numpy window mismatch: {mismatches[:5]}" + cases.append(_count_mismatch_case(window_length)) + rows = [(k, s, e) for k, (s, e) in enumerate(cases)] + windows = window_intervals_udf(window_length) + batch_dtype = _batch_dtype_udf() + + def _event_fact(batch_rows) -> tuple[dict, set]: + # One partition, so one Arrow batch: a null anywhere turns the whole batch float64. + df = spark.createDataFrame(batch_rows, "k int, start_ts long, stop_ts long").coalesce(1) + out = df.select( + "k", + windows(F.col("start_ts"), F.col("stop_ts")).alias("w"), + batch_dtype(F.col("start_ts")).alias("dtype"), + ).collect() + return {r.k: _as_list(r.w) for r in out if r.k >= 0}, {r.dtype for r in out} + + without_nulls, dtypes_int = _event_fact(rows) + with_null, dtypes_float = _event_fact([*rows, (-1, None, None)]) + assert dtypes_int == {"int64"} and dtypes_float == {"float64"} + + mismatches = [] + for k, (start, stop) in enumerate(cases): + solve = _as_list(_build(np.float64(start), np.float64(stop), window_length).get_data()) + assert solve, f"case {k} produced no windows" + if not (without_nulls[k] == with_null[k] == solve): + mismatches.append((k, start, stop)) + assert not mismatches, f"event fact / solve window mismatch: {mismatches[:5]}" def test_stats_aggregator_windows_equal_helper_windows(spark): # noqa: F811 """A StatsAggregator scoped to a TimeWindowExpression emits exactly the helper's windows (no merging of touching windows, no extra drops).""" start, stop, w = 1_700_000_000_000_000_123, 1_700_000_007_000_000_049, 1_000_000_007.0 - _, spark_windows = _spark_windows(spark, [(0, start, stop)], w) + expected = list(zip(*tile_windows(start, stop, w), strict=True)) # One channel sampled across the whole container, so every window has data. ts = np.linspace(float(start), float(stop), 50) @@ -367,7 +351,7 @@ def test_stats_aggregator_windows_equal_helper_windows(spark): # noqa: F811 cache = _FakeCache({_WINDOW_START: np.float64(start), _WINDOW_STOP: np.float64(stop)}) event_timestamps, numeric_values, _, _ = agg.build(cache) - assert len(spark_windows[0]) == 7 + assert len(expected) == 7 # Same windows in the same order: event_timestamps' position is the window index. - assert _as_list(event_timestamps) == _as_list(spark_windows[0]) + assert _as_list(event_timestamps) == _as_list(expected) assert len(numeric_values[0]) == len(event_timestamps) From 7b5442b41b1053a2cdeea8dc133fcc4c2accb7c0 Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 7 Oct 2026 17:28:08 +0200 Subject: [PATCH 19/27] feat(reporting, query-engine): hash TimeWindowEvent event_instance_id by window boundaries Revert `TimeWindowEvent` `event_instance_id` generation from the window-index hash back to the standard interval-event hash (`container_id::event_name::start_ts::end_ts`). Since the reporting event fact and the solve-side `TimeWindowExpression` now share the same `tile_windows` implementation and read the same Spark-computed window-bound columns, they produce identical windows and matching ids without relying on positional indexing. - Remove the `TimeWindowEvent` special case from `generate_event_instance_id_column` and drop the `window_index_col` parameter. - Stop emitting `interval_index` in `StatsAggregator` and remove the `posexplode`/`window_index` column from `TimeWindowEvent.determine_events`. - Update docstrings, API references, skills, and tests to describe boundary-based ids and remove references to window position/index hashing. - Strengthen `TimeWindowEvent` unit and integration tests to verify stats values join to the correct windows under the new id scheme. --- .../events/time_window_event.md | 5 +- docs/impulse/docs/references/report/event.md | 8 +- skills/impulse-events/SKILL.md | 2 +- .../query/events/time_window_expression.py | 6 +- .../aggregations/stats_aggregator.py | 43 +++----- .../events/time_window_event.py | 16 ++- .../util/event_instance_util.py | 27 +---- .../events/time_window_expression_test.py | 4 +- .../integration/time_window_event_test.py | 28 ++--- .../aggregations/stats_aggregator_test.py | 101 ++++++++++++++---- 10 files changed, 124 insertions(+), 116 deletions(-) diff --git a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md index df8c3f9e..1128266d 100644 --- a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md +++ b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md @@ -152,9 +152,8 @@ Resolves the matching containers via the solver's filter pipeline (like containers' ``start_ts`` / ``stop_ts`` in the channel time frame (``SolverConfig.with_window_bounds``), so every filtered container gets windows. Each window becomes one event instance (``start_ts < end_ts``) whose -``event_instance_id`` hashes its position among the container's windows. The solve -uses the same window function for scoped aggregations (see -:func:`window_intervals_udf`), so the ids match. +``event_instance_id`` hashes its boundaries. The solve uses the same window function +for scoped aggregations (see :func:`window_intervals_udf`), so the ids match. **Arguments**: diff --git a/docs/impulse/docs/references/report/event.md b/docs/impulse/docs/references/report/event.md index 7645a359..7e60e1e2 100644 --- a/docs/impulse/docs/references/report/event.md +++ b/docs/impulse/docs/references/report/event.md @@ -261,8 +261,8 @@ containers in incremental mode. zero-length trailing slice is dropped (every instance satisfies `start_ts < end_ts`). Containers whose `start_ts` or `stop_ts` is null, NaN or infinite get no windows. 3. Each window becomes one **event instance**, written to the shared `event_instance_fact` table. - Its `event_instance_id` hashes the container, the event name and the window's position in - the container (0, 1, 2, ...). + Its `event_instance_id` hashes the container, the event name and the window's start and end, + like for other interval events. 4. An aggregation scoped to the event (`StatsAggregator(..., event=time_window_event)`) computes its statistic **once per window** and joins back to those instances. @@ -277,8 +277,8 @@ samples' time base (as they do in real measurement data). :::note Window boundaries are stored as doubles (`start_ts` / `end_ts`), like every other event type. Epoch timestamps in nanoseconds exceed the range doubles represent exactly, so their window boundaries -are rounded to about 256 ns. The `event_instance_id` depends on the window's position, not on its -boundaries, so the rounding does not affect how aggregations join to the windows. +are rounded to about 256 ns. The event and its aggregations use the same window function, so the +rounding is the same on both sides and their `event_instance_id` values still match. ::: ## Event output schema diff --git a/skills/impulse-events/SKILL.md b/skills/impulse-events/SKILL.md index 0efd32c0..dd733d64 100644 --- a/skills/impulse-events/SKILL.md +++ b/skills/impulse-events/SKILL.md @@ -155,7 +155,7 @@ report.add_event(ten_minute) Windows are computed from `container_metrics` for every container matching the report's filters, with or without channel data or a scoped aggregation. Pair it with an aggregation scoped to the event (e.g. `StatsAggregator(..., event=...)`) to compute one statistic per window; those rows carry the same -`event_instance_id` values as the windows (the id hashes container, event name and window position). +`event_instance_id` values as the windows (both sides use the same window function). Because the windows come from `container_metrics`, those boundaries must share the channel samples' time base for the per-window values to be meaningful. If `container_metrics.start_ts`/`stop_ts` are `TIMESTAMP` columns, set `query_engine.solver_config.channel_time_unit` to the unit of the channel diff --git a/src/impulse_query_engine/analyze/query/events/time_window_expression.py b/src/impulse_query_engine/analyze/query/events/time_window_expression.py index 1fa905c4..aa61d394 100644 --- a/src/impulse_query_engine/analyze/query/events/time_window_expression.py +++ b/src/impulse_query_engine/analyze/query/events/time_window_expression.py @@ -72,9 +72,9 @@ def tile_windows( The one window implementation behind ``TimeWindowEvent``: the solve calls it through :meth:`TimeWindowExpression.build` (scoped aggregations), the event fact through - :func:`window_intervals_udf`. ``event_instance_id`` hashes a window's position, so both - sides must produce the same windows in the same order, which a single function - guarantees as long as both pass in the same values. Both read the same Spark-computed + :func:`window_intervals_udf`. ``event_instance_id`` hashes each window's boundaries, so + both sides must produce identical windows, which a single function guarantees as long as + both pass in the same values. Both read the same Spark-computed bounds (``SolverConfig.with_window_bounds``), but pandas hands them over as ``int64`` or ``float64`` (nulls force ``float64``), or as ``None`` / ``NaN``. The bounds are therefore converted to ``float`` first: ``int64`` -> ``float64`` rounds to the nearest diff --git a/src/impulse_reporting/aggregations/stats_aggregator.py b/src/impulse_reporting/aggregations/stats_aggregator.py index 31c5df36..e5a6eb7f 100644 --- a/src/impulse_reporting/aggregations/stats_aggregator.py +++ b/src/impulse_reporting/aggregations/stats_aggregator.py @@ -481,9 +481,7 @@ def _explode_stats_values(df: DataFrame) -> DataFrame: Returns ------- pyspark.sql.DataFrame - DataFrame with exploded statistics for each signal and interval, carrying the - interval's position as ``interval_index`` (the window index of a - ``TimeWindowEvent``). + DataFrame with exploded statistics for each signal and interval. """ # Step 1: Explode by signal index to get one row per signal. # @@ -528,7 +526,6 @@ def _explode_stats_values(df: DataFrame) -> DataFrame: "event_id", "event_name", "signal_index", - "interval_index", f.col("zipped.event_timestamps").getItem(0).alias("start_ts"), f.col("zipped.event_timestamps").getItem(1).alias("end_ts"), f.col("zipped.signal_stats_per_interval").alias("statistics"), @@ -541,7 +538,6 @@ def _explode_stats_values(df: DataFrame) -> DataFrame: "event_name", "event_id", "signal_index", - "interval_index", "start_ts", "end_ts", f.explode(f.col("statistics")).alias("aggregation_label", "statistic_value"), @@ -667,9 +663,8 @@ def _add_event_instance_id_column( Add an event_instance_id column, matching ``event_instance_fact``. The id comes from ``generate_event_instance_id_column``: a ``ContainerEvent`` - gets ``xxhash64(container_id)`` (one id per container), a ``TimeWindowEvent`` the - window-index hash over ``interval_index``, all other event types the - timestamp-based hash. The event-type cases are applied per row (keyed on + gets ``xxhash64(container_id)`` (one id per container), all other event types get + the timestamp-based hash. The container-event case is applied per row (keyed on ``stats_name``) since a frame may mix event types. Parameters @@ -683,34 +678,22 @@ def _add_event_instance_id_column( Function that adds the event_instance_id column to a DataFrame. """ from impulse_reporting.events.container_event import ContainerEvent - from impulse_reporting.events.time_window_event import TimeWindowEvent def _(df: DataFrame) -> DataFrame: - def stats_names_scoped_to(event_cls: type) -> list[str]: - return [ - agg.get_name() - for agg in aggregations - if agg and isinstance(agg.get_event(), event_cls) - ] - - container_event_stats_names = stats_names_scoped_to(ContainerEvent) - time_window_stats_names = stats_names_scoped_to(TimeWindowEvent) - - event_instance_id_column = generate_event_instance_id_column() - # Only reference interval_index when a TimeWindowEvent is in play: frames of - # other aggregation types (e.g. PointValueAggregator) do not carry it. - if time_window_stats_names: - event_instance_id_column = f.when( - f.col("stats_name").isin(time_window_stats_names), - generate_event_instance_id_column( - event_type=TimeWindowEvent, window_index_col="interval_index" - ), - ).otherwise(event_instance_id_column) + container_event_stats_names = [ + agg.get_name() + for agg in aggregations + if agg and isinstance(agg.get_event(), ContainerEvent) + ] + + timestamp_based_id = generate_event_instance_id_column() if container_event_stats_names: event_instance_id_column = f.when( f.col("stats_name").isin(container_event_stats_names), generate_event_instance_id_column(event_type=ContainerEvent), - ).otherwise(event_instance_id_column) + ).otherwise(timestamp_based_id) + else: + event_instance_id_column = timestamp_based_id return df.withColumn("event_instance_id", event_instance_id_column) diff --git a/src/impulse_reporting/events/time_window_event.py b/src/impulse_reporting/events/time_window_event.py index 9883b52d..029f781b 100644 --- a/src/impulse_reporting/events/time_window_event.py +++ b/src/impulse_reporting/events/time_window_event.py @@ -37,8 +37,8 @@ class TimeWindowEvent(ContainerBoundaryEvent): :func:`window_intervals_udf`), so every filtered container gets windows regardless of its channel data. Aggregations scoped to this event evaluate the :class:`TimeWindowExpression` in the solve. Both use the same window function - (``tile_windows``), and ``event_instance_id`` hashes the window's position, so the - ids match on both sides. + (``tile_windows``), so they produce identical windows, and the timestamp-based + ``event_instance_id`` (like for other interval events) matches on both sides. """ def __init__( @@ -209,9 +209,8 @@ def determine_events( containers' ``start_ts`` / ``stop_ts`` in the channel time frame (``SolverConfig.with_window_bounds``), so every filtered container gets windows. Each window becomes one event instance (``start_ts < end_ts``) whose - ``event_instance_id`` hashes its position among the container's windows. The solve - uses the same window function for scoped aggregations (see - :func:`window_intervals_udf`), so the ids match. + ``event_instance_id`` hashes its boundaries. The solve uses the same window function + for scoped aggregations (see :func:`window_intervals_udf`), so the ids match. Parameters ---------- @@ -244,9 +243,8 @@ def determine_events( stop_ts = f.col(solver.config.window_stop_col) # One (event_name, windows) struct per event, exploded in a single pass over the - # containers. posexplode yields each window's position, which the - # event_instance_id hashes (scoped aggregations use the same position). The windows - # UDF only sees this container-level plan, never the channels table. + # containers. The windows UDF only sees this container-level plan, never the + # channels table. per_event = f.array( *[ f.struct( @@ -267,7 +265,7 @@ def determine_events( .select( "container_id", f.col("event.event_name").alias("event_name"), - f.posexplode(f.col("event.windows")).alias("window_index", "event_instance"), + f.explode(f.col("event.windows")).alias("event_instance"), ) .withColumn("start_ts", f.col("event_instance").getItem(0)) .withColumn("end_ts", f.col("event_instance").getItem(1)) diff --git a/src/impulse_reporting/util/event_instance_util.py b/src/impulse_reporting/util/event_instance_util.py index f104b294..64763986 100644 --- a/src/impulse_reporting/util/event_instance_util.py +++ b/src/impulse_reporting/util/event_instance_util.py @@ -17,26 +17,21 @@ def generate_event_instance_id_column( event_name_col: str = "event_name", start_ts_col: str = "start_ts", end_ts_col: str = "end_ts", - window_index_col: str = "window_index", ) -> Column: """ Generate an event_instance_id column. The id is an xxHash64 of ``container_id::event_name::start_ts::end_ts``. For ``ContainerEvent`` only ``container_id`` is hashed, since a container - event produces exactly one instance per container. For ``TimeWindowEvent`` - the window's position replaces the timestamps - (``container_id::event_name::window_index``), so the event fact and the - aggregations scoped to it agree without bit-identical window boundaries. - The result is a signed 64-bit long (may be negative), wide enough to keep - this merge/join key collision-free at scale. + event produces exactly one instance per container. The result is a signed + 64-bit long (may be negative), wide enough to keep this merge/join key + collision-free at scale. Parameters ---------- event_type : type[Event] or None, optional The event class. When the class is ``ContainerEvent``, the - ``container_id`` column is hashed; for ``TimeWindowEvent`` the - window-index hash is returned. For any other value (including + ``container_id`` column is hashed. For any other value (including ``None`` for backward-compatibility) the timestamp-based hash column is returned. container_id_col : str, optional @@ -47,9 +42,6 @@ def generate_event_instance_id_column( Name of the start timestamp column, defaults to "start_ts". end_ts_col : str, optional Name of the end timestamp column, defaults to "end_ts". - window_index_col : str, optional - Name of the window position column (``TimeWindowEvent`` only), defaults - to "window_index". Returns ------- @@ -57,21 +49,10 @@ def generate_event_instance_id_column( A column expression for the event_instance_id. """ from impulse_reporting.events.container_event import ContainerEvent - from impulse_reporting.events.time_window_event import TimeWindowEvent if event_type is ContainerEvent: return f.xxhash64(f.col(container_id_col).cast("string")) - if event_type is TimeWindowEvent: - return f.xxhash64( - f.concat_ws( - "::", - f.col(container_id_col), - f.col(event_name_col), - f.col(window_index_col), - ) - ) - return f.xxhash64( f.concat_ws( "::", diff --git a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py index 1b275744..b7b27139 100644 --- a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py @@ -255,7 +255,7 @@ def test_window_intervals_udf_invalid_max_windows_raises(): def _as_list(windows) -> list[tuple[float, float]]: - """Windows as ordered (start, end) pairs: the order is the window index the ids hash.""" + """Windows as ordered (start, end) pairs.""" return [(float(s), float(e)) for s, e in windows] @@ -352,6 +352,6 @@ def test_stats_aggregator_windows_equal_helper_windows(spark): # noqa: F811 event_timestamps, numeric_values, _, _ = agg.build(cache) assert len(expected) == 7 - # Same windows in the same order: event_timestamps' position is the window index. + # The same windows, in the same order. assert _as_list(event_timestamps) == _as_list(expected) assert len(numeric_values[0]) == len(event_timestamps) diff --git a/tests/impulse_reporting/integration/time_window_event_test.py b/tests/impulse_reporting/integration/time_window_event_test.py index 65470e39..cee850e5 100644 --- a/tests/impulse_reporting/integration/time_window_event_test.py +++ b/tests/impulse_reporting/integration/time_window_event_test.py @@ -81,7 +81,10 @@ def _expected_window_count(container_id: int) -> int: def test_time_window_event_in_report(spark, basic_narrow_db): - """A TimeWindowEvent (alongside a scoped aggregation) tiles each container into windows.""" + """A TimeWindowEvent registered on a report tiles each container exactly into windows, + with one id per window, and writes its event_dimension row. Per-window statistics are + covered by test_time_window_event_aggregation_join (the basic fixture's container + boundaries don't overlap its samples, so they would all be NaN here).""" my_report = Report( name="time_window_event_report", spark=spark, @@ -94,20 +97,6 @@ def test_time_window_event_in_report(spark, basic_narrow_db): ) my_report.add_event(window_evt) - query = my_report.get_db().query - page = Page(page_number=1) - my_report.add_page(page) - page.add_aggregation( - StatsAggregator( - name="rpm_stats_per_window", - input_expressions=[query.channel(channel_name="Engine RPM")], - channel_names=["Engine RPM"], - statistics=["min", "max", "mean"], - event=window_evt, - desc="Engine RPM stats per window", - ) - ) - my_report.determine_report() event_dfs = my_report.event_dfs @@ -360,9 +349,8 @@ def _assert_window_stats_match_samples( and windows without RPM samples carry no value (RPM only covers each container's first minute, so most windows are empty). - The ids hash the window's position, so a position that drifted between the event fact - and the solve would attach a neighbouring window's values; this pins every stats row to - the window whose boundaries event_instance_fact stores. + This pins every stats row to the window whose boundaries event_instance_fact stores, so + windows that differed between the event fact and the solve would show up here. """ rpm_channels = ( spark.read.table(f"{schema}.channel_metrics") @@ -544,8 +532,8 @@ def _assert_windows_tile_containers(rows, boundaries: dict, window_length: float def test_multiple_time_window_events_coexist(spark, setup_tw_aligned_db): """Two TimeWindowEvents with different window lengths coexist in one report: each tiles every container with its own windows, and the statistics scoped to each event carry - the values of that event's windows. The ids hash the event name, so window k of one - event never joins window k of the other.""" + the values of that event's windows. The ids include the event name, so a window of one + event never joins the other event's windows, even where their boundaries coincide.""" schema, window_length, _ = setup_tw_aligned_db table_prefix = "time_window_multi_test" my_report = Report( diff --git a/tests/impulse_reporting/unit/aggregations/stats_aggregator_test.py b/tests/impulse_reporting/unit/aggregations/stats_aggregator_test.py index d36d0e77..a12db25e 100644 --- a/tests/impulse_reporting/unit/aggregations/stats_aggregator_test.py +++ b/tests/impulse_reporting/unit/aggregations/stats_aggregator_test.py @@ -4,6 +4,8 @@ Tests follow the same pattern as histogram_test.py. """ +import math + import pyspark.sql.functions as f import pyspark.sql.types as T import pytest @@ -14,6 +16,7 @@ PerChannelStatistic, ) from impulse_query_engine.analyze.query.solvers.default_solver import DefaultSolver +from impulse_query_engine.measurement_db import MeasurementDB, MeasurementDBConfig from impulse_reporting.aggregations.stats_aggregator import StatsAggregator from impulse_reporting.events.basic_event import BasicEvent from impulse_reporting.events.container_event import ContainerEvent @@ -531,13 +534,32 @@ def test_determine_aggregations_container_event_instance_id(spark, basic_narrow_ assert all(row.event_instance_id != row.expected_container_id for row in basic_rows) -def test_determine_aggregations_time_window_event_instance_id(spark, basic_narrow_db): - """Time-window stats rows use the window-index id that ``TimeWindowEvent.determine_events`` - writes to ``event_instance_fact``: one id per window, all of them materialized. Basic-event - stats in the same frame keep the timestamp-based id.""" - eng_rpm = basic_narrow_db.query.channel(channel_name="Engine RPM") +def _aligned_boundaries_db(basic_narrow_db: MeasurementDB) -> MeasurementDB: + """Clone of basic_narrow_db whose container_metrics start_ts / stop_ts span each + container's channel samples (µs), so time windows actually overlap the data. In the + original, the boundaries (2025, epoch ms) and the samples (2017, epoch µs) never meet.""" + tables = dict(basic_narrow_db.config.debug_tables) + bounds = ( + tables["channels"] + .groupBy("container_id") + .agg(f.min("tstart").alias("start_ts"), f.max("tend").alias("stop_ts")) + ) + tables["container_metrics"] = ( + tables["container_metrics"].drop("start_ts", "stop_ts").join(bounds, "container_id") + ) + return MeasurementDB(MeasurementDBConfig.for_debug(tables), ws=basic_narrow_db.ws) - window_event = TimeWindowEvent(name="ten_s", window_length=10_000) + +def test_determine_aggregations_time_window_event_instance_id(spark, basic_narrow_db): + """Time-window stats rows carry the timestamp-based id that + ``TimeWindowEvent.determine_events`` writes to ``event_instance_fact`` (one id per window, + all of them materialized) and the statistics of their own window. Basic-event stats in the + same frame keep their own ids.""" + db = _aligned_boundaries_db(basic_narrow_db) + window_length = 600_000_000 # 10 min in µs + eng_rpm = db.query.channel(channel_name="Engine RPM") + + window_event = TimeWindowEvent(name="ten_min", window_length=window_length) basic_event = BasicEvent(name="rpm_event", expr=eng_rpm > 500) window_stats = StatsAggregator( name="window_stats", @@ -555,30 +577,67 @@ def test_determine_aggregations_time_window_event_instance_id(spark, basic_narro ) solver = DefaultSolver(spark) - solved_df = basic_narrow_db.query.select( - window_stats.get_expression(), basic_stats.get_expression() - ).solve(spark, solver) + solved_df = db.query.select(window_stats.get_expression(), basic_stats.get_expression()).solve( + spark, solver + ) df = StatsAggregator.determine_aggregations( spark=spark, aggregations=[window_stats, basic_stats], solved_df=solved_df ) windows = TimeWindowEvent.determine_events( - spark, [window_event], query=basic_narrow_db.query, solver=solver + spark, [window_event], query=db.query, solver=solver ) - window_ids = {r.event_instance_id for r in windows.collect()} + window_rows = windows.collect() + window_ids = {r.event_instance_id for r in window_rows} window_count = { - r.container_id: r.n - for r in windows.groupBy("container_id").count().withColumnRenamed("count", "n").collect() + cid: sum(1 for r in window_rows if r.container_id == cid) + for cid in {r.container_id for r in window_rows} } - window_rows = df.filter(f.col("visual_id") == window_stats.get_id()) - assert window_rows.count() > 0 - assert {r.event_instance_id for r in window_rows.collect()} <= window_ids - # Every window of every solved container gets its own id (no collapsed indices). - per_container = window_rows.groupBy("container_id").agg( - f.countDistinct("event_instance_id").alias("n") + stats_rows = df.filter(f.col("visual_id") == window_stats.get_id()).collect() + assert stats_rows + assert {r.event_instance_id for r in stats_rows} <= window_ids + # Every window of every solved container gets its own id. + for cid, n in window_count.items(): + assert len({r.event_instance_id for r in stats_rows if r.container_id == cid}) == n + + # Real values. The RPM channel only covers each container's first minute, so only the + # first window holds samples: its min / max are those of the RPM samples, all other + # windows carry no value. + rpm_ids = ( + db.channel_metrics(spark) + .filter(f.col("channel_name") == "Engine RPM") + .select("container_id", "channel_id") ) - for row in per_container.collect(): - assert row.n == window_count[row.container_id], row + rpm = { + r.container_id: r + for r in db.channels(spark) + .join(rpm_ids, ["container_id", "channel_id"]) + .groupBy("container_id") + .agg( + f.min(f.col("value").cast("double")).alias("min"), + f.max(f.col("value").cast("double")).alias("max"), + f.max("tend").alias("last_tend"), + ) + .collect() + } + first_window = {} + for r in window_rows: + if r.container_id not in first_window or r.start_ts < first_window[r.container_id][0]: + first_window[r.container_id] = (r.start_ts, r.event_instance_id) + for cid, (start, first_id) in first_window.items(): + assert rpm[cid].last_tend - start < window_length, "fixture: RPM beyond window 0" + values = { + r.aggregation_label: r.statistic_value + for r in stats_rows + if r.event_instance_id == first_id + } + assert values == {"min": rpm[cid].min, "max": rpm[cid].max}, (cid, values) + later = [ + r for r in stats_rows if r.event_instance_id not in {i for _, i in first_window.values()} + ] + # No samples: the statistic is null (or NaN). + assert later + assert all(r.statistic_value is None or math.isnan(r.statistic_value) for r in later) basic_rows = df.filter(f.col("visual_id") == basic_stats.get_id()).collect() assert len(basic_rows) > 0 From b4ee4b7689b5dfa281c35f7aa7e27646db6d9d71 Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 7 Oct 2026 17:46:11 +0200 Subject: [PATCH 20/27] refactor(query-engine, reporting): extract with_window_bounds from SolverConfig to a standalone utility Move `SolverConfig.with_window_bounds` into a new `solvers.utils.window_bounds` module as a standalone function that takes a `SolverConfig` argument. This decouples the window-bound computation from the configuration model and makes it easier to share between the query engine solve path and the reporting event fact. Update all call sites, docstrings, API references, and tests to reference the new location. --- .../analyze/query/solvers/solver_config.md | 58 +------ .../events/time_window_event.md | 8 +- .../query/events/time_window_expression.py | 21 +-- .../analyze/query/solvers/default_solver.py | 3 +- .../analyze/query/solvers/solver_config.py | 145 +---------------- .../query/solvers/utils/window_bounds.py | 149 ++++++++++++++++++ .../events/time_window_event.py | 11 +- .../window_bounds_test.py} | 23 +-- 8 files changed, 202 insertions(+), 216 deletions(-) create mode 100644 src/impulse_query_engine/analyze/query/solvers/utils/window_bounds.py rename tests/impulse_query_engine/unit/analyze/query/solvers/{container_boundaries_test.py => utils/window_bounds_test.py} (91%) diff --git a/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md b/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md index a8db9e9d..adb1aa08 100644 --- a/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md +++ b/docs/impulse/docs/references/api/impulse_query_engine/analyze/query/solvers/solver_config.md @@ -128,10 +128,10 @@ override for the channel mapping (alias) table. - `unit_conversion` (`TableConfig`): Column mappings and filters for the unit conversion table. - `channel_time_unit` (`{"s", "ms", "us", "ns"} or None`): Time unit of the timestamps in the ``channels`` table (``tstart`` / ``tend``, or ``timestamp`` for RAW data). Only used to compute ``TimeWindowEvent`` windows in -that unit (see :meth:`with_window_bounds`); required when ``container_metrics`` -``start_ts`` / ``stop_ts`` are ``TIMESTAMP`` columns. Nothing else is converted: -channel timestamps, and the ``start_ts`` / ``stop_ts`` seen by UDFs, -``ContainerEvent`` and ``measurement_dimension``, keep their original values. +that unit (see ``solvers.utils.window_bounds.with_window_bounds``); required when +``container_metrics`` ``start_ts`` / ``stop_ts`` are ``TIMESTAMP`` columns. Nothing +else is converted: channel timestamps, and the ``start_ts`` / ``stop_ts`` seen by +UDFs, ``ContainerEvent`` and ``measurement_dimension``, keep their original values. - `channel_time_origin` (`{"epoch", "container_start"}`): Origin of the channel timestamps: absolute epoch (default), or relative to the container's ``start_ts``. Like :attr:`channel_time_unit`, only used for the ``TimeWindowEvent`` windows. @@ -238,8 +238,8 @@ def window_start_col() -> str Internal column name for the container start in the channel time frame. -Added by :meth:`with_window_bounds`; prefixed so it cannot clash with a customer -column. +Added by ``solvers.utils.window_bounds.with_window_bounds``; prefixed so it cannot +clash with a customer column. #### window\_stop\_col @@ -250,8 +250,8 @@ def window_stop_col() -> str Internal column name for the container stop in the channel time frame. -Added by :meth:`with_window_bounds`; prefixed so it cannot clash with a customer -column. +Added by ``solvers.utils.window_bounds.with_window_bounds``; prefixed so it cannot +clash with a customer column. #### stop\_ts\_col @@ -509,48 +509,6 @@ samples instead of splitting them; use drop_implausible_data instead. No-op when not raw. -#### with\_window\_bounds - -```python -def with_window_bounds(df: DataFrame) -> DataFrame -``` - -Add the container start/stop in the channel time frame, for ``TimeWindowEvent``. - -A ``TimeWindowEvent`` tiles each container into windows that must be in the same -time frame as the channel timestamps (:attr:`channel_time_unit`, -:attr:`channel_time_origin`). This adds :attr:`window_start_col` / -:attr:`window_stop_col`, derived from the raw ``start_ts`` / ``stop_ts``, which stay -unchanged for UDFs, ``ContainerEvent`` and ``measurement_dimension``: - -- origin ``"epoch"``: ``TIMESTAMP`` boundaries as epoch numbers in - :attr:`channel_time_unit`; numeric boundaries converted from - :attr:`container_time_unit` to :attr:`channel_time_unit` (as they are when unset); -- origin ``"container_start"``: ``0`` and ``stop_ts - start_ts``, converted the same - way (the difference is taken first, in the boundaries' own unit). - -``TIMESTAMP`` values are converted via ``unix_micros``, which is exact and -independent of the session time zone; ``"s"`` / ``"ms"`` give doubles, ``"us"`` / -``"ns"`` longs. Numeric boundaries converted to a finer unit are multiplied by an -integer (exact, keeping longs), to a coarser unit divided (doubles). The event fact and the solve both call this, so their windows use -the same bounds. The types are checked on the schema, so a missing setting fails -before any Spark job runs. - -**Arguments**: - -- `df` (`pyspark.sql.DataFrame`): Column-mapped ``container_metrics`` frame (or a projection of it) with -``start_ts`` and ``stop_ts``. - -**Raises**: - -- `ValueError`: If ``start_ts`` / ``stop_ts`` are missing, are ``TIMESTAMP_NTZ`` or ``DATE``, -mix ``TIMESTAMP`` and numeric types, or are ``TIMESTAMP`` while -:attr:`channel_time_unit` is unset or :attr:`container_time_unit` is set. - -**Returns**: - -`pyspark.sql.DataFrame`: *df* with the two window-bound columns added. - #### validate\_container\_time\_unit\_requires\_channel\_time\_unit ```python diff --git a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md index 1128266d..a443070e 100644 --- a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md +++ b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md @@ -109,8 +109,9 @@ Calculate definition hash for the time-window event. Only includes the expression string, which encodes the attributes that affect the event results: ``window_length`` and the channel time frame (``channel_time_unit``, -``channel_time_origin``, ``container_time_unit``; omitted while unset / default). Resizing the window or -changing the time frame therefore forces a full recompute in incremental mode. +``channel_time_origin``, ``container_time_unit``; omitted while unset / default). +Resizing the window or changing the time frame therefore forces a full recompute in +incremental mode. Excludes: name, description, required_channels, max_windows_per_container, report_id @@ -150,7 +151,8 @@ Extract the event fact table for the given list of TimeWindowEvent objects. Resolves the matching containers via the solver's filter pipeline (like ``ContainerEvent``) and computes each event's windows natively from the containers' ``start_ts`` / ``stop_ts`` in the channel time frame -(``SolverConfig.with_window_bounds``), so every filtered container gets windows. +(``solvers.utils.window_bounds.with_window_bounds``), so every filtered container +gets windows. Each window becomes one event instance (``start_ts < end_ts``) whose ``event_instance_id`` hashes its boundaries. The solve uses the same window function for scoped aggregations (see :func:`window_intervals_udf`), so the ids match. diff --git a/src/impulse_query_engine/analyze/query/events/time_window_expression.py b/src/impulse_query_engine/analyze/query/events/time_window_expression.py index aa61d394..880571f0 100644 --- a/src/impulse_query_engine/analyze/query/events/time_window_expression.py +++ b/src/impulse_query_engine/analyze/query/events/time_window_expression.py @@ -17,8 +17,9 @@ from impulse_query_engine.model.series.intervals import Intervals # Reuse SolverConfig's internal column names for the container bounds in the channel time -# frame (see SolverConfig.with_window_bounds) rather than re-declaring the literals here. -# These are the keys under which the solve exposes them via ``SeriesCache.container_metrics``. +# frame (see solvers.utils.window_bounds.with_window_bounds) rather than re-declaring the +# literals here. These are the keys under which the solve exposes them via +# ``SeriesCache.container_metrics``. # A default instance suffices since the names are config-invariant. _SOLVER_CONFIG = SolverConfig() @@ -74,9 +75,9 @@ def tile_windows( :meth:`TimeWindowExpression.build` (scoped aggregations), the event fact through :func:`window_intervals_udf`. ``event_instance_id`` hashes each window's boundaries, so both sides must produce identical windows, which a single function guarantees as long as - both pass in the same values. Both read the same Spark-computed - bounds (``SolverConfig.with_window_bounds``), but pandas hands them over as ``int64`` or - ``float64`` (nulls force ``float64``), or as ``None`` / ``NaN``. The bounds are + both pass in the same values. Both read the same Spark-computed bounds + (``solvers.utils.window_bounds.with_window_bounds``), but pandas hands them over as + ``int64`` or ``float64`` (nulls force ``float64``), or as ``None`` / ``NaN``. The bounds are therefore converted to ``float`` first: ``int64`` -> ``float64`` rounds to the nearest double on either path, so the arithmetic below runs on identical doubles. @@ -171,10 +172,10 @@ class TimeWindowExpression(TimeSeriesExpression): The windows are derived purely from the container's ``start_ts`` / ``stop_ts`` metadata (no channel data), so the expression declares no selectors and instead requests the container bounds in the channel time frame via :meth:`required_container_metrics` - (computed by ``SolverConfig.with_window_bounds``). Windows tile those bounds with a - fixed length ``window_length`` (expressed in the same time unit as the channel - timestamps); the final window is clamped to the stop bound when the last full window - would overrun it. + (computed by ``solvers.utils.window_bounds.with_window_bounds``). Windows tile those + bounds with a fixed length ``window_length`` (expressed in the same time unit as the + channel timestamps); the final window is clamped to the stop bound when the last full + window would overrun it. Visual timeline (window_length = W):: @@ -310,7 +311,7 @@ def required_container_metrics(self) -> set[str]: ------- set of str The container start/stop in the channel time frame, which the solver derives - from ``start_ts`` / ``stop_ts`` (``SolverConfig.with_window_bounds``). + from ``start_ts`` / ``stop_ts`` (``solvers.utils.window_bounds.with_window_bounds``). """ return {_SOLVER_CONFIG.window_start_col, _SOLVER_CONFIG.window_stop_col} diff --git a/src/impulse_query_engine/analyze/query/solvers/default_solver.py b/src/impulse_query_engine/analyze/query/solvers/default_solver.py index 5e0b9272..f9e6d7b0 100644 --- a/src/impulse_query_engine/analyze/query/solvers/default_solver.py +++ b/src/impulse_query_engine/analyze/query/solvers/default_solver.py @@ -24,6 +24,7 @@ from .solver_config import RawEncoder, SolverConfig from .utils.interval_encoder import IntervalEncoder from .utils.rle_encoder import RleEncoder +from .utils.window_bounds import with_window_bounds if TYPE_CHECKING: from impulse_query_engine.measurement_db import MeasurementDB @@ -1310,7 +1311,7 @@ def _build_container_metadata_df( # TimeWindowExpression reads the container bounds in the channel time # frame, computed exactly like the TimeWindowEvent fact does. The raw # start_ts/stop_ts stay unchanged for any other expression (e.g. UDFs). - metrics = self.config.with_window_bounds(metrics) + metrics = with_window_bounds(metrics, self.config) missing = [c for c in metric_cols if c not in metrics.columns] if missing: raise ValueError( diff --git a/src/impulse_query_engine/analyze/query/solvers/solver_config.py b/src/impulse_query_engine/analyze/query/solvers/solver_config.py index 7a419ea2..559289ba 100644 --- a/src/impulse_query_engine/analyze/query/solvers/solver_config.py +++ b/src/impulse_query_engine/analyze/query/solvers/solver_config.py @@ -18,13 +18,7 @@ from enum import StrEnum from typing import Literal -import pyspark.sql.functions as F -import pyspark.sql.types as T from pydantic import BaseModel, model_validator -from pyspark.sql import Column, DataFrame - -# Nanoseconds per time unit, for converting container boundaries into the channel unit. -_NANOS_PER_UNIT = {"s": 10**9, "ms": 10**6, "us": 10**3, "ns": 1} class RawEncoder(StrEnum): @@ -143,10 +137,10 @@ class SolverConfig(BaseModel): channel_time_unit : {"s", "ms", "us", "ns"} or None Time unit of the timestamps in the ``channels`` table (``tstart`` / ``tend``, or ``timestamp`` for RAW data). Only used to compute ``TimeWindowEvent`` windows in - that unit (see :meth:`with_window_bounds`); required when ``container_metrics`` - ``start_ts`` / ``stop_ts`` are ``TIMESTAMP`` columns. Nothing else is converted: - channel timestamps, and the ``start_ts`` / ``stop_ts`` seen by UDFs, - ``ContainerEvent`` and ``measurement_dimension``, keep their original values. + that unit (see ``solvers.utils.window_bounds.with_window_bounds``); required when + ``container_metrics`` ``start_ts`` / ``stop_ts`` are ``TIMESTAMP`` columns. Nothing + else is converted: channel timestamps, and the ``start_ts`` / ``stop_ts`` seen by + UDFs, ``ContainerEvent`` and ``measurement_dimension``, keep their original values. channel_time_origin : {"epoch", "container_start"} Origin of the channel timestamps: absolute epoch (default), or relative to the container's ``start_ts``. Like :attr:`channel_time_unit`, only used for the @@ -254,8 +248,8 @@ def start_ts_col(self) -> str: def window_start_col(self) -> str: """Internal column name for the container start in the channel time frame. - Added by :meth:`with_window_bounds`; prefixed so it cannot clash with a customer - column. + Added by ``solvers.utils.window_bounds.with_window_bounds``; prefixed so it cannot + clash with a customer column. """ return "__window_start" @@ -263,8 +257,8 @@ def window_start_col(self) -> str: def window_stop_col(self) -> str: """Internal column name for the container stop in the channel time frame. - Added by :meth:`with_window_bounds`; prefixed so it cannot clash with a customer - column. + Added by ``solvers.utils.window_bounds.with_window_bounds``; prefixed so it cannot + clash with a customer column. """ return "__window_stop" @@ -464,119 +458,6 @@ def reject_implausible_channels_filter_in_raw(self, is_raw: bool) -> None: "implausible points inside the encoder with correct interval boundaries." ) - def _boundary_fields(self, df: DataFrame) -> list[T.StructField]: - """Return the container start/stop timestamp fields present on *df*.""" - names = {self.start_ts_col, self.stop_ts_col} - return [field for field in df.schema.fields if field.name in names] - - def with_window_bounds(self, df: DataFrame) -> DataFrame: - """Add the container start/stop in the channel time frame, for ``TimeWindowEvent``. - - A ``TimeWindowEvent`` tiles each container into windows that must be in the same - time frame as the channel timestamps (:attr:`channel_time_unit`, - :attr:`channel_time_origin`). This adds :attr:`window_start_col` / - :attr:`window_stop_col`, derived from the raw ``start_ts`` / ``stop_ts``, which stay - unchanged for UDFs, ``ContainerEvent`` and ``measurement_dimension``: - - - origin ``"epoch"``: ``TIMESTAMP`` boundaries as epoch numbers in - :attr:`channel_time_unit`; numeric boundaries converted from - :attr:`container_time_unit` to :attr:`channel_time_unit` (as they are when unset); - - origin ``"container_start"``: ``0`` and ``stop_ts - start_ts``, converted the same - way (the difference is taken first, in the boundaries' own unit). - - ``TIMESTAMP`` values are converted via ``unix_micros``, which is exact and - independent of the session time zone; ``"s"`` / ``"ms"`` give doubles, ``"us"`` / - ``"ns"`` longs. Numeric boundaries converted to a finer unit are multiplied by an - integer (exact, keeping longs), to a coarser unit divided (doubles). The event fact and the solve both call this, so their windows use - the same bounds. The types are checked on the schema, so a missing setting fails - before any Spark job runs. - - Parameters - ---------- - df : pyspark.sql.DataFrame - Column-mapped ``container_metrics`` frame (or a projection of it) with - ``start_ts`` and ``stop_ts``. - - Returns - ------- - pyspark.sql.DataFrame - *df* with the two window-bound columns added. - - Raises - ------ - ValueError - If ``start_ts`` / ``stop_ts`` are missing, are ``TIMESTAMP_NTZ`` or ``DATE``, - mix ``TIMESTAMP`` and numeric types, or are ``TIMESTAMP`` while - :attr:`channel_time_unit` is unset or :attr:`container_time_unit` is set. - """ - types = {field.name: field.dataType for field in self._boundary_fields(df)} - missing = [c for c in (self.start_ts_col, self.stop_ts_col) if c not in types] - if missing: - raise ValueError( - f"TimeWindowEvent needs the container_metrics columns {missing} to compute " - f"its windows. Available columns: {df.columns}" - ) - for name, dtype in types.items(): - if isinstance(dtype, (T.TimestampNTZType, T.DateType)): - raise ValueError( - f"container_metrics column '{name}' has type {dtype.simpleString()}, " - "which cannot be converted to an epoch unambiguously (it carries no time " - "zone). Use a TIMESTAMP or epoch-number column." - ) - is_timestamp = {isinstance(dtype, T.TimestampType) for dtype in types.values()} - if len(is_timestamp) > 1: - raise ValueError( - f"container_metrics columns '{self.start_ts_col}' and '{self.stop_ts_col}' " - "must both be TIMESTAMP or both be numeric to compute TimeWindowEvent windows." - ) - timestamps = is_timestamp.pop() - if timestamps and self.channel_time_unit is None: - raise ValueError( - f"TimeWindowEvent needs its windows in the channel time frame, but " - f"container_metrics '{self.start_ts_col}' / '{self.stop_ts_col}' are " - "TIMESTAMP columns. Set query_engine.solver_config.channel_time_unit to the " - "unit of the channel timestamps (one of 's', 'ms', 'us', 'ns'), and " - "channel_time_origin to 'container_start' if they are relative to the " - "container start." - ) - if timestamps and self.container_time_unit is not None: - raise ValueError( - f"container_time_unit only applies to numeric container_metrics " - f"'{self.start_ts_col}' / '{self.stop_ts_col}', but they are TIMESTAMP " - "columns, which carry their own unit. Remove container_time_unit." - ) - - start, stop = F.col(self.start_ts_col), F.col(self.stop_ts_col) - if self.channel_time_origin == "container_start": - window_start = F.lit(0) - # Subtract exactly in microseconds before scaling to the channel unit. - window_stop = ( - self._micros_in_unit(F.unix_micros(stop) - F.unix_micros(start)) - if timestamps - else self._container_to_channel_unit(stop - start) - ) - elif timestamps: - window_start = self._micros_in_unit(F.unix_micros(start)) - window_stop = self._micros_in_unit(F.unix_micros(stop)) - else: - window_start = self._container_to_channel_unit(start) - window_stop = self._container_to_channel_unit(stop) - return df.withColumn(self.window_start_col, window_start).withColumn( - self.window_stop_col, window_stop - ) - - def _container_to_channel_unit(self, col: Column) -> Column: - """Numeric boundary *col* converted from :attr:`container_time_unit` to - :attr:`channel_time_unit` (unchanged when unset or equal).""" - if self.container_time_unit is None or self.container_time_unit == self.channel_time_unit: - return col - source = _NANOS_PER_UNIT[self.container_time_unit] - target = _NANOS_PER_UNIT[self.channel_time_unit] - if source > target: - # Finer target unit: an integer factor keeps long boundaries exact. - return col * F.lit(source // target) - return col / F.lit(float(target // source)) - @model_validator(mode="after") def validate_container_time_unit_requires_channel_time_unit(self): """``container_time_unit`` converts into ``channel_time_unit``, so it needs one.""" @@ -587,13 +468,3 @@ def validate_container_time_unit_requires_channel_time_unit(self): "timestamps." ) return self - - def _micros_in_unit(self, micros: Column) -> Column: - """Microseconds converted to :attr:`channel_time_unit`.""" - if self.channel_time_unit == "s": - return micros / F.lit(1e6) - if self.channel_time_unit == "ms": - return micros / F.lit(1e3) - if self.channel_time_unit == "ns": - return micros * F.lit(1000) - return micros diff --git a/src/impulse_query_engine/analyze/query/solvers/utils/window_bounds.py b/src/impulse_query_engine/analyze/query/solvers/utils/window_bounds.py new file mode 100644 index 00000000..5e96696d --- /dev/null +++ b/src/impulse_query_engine/analyze/query/solvers/utils/window_bounds.py @@ -0,0 +1,149 @@ +"""Container bounds in the channel time frame, for ``TimeWindowEvent`` windows. + +Both the ``TimeWindowEvent`` event fact and the solve (``TimeWindowExpression`` via the +container metadata) derive their window bounds here, from the raw ``container_metrics`` +``start_ts`` / ``stop_ts`` and the ``SolverConfig`` channel time settings +(``channel_time_unit``, ``channel_time_origin``, ``container_time_unit``). +""" + +import pyspark.sql.functions as F +import pyspark.sql.types as T +from pyspark.sql import Column, DataFrame + +from impulse_query_engine.analyze.query.solvers.solver_config import SolverConfig + +# Nanoseconds per time unit, for converting container boundaries into the channel unit. +_NANOS_PER_UNIT = {"s": 10**9, "ms": 10**6, "us": 10**3, "ns": 1} + + +def with_window_bounds(df: DataFrame, config: SolverConfig) -> DataFrame: + """Add the container start/stop in the channel time frame, for ``TimeWindowEvent``. + + A ``TimeWindowEvent`` tiles each container into windows that must be in the same time + frame as the channel timestamps (``config.channel_time_unit``, + ``config.channel_time_origin``). This adds ``config.window_start_col`` / + ``config.window_stop_col``, derived from the raw ``start_ts`` / ``stop_ts``, which stay + unchanged for UDFs, ``ContainerEvent`` and ``measurement_dimension``: + + - origin ``"epoch"``: ``TIMESTAMP`` boundaries as epoch numbers in + ``channel_time_unit``; numeric boundaries converted from ``container_time_unit`` to + ``channel_time_unit`` (as they are when unset); + - origin ``"container_start"``: ``0`` and ``stop_ts - start_ts``, converted the same way + (the difference is taken first, in the boundaries' own unit). + + ``TIMESTAMP`` values are converted via ``unix_micros``, which is exact and independent of + the session time zone; ``"s"`` / ``"ms"`` give doubles, ``"us"`` / ``"ns"`` longs. + Numeric boundaries converted to a finer unit are multiplied by an integer (exact, + keeping longs), to a coarser unit divided (doubles). The event fact and the solve both + call this, so their windows use the same bounds. The types are checked on the schema, + so a missing setting fails before any Spark job runs. + + Parameters + ---------- + df : pyspark.sql.DataFrame + Column-mapped ``container_metrics`` frame (or a projection of it) with ``start_ts`` + and ``stop_ts``. + config : SolverConfig + Solver configuration holding the channel time settings and column names. + + Returns + ------- + pyspark.sql.DataFrame + *df* with the two window-bound columns added. + + Raises + ------ + ValueError + If ``start_ts`` / ``stop_ts`` are missing, are ``TIMESTAMP_NTZ`` or ``DATE``, mix + ``TIMESTAMP`` and numeric types, or are ``TIMESTAMP`` while ``channel_time_unit`` is + unset or ``container_time_unit`` is set. + """ + types = {field.name: field.dataType for field in _boundary_fields(df, config)} + missing = [c for c in (config.start_ts_col, config.stop_ts_col) if c not in types] + if missing: + raise ValueError( + f"TimeWindowEvent needs the container_metrics columns {missing} to compute " + f"its windows. Available columns: {df.columns}" + ) + for name, dtype in types.items(): + if isinstance(dtype, (T.TimestampNTZType, T.DateType)): + raise ValueError( + f"container_metrics column '{name}' has type {dtype.simpleString()}, " + "which cannot be converted to an epoch unambiguously (it carries no time " + "zone). Use a TIMESTAMP or epoch-number column." + ) + is_timestamp = {isinstance(dtype, T.TimestampType) for dtype in types.values()} + if len(is_timestamp) > 1: + raise ValueError( + f"container_metrics columns '{config.start_ts_col}' and '{config.stop_ts_col}' " + "must both be TIMESTAMP or both be numeric to compute TimeWindowEvent windows." + ) + timestamps = is_timestamp.pop() + if timestamps and config.channel_time_unit is None: + raise ValueError( + f"TimeWindowEvent needs its windows in the channel time frame, but " + f"container_metrics '{config.start_ts_col}' / '{config.stop_ts_col}' are " + "TIMESTAMP columns. Set query_engine.solver_config.channel_time_unit to the " + "unit of the channel timestamps (one of 's', 'ms', 'us', 'ns'), and " + "channel_time_origin to 'container_start' if they are relative to the " + "container start." + ) + if timestamps and config.container_time_unit is not None: + raise ValueError( + f"container_time_unit only applies to numeric container_metrics " + f"'{config.start_ts_col}' / '{config.stop_ts_col}', but they are TIMESTAMP " + "columns, which carry their own unit. Remove container_time_unit." + ) + + start, stop = F.col(config.start_ts_col), F.col(config.stop_ts_col) + unit = config.channel_time_unit + if config.channel_time_origin == "container_start": + window_start = F.lit(0) + # Subtract exactly in microseconds before scaling to the channel unit. + window_stop = ( + _micros_in_unit(F.unix_micros(stop) - F.unix_micros(start), unit) + if timestamps + else _container_to_channel_unit(stop - start, config) + ) + elif timestamps: + window_start = _micros_in_unit(F.unix_micros(start), unit) + window_stop = _micros_in_unit(F.unix_micros(stop), unit) + else: + window_start = _container_to_channel_unit(start, config) + window_stop = _container_to_channel_unit(stop, config) + return df.withColumn(config.window_start_col, window_start).withColumn( + config.window_stop_col, window_stop + ) + + +def _boundary_fields(df: DataFrame, config: SolverConfig) -> list[T.StructField]: + """Return the container start/stop timestamp fields present on *df*.""" + names = {config.start_ts_col, config.stop_ts_col} + return [field for field in df.schema.fields if field.name in names] + + +def _container_to_channel_unit(col: Column, config: SolverConfig) -> Column: + """Numeric boundary *col* converted from ``container_time_unit`` to ``channel_time_unit`` + (unchanged when unset or equal).""" + if ( + config.container_time_unit is None + or config.container_time_unit == config.channel_time_unit + ): + return col + source = _NANOS_PER_UNIT[config.container_time_unit] + target = _NANOS_PER_UNIT[config.channel_time_unit] + if source > target: + # Finer target unit: an integer factor keeps long boundaries exact. + return col * F.lit(source // target) + return col / F.lit(float(target // source)) + + +def _micros_in_unit(micros: Column, unit: str | None) -> Column: + """Microseconds converted to *unit*.""" + if unit == "s": + return micros / F.lit(1e6) + if unit == "ms": + return micros / F.lit(1e3) + if unit == "ns": + return micros * F.lit(1000) + return micros diff --git a/src/impulse_reporting/events/time_window_event.py b/src/impulse_reporting/events/time_window_event.py index 029f781b..e1df88d6 100644 --- a/src/impulse_reporting/events/time_window_event.py +++ b/src/impulse_reporting/events/time_window_event.py @@ -19,6 +19,7 @@ ) from impulse_query_engine.analyze.query.query_builder import QueryBuilder from impulse_query_engine.analyze.query.solvers.query_solver import QuerySolver +from impulse_query_engine.analyze.query.solvers.utils.window_bounds import with_window_bounds from impulse_reporting.events.container_boundary_event import ContainerBoundaryEvent from impulse_reporting.persist.fact_schema import EVENT_INSTANCE_FACT_SCHEMA from impulse_reporting.util.event_instance_util import generate_event_instance_id_column @@ -152,8 +153,9 @@ def determine_definition_hash(self) -> int: Only includes the expression string, which encodes the attributes that affect the event results: ``window_length`` and the channel time frame (``channel_time_unit``, - ``channel_time_origin``, ``container_time_unit``; omitted while unset / default). Resizing the window or - changing the time frame therefore forces a full recompute in incremental mode. + ``channel_time_origin``, ``container_time_unit``; omitted while unset / default). + Resizing the window or changing the time frame therefore forces a full recompute in + incremental mode. Excludes: name, description, required_channels, max_windows_per_container, report_id @@ -207,7 +209,8 @@ def determine_events( Resolves the matching containers via the solver's filter pipeline (like ``ContainerEvent``) and computes each event's windows natively from the containers' ``start_ts`` / ``stop_ts`` in the channel time frame - (``SolverConfig.with_window_bounds``), so every filtered container gets windows. + (``solvers.utils.window_bounds.with_window_bounds``), so every filtered container + gets windows. Each window becomes one event instance (``start_ts < end_ts``) whose ``event_instance_id`` hashes its boundaries. The solve uses the same window function for scoped aggregations (see :func:`window_intervals_udf`), so the ids match. @@ -238,7 +241,7 @@ def determine_events( # The windows are computed in the channel time frame, from the same bounds the solve # uses for scoped aggregations (fails fast on the schema, e.g. when TIMESTAMP # boundaries lack solver_config.channel_time_unit). - container_metrics_df = solver.config.with_window_bounds(container_metrics_df) + container_metrics_df = with_window_bounds(container_metrics_df, solver.config) start_ts = f.col(solver.config.window_start_col) stop_ts = f.col(solver.config.window_stop_col) diff --git a/tests/impulse_query_engine/unit/analyze/query/solvers/container_boundaries_test.py b/tests/impulse_query_engine/unit/analyze/query/solvers/utils/window_bounds_test.py similarity index 91% rename from tests/impulse_query_engine/unit/analyze/query/solvers/container_boundaries_test.py rename to tests/impulse_query_engine/unit/analyze/query/solvers/utils/window_bounds_test.py index 7b6e7fbc..91b58e75 100644 --- a/tests/impulse_query_engine/unit/analyze/query/solvers/container_boundaries_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/solvers/utils/window_bounds_test.py @@ -1,5 +1,5 @@ # pylint: disable=missing-function-docstring, redefined-outer-name -"""Tests for SolverConfig.with_window_bounds. +"""Tests for solvers.utils.window_bounds.with_window_bounds. ``TimeWindowEvent`` windows are computed in the channel time frame (``channel_time_unit`` / ``channel_time_origin``). ``with_window_bounds`` derives the container @@ -15,6 +15,7 @@ from pyspark.sql import SparkSession from impulse_query_engine.analyze.query.solvers.solver_config import SolverConfig +from impulse_query_engine.analyze.query.solvers.utils.window_bounds import with_window_bounds from tests.conftest import spark # noqa: F401 (pytest fixture) # 2025-07-03 07:41:41.483456 UTC @@ -38,7 +39,7 @@ def _boundaries_df(spark: SparkSession): # noqa: F811 def _bounds(cfg: SolverConfig, df) -> dict: - out = cfg.with_window_bounds(df) + out = with_window_bounds(df, cfg) return {r.container_id: (r[_START], r[_STOP]) for r in out.collect()} @@ -63,7 +64,7 @@ def test_epoch_origin_converts_timestamps_to_unit( previous_tz = spark.conf.get("spark.sql.session.timeZone") spark.conf.set("spark.sql.session.timeZone", session_tz) try: - out = SolverConfig(channel_time_unit=unit).with_window_bounds(_boundaries_df(spark)) + out = with_window_bounds(_boundaries_df(spark), SolverConfig(channel_time_unit=unit)) rows = {r.container_id: r for r in out.collect()} finally: spark.conf.set("spark.sql.session.timeZone", previous_tz) @@ -132,7 +133,7 @@ def _ms_boundaries_df(spark: SparkSession): # noqa: F811 def test_numeric_ms_boundaries_converted_to_finer_channel_unit_exactly(spark): # noqa: F811 # Boundaries in epoch ms, channels in µs: an integer factor keeps the longs exact. cfg = SolverConfig(channel_time_unit="us", container_time_unit="ms") - out = cfg.with_window_bounds(_ms_boundaries_df(spark)) + out = with_window_bounds(_ms_boundaries_df(spark), cfg) start_ms = _EPOCH_MICROS // 1000 stop_ms = (_EPOCH_MICROS + _SPAN_MICROS) // 1000 assert out.schema[_START].dataType == T.LongType() @@ -152,7 +153,7 @@ def test_numeric_boundaries_converted_to_coarser_channel_unit(spark): # noqa: F "container_id int, start_ts long, stop_ts long", ) cfg = SolverConfig(channel_time_unit="ms", container_time_unit="us") - out = cfg.with_window_bounds(df) + out = with_window_bounds(df, cfg) assert out.schema[_START].dataType == T.DoubleType() assert _bounds(cfg, df) == { 1: (_EPOCH_MICROS / 1000.0, (_EPOCH_MICROS + _SPAN_MICROS) / 1000.0) @@ -169,7 +170,7 @@ def test_numeric_boundaries_unchanged_for_equal_or_unset_container_unit(spark): def test_container_time_unit_rejected_for_timestamp_boundaries(spark): # noqa: F811 cfg = SolverConfig(channel_time_unit="s", container_time_unit="ms") with pytest.raises(ValueError, match="container_time_unit only applies to numeric"): - cfg.with_window_bounds(_boundaries_df(spark)) + with_window_bounds(_boundaries_df(spark), cfg) def test_container_time_unit_requires_channel_time_unit(): @@ -182,7 +183,7 @@ def test_container_time_unit_requires_channel_time_unit(): def test_raw_boundaries_stay_unchanged(spark): # noqa: F811 df = _boundaries_df(spark) cfg = SolverConfig(channel_time_unit="s", channel_time_origin="container_start") - out = cfg.with_window_bounds(df) + out = with_window_bounds(df, cfg) assert isinstance(out.schema["start_ts"].dataType, T.TimestampType) assert out.select("container_id", "start_ts", "stop_ts").collect() == df.collect() @@ -190,7 +191,7 @@ def test_raw_boundaries_stay_unchanged(spark): # noqa: F811 def test_timestamp_boundaries_without_unit_rejected(spark): # noqa: F811 for origin in ("epoch", "container_start"): with pytest.raises(ValueError, match=r"TimeWindowEvent.*channel_time_unit"): - SolverConfig(channel_time_origin=origin).with_window_bounds(_boundaries_df(spark)) + with_window_bounds(_boundaries_df(spark), SolverConfig(channel_time_origin=origin)) @pytest.mark.parametrize( @@ -202,15 +203,15 @@ def test_zone_less_types_rejected(spark, value, ddl): # noqa: F811 [(1, value, value)], f"container_id int, start_ts {ddl}, stop_ts {ddl}" ) with pytest.raises(ValueError, match="start_ts"): - SolverConfig(channel_time_unit="s").with_window_bounds(df) + with_window_bounds(df, SolverConfig(channel_time_unit="s")) def test_mixed_and_missing_boundaries_rejected(spark): # noqa: F811 mixed = _boundaries_df(spark).withColumn("stop_ts", F.lit(1.0)) with pytest.raises(ValueError, match="both be TIMESTAMP or both be numeric"): - SolverConfig(channel_time_unit="s").with_window_bounds(mixed) + with_window_bounds(mixed, SolverConfig(channel_time_unit="s")) with pytest.raises(ValueError, match="stop_ts"): - SolverConfig().with_window_bounds(_boundaries_df(spark).drop("stop_ts")) + with_window_bounds(_boundaries_df(spark).drop("stop_ts"), SolverConfig()) def test_channel_time_settings_validated(): From 44129db90c1dbf0c4de2e24c5bad337cd8ad186c Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 7 Oct 2026 17:50:57 +0200 Subject: [PATCH 21/27] refactor(reporting): move as_dict, required_channels, and hash helpers to ContainerBoundaryEvent Hoist `as_dict`, `required_channels`, and shared SHA-256 hashing/normalization helpers into `ContainerBoundaryEvent` to remove duplication between `ContainerEvent` and `TimeWindowEvent`. Subclasses now only customize `description`, `attributes`, and `required_channels` initialization. Update API reference docs to drop the subclass `as_dict` entries that are now inherited. --- .../events/container_event.md | 12 ------ .../events/time_window_event.md | 12 ------ .../events/container_boundary_event.py | 40 +++++++++++++++++++ .../events/container_event.py | 29 +------------- .../events/time_window_event.py | 35 ++-------------- 5 files changed, 45 insertions(+), 83 deletions(-) diff --git a/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md b/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md index 5e710ba1..af2911e8 100644 --- a/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md +++ b/docs/impulse/docs/references/api/impulse_reporting/events/container_event.md @@ -74,18 +74,6 @@ so the name of the event is hashed. `int`: Hash value representing the computation definition. -#### as\_dict - -```python -def as_dict() -> dict -``` - -Return a dictionary representation of the event. - -**Returns**: - -`dict`: - #### determine\_events ```python diff --git a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md index a443070e..07da0708 100644 --- a/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md +++ b/docs/impulse/docs/references/api/impulse_reporting/events/time_window_event.md @@ -120,18 +120,6 @@ report_id `int`: Hash value representing the computation definition. -#### as\_dict - -```python -def as_dict() -> dict -``` - -Get a dictionary representation of the event. - -**Returns**: - -`dict`: Dictionary containing event metadata. - #### determine\_events ```python diff --git a/src/impulse_reporting/events/container_boundary_event.py b/src/impulse_reporting/events/container_boundary_event.py index 21d8c58c..e92ab580 100644 --- a/src/impulse_reporting/events/container_boundary_event.py +++ b/src/impulse_reporting/events/container_boundary_event.py @@ -2,7 +2,9 @@ from __future__ import annotations +import hashlib import zlib +from collections.abc import Mapping from pyspark.sql import DataFrame, Row, SparkSession @@ -20,8 +22,46 @@ class ContainerBoundaryEvent(Event): filtered container yields instances regardless of its channel data. The report therefore excludes these event types from the solvable expressions and dispatches them with ``query`` / ``solver`` rather than ``solved_df``. + + Subclasses set ``description`` and ``attributes`` (via :meth:`_normalize_attributes`), + and ``required_channels`` when they have any; :meth:`as_dict` writes them to + ``event_dimension``. """ + required_channels: list[str] | None = None + + @staticmethod + def _normalize_attributes(attributes: Mapping[str, str] | None) -> dict[str, str]: + """Return *attributes* with string keys and values (empty when ``None``).""" + return {str(k): str(v) for k, v in (attributes or {}).items()} + + @staticmethod + def _sha256_long(text: str) -> int: + """SHA-256 of *text*, truncated to a signed 64-bit int (the ``definition_hash`` + type).""" + hash_bytes = hashlib.sha256(text.encode()).digest() + return int.from_bytes(hash_bytes[:8], byteorder="big", signed=True) + + def as_dict(self) -> dict: + """Return the event's ``event_dimension`` row as a dictionary. + + Returns + ------- + dict + Event metadata keyed by the ``event_dimension`` column names. + """ + return { + "event_id": self.get_id(), + "report_id": self.report_id, + "event_type": self.get_event_type_str(), + "event_name": self.name, + "event_description": self.description, + "required_channels": self.required_channels, + "event_expression": self.get_expression_str(), + "definition_hash": self.determine_definition_hash(), + "attributes": self.attributes, + } + def get_id(self) -> int: """Return a unique identifier derived from the event name. diff --git a/src/impulse_reporting/events/container_event.py b/src/impulse_reporting/events/container_event.py index 74a509d8..66ad7cda 100644 --- a/src/impulse_reporting/events/container_event.py +++ b/src/impulse_reporting/events/container_event.py @@ -2,7 +2,6 @@ from __future__ import annotations -import hashlib from typing import TYPE_CHECKING import pyspark.sql.functions as f @@ -44,10 +43,7 @@ def __init__(self, name: str, desc: str = None, attributes: dict[str, str] = Non """ super().__init__(name) self.description = desc - normalized_attributes: dict[str, str] = {} - if attributes is not None: - normalized_attributes = {str(k): str(v) for k, v in attributes.items()} - self.attributes = normalized_attributes + self.attributes = self._normalize_attributes(attributes) # ------------------------------------------------------------------ # Instance methods @@ -85,28 +81,7 @@ def determine_definition_hash(self) -> int: int Hash value representing the computation definition. """ - hash_input = self.name - hash_bytes = hashlib.sha256(hash_input.encode()).digest() - return int.from_bytes(hash_bytes[:8], byteorder="big", signed=True) - - def as_dict(self) -> dict: - """Return a dictionary representation of the event. - - Returns - ------- - dict - """ - return { - "event_id": self.get_id(), - "report_id": self.report_id, - "event_type": self.get_event_type_str(), - "event_name": self.name, - "event_description": self.description, - "required_channels": None, - "event_expression": self.get_expression_str(), - "definition_hash": self.determine_definition_hash(), - "attributes": self.attributes, - } + return self._sha256_long(self.name) # ------------------------------------------------------------------ # Class methods diff --git a/src/impulse_reporting/events/time_window_event.py b/src/impulse_reporting/events/time_window_event.py index e1df88d6..10f77b8d 100644 --- a/src/impulse_reporting/events/time_window_event.py +++ b/src/impulse_reporting/events/time_window_event.py @@ -2,7 +2,6 @@ from __future__ import annotations -import hashlib from collections.abc import Mapping import pyspark.sql.functions as f @@ -96,13 +95,10 @@ def __init__( self.max_windows_per_container = self.expression.max_windows self.description = desc self.required_channels = required_channels - normalized_attributes: dict[str, str] = {} - if attributes is not None: - normalized_attributes = {str(k): str(v) for k, v in attributes.items()} + self.attributes = self._normalize_attributes(attributes) # Surface the window length for traceability in event_dimension, without # clobbering an explicit user-supplied attribute of the same key. - normalized_attributes.setdefault("window_length", str(self.window_length)) - self.attributes = normalized_attributes + self.attributes.setdefault("window_length", str(self.window_length)) def set_channel_time( self, unit: str | None, origin: str = "epoch", container_unit: str | None = None @@ -165,32 +161,7 @@ def determine_definition_hash(self) -> int: int Hash value representing the computation definition. """ - hash_input = self.get_expression_str() - - # Use SHA-256 and return as int (truncated to fit LongType) - hash_bytes = hashlib.sha256(hash_input.encode()).digest() - return int.from_bytes(hash_bytes[:8], byteorder="big", signed=True) - - def as_dict(self) -> dict: - """ - Get a dictionary representation of the event. - - Returns - ------- - dict - Dictionary containing event metadata. - """ - return { - "event_id": self.get_id(), - "report_id": self.report_id, - "event_type": self.get_event_type_str(), - "event_name": self.name, - "event_description": self.description, - "required_channels": self.required_channels, - "event_expression": self.get_expression_str(), - "definition_hash": self.determine_definition_hash(), - "attributes": self.attributes, - } + return self._sha256_long(self.get_expression_str()) @classmethod def determine_events( From b632d5188db1e0220a92ac9469e7fa6e7d64ed8b Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 7 Oct 2026 18:16:53 +0200 Subject: [PATCH 22/27] docs(query-engine, skills): clarify container_metrics.start_ts/stop_ts references in TimeWindowEvent docs Update configuration docs, event reference, and skills to consistently refer to `container_metrics.start_ts`/`stop_ts` when describing the source of `TimeWindowEvent` container boundaries and the columns that remain unchanged for `ContainerEvent`, `measurement_dimension`, and UDFs. --- docs/impulse/docs/config/configuration.md | 8 ++--- docs/impulse/docs/references/report/event.md | 32 +++++++++++--------- skills/impulse-config/SKILL.md | 7 +++-- skills/impulse-events/SKILL.md | 8 ++--- 4 files changed, 30 insertions(+), 25 deletions(-) diff --git a/docs/impulse/docs/config/configuration.md b/docs/impulse/docs/config/configuration.md index 85eb3934..d8199953 100644 --- a/docs/impulse/docs/config/configuration.md +++ b/docs/impulse/docs/config/configuration.md @@ -173,8 +173,8 @@ Top-level fields on `SolverConfig`: - `channel_time_unit` (`"s"` | `"ms"` | `"us"` | `"ns"`, optional) and `channel_time_origin` (`"epoch"` (default) | `"container_start"`): the time frame of the channel timestamps (`tstart`/`tend`, or `timestamp` when `data_type = "RAW"`). `"epoch"` means absolute epoch - numbers; `"container_start"` means time relative to the container's `start_ts` (e.g. seconds - since the recording started). Channel timestamps are never converted; they must already be + numbers; `"container_start"` means time relative to the container's start + (`container_metrics.start_ts`, e.g. seconds since the recording started). Channel timestamps are never converted; they must already be numbers in that frame. - `container_time_unit` (`"s"` | `"ms"` | `"us"` | `"ns"`, optional): the unit of **numeric** @@ -193,8 +193,8 @@ Top-level fields on `SolverConfig`: `channel_time_unit` is required when the boundaries are `TIMESTAMP` columns; a report with a `TimeWindowEvent` fails with a clear error until it is set. `TIMESTAMP_NTZ` and `DATE` boundaries - are not supported. Everything else sees the original `start_ts`/`stop_ts`: `ContainerEvent`, - `measurement_dimension`, container filters, and UDFs that request them via + are not supported. Everything else sees the original `container_metrics.start_ts`/`stop_ts`: + `ContainerEvent`, `measurement_dimension`, container filters, and UDFs that request them via `apply(..., container_metrics=[...])` (a `TIMESTAMP` arrives there as a `pd.Timestamp`). The channel time frame is part of the definition hash of every `TimeWindowEvent` and of the diff --git a/docs/impulse/docs/references/report/event.md b/docs/impulse/docs/references/report/event.md index 7e60e1e2..7dca0614 100644 --- a/docs/impulse/docs/references/report/event.md +++ b/docs/impulse/docs/references/report/event.md @@ -240,13 +240,14 @@ The windows are computed in the time frame of the channel timestamps, set by setting. - If the channel timestamps are relative to the container start (e.g. seconds since the recording started), also set `channel_time_origin="container_start"`. The windows then run from - `0` to `stop_ts - start_ts`. -- If numeric `start_ts`/`stop_ts` are in another unit than the channel timestamps (e.g. epoch ms - boundaries, µs samples), set `container_time_unit` to their unit (e.g. `"ms"`) and - `channel_time_unit` to the channels' (e.g. `"us"`). Otherwise no window overlaps the samples. + `0` to the container's duration (`stop_ts - start_ts` of `container_metrics`). +- If numeric `container_metrics.start_ts`/`stop_ts` are in another unit than the channel + timestamps (e.g. epoch ms boundaries, µs samples), set `container_time_unit` to their unit + (e.g. `"ms"`) and `channel_time_unit` to the channels' (e.g. `"us"`). Otherwise no window + overlaps the samples. Only the windows use these settings: `ContainerEvent`, `measurement_dimension` and UDFs that read -`start_ts`/`stop_ts` keep seeing the original values. The channel time frame is part of the +`container_metrics.start_ts`/`stop_ts` keep seeing the original values. The channel time frame is part of the event's definition (and of the aggregations scoped to it), so changing it recomputes them over all containers in incremental mode. ::: @@ -255,11 +256,12 @@ containers in incremental mode. 1. The event resolves the matching containers through the report's container filters (like `ContainerEvent`), reads `start_ts` and `stop_ts` from the `container_metrics` table, and - tiles `[start_ts, stop_ts]`, in the channel time frame, into consecutive windows of length + tiles that span, in the channel time frame, into consecutive windows of length `window_length`. The window instances in `event_instance_fact` are in that frame too. -2. The **final window is clamped** to `stop_ts` when the last full window would overrun it; any - zero-length trailing slice is dropped (every instance satisfies `start_ts < end_ts`). - Containers whose `start_ts` or `stop_ts` is null, NaN or infinite get no windows. +2. The **final window is clamped** to the container's `stop_ts` when the last full window would + overrun it; any zero-length trailing slice is dropped (every window instance in + `event_instance_fact` satisfies `start_ts < end_ts`). Containers whose + `container_metrics.start_ts` or `stop_ts` is null, NaN or infinite get no windows. 3. Each window becomes one **event instance**, written to the shared `event_instance_fact` table. Its `event_instance_id` hashes the container, the event name and the window's start and end, like for other interval events. @@ -270,14 +272,16 @@ containers in incremental mode. The windows are computed from `container_metrics` alone, so **every** container that matches the report's filters gets windows, whether or not it has channel data and whether or not an aggregation is scoped to the event. An aggregation scoped to the event uses the same window -function in the query engine, so its per-window rows carry the same `event_instance_id` values. For the per-window values to be meaningful, the container boundaries must share the channel -samples' time base (as they do in real measurement data). +function in the query engine, so its per-window rows carry the same `event_instance_id` values. +For the per-window values to be meaningful, the container boundaries (`container_metrics.start_ts` +/ `stop_ts`) must share the channel samples' time base, or be converted into it with the settings +above. ::: :::note -Window boundaries are stored as doubles (`start_ts` / `end_ts`), like every other event type. Epoch -timestamps in nanoseconds exceed the range doubles represent exactly, so their window boundaries -are rounded to about 256 ns. The event and its aggregations use the same window function, so the +Window boundaries are stored in `event_instance_fact` as doubles (its `start_ts` / `end_ts` +columns), like every other event type. Epoch timestamps in nanoseconds exceed the range doubles +represent exactly, so their window boundaries are rounded to about 256 ns. The event and its aggregations use the same window function, so the rounding is the same on both sides and their `event_instance_id` values still match. ::: diff --git a/skills/impulse-config/SKILL.md b/skills/impulse-config/SKILL.md index 30f06463..81568e03 100644 --- a/skills/impulse-config/SKILL.md +++ b/skills/impulse-config/SKILL.md @@ -127,13 +127,14 @@ table that has one (`container_tags`, `container_metrics`, `channel_mapping`). O Top-level `channel_time_unit` (`"s"` | `"ms"` | `"us"` | `"ns"`, optional) and `channel_time_origin` (`"epoch"` default | `"container_start"`) describe the time frame of the `channels` timestamps (`tstart`/`tend`, or `timestamp` with `data_type="RAW"`): absolute epoch numbers, or time relative -to the container's `start_ts`. Only `TimeWindowEvent` uses them, to compute its windows in that +to the container's start (`container_metrics.start_ts`). Only `TimeWindowEvent` uses them, to compute its windows in that frame from `container_metrics.start_ts`/`stop_ts` (origin `"container_start"`: from `0` to `stop_ts - start_ts`). `channel_time_unit` is required when those are `TIMESTAMP` columns. If they are numeric but in another unit than the channels (e.g. epoch ms boundaries, µs samples), also set `container_time_unit` (e.g. `"ms"`; requires `channel_time_unit`, not allowed for `TIMESTAMP`). -Nothing is converted in place: the channel timestamps, and the `start_ts`/`stop_ts` seen by -`ContainerEvent`, `measurement_dimension` and UDFs (a `pd.Timestamp`), keep their original values. +Nothing is converted in place: the channel timestamps, and the `container_metrics.start_ts` / +`stop_ts` seen by `ContainerEvent`, `measurement_dimension` and UDFs (a `pd.Timestamp`), keep their +original values. ```python "query_engine": { diff --git a/skills/impulse-events/SKILL.md b/skills/impulse-events/SKILL.md index dd733d64..542caddf 100644 --- a/skills/impulse-events/SKILL.md +++ b/skills/impulse-events/SKILL.md @@ -160,10 +160,10 @@ Because the windows come from `container_metrics`, those boundaries must share t time base for the per-window values to be meaningful. If `container_metrics.start_ts`/`stop_ts` are `TIMESTAMP` columns, set `query_engine.solver_config.channel_time_unit` to the unit of the channel timestamps (`tstart`/`tend`, or `timestamp` for RAW), and `channel_time_origin="container_start"` if -they are relative to the container start (windows then run from `0`). Numeric boundaries in another -unit than the channels (e.g. epoch ms vs. µs samples) need `container_time_unit` as well. Only the -windows use these settings; `start_ts`/`stop_ts` keep their original values for `ContainerEvent` and -UDFs. +they are relative to the container start (windows then run from `0`). Numeric `container_metrics` +boundaries in another unit than the channels (e.g. epoch ms vs. µs samples) need +`container_time_unit` as well. Only the windows use these settings; `container_metrics.start_ts` / +`stop_ts` keep their original values for `ContainerEvent` and UDFs. Containers with null, NaN or infinite boundaries get no windows. ## Output schema From 5a9df2c9b06fdd66ace170860ea35d0c9e53410e Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 7 Oct 2026 18:27:35 +0200 Subject: [PATCH 23/27] fix(query-engine): cast window-bound conversion factor to long to prevent int overflow When converting integer container boundaries from a coarser unit to a finer channel unit (e.g. epoch seconds to ms), multiply by a long-typed factor so Spark widens the result instead of keeping int*int as int. This avoids ARITHMETIC_OVERFLOW under ANSI mode and silent wrapping otherwise. Doubles and decimals retain their original types. Add unit tests covering int overflow behavior and double fraction preservation. --- .../query/solvers/utils/window_bounds.py | 6 ++- .../query/solvers/utils/window_bounds_test.py | 41 +++++++++++++++++++ 2 files changed, 45 insertions(+), 2 deletions(-) diff --git a/src/impulse_query_engine/analyze/query/solvers/utils/window_bounds.py b/src/impulse_query_engine/analyze/query/solvers/utils/window_bounds.py index 5e96696d..37434a94 100644 --- a/src/impulse_query_engine/analyze/query/solvers/utils/window_bounds.py +++ b/src/impulse_query_engine/analyze/query/solvers/utils/window_bounds.py @@ -133,8 +133,10 @@ def _container_to_channel_unit(col: Column, config: SolverConfig) -> Column: source = _NANOS_PER_UNIT[config.container_time_unit] target = _NANOS_PER_UNIT[config.channel_time_unit] if source > target: - # Finer target unit: an integer factor keeps long boundaries exact. - return col * F.lit(source // target) + # Finer target unit: an integer factor keeps long boundaries exact. A long literal + # widens INT boundaries to long (int * int would stay int and overflow, e.g. epoch + # seconds * 1000); doubles and decimals keep their type. + return col * F.lit(source // target).cast(T.LongType()) return col / F.lit(float(target // source)) diff --git a/tests/impulse_query_engine/unit/analyze/query/solvers/utils/window_bounds_test.py b/tests/impulse_query_engine/unit/analyze/query/solvers/utils/window_bounds_test.py index 91b58e75..3a70bb03 100644 --- a/tests/impulse_query_engine/unit/analyze/query/solvers/utils/window_bounds_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/solvers/utils/window_bounds_test.py @@ -146,6 +146,47 @@ def test_numeric_ms_boundaries_converted_to_finer_channel_unit_exactly(spark): assert _bounds(relative, _ms_boundaries_df(spark)) == {1: (0, (stop_ms - start_ms) * 1000)} +@pytest.mark.parametrize("ansi", ["true", "false"]) +def test_int_boundaries_widen_to_long_instead_of_overflowing(spark, ansi): # noqa: F811 + """INT epoch seconds * 1000 exceeds int32. Spark keeps int * int as int, which raised + ARITHMETIC_OVERFLOW under ANSI and silently wrapped to negative bounds without it.""" + int_seconds = ( + _boundaries_df(spark) + .filter(F.col("container_id") == 1) + .select( + "container_id", + F.unix_seconds("start_ts").cast("int").alias("start_ts"), + F.unix_seconds("stop_ts").cast("int").alias("stop_ts"), + ) + ) + start_s = _EPOCH_MICROS // 1_000_000 + stop_s = (_EPOCH_MICROS + _SPAN_MICROS) // 1_000_000 + previous_ansi = spark.conf.get("spark.sql.ansi.enabled") + spark.conf.set("spark.sql.ansi.enabled", ansi) + try: + cfg = SolverConfig(channel_time_unit="ms", container_time_unit="s") + out = with_window_bounds(int_seconds, cfg) + assert out.schema[_START].dataType == T.LongType() + assert _bounds(cfg, int_seconds) == {1: (start_s * 1000, stop_s * 1000)} + + relative = SolverConfig( + channel_time_unit="ms", channel_time_origin="container_start", container_time_unit="s" + ) + assert _bounds(relative, int_seconds) == {1: (0, (stop_s - start_s) * 1000)} + finally: + spark.conf.set("spark.sql.ansi.enabled", previous_ansi) + + +def test_double_boundaries_keep_fractions_when_converted_to_finer_unit(spark): # noqa: F811 + df = spark.createDataFrame( + [(1, 1000.25, 4600.75)], "container_id int, start_ts double, stop_ts double" + ) + cfg = SolverConfig(channel_time_unit="ms", container_time_unit="s") + out = with_window_bounds(df, cfg) + assert out.schema[_START].dataType == T.DoubleType() + assert _bounds(cfg, df) == {1: (1_000_250.0, 4_600_750.0)} + + def test_numeric_boundaries_converted_to_coarser_channel_unit(spark): # noqa: F811 # Boundaries in epoch µs, channels in ms: a division, giving doubles. df = spark.createDataFrame( From e6105d0ec55f58f8b8eb22425d1ae2cb3da89836 Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 7 Oct 2026 18:47:55 +0200 Subject: [PATCH 24/27] docs(reporting): document final-window rounding edge case for fractional window lengths Add a note to the TimeWindowEvent docs explaining that when `window_length` is not a whole number in the channel unit or not a multiple of 256 ns for nanosecond epochs, rounding can produce a final window only a few ulps long. --- docs/impulse/docs/references/report/event.md | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/docs/impulse/docs/references/report/event.md b/docs/impulse/docs/references/report/event.md index 7dca0614..fa4671be 100644 --- a/docs/impulse/docs/references/report/event.md +++ b/docs/impulse/docs/references/report/event.md @@ -281,8 +281,11 @@ above. :::note Window boundaries are stored in `event_instance_fact` as doubles (its `start_ts` / `end_ts` columns), like every other event type. Epoch timestamps in nanoseconds exceed the range doubles -represent exactly, so their window boundaries are rounded to about 256 ns. The event and its aggregations use the same window function, so the -rounding is the same on both sides and their `event_instance_id` values still match. +represent exactly, so their window boundaries are rounded to about 256 ns. The event and its +aggregations use the same window function, so the rounding is the same on both sides and their +`event_instance_id` values still match. If `window_length` is not a whole number in the channel +unit (e.g. `0.2` or `1.7` over timestamps in seconds), or not a multiple of 256 ns for nanosecond +epochs, rounding can add a final window only a few ulps long. ::: ## Event output schema From 903291029e3f6dcd1c0b6fa6297498240c02f801 Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Wed, 7 Oct 2026 18:54:32 +0200 Subject: [PATCH 25/27] test(query-engine): use basic_narrow_db fixture for window_bounds tests Replace hard-coded timestamps in window_bounds tests with the shared basic_narrow_db fixture's container_metrics boundaries. This makes the tests data-driven and consistent with the rest of the suite while still covering TIMESTAMP, numeric, and null boundary cases. --- .../query/solvers/utils/window_bounds_test.py | 250 ++++++++++-------- 1 file changed, 134 insertions(+), 116 deletions(-) diff --git a/tests/impulse_query_engine/unit/analyze/query/solvers/utils/window_bounds_test.py b/tests/impulse_query_engine/unit/analyze/query/solvers/utils/window_bounds_test.py index 3a70bb03..b5ed9401 100644 --- a/tests/impulse_query_engine/unit/analyze/query/solvers/utils/window_bounds_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/solvers/utils/window_bounds_test.py @@ -5,40 +5,59 @@ (``channel_time_unit`` / ``channel_time_origin``). ``with_window_bounds`` derives the container start/stop in that frame as two extra columns and leaves the raw ``start_ts`` / ``stop_ts`` untouched for everyone else (UDFs, ``ContainerEvent``, ``measurement_dimension``). -""" -import datetime as dt +All frames are the ``basic_narrow_db`` fixture's ``container_metrics`` boundaries (epoch-ms +longs), recast per test; container 2's boundaries are nulled to cover missing values. +""" import pyspark.sql.functions as F import pyspark.sql.types as T import pytest -from pyspark.sql import SparkSession +from pyspark.sql import DataFrame, SparkSession from impulse_query_engine.analyze.query.solvers.solver_config import SolverConfig from impulse_query_engine.analyze.query.solvers.utils.window_bounds import with_window_bounds -from tests.conftest import spark # noqa: F401 (pytest fixture) +from impulse_query_engine.measurement_db import MeasurementDB +from tests.conftest import basic_narrow_db, spark # noqa: F401 (pytest fixtures) -# 2025-07-03 07:41:41.483456 UTC -_EPOCH_MICROS = 1_751_528_501_483_456 -# One hour and half a second later. -_SPAN_MICROS = 3_600_500_000 _START, _STOP = "__window_start", "__window_stop" +_NULL_CONTAINER = 2 + + +def _ms_boundaries(spark: SparkSession, db: MeasurementDB) -> DataFrame: # noqa: F811 + """The fixture's container boundaries (epoch-ms longs), container 2's set to null.""" + def unless_null_container(name: str): + return F.when(F.col("container_id") != _NULL_CONTAINER, F.col(name)).alias(name) -def _boundaries_df(spark: SparkSession): # noqa: F811 - """container_metrics-like frame with TIMESTAMP boundaries (and a null row).""" - df = spark.createDataFrame( - [(1, _EPOCH_MICROS, _EPOCH_MICROS + _SPAN_MICROS), (2, None, None)], - "container_id int, start_us long, stop_us long", + return db.container_metrics(spark).select( + "container_id", unless_null_container("start_ts"), unless_null_container("stop_ts") ) + + +def _recast(df: DataFrame, cast) -> DataFrame: + """*df* with ``start_ts`` / ``stop_ts`` passed through *cast* (a Column -> Column).""" return df.select( "container_id", - F.timestamp_micros("start_us").alias("start_ts"), - F.timestamp_micros("stop_us").alias("stop_ts"), + cast(F.col("start_ts")).alias("start_ts"), + cast(F.col("stop_ts")).alias("stop_ts"), ) -def _bounds(cfg: SolverConfig, df) -> dict: +def _timestamp_boundaries(spark: SparkSession, db: MeasurementDB) -> DataFrame: # noqa: F811 + return _recast(_ms_boundaries(spark, db), F.timestamp_millis) + + +def _raw_ms(spark: SparkSession, db: MeasurementDB) -> dict: # noqa: F811 + """``{container_id: (start_ms, stop_ms)}`` of the containers with boundaries.""" + return { + r.container_id: (r.start_ts, r.stop_ts) + for r in _ms_boundaries(spark, db).collect() + if r.start_ts is not None + } + + +def _bounds(cfg: SolverConfig, df: DataFrame) -> dict: out = with_window_bounds(df, cfg) return {r.container_id: (r[_START], r[_STOP]) for r in out.collect()} @@ -49,169 +68,169 @@ def test_window_bound_column_names(): @pytest.mark.parametrize( - "unit, expected_type, expected_start", + "unit, expected_type, from_micros", [ - ("s", T.DoubleType(), _EPOCH_MICROS / 1e6), - ("ms", T.DoubleType(), _EPOCH_MICROS / 1e3), - ("us", T.LongType(), _EPOCH_MICROS), - ("ns", T.LongType(), _EPOCH_MICROS * 1000), + ("s", T.DoubleType(), lambda us: us / 1e6), + ("ms", T.DoubleType(), lambda us: us / 1e3), + ("us", T.LongType(), lambda us: us), + ("ns", T.LongType(), lambda us: us * 1000), ], ) @pytest.mark.parametrize("session_tz", ["UTC", "Europe/Berlin"]) def test_epoch_origin_converts_timestamps_to_unit( - spark, unit, expected_type, expected_start, session_tz # noqa: F811 + spark, basic_narrow_db, unit, expected_type, from_micros, session_tz # noqa: F811 ): previous_tz = spark.conf.get("spark.sql.session.timeZone") spark.conf.set("spark.sql.session.timeZone", session_tz) try: - out = with_window_bounds(_boundaries_df(spark), SolverConfig(channel_time_unit=unit)) - rows = {r.container_id: r for r in out.collect()} + df = _timestamp_boundaries(spark, basic_narrow_db) + out = with_window_bounds(df, SolverConfig(channel_time_unit=unit)) + bounds = {r.container_id: (r[_START], r[_STOP]) for r in out.collect()} finally: spark.conf.set("spark.sql.session.timeZone", previous_tz) assert out.schema[_START].dataType == expected_type - assert rows[1][_START] == expected_start # exact, independent of the session time zone - assert rows[2][_START] is None and rows[2][_STOP] is None + # Exact and independent of the session time zone. + for cid, (start_ms, stop_ms) in _raw_ms(spark, basic_narrow_db).items(): + assert bounds[cid] == (from_micros(start_ms * 1000), from_micros(stop_ms * 1000)) + assert bounds[_NULL_CONTAINER] == (None, None) -def test_epoch_seconds_match_spark_cast_to_double(spark): # noqa: F811 +def test_epoch_seconds_match_spark_cast_to_double(spark, basic_narrow_db): # noqa: F811 # "s" equals Spark's cast(timestamp as double), bit for bit. - df = _boundaries_df(spark) + df = _timestamp_boundaries(spark, basic_narrow_db) casted = { - r.container_id: (r.s, r.e) - for r in df.select( - "container_id", - F.col("start_ts").cast("double").alias("s"), - F.col("stop_ts").cast("double").alias("e"), - ).collect() + r.container_id: (r.start_ts, r.stop_ts) + for r in _recast(df, lambda c: c.cast("double")).collect() } assert _bounds(SolverConfig(channel_time_unit="s"), df) == casted @pytest.mark.parametrize( - "unit, expected_stop", [("s", 3600.5), ("ms", 3_600_500.0), ("us", _SPAN_MICROS)] + "unit, from_micros", + [("s", lambda us: us / 1e6), ("ms", lambda us: us / 1e3), ("us", lambda us: us)], ) @pytest.mark.parametrize("session_tz", ["UTC", "Europe/Berlin"]) def test_container_start_origin_gives_relative_bounds( - spark, unit, expected_stop, session_tz # noqa: F811 + spark, basic_narrow_db, unit, from_micros, session_tz # noqa: F811 ): previous_tz = spark.conf.get("spark.sql.session.timeZone") spark.conf.set("spark.sql.session.timeZone", session_tz) try: cfg = SolverConfig(channel_time_unit=unit, channel_time_origin="container_start") - bounds = _bounds(cfg, _boundaries_df(spark)) + bounds = _bounds(cfg, _timestamp_boundaries(spark, basic_narrow_db)) finally: spark.conf.set("spark.sql.session.timeZone", previous_tz) - assert bounds[1] == (0, expected_stop) + for cid, (start_ms, stop_ms) in _raw_ms(spark, basic_narrow_db).items(): + assert bounds[cid] == (0, from_micros((stop_ms - start_ms) * 1000)) # A null boundary leaves a null stop bound, so the container gets no windows. - assert bounds[2][1] is None + assert bounds[_NULL_CONTAINER][1] is None -def test_numeric_boundaries_epoch_as_is_and_container_start_shifted(spark): # noqa: F811 - df = spark.createDataFrame( - [(1, 1000.5, 4601.0)], "container_id int, start_ts double, stop_ts double" - ) - assert _bounds(SolverConfig(), df) == {1: (1000.5, 4601.0)} +def test_numeric_boundaries_epoch_as_is_and_container_start_shifted( + spark, basic_narrow_db # noqa: F811 +): + df = _ms_boundaries(spark, basic_narrow_db) + raw = _raw_ms(spark, basic_narrow_db) + epoch = _bounds(SolverConfig(), df) + assert all(epoch[cid] == bounds for cid, bounds in raw.items()) # Shift only: numeric boundaries are already in the channels' unit, so no unit is needed. - assert _bounds(SolverConfig(channel_time_origin="container_start"), df) == {1: (0, 3600.5)} - - -def _ms_boundaries_df(spark: SparkSession): # noqa: F811 - """The TIMESTAMP boundaries of _boundaries_df as epoch-ms longs (container 1 only).""" - return ( - _boundaries_df(spark) - .filter(F.col("container_id") == 1) - .select( - "container_id", - F.unix_millis("start_ts").alias("start_ts"), - F.unix_millis("stop_ts").alias("stop_ts"), - ) - ) + relative = _bounds(SolverConfig(channel_time_origin="container_start"), df) + assert all(relative[cid] == (0, stop - start) for cid, (start, stop) in raw.items()) -def test_numeric_ms_boundaries_converted_to_finer_channel_unit_exactly(spark): # noqa: F811 +def test_numeric_ms_boundaries_converted_to_finer_channel_unit_exactly( + spark, basic_narrow_db # noqa: F811 +): # Boundaries in epoch ms, channels in µs: an integer factor keeps the longs exact. + df = _ms_boundaries(spark, basic_narrow_db) + raw = _raw_ms(spark, basic_narrow_db) cfg = SolverConfig(channel_time_unit="us", container_time_unit="ms") - out = with_window_bounds(_ms_boundaries_df(spark), cfg) - start_ms = _EPOCH_MICROS // 1000 - stop_ms = (_EPOCH_MICROS + _SPAN_MICROS) // 1000 - assert out.schema[_START].dataType == T.LongType() - assert _bounds(cfg, _ms_boundaries_df(spark)) == {1: (start_ms * 1000, stop_ms * 1000)} + assert with_window_bounds(df, cfg).schema[_START].dataType == T.LongType() + bounds = _bounds(cfg, df) + assert all(bounds[cid] == (start * 1000, stop * 1000) for cid, (start, stop) in raw.items()) relative = SolverConfig( channel_time_unit="us", channel_time_origin="container_start", container_time_unit="ms" ) # The difference is taken in ms first, then converted. - assert _bounds(relative, _ms_boundaries_df(spark)) == {1: (0, (stop_ms - start_ms) * 1000)} + bounds = _bounds(relative, df) + assert all(bounds[cid] == (0, (stop - start) * 1000) for cid, (start, stop) in raw.items()) @pytest.mark.parametrize("ansi", ["true", "false"]) -def test_int_boundaries_widen_to_long_instead_of_overflowing(spark, ansi): # noqa: F811 +def test_int_boundaries_widen_to_long_instead_of_overflowing( + spark, basic_narrow_db, ansi # noqa: F811 +): """INT epoch seconds * 1000 exceeds int32. Spark keeps int * int as int, which raised ARITHMETIC_OVERFLOW under ANSI and silently wrapped to negative bounds without it.""" - int_seconds = ( - _boundaries_df(spark) - .filter(F.col("container_id") == 1) - .select( - "container_id", - F.unix_seconds("start_ts").cast("int").alias("start_ts"), - F.unix_seconds("stop_ts").cast("int").alias("stop_ts"), - ) + int_seconds = _recast( + _ms_boundaries(spark, basic_narrow_db), lambda c: (c / F.lit(1000)).cast("int") ) - start_s = _EPOCH_MICROS // 1_000_000 - stop_s = (_EPOCH_MICROS + _SPAN_MICROS) // 1_000_000 + raw_s = { + cid: (start // 1000, stop // 1000) + for cid, (start, stop) in _raw_ms(spark, basic_narrow_db).items() + } previous_ansi = spark.conf.get("spark.sql.ansi.enabled") spark.conf.set("spark.sql.ansi.enabled", ansi) try: + assert int_seconds.schema["start_ts"].dataType == T.IntegerType() cfg = SolverConfig(channel_time_unit="ms", container_time_unit="s") - out = with_window_bounds(int_seconds, cfg) - assert out.schema[_START].dataType == T.LongType() - assert _bounds(cfg, int_seconds) == {1: (start_s * 1000, stop_s * 1000)} + assert with_window_bounds(int_seconds, cfg).schema[_START].dataType == T.LongType() + bounds = _bounds(cfg, int_seconds) + assert all(bounds[cid] == (s * 1000, e * 1000) for cid, (s, e) in raw_s.items()) relative = SolverConfig( channel_time_unit="ms", channel_time_origin="container_start", container_time_unit="s" ) - assert _bounds(relative, int_seconds) == {1: (0, (stop_s - start_s) * 1000)} + bounds = _bounds(relative, int_seconds) + assert all(bounds[cid] == (0, (e - s) * 1000) for cid, (s, e) in raw_s.items()) finally: spark.conf.set("spark.sql.ansi.enabled", previous_ansi) -def test_double_boundaries_keep_fractions_when_converted_to_finer_unit(spark): # noqa: F811 - df = spark.createDataFrame( - [(1, 1000.25, 4600.75)], "container_id int, start_ts double, stop_ts double" - ) +def test_double_boundaries_keep_fractions_when_converted_to_finer_unit( + spark, basic_narrow_db # noqa: F811 +): + # Seconds as doubles (the fixture's ms values / 1000, so with a fractional part). + seconds = _recast(_ms_boundaries(spark, basic_narrow_db), lambda c: c / F.lit(1000.0)) cfg = SolverConfig(channel_time_unit="ms", container_time_unit="s") - out = with_window_bounds(df, cfg) - assert out.schema[_START].dataType == T.DoubleType() - assert _bounds(cfg, df) == {1: (1_000_250.0, 4_600_750.0)} + assert with_window_bounds(seconds, cfg).schema[_START].dataType == T.DoubleType() + bounds = _bounds(cfg, seconds) + for cid, (start_ms, stop_ms) in _raw_ms(spark, basic_narrow_db).items(): + assert bounds[cid] == ((start_ms / 1000.0) * 1000, (stop_ms / 1000.0) * 1000) + # Not truncated to whole seconds before the conversion. + assert bounds[cid][0] != (start_ms // 1000) * 1000 -def test_numeric_boundaries_converted_to_coarser_channel_unit(spark): # noqa: F811 +def test_numeric_boundaries_converted_to_coarser_channel_unit( + spark, basic_narrow_db # noqa: F811 +): # Boundaries in epoch µs, channels in ms: a division, giving doubles. - df = spark.createDataFrame( - [(1, _EPOCH_MICROS, _EPOCH_MICROS + _SPAN_MICROS)], - "container_id int, start_ts long, stop_ts long", - ) + micros = _recast(_ms_boundaries(spark, basic_narrow_db), lambda c: c * F.lit(1000)) cfg = SolverConfig(channel_time_unit="ms", container_time_unit="us") - out = with_window_bounds(df, cfg) - assert out.schema[_START].dataType == T.DoubleType() - assert _bounds(cfg, df) == { - 1: (_EPOCH_MICROS / 1000.0, (_EPOCH_MICROS + _SPAN_MICROS) / 1000.0) - } + assert with_window_bounds(micros, cfg).schema[_START].dataType == T.DoubleType() + bounds = _bounds(cfg, micros) + for cid, (start_ms, stop_ms) in _raw_ms(spark, basic_narrow_db).items(): + assert bounds[cid] == (float(start_ms), float(stop_ms)) -def test_numeric_boundaries_unchanged_for_equal_or_unset_container_unit(spark): # noqa: F811 - df = _ms_boundaries_df(spark) +def test_numeric_boundaries_unchanged_for_equal_or_unset_container_unit( + spark, basic_narrow_db # noqa: F811 +): + df = _ms_boundaries(spark, basic_narrow_db) raw = {r.container_id: (r.start_ts, r.stop_ts) for r in df.collect()} assert _bounds(SolverConfig(channel_time_unit="ms", container_time_unit="ms"), df) == raw assert _bounds(SolverConfig(channel_time_unit="us"), df) == raw -def test_container_time_unit_rejected_for_timestamp_boundaries(spark): # noqa: F811 +def test_container_time_unit_rejected_for_timestamp_boundaries( + spark, basic_narrow_db # noqa: F811 +): cfg = SolverConfig(channel_time_unit="s", container_time_unit="ms") with pytest.raises(ValueError, match="container_time_unit only applies to numeric"): - with_window_bounds(_boundaries_df(spark), cfg) + with_window_bounds(_timestamp_boundaries(spark, basic_narrow_db), cfg) def test_container_time_unit_requires_channel_time_unit(): @@ -221,38 +240,37 @@ def test_container_time_unit_requires_channel_time_unit(): assert (cfg.channel_time_unit, cfg.container_time_unit) == ("us", "ms") -def test_raw_boundaries_stay_unchanged(spark): # noqa: F811 - df = _boundaries_df(spark) +def test_raw_boundaries_stay_unchanged(spark, basic_narrow_db): # noqa: F811 + df = _timestamp_boundaries(spark, basic_narrow_db) cfg = SolverConfig(channel_time_unit="s", channel_time_origin="container_start") out = with_window_bounds(df, cfg) assert isinstance(out.schema["start_ts"].dataType, T.TimestampType) assert out.select("container_id", "start_ts", "stop_ts").collect() == df.collect() -def test_timestamp_boundaries_without_unit_rejected(spark): # noqa: F811 +def test_timestamp_boundaries_without_unit_rejected(spark, basic_narrow_db): # noqa: F811 for origin in ("epoch", "container_start"): with pytest.raises(ValueError, match=r"TimeWindowEvent.*channel_time_unit"): - with_window_bounds(_boundaries_df(spark), SolverConfig(channel_time_origin=origin)) + with_window_bounds( + _timestamp_boundaries(spark, basic_narrow_db), + SolverConfig(channel_time_origin=origin), + ) -@pytest.mark.parametrize( - "value, ddl", - [(dt.datetime(2025, 7, 3, 7, 41, 41), "timestamp_ntz"), (dt.date(2025, 7, 3), "date")], -) -def test_zone_less_types_rejected(spark, value, ddl): # noqa: F811 - df = spark.createDataFrame( - [(1, value, value)], f"container_id int, start_ts {ddl}, stop_ts {ddl}" - ) +@pytest.mark.parametrize("zone_less_type", ["timestamp_ntz", "date"]) +def test_zone_less_types_rejected(spark, basic_narrow_db, zone_less_type): # noqa: F811 + df = _recast(_timestamp_boundaries(spark, basic_narrow_db), lambda c: c.cast(zone_less_type)) with pytest.raises(ValueError, match="start_ts"): with_window_bounds(df, SolverConfig(channel_time_unit="s")) -def test_mixed_and_missing_boundaries_rejected(spark): # noqa: F811 - mixed = _boundaries_df(spark).withColumn("stop_ts", F.lit(1.0)) +def test_mixed_and_missing_boundaries_rejected(spark, basic_narrow_db): # noqa: F811 + timestamps = _timestamp_boundaries(spark, basic_narrow_db) + mixed = timestamps.withColumn("stop_ts", F.unix_millis("stop_ts")) with pytest.raises(ValueError, match="both be TIMESTAMP or both be numeric"): with_window_bounds(mixed, SolverConfig(channel_time_unit="s")) with pytest.raises(ValueError, match="stop_ts"): - with_window_bounds(_boundaries_df(spark).drop("stop_ts"), SolverConfig()) + with_window_bounds(timestamps.drop("stop_ts"), SolverConfig()) def test_channel_time_settings_validated(): From e425ec060dbe91ee76f6708bf4385ff7d5181ec7 Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Thu, 8 Oct 2026 10:52:56 +0200 Subject: [PATCH 26/27] fix(query-engine, reporting): return struct of starts/ends from window_intervals_udf Change the pandas UDF return type from `array>` to `struct, ends: array>` so each row reliably carries one array per column, even when all rows have the same window count. Update `TimeWindowEvent.determine_events` to zip the struct fields and adjust unit tests accordingly. --- .../query/events/time_window_expression.py | 30 +++++++++++-------- .../events/time_window_event.py | 12 ++++---- .../events/time_window_expression_test.py | 25 +++++++++++++--- 3 files changed, 45 insertions(+), 22 deletions(-) diff --git a/src/impulse_query_engine/analyze/query/events/time_window_expression.py b/src/impulse_query_engine/analyze/query/events/time_window_expression.py index 880571f0..3c7072bd 100644 --- a/src/impulse_query_engine/analyze/query/events/time_window_expression.py +++ b/src/impulse_query_engine/analyze/query/events/time_window_expression.py @@ -131,9 +131,8 @@ def tile_windows( def window_intervals_udf(window_length: float, max_windows: int = MAX_WINDOWS_PER_CONTAINER): """Scalar pandas UDF giving each container's windows via :func:`tile_windows`. - Used by the reporting ``TimeWindowEvent`` for its event fact, on one row per container - (its plan reads only ``container_metrics`` / ``container_tags``, never the channels - table), so the event fact and the solve share one window implementation. + Used by the reporting ``TimeWindowEvent`` for its event fact (one row per container), so + the event fact and the solve share one window implementation. Parameters ---------- @@ -147,20 +146,25 @@ def window_intervals_udf(window_length: float, max_windows: int = MAX_WINDOWS_PE Returns ------- callable - A pandas UDF ``(start, stop) -> array>`` with one ``[start, end]`` pair - per window, in order; empty when a bound is null, NaN or infinite, or the span is not - strictly positive. + A pandas UDF ``(start, stop) -> struct, ends: array>`` + with the window starts and ends, in order; empty when a bound is null, NaN or + infinite, or the span is not strictly positive. """ window_length = float(window_length) max_windows = validate_max_windows(max_windows) - @F.pandas_udf("array>") - def windows(start: pd.Series, stop: pd.Series) -> pd.Series: - return pd.Series( - [ - np.column_stack(tile_windows(s, e, window_length, max_windows)).tolist() - for s, e in zip(start, stop, strict=True) - ] + @F.pandas_udf("struct, ends: array>") + def windows(start: pd.Series, stop: pd.Series) -> pd.DataFrame: + pairs = [ + tile_windows(s, e, window_length, max_windows) + for s, e in zip(start, stop, strict=True) + ] + # object dtype keeps one array per row, also when all rows have equal window counts. + return pd.DataFrame( + { + "starts": pd.Series([p[0] for p in pairs], dtype=object), + "ends": pd.Series([p[1] for p in pairs], dtype=object), + } ) return windows diff --git a/src/impulse_reporting/events/time_window_event.py b/src/impulse_reporting/events/time_window_event.py index 10f77b8d..38367081 100644 --- a/src/impulse_reporting/events/time_window_event.py +++ b/src/impulse_reporting/events/time_window_event.py @@ -217,8 +217,7 @@ def determine_events( stop_ts = f.col(solver.config.window_stop_col) # One (event_name, windows) struct per event, exploded in a single pass over the - # containers. The windows UDF only sees this container-level plan, never the - # channels table. + # containers. per_event = f.array( *[ f.struct( @@ -239,10 +238,13 @@ def determine_events( .select( "container_id", f.col("event.event_name").alias("event_name"), - f.explode(f.col("event.windows")).alias("event_instance"), + f.inline( + f.arrays_zip( + f.col("event.windows.starts").alias("start_ts"), + f.col("event.windows.ends").alias("end_ts"), + ) + ), ) - .withColumn("start_ts", f.col("event_instance").getItem(0)) - .withColumn("end_ts", f.col("event_instance").getItem(1)) .withColumn( "event_instance_id", generate_event_instance_id_column(event_type=TimeWindowEvent), diff --git a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py index b7b27139..3e946121 100644 --- a/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/events/time_window_expression_test.py @@ -208,10 +208,13 @@ def test_tile_windows_none_and_na_yield_empty(): # window_intervals_udf: tile_windows on the event fact side (one row per container) # --------------------------------------------------------------------------- def _udf_windows(spark, rows, window_length, ts_type="long", **kwargs): # noqa: F811 - df = spark.createDataFrame(rows, f"k int, start_ts {ts_type}, stop_ts {ts_type}") + # One partition, so all rows reach the UDF in one Arrow batch. + df = spark.createDataFrame(rows, f"k int, start_ts {ts_type}, stop_ts {ts_type}").coalesce(1) windows = window_intervals_udf(window_length, **kwargs) out = df.select("k", windows(F.col("start_ts"), F.col("stop_ts")).alias("w")) - return out, {r.k: [list(p) for p in r.w] for r in out.collect()} + return out, { + r.k: [[s, e] for s, e in zip(r.w.starts, r.w.ends, strict=True)] for r in out.collect() + } def test_window_intervals_udf_edge_cases(spark): # noqa: F811 @@ -227,7 +230,9 @@ def test_window_intervals_udf_edge_cases(spark): # noqa: F811 ] out, w = _udf_windows(spark, rows, 10) - assert out.schema["w"].dataType.simpleString() == "array>" + assert out.schema["w"].dataType.simpleString() == ( + "struct,ends:array>" + ) assert w[0] == [[float(s), float(s + 10)] for s in range(0, 100, 10)] assert w[1][-1] == [100.0, 105.0] and len(w[1]) == 11 assert w[2] == [[1000.0, 1010.0]] @@ -254,6 +259,15 @@ def test_window_intervals_udf_invalid_max_windows_raises(): window_intervals_udf(10, max_windows=0) +def test_window_intervals_udf_uniform_and_empty_batches(spark): # noqa: F811 + # Equal window counts across a batch still give one array per row. + _, w = _udf_windows(spark, [(k, 100 * k, 100 * k + 30) for k in range(4)], 10) + assert w == {k: [[100.0 * k + s, 100.0 * k + s + 10] for s in (0, 10, 20)] for k in range(4)} + # A batch without any windows. + _, w = _udf_windows(spark, [(0, None, None), (1, 50, 50)], 10) + assert w == {0: [], 1: []} + + def _as_list(windows) -> list[tuple[float, float]]: """Windows as ordered (start, end) pairs.""" return [(float(s), float(e)) for s, e in windows] @@ -315,7 +329,10 @@ def _event_fact(batch_rows) -> tuple[dict, set]: windows(F.col("start_ts"), F.col("stop_ts")).alias("w"), batch_dtype(F.col("start_ts")).alias("dtype"), ).collect() - return {r.k: _as_list(r.w) for r in out if r.k >= 0}, {r.dtype for r in out} + per_container = { + r.k: _as_list(zip(r.w.starts, r.w.ends, strict=True)) for r in out if r.k >= 0 + } + return per_container, {r.dtype for r in out} without_nulls, dtypes_int = _event_fact(rows) with_null, dtypes_float = _event_fact([*rows, (-1, None, None)]) From 986c125804c74d059183745f7cd2d875f3ad032c Mon Sep 17 00:00:00 2001 From: "tom.bonfert" Date: Thu, 8 Oct 2026 12:29:05 +0200 Subject: [PATCH 27/27] fix(query-engine): improve missing container_metrics column error message When `with_window_bounds` cannot find required `container_metrics` columns, include the unmapped physical column name in the error and point users to `solver_config.container_metrics.column_name_mapping`. Add a unit test verifying the message for an unmapped `stop_ts` physical name. --- .../analyze/query/solvers/utils/window_bounds.py | 4 +++- .../unit/analyze/query/solvers/utils/window_bounds_test.py | 6 ++++++ 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/src/impulse_query_engine/analyze/query/solvers/utils/window_bounds.py b/src/impulse_query_engine/analyze/query/solvers/utils/window_bounds.py index 37434a94..068d2024 100644 --- a/src/impulse_query_engine/analyze/query/solvers/utils/window_bounds.py +++ b/src/impulse_query_engine/analyze/query/solvers/utils/window_bounds.py @@ -63,7 +63,9 @@ def with_window_bounds(df: DataFrame, config: SolverConfig) -> DataFrame: if missing: raise ValueError( f"TimeWindowEvent needs the container_metrics columns {missing} to compute " - f"its windows. Available columns: {df.columns}" + f"its windows. Available columns: {df.columns}. If they have other physical " + "names, map them via " + "query_engine.solver_config.container_metrics.column_name_mapping." ) for name, dtype in types.items(): if isinstance(dtype, (T.TimestampNTZType, T.DateType)): diff --git a/tests/impulse_query_engine/unit/analyze/query/solvers/utils/window_bounds_test.py b/tests/impulse_query_engine/unit/analyze/query/solvers/utils/window_bounds_test.py index b5ed9401..1254833e 100644 --- a/tests/impulse_query_engine/unit/analyze/query/solvers/utils/window_bounds_test.py +++ b/tests/impulse_query_engine/unit/analyze/query/solvers/utils/window_bounds_test.py @@ -271,6 +271,12 @@ def test_mixed_and_missing_boundaries_rejected(spark, basic_narrow_db): # noqa: with_window_bounds(mixed, SolverConfig(channel_time_unit="s")) with pytest.raises(ValueError, match="stop_ts"): with_window_bounds(timestamps.drop("stop_ts"), SolverConfig()) + # An unmapped physical name: the error lists it and points to the column mapping. + unmapped = timestamps.withColumnRenamed("stop_ts", "measurement_end") + with pytest.raises( + ValueError, match=r"'measurement_end'.*container_metrics\.column_name_mapping" + ): + with_window_bounds(unmapped, SolverConfig(channel_time_unit="s")) def test_channel_time_settings_validated():