From 0611b2e9b866febb186a0c659a88f9b1a547458a Mon Sep 17 00:00:00 2001 From: Jeff Brennan Date: Sat, 12 Sep 2026 20:20:47 -0400 Subject: [PATCH] feat(event-logs): coverage for partial, rolled, and retried logs Implements brief 04, increments 1-3. Plan fallback and partial parsing: parse_source() streams events and requires neither ApplicationStart nor an adaptive update. Plans rank final adaptive > in-progress adaptive > the plan on SQLExecutionStart, so a non-AQE workload no longer yields "No queries found". A bad line is corrupt_line when more lines follow and truncated_tail when it is last; strict mode raises with URI and line. Missing task metrics stay None, unfinished queries and jobs keep null end timestamps, unknown events are counted. Identity and attempts: stage attempts are keyed by (stage_id, stage_attempt_id); task status, failure reason and partition id reach the task frame. Job/stage and query/stage relations move to association tables so they cannot duplicate task rows -- nested_final_plans was reporting 31 rows for 19 tasks, double-counting 12. Output totals count retained outputs (one attempt per partition, resolving speculative races and recomputed partitions); resource totals count every attempt. Source discovery and streaming: new sparkparse/eventlog.py discovers logical sources, collapsing rolled eventlog_v2_* directories, flagging .inprogress and ignoring markers. Explicit selection by name, app id, rolling dir or path, with a documented newest-mtime default; new `sparkparse logs` and `get --all-apps`. Readable codecs are none, gz and zstd; Spark's Java-framed lz4/lzf/snappy fail by name. Reads never copy a log whole, asserted directly and benchmarked by tests/benchmark_ingestion.py. Golden fixtures are regenerated: more queries parse, duplicate task rows are gone, and accumulator ordering is now deterministic. --- CLAUDE.md | 53 +- README.md | 26 +- plans/improvements-2026-09/04-event-logs.md | 3 +- plans/improvements-2026-09/README.md | 2 +- pyproject.toml | 4 + sparkparse/analyze.py | 83 +- sparkparse/app.py | 69 +- sparkparse/capture.py | 24 +- sparkparse/clean.py | 383 +- sparkparse/connect.py | 83 +- sparkparse/eventlog.py | 296 ++ sparkparse/history.py | 17 +- sparkparse/models.py | 119 +- sparkparse/pages/home.py | 66 +- sparkparse/parse.py | 570 ++- sparkparse/schemas.py | 81 + tests/benchmark_ingestion.py | 69 + .../expected_nested_final_plans.json | 3200 +++++++---------- .../expected_nested_loop_join.json | 1446 +++++--- tests/eventlog_fixtures.py | 168 + tests/test_capture_contract.py | 4 +- tests/test_eventlog.py | 357 ++ tests/test_parse_partial.py | 601 ++++ uv.lock | 82 +- 24 files changed, 5262 insertions(+), 2544 deletions(-) create mode 100644 sparkparse/eventlog.py create mode 100644 sparkparse/schemas.py create mode 100644 tests/benchmark_ingestion.py create mode 100644 tests/eventlog_fixtures.py create mode 100644 tests/test_eventlog.py create mode 100644 tests/test_parse_partial.py diff --git a/CLAUDE.md b/CLAUDE.md index fa3bb24..65f5fb4 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -15,6 +15,9 @@ Dash dashboard for identifying performance bottlenecks. Spark event log (JSONL) │ ▼ + sparkparse/eventlog.py – discover logical log sources (rolled, compressed, + │ in-progress) and stream their events line by line + ▼ sparkparse/parse.py – parse raw log into ParsedLog (Pydantic model) │ ▼ @@ -36,7 +39,9 @@ Spark event log (JSONL) | File | Purpose | | ----------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------- | | `sparkparse/models.py` | All Pydantic models. `NodeType` enum (30+ values), `ParsedLog`, `ParsedLogDataFrames`, `NODE_TYPE_DETAIL_MAP` (node type → detail model class) | -| `sparkparse/parse.py` | `get_parsed_metrics()` is the main entry point. `parse_spark_ui_tree()` converts indented ASCII plans to node graphs. `parse_log()` orchestrates everything. | +| `sparkparse/parse.py` | `get_parsed_metrics()` is the main entry point (`get_all_parsed_metrics()` for every application in a directory). `parse_source()` parses one logical log incrementally; `parse_spark_ui_tree()` converts indented ASCII plans to node graphs. | +| `sparkparse/eventlog.py`| Event-log source discovery and streaming. `discover_sources()` collapses rolled `eventlog_v2_*` directories into one source and skips markers/checksums; `resolve_source()` implements explicit selection and the newest-source default; `iter_lines()` streams segments. Readable codecs: none, `zstd`, `gz`. | +| `sparkparse/schemas.py` | Canonical typed empty frame schemas (`DAG_SCHEMA`, `COMBINED_SCHEMA`) shared by the event-log and Connect paths. | | `sparkparse/clean.py` | `log_to_dag_df()` and `log_to_combined_df()` produce the two output DataFrames. `get_readable_size()` and `get_readable_timing()` are Polars expression helpers. | | `sparkparse/app.py` | Typer CLI. `get` → parses and writes output files. `viz` → launches dashboard. | | `sparkparse/connect.py` | Spark Connect adapter. Intercepts client action boundaries and `_build_metrics`, attributes metrics per execution/thread, and builds the dag frame. `probe_connect_support()` reports the client surface; `SparkConnectCapture.from_plan_metrics()` replays recorded executions offline. | @@ -56,14 +61,53 @@ Spark event log (JSONL) ### `combined` DataFrame columns (per task) -- `log_name`, `parsed_log_name`, `query_id`, `query_function` +- `log_name`, `parsed_log_name`, `query_id`, `query_function`, `query_count` - `query/job/stage/task` start/end timestamps and duration_seconds -- `task_id`, `executor_id`, `nodes` (list of physical plan nodes for this task) +- `stage_attempt_id`, `stage_num_tasks`, `stage_status`, `stage_failure_reason` +- `task_id`, `partition_id`, `attempt`, `task_status`, `task_succeeded`, + `task_failure_reason`, `executor_id`, `nodes` (plan nodes this task fed) - Executor metrics: `executor_run_time_seconds`, `executor_cpu_time_seconds`, `jvm_gc_time_seconds`, `peak_execution_memory_bytes` - Input/output: `bytes_read`, `records_read`, `bytes_written`, `records_written` - Shuffle: `shuffle_remote_bytes_read`, `shuffle_local_bytes_read`, `shuffle_bytes_written` - Spill: `memory_bytes_spilled`, `disk_bytes_spilled` +### Attempt and association contract + +- One row per task **attempt**, keyed by `(stage_id, stage_attempt_id, task_id)`. + A retried stage keeps both attempts; keying on `stage_id` alone would fold a + retry's metrics into the original. +- Resource usage (`executor_run_time_seconds`, spill, GC) counts every attempt; + output accounting (`bytes_read/written`, `records_*`, shuffle bytes) counts + only *retained* outputs. Success alone is not enough: a losing speculative + copy and a partition recomputed in a later stage attempt both end + successfully, so `analyze.retained_outputs()` keeps one attempt per + `(stage_id, partition)` — latest stage attempt, earliest finish within it. + `to_plan_summary()["total_basis"]` records which basis each total used. +- A stage can belong to several jobs and serve several queries. Those relations + live in `ParsedLogDataFrames.job_stage` / `.query_stage`; joining them into + `combined` would duplicate task rows and double-count their metrics. The task + frame carries the earliest attributed query plus `query_count`. +- Tasks no SQL execution claims (schema inference, RDD work) stay in `combined` + with a null `query_id` rather than being dropped. + +### Event-log source contract + +- `parse_source()` requires neither `SparkListenerApplicationStart` nor an + adaptive execution update. The plan comes from the strongest event seen: + final adaptive plan > in-progress adaptive plan > the plan on + `SQLExecutionStart`. A non-AQE workload is normal, not an error. +- A line that fails to decode is `corrupt_line` when more lines follow and + `truncated_tail` when it is the last line of the last segment. Tolerant mode + records both in `ParsedLog.diagnostics`; `strict=True` raises with URI and + line number. +- Missing data stays missing: a task with no `Task Metrics` has `metrics=None`, + a query with no end event has a null end timestamp and duration. +- Reads are incremental: segments stream line by line and the log text is never + copied whole (`tests/test_eventlog.py` asserts no `read()`/`readlines()` on a + log handle). Peak RSS still grows with retained model state — roughly 36 KB + per task on the recorded fixtures. `python -m tests.benchmark_ingestion + [log_file]` reports elapsed time and peak RSS for a single log. + ### Metric and findings contract - Raw operator metrics are normalized through `sparkparse/metrics.py`. Use @@ -204,6 +248,9 @@ uv run pyrefly check sparkparse/ tests/ # type check ## Known quirks +- Spark's `lz4`, `lzf` and `snappy` event-log codecs use Java-specific block + framing that no Python codec reads; those raise `UnsupportedCodecError` naming + the codec. `zstd` needs the `zstandard` package (`sparkparse[zstd]`). - `capture.py` borrows supplied SparkSessions and requires event logging to be enabled before capture. Use `cap.spark` inside the context; opt into an owned session explicitly. - `test.py` and `test_capture.py` in `tests/` are integration tests that spin up a local diff --git a/README.md b/README.md index 760f8a8..4582c9c 100644 --- a/README.md +++ b/README.md @@ -25,9 +25,18 @@ pip install sparkparse ### CLI ```bash -# parse logs and write output files +# list the event-log sources found in a directory +sparkparse logs ./logs + +# parse the newest log and write output files sparkparse get --log-dir ./logs --out-format parquet +# parse one application explicitly (file name, rolling-log dir, or app id) +sparkparse get ./logs --log-file eventlog_v2_app-20260912-0001 + +# parse every application in the directory +sparkparse get ./logs --all-apps + # launch the dashboard sparkparse viz --log-dir ./logs @@ -57,6 +66,21 @@ sparkparse should own a local session, opt in explicitly with with unavailable task/stage telemetry called out in `result.capabilities` rather than represented as zeroes. +### event logs + +Rolled logs (`spark.eventLog.rolling.enabled=true`) are read as one logical source +with ordered segments; `.inprogress` logs are read as far as they go and reported as +incomplete. Segments are streamed line by line, never copied into memory whole. +Zstd-compressed logs need `sparkparse[zstd]`; Spark's `lz4`, `lzf` and `snappy` +codecs use Java-specific framing that Python cannot decode, and fail with an error +naming the codec. + +Neither `SparkListenerApplicationStart` nor adaptive execution is required. A +non-AQE query keeps the plan from its `SQLExecutionStart` event, a query with no +end event keeps a null duration, and a truncated final line is reported as +truncation rather than corruption. `strict=True` turns those diagnostics into +errors. + Use `backend="classic"` or `backend="connect"` to override detection. For post-run ingestion without a Spark session, use `backend="event_log", log_file="/path/to/log"`. Borrowed captures select the current application's log; use an explicit `log_file` diff --git a/plans/improvements-2026-09/04-event-logs.md b/plans/improvements-2026-09/04-event-logs.md index c416dd4..f31ccf8 100644 --- a/plans/improvements-2026-09/04-event-logs.md +++ b/plans/improvements-2026-09/04-event-logs.md @@ -1,6 +1,7 @@ # 04 — Event-log coverage and scalable ingestion -Status: proposed. Priority: P1, with non-AQE support an early correctness fix. +Status: implemented offline (increments 1-3); live rolled/compressed Databricks +validation outstanding. Priority: P1, with non-AQE support an early correctness fix. Depends on 01's identity/partial-result contract. Files: `parse.py`, `clean.py`, `models.py`, `storage.py` and event-log fixtures. diff --git a/plans/improvements-2026-09/README.md b/plans/improvements-2026-09/README.md index 09f60e4..084d823 100644 --- a/plans/improvements-2026-09/README.md +++ b/plans/improvements-2026-09/README.md @@ -24,7 +24,7 @@ access mode. Spark Connect is a transport, not proof that compute is serverless. | P0 | [01 — Capture and capabilities](01-capture-and-capabilities.md) | Safe session lifecycle, common finalization, explicit missing data | Implemented + smoke-tested | | P0 | [02 — Connect correctness](02-connect-correctness.md) | Reliable query attribution and tolerant operator handling | Implemented offline; live validation outstanding | | P0 | [03 — Analysis correctness and depth](03-analysis.md) | Accurate metrics and evidence-based findings | Implemented offline; increments 1–3 | -| P1 | [04 — Event-log robustness](04-event-logs.md) | Non-AQE, partial, rolled, retried, and larger workloads | Large; 01 identities | +| P1 | [04 — Event-log robustness](04-event-logs.md) | Non-AQE, partial, rolled, retried, and larger workloads | Implemented offline; increments 1–3 | | P1 | [05 — Developer experience and validation](05-developer-experience.md) | Installable CLI, reproducible checks, serverless-friendly reports | Medium; packaging can start immediately | | P1 | [06 — History and comparisons](06-history.md) | Comparable runs and meaningful regression alerts | Medium; 01 and 03 | diff --git a/pyproject.toml b/pyproject.toml index 0edd3cc..f3b542f 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -27,6 +27,9 @@ azure = ["adlfs>=2024.1.0"] gcs = ["gcsfs>=2024.1.0"] cloud = ["s3fs>=2024.1.0", "adlfs>=2024.1.0", "gcsfs>=2024.1.0"] delta = ["deltalake>=0.25"] +# Zstd-compressed event logs. Spark's lz4/lzf/snappy codecs use Java-specific +# framing and cannot be read from Python at all. +zstd = ["zstandard>=0.22"] [tool.hatch.build.targets.wheel] @@ -50,6 +53,7 @@ dev-dependencies = [ "pyspark>=3.5.5", "pytest>=8.3.5", "ruff>=0.9.9", + "zstandard>=0.25.0", ] [tool.ruff] diff --git a/sparkparse/analyze.py b/sparkparse/analyze.py index 4e7062e..2fff4ba 100644 --- a/sparkparse/analyze.py +++ b/sparkparse/analyze.py @@ -446,6 +446,78 @@ def expr(self, value: Any) -> Any: _COMPACT_DEFAULT_NODES = 25 +# Output that a failed or killed attempt produced is discarded and recomputed, +# so counting it would inflate the ledger. The resources that attempt burned +# were still spent, so those are counted for every attempt. +OUTPUT_ACCOUNTING_COLUMNS: frozenset[str] = frozenset( + { + "bytes_read", + "records_read", + "bytes_written", + "records_written", + "shuffle_bytes_read", + "shuffle_bytes_written", + } +) + + +def _partition_key(columns: set[str]) -> pl.Expr | None: + """Expression identifying the partition a task attempt computed.""" + available = [name for name in ("partition_id", "index") if name in columns] + if not available: + return None + return pl.coalesce([pl.col(name) for name in available]).alias("_partition") + + +def retained_outputs(combined: pl.DataFrame) -> pl.DataFrame: + """Rows for the task attempts whose output was actually kept. + + Success is necessary but not sufficient. Two successful attempts can exist + for the same partition — a speculative copy that finished after the commit + was already awarded, or a partition recomputed in a later stage attempt + after its output was lost — and only one of them contributes bytes and + rows. Counting both inflates every output total. + + One attempt survives per ``(stage_id, partition)``: the latest stage + attempt (its output supersedes the lost one), and within it the attempt + that finished first, which is the one Spark's commit coordinator would have + authorized. Every attempt stays in ``combined`` for resource accounting. + + Sources that do not report task status (Spark Connect plan metrics) have no + ``task_succeeded`` column; every row they do report is treated as kept. + """ + columns = set(combined.columns) + if "task_succeeded" not in columns: + return combined + + kept = combined.filter(pl.col("task_succeeded").fill_null(True)) + partition = _partition_key(columns) + if partition is None or "stage_id" not in columns or kept.is_empty(): + return kept + + order = [("stage_id", False), ("_partition", False)] + if "stage_attempt_id" in columns: + order.append(("stage_attempt_id", True)) + if "task_end_timestamp" in columns: + order.append(("task_end_timestamp", False)) + if "task_id" in columns: + order.append(("task_id", False)) + + return ( + kept.with_columns(partition) + .sort( + [name for name, _ in order], + descending=[descending for _, descending in order], + nulls_last=True, + ) + .unique(subset=["stage_id", "_partition"], keep="first", maintain_order=True) + .drop("_partition") + ) + + +def total_basis(column: str) -> str: + return "retained_outputs" if column in OUTPUT_ACCOUNTING_COLUMNS else "all_attempts" + def _safe_node_name(row: dict[str, Any], redactor: Redactor) -> str | None: """Return the node's display name, redacted when it carries workload text. @@ -606,9 +678,11 @@ def to_plan_summary( # An empty task table is a valid Connect result, not evidence of zero work. agg: dict[str, Any] = dict.fromkeys(_TOTAL_COLUMNS) else: - agg = combined.select( - *(pl.sum(column).alias(column) for column in _TOTAL_COLUMNS) - ).row(0, named=True) + kept = retained_outputs(combined) + agg = {} + for column in _TOTAL_COLUMNS: + frame = kept if column in OUTPUT_ACCOUNTING_COLUMNS else combined + agg[column] = frame.select(pl.sum(column)).item() if frame.height else None summary: dict[str, Any] = { "schema_version": SUMMARY_SCHEMA_VERSION, @@ -618,6 +692,9 @@ def to_plan_summary( "total_units": { column: _TOTAL_UNITS[column].value for column in _TOTAL_COLUMNS }, + # Which attempts each total covers: resource usage counts every + # attempt, output accounting counts only the attempts that were kept. + "total_basis": {column: total_basis(column) for column in _TOTAL_COLUMNS}, "coverage": capabilities.model_dump(mode="json"), } diff --git a/sparkparse/app.py b/sparkparse/app.py index 336e379..33976f9 100644 --- a/sparkparse/app.py +++ b/sparkparse/app.py @@ -9,8 +9,9 @@ from sparkparse import alerts, history from sparkparse.analyze import to_analysis_export, to_plan_summary from sparkparse.dashboard import init_dashboard, run_app +from sparkparse.eventlog import discover_sources from sparkparse.models import OutputFormat, ParsedLogDataFrames, RunRecord -from sparkparse.parse import get_parsed_metrics +from sparkparse.parse import get_all_parsed_metrics, get_parsed_metrics from sparkparse.storage import ( get_path_name, get_path_stem, @@ -75,8 +76,19 @@ def get( ] = "data/logs/raw", log_file: Annotated[ str | None, - typer.Option(help="Parse a single log file instead of the whole directory."), + typer.Option( + help="Event log to parse: file name, rolling-log directory, " + "application id, or full path. Default: the newest source in " + "log_dir (most recent mtime, name order when there is none)." + ), ] = None, + all_apps: Annotated[ + bool, + typer.Option( + "--all-apps", + help="Parse every application in log_dir instead of just the newest.", + ), + ] = False, out_dir: Annotated[ str | None, typer.Option(help="Directory to write parsed output files.") ] = "data/logs/parsed", @@ -94,6 +106,22 @@ def get( ] = False, ) -> ParsedLogDataFrames: """Parse Spark event logs and write structured DataFrames to disk.""" + if all_apps: + if log_file is not None: + raise typer.BadParameter("--all-apps cannot be combined with --log-file") + results = get_all_parsed_metrics( + log_dir=log_dir, + out_dir=out_dir, + out_format=out_format, + verbose=verbose, + strict=strict, + ) + # Query ids restart per application, so the results stay separate. The + # command returns the newest one for interactive use; every one of them + # is written to out_dir. + typer.echo(f"Parsed {len(results)} application(s): {', '.join(results)}") + return results[sorted(results)[-1]] + return get_parsed_metrics( log_dir=log_dir, log_file=log_file, @@ -105,6 +133,43 @@ def get( ) +@app.command("logs") +def logs( + log_dir: Annotated[ + str, typer.Argument(help="Directory containing raw Spark event logs.") + ] = "data/logs/raw", + format: Annotated[ + AnalysisFormat, + typer.Option(help="Output format: 'text' (default) or 'json'."), + ] = AnalysisFormat.text, +) -> None: + """List the event-log sources discovered in a directory. + + Rolling logs collapse into one source with ordered segments; marker files + and checksums are ignored. The last row is the one a bare parse selects. + """ + sources = discover_sources(log_dir) + if not sources: + typer.echo(f"No event log sources found in {log_dir}") + raise typer.Exit(1) + + ordered = sorted(sources, key=lambda item: (item.modified or 0, item.name)) + if format == AnalysisFormat.json: + typer.echo(json.dumps([s.model_dump(mode="json") for s in ordered], indent=2)) + return + + for source in ordered: + codecs = ",".join(codec.value for codec in source.codecs) + typer.echo( + f"{source.name}\t" + f"segments={len(source.segments)}\t" + f"rolling={source.rolling}\t" + f"complete={source.complete}\t" + f"codec={codecs}" + ) + typer.echo(f"\nNewest (default selection): {ordered[-1].name}", err=True) + + @app.command("analyze") def analyze( log_dir: Annotated[ diff --git a/sparkparse/capture.py b/sparkparse/capture.py index ecc59f6..fb4255f 100644 --- a/sparkparse/capture.py +++ b/sparkparse/capture.py @@ -18,6 +18,7 @@ from sparkparse import alerts, history from sparkparse.analyze import to_analysis_export, to_plan_summary +from sparkparse.eventlog import discover_sources from sparkparse.models import ( CapabilityStatus, CaptureCapabilities, @@ -602,16 +603,27 @@ def _select_log(self) -> str: return self.log_file assert self._log_dir is not None app_id = self._metadata.source_application_id + # Rolling logs are a directory and compressed logs carry a codec + # suffix, so match on the discovered application identity rather than + # on an exact file name. candidates = [ - Path(f).name - for f in list_files(self._log_dir) - if Path(f).name in {app_id, f"{app_id}.inprogress"} + source + for source in discover_sources(self._log_dir) + if source.application_id is not None + and app_id is not None + and ( + source.application_id == app_id + # Rolling dirs and multi-attempt logs append _. + or source.application_id.startswith(f"{app_id}_") + ) ] - if app_id is None or len(candidates) != 1: + if len(candidates) != 1: raise ValueError( - "Cannot uniquely identify this application log; supply log_file explicitly. Rolled/compressed logs require post-run ingestion." + "Cannot uniquely identify this application log; supply log_file " + f"explicitly. Found {len(candidates)} candidate(s) for " + f"application {app_id!r} in {self._log_dir}." ) - return candidates[0] + return candidates[0].root_uri or candidates[0].name def _finalize_capture(self, dfs: ParsedLogDataFrames) -> None: self._set_result(dfs) diff --git a/sparkparse/clean.py b/sparkparse/clean.py index 7f8ae6a..f0253ad 100644 --- a/sparkparse/clean.py +++ b/sparkparse/clean.py @@ -1,26 +1,66 @@ import logging from pathlib import Path +from typing import Any import polars as pl +from pydantic import BaseModel from sparkparse.common import resolve_dir, timeit, write_dataframe from sparkparse.models import ( Job, + Metrics, OutputFormat, ParsedLog, PhysicalPlan, QueryEvent, Stage, + StageStatus, Task, + TaskStatus, ) +from sparkparse.schemas import COMBINED_SCHEMA, DAG_SCHEMA from sparkparse.storage import is_cloud_path, join_path +_JOB_SCHEMA: dict[str, Any] = { + "job_id": pl.Int64, + "stage_id": pl.Int64, + "job_start_timestamp": pl.Int64, + "job_end_timestamp": pl.Int64, + "job_duration_seconds": pl.Float64, +} + +_STAGE_SCHEMA: dict[str, Any] = { + "stage_id": pl.Int64, + "stage_attempt_id": pl.Int64, + "stage_start_timestamp": pl.Int64, + "stage_end_timestamp": pl.Int64, + "stage_duration_seconds": pl.Float64, + "stage_num_tasks": pl.Int64, + "stage_failure_reason": pl.Utf8, + "stage_status": pl.Utf8, +} + def clean_jobs(jobs: list[Job]) -> pl.DataFrame: + """One row per (job, stage) declared by JobStart, with the job's timings. + + A job still running when the log ended contributes no ``end`` row, so the + pivot has no ``end`` column at all; the duration is unknown, not zero. + """ + if not jobs: + return pl.DataFrame(schema=_JOB_SCHEMA) + jobs_df = pl.DataFrame(jobs) jobs_with_duration = ( - jobs_df.select("job_id", "event_type", "job_timestamp") - .pivot("event_type", index="job_id", values="job_timestamp") + _with_missing_columns( + jobs_df.select("job_id", "event_type", "job_timestamp").pivot( + "event_type", + index="job_id", + values="job_timestamp", + aggregate_function="first", + ), + {"start": pl.Int64, "end": pl.Int64}, + ) .with_columns( (pl.col("end") - pl.col("start")) .mul(1 / 1_000) @@ -32,36 +72,123 @@ def clean_jobs(jobs: list[Job]) -> pl.DataFrame: jobs_df.select("job_id", "stages") .explode("stages") .rename({"stages": "stage_id"}) + # JobEnd events carry no stage list; their null row is not a relation. + .filter(pl.col("stage_id").is_not_null()) .join(jobs_with_duration, on="job_id", how="left") ) return jobs_final +def _with_missing_columns(df: pl.DataFrame, columns: dict[str, Any]) -> pl.DataFrame: + """Add any of ``columns`` the frame lacks, typed and null-filled. + + A stage that never completed contributes no ``end`` row, so the pivot has + no ``end`` column at all. The column has to exist and be null — its absence + is missing data, not a zero-length stage. + """ + missing = [ + pl.lit(None, dtype=dtype).alias(name) + for name, dtype in columns.items() + if name not in df.columns + ] + return df.with_columns(missing) if missing else df + + def clean_stages(stages: list[Stage]) -> pl.DataFrame: + """One row per stage *attempt*. + + ``stage_id`` alone repeats across retries, so the key is + ``(stage_id, stage_attempt_id)``; keying on stage id alone would fold a + retry's task metrics into the original attempt. + """ + if not stages: + return pl.DataFrame(schema=_STAGE_SCHEMA) + stages_df = pl.DataFrame(stages) + index = ["stage_id", "stage_attempt_id"] + + timings = _with_missing_columns( + stages_df.pivot( + "event_type", + index=index, + values="stage_timestamp", + aggregate_function="first", + ), + {"start": pl.Int64, "end": pl.Int64}, + ) + + attributes = stages_df.group_by(index).agg( + pl.col("num_tasks").drop_nulls().last().alias("stage_num_tasks"), + pl.col("failure_reason").drop_nulls().last().alias("stage_failure_reason"), + pl.when(pl.col("event_type").eq("end")) + .then(pl.col("status")) + .otherwise(None) + .drop_nulls() + .last() + .alias("stage_status"), + ) + stages_final = ( - stages_df.pivot("event_type", index="stage_id", values="stage_timestamp") - .with_columns( + timings.with_columns( (pl.col("end") - pl.col("start")) .mul(1 / 1000) .alias("stage_duration_seconds") ) .rename({"start": "stage_start_timestamp", "end": "stage_end_timestamp"}) + .join(attributes, on=index, how="left") + .with_columns( + pl.col("stage_status").fill_null(StageStatus.running.value), + ) ) return stages_final +def _null_metrics_template() -> dict: + """Nested all-null dict matching the ``Metrics`` model shape. + + Used only when *no* task in the log reported metrics: polars would infer a + Null column and refuse to unnest it. Every leaf stays null, so the frame + says "not measured" rather than "measured zero". + """ + + def fields(model: type[BaseModel]) -> dict: + out: dict = {} + for name, field in model.model_fields.items(): + annotation = field.annotation + if isinstance(annotation, type) and issubclass(annotation, BaseModel): + out[name] = fields(annotation) + else: + out[name] = None + return out + + return fields(Metrics) + + def clean_tasks(tasks: list[Task]) -> pl.DataFrame: + """One row per task *attempt*, including failed and speculative attempts.""" task_df = pl.DataFrame(tasks) - tasks_final = task_df.with_columns( - (pl.col("task_finish_time") - pl.col("task_start_time")) - .mul(1 / 1_000) - .alias("task_duration_seconds") - ).rename( - { - "task_start_time": "task_start_timestamp", - "task_finish_time": "task_end_timestamp", - } + if task_df.height and task_df.schema["metrics"] == pl.Null: + task_df = task_df.with_columns( + pl.Series("metrics", [_null_metrics_template()] * task_df.height) + ) + + tasks_final = ( + task_df.with_columns( + (pl.col("task_finish_time") - pl.col("task_start_time")) + .mul(1 / 1_000) + .alias("task_duration_seconds") + ) + .with_columns( + pl.col("status").eq(TaskStatus.success.value).alias("task_succeeded") + ) + .rename( + { + "task_start_time": "task_start_timestamp", + "task_finish_time": "task_end_timestamp", + "status": "task_status", + "failure_reason": "task_failure_reason", + } + ) ) return tasks_final @@ -69,21 +196,37 @@ def clean_tasks(tasks: list[Task]) -> pl.DataFrame: def clean_plan( query_times: list[QueryEvent], queries: list[PhysicalPlan] ) -> pl.DataFrame: - plan = pl.DataFrame() - for query in queries: - plan = pl.concat( - [ - plan, - pl.DataFrame([node.model_dump() for node in query.nodes]).with_columns( - pl.lit(query.query_id).alias("query_id") - ), - ] - ) + # One frame for every query at once: per-query frames disagree on dtype + # whenever a column is all-null in one plan (an initial plan with no + # codegen ids) and populated in another. + node_rows = [ + {**node.model_dump(), "query_id": query.query_id} + for query in queries + for node in query.nodes + ] + plan = pl.DataFrame( + node_rows, + schema_overrides={ + "node_id": pl.Int64, + "whole_stage_codegen_id": pl.Int64, + "query_id": pl.Int64, + }, + ) query_times_df = pl.DataFrame([qt.model_dump() for qt in query_times]) query_times_pivoted = ( ( - query_times_df.pivot("event_type", index="query_id", values="query_time") + _with_missing_columns( + query_times_df.pivot( + "event_type", + index="query_id", + values="query_time", + aggregate_function="first", + ), + # A query that failed or was still running when the log ended + # has no end event at all. + {"start": pl.Int64, "end": pl.Int64}, + ) .rename({"start": "query_start_timestamp", "end": "query_end_timestamp"}) .with_columns( (pl.col("query_end_timestamp") - pl.col("query_start_timestamp")) @@ -100,16 +243,19 @@ def clean_plan( how="left", ) .with_columns( + # A query still running when the log ended has no duration. The + # label has to say so; concat_str would null the whole header. pl.concat_str( [ pl.col("query_id"), pl.lit(" - "), - pl.col("query_function"), + pl.col("query_function").fill_null("unknown"), pl.lit(" ["), pl.col("query_duration_seconds") .mul(1 / 60) .round(2) - .cast(pl.String), + .cast(pl.String) + .fill_null("unknown"), pl.lit(" min]"), ] ).alias("query_header") @@ -399,7 +545,10 @@ def get_node_metrics( .replace_strict(metric_type_order) .alias("metric_order") ) - .sort("metric_order", "metric_name") + # Tie-break on the physical identity so the accumulator list order is + # reproducible: metric type and name alone leave ties, and the input + # order shifts whenever an unrelated query is added to the log. + .sort("metric_order", "metric_name", "stage_id", "task_id", "accumulator_id") .drop("metric_order") .group_by(*dag_base_cols) .agg(pl.col("accumulators")) @@ -421,7 +570,7 @@ def get_node_metrics( .replace_strict(metric_type_order) .alias("metric_order") ) - .sort("metric_order", "metric_name") + .sort("metric_order", "metric_name", "accumulator_id") .drop("metric_order") .group_by("query_id", "node_id") .agg(pl.col("accumulator_totals")) @@ -554,6 +703,11 @@ def log_to_dag_df(result: ParsedLog) -> pl.DataFrame: extra_cols = ["details", "query_function", "query_header"] + if not result.queries: + # No SQL executions: jobs and tasks may still exist, but there is no + # plan to hang them on. An empty typed frame says that explicitly. + return pl.DataFrame(schema=DAG_SCHEMA) + plan = clean_plan(result.query_times, result.queries) dag_long = get_dag_long(result, plan) @@ -594,24 +748,42 @@ def log_to_dag_df(result: ParsedLog) -> pl.DataFrame: ) ) + # Per-task accumulator rows should outnumber the per-node totals. That + # holds for any query with more than one task, but not for a plan-only or + # single-task query, so it is a signal rather than an invariant. total_same_as_task = dag_final.filter( pl.col("n_accumulators") == pl.col("n_accumulator_totals") ) - assert total_same_as_task.shape[0] < dag_final.shape[0] + if total_same_as_task.shape[0] >= dag_final.shape[0]: + logging.debug( + "Every node has as many task accumulators as totals; the log may " + "cover only single-task stages or carry no task metrics." + ) + # A join that multiplied plan rows would silently double every metric. assert dag_final.shape[0] == plan.shape[0] return dag_final -@timeit -def log_to_combined_df( - result: ParsedLog, dag: pl.DataFrame, log_name: str -) -> pl.DataFrame: - stages_final = clean_stages(result.stages) - jobs_final = clean_jobs(result.jobs) +_QUERY_TASK_SCHEMA: dict[str, Any] = { + "query_id": pl.Int64, + "query_function": pl.Utf8, + "query_start_timestamp": pl.Utf8, + "query_end_timestamp": pl.Utf8, + "query_duration_seconds": pl.Float64, + "stage_id": pl.Int64, + "task_id": pl.Int64, + "nodes": pl.List(pl.Utf8), +} - tasks = clean_tasks(result.tasks) - query_task_lookup = ( + +def _query_task_nodes(dag: pl.DataFrame) -> pl.DataFrame: + """One row per (query, stage, task) with the plan nodes that task fed.""" + if dag.is_empty() or "accumulators" not in dag.columns: + # No plan to attribute tasks to; every task stays unattributed rather + # than disappearing from the frame. + return pl.DataFrame(schema=_QUERY_TASK_SCHEMA) + return ( dag.filter(pl.col("node_id").le(100_000)) .explode("accumulators") .with_columns( @@ -636,9 +808,121 @@ def log_to_combined_df( .agg(pl.col("node_name").alias("nodes")) ) + +def query_stage_associations(dag: pl.DataFrame) -> pl.DataFrame: + """Every (query, stage) pair observed in the plan accumulators. + + A stage reused by several queries appears once per query. This is kept as + its own table because joining it into the task frame would duplicate task + rows and double-count their metrics. + """ + if dag.is_empty() or "accumulators" not in dag.columns: + return pl.DataFrame(schema={"query_id": pl.Int64, "stage_id": pl.Int64}) + return ( + dag.filter(pl.col("node_id").le(100_000)) + .explode("accumulators") + .with_columns(pl.col("accumulators").struct.field("stage_id").alias("stage_id")) + .filter(pl.col("stage_id").is_not_null()) + .select("query_id", "stage_id") + .unique() + .sort("query_id", "stage_id") + ) + + +def job_stage_associations(result: ParsedLog) -> pl.DataFrame: + """Every (job, stage) pair declared by JobStart events. + + A stage skipped because its output was already available is still listed by + the later job, so this relation is genuinely many-to-many. + """ + jobs_with_stages = [job for job in result.jobs if job.stages] + if not jobs_with_stages: + return pl.DataFrame(schema={"job_id": pl.Int64, "stage_id": pl.Int64}) + return ( + pl.DataFrame( + [{"job_id": job.job_id, "stage_id": job.stages} for job in jobs_with_stages] + ) + .explode("stage_id") + .unique() + .sort("job_id", "stage_id") + ) + + +def _primary_job_per_stage( + jobs_final: pl.DataFrame, stages_final: pl.DataFrame +) -> pl.DataFrame: + """Pick one job per stage attempt so the task frame stays one row per task. + + The full relation lives in :func:`job_stage_associations`. The job chosen + here is the lowest-numbered job whose window contains the stage's start — + the job that actually ran it, rather than a later job that merely listed + it as already-computed. + """ + if jobs_final.is_empty(): + return jobs_final + stage_starts = stages_final.select( + "stage_id", "stage_attempt_id", "stage_start_timestamp" + ) + ranked = ( + jobs_final.join(stage_starts, on="stage_id", how="left") + .with_columns( + ( + pl.col("stage_start_timestamp").is_not_null() + & pl.col("job_start_timestamp").le(pl.col("stage_start_timestamp")) + & ( + pl.col("job_end_timestamp").is_null() + | pl.col("job_end_timestamp").ge(pl.col("stage_start_timestamp")) + ) + ).alias("ran_in_job") + ) + .sort( + ["stage_id", "stage_attempt_id", "ran_in_job", "job_id"], + descending=[False, False, True, False], + nulls_last=True, + ) + .unique( + subset=["stage_id", "stage_attempt_id"], keep="first", maintain_order=True + ) + ) + return ranked.drop("ran_in_job", "stage_start_timestamp") + + +@timeit +def log_to_combined_df( + result: ParsedLog, dag: pl.DataFrame, log_name: str +) -> pl.DataFrame: + if not result.tasks: + return pl.DataFrame(schema=COMBINED_SCHEMA) + + stages_final = clean_stages(result.stages) + jobs_final = clean_jobs(result.jobs) + primary_jobs = _primary_job_per_stage(jobs_final, stages_final) + + tasks = clean_tasks(result.tasks) + query_task_nodes = _query_task_nodes(dag) + + # One task belongs to one physical stage attempt, but a shared stage can + # serve several queries. Attribute the task to the earliest query and keep + # the whole relation in query_stage instead of duplicating the task row. + query_task_lookup = ( + query_task_nodes.sort("query_id") + .group_by("stage_id", "task_id") + .agg( + pl.col("query_id").first(), + pl.col("query_function").first(), + pl.col("query_start_timestamp").first(), + pl.col("query_end_timestamp").first(), + pl.col("query_duration_seconds").first(), + # Already node_id-ordered per query; keep that order across the + # queries a shared task serves instead of re-sorting as text. + pl.col("nodes").flatten().unique(maintain_order=True).alias("nodes"), + pl.col("query_id").n_unique().alias("query_count"), + ) + ) + combined = ( - tasks.join(stages_final, on="stage_id", how="left") - .join(jobs_final, on="stage_id", how="left") + tasks.join(stages_final, on=["stage_id", "stage_attempt_id"], how="left") + .join(primary_jobs, on=["stage_id", "stage_attempt_id"], how="left") .sort("job_id", "stage_id", "task_id") .unnest("metrics") .unnest("task_metrics") @@ -662,14 +946,19 @@ def log_to_combined_df( "query_start_timestamp", "query_end_timestamp", "query_duration_seconds", + "query_count", "job_id", "stage_id", + "stage_attempt_id", "job_start_timestamp", "job_end_timestamp", "job_duration_seconds", "stage_start_timestamp", "stage_end_timestamp", "stage_duration_seconds", + "stage_num_tasks", + "stage_status", + "stage_failure_reason", # core task info "task_id", "task_start_timestamp", @@ -722,7 +1011,13 @@ def log_to_combined_df( "executor_id", "host", "index", + "partition_id", "attempt", + # Resource usage is per attempt; output accounting is per *successful* + # attempt. Keeping both means a retried task is not counted twice. + "task_status", + "task_succeeded", + "task_failure_reason", "failed", "killed", "speculative", @@ -779,8 +1074,14 @@ def log_to_combined_df( ] ) .join(query_task_lookup, on=["stage_id", "task_id"], how="left") - .filter(pl.col("query_id").is_not_null()) - .sort("query_id", "stage_id", "task_id") + # Tasks from jobs no SQL execution claims (schema inference, RDD work) + # keep a null query_id. Dropping them would erase real resource usage + # from the ledger. + .with_columns( + pl.col("query_count").fill_null(0), + pl.col("nodes").fill_null(pl.lit([], dtype=pl.List(pl.Utf8))), + ) + .sort("query_id", "stage_id", "task_id", nulls_last=True) ) final = combined_clean.select(final_cols) diff --git a/sparkparse/connect.py b/sparkparse/connect.py index 2603310..d0291da 100644 --- a/sparkparse/connect.py +++ b/sparkparse/connect.py @@ -30,6 +30,11 @@ import polars as pl from sparkparse.models import CaptureDiagnostic, NodeType, ParsedLogDataFrames +from sparkparse.schemas import ( + COMBINED_SCHEMA, + DAG_SCHEMA, + empty_capture_dataframes, +) if TYPE_CHECKING: from pyspark.sql import SparkSession @@ -63,76 +68,6 @@ "PhotonColumnarToRow": NodeType.ColumnarToRow, } -_ACCUM_TOTALS_STRUCT = pl.Struct( - { - "metric_name": pl.Utf8, - "metric_type": pl.Utf8, - "value": pl.Float64, - # Exact integer counterpart of ``value``; null when the metric is - # fractional or was rescaled. ``value`` is float64 and cannot hold a - # count above 2**53 without rounding it. - "value_exact": pl.Int64, - "readable_value": pl.Float64, - "readable_unit": pl.Utf8, - "readable_str": pl.Utf8, - } -) - -_DAG_SCHEMA: dict[str, pl.PolarsDataType] = { - "log_name": pl.Utf8, - "parsed_log_name": pl.Utf8, - "query_id": pl.Int64, - "query_function": pl.Utf8, - "query_header": pl.Utf8, - "query_start_timestamp": pl.Utf8, - "query_end_timestamp": pl.Utf8, - "query_duration_seconds": pl.Float64, - "source_execution_id": pl.Utf8, - "node_id": pl.Int64, - "node_type": pl.Utf8, - "node_name": pl.Utf8, - "child_nodes": pl.Utf8, - "whole_stage_codegen_id": pl.Int64, - "details": pl.Utf8, - "accumulator_totals": pl.List(_ACCUM_TOTALS_STRUCT), - "n_accumulator_totals": pl.Int64, - "node_duration_minutes": pl.Float64, - "n_accumulators": pl.Int64, - "node_id_adj": pl.Int64, -} - -_COMBINED_SCHEMA: dict[str, pl.PolarsDataType] = { - "log_name": pl.Utf8, - "parsed_log_name": pl.Utf8, - "query_id": pl.Int64, - "stage_id": pl.Int64, - "task_id": pl.Int64, - "task_duration_seconds": pl.Float64, - "bytes_read": pl.Int64, - "records_read": pl.Int64, - "bytes_written": pl.Int64, - "records_written": pl.Int64, - "memory_bytes_spilled": pl.Int64, - "disk_bytes_spilled": pl.Int64, - "shuffle_bytes_read": pl.Int64, - "shuffle_bytes_written": pl.Int64, - "shuffle_remote_bytes_read": pl.Int64, - "shuffle_local_bytes_read": pl.Int64, - "executor_run_time_seconds": pl.Float64, - "jvm_gc_time_seconds": pl.Float64, - "executor_id": pl.Utf8, - "nodes": pl.List(pl.Utf8), -} - - -def empty_capture_dataframes() -> ParsedLogDataFrames: - """Return the canonical typed empty frames used by capture backends.""" - return ParsedLogDataFrames( - dag=pl.DataFrame(schema=_DAG_SCHEMA), - combined=pl.DataFrame(schema=_COMBINED_SCHEMA), - ) - - _JOIN_NODE_TYPES: frozenset[NodeType] = frozenset( { NodeType.BroadcastHashJoin, @@ -1121,7 +1056,7 @@ def executions(self) -> list[dict[str, Any]]: def _build_dataframes(self) -> ParsedLogDataFrames: return ParsedLogDataFrames( - dag=self._build_dag_df(), combined=pl.DataFrame(schema=_COMBINED_SCHEMA) + dag=self._build_dag_df(), combined=pl.DataFrame(schema=COMBINED_SCHEMA) ) def _resolve_join_details( @@ -1200,9 +1135,9 @@ def _build_dag_df(self) -> pl.DataFrame: ) if not rows: - return pl.DataFrame(schema=_DAG_SCHEMA) - frame = pl.DataFrame(rows, schema_overrides=_DAG_SCHEMA) - return frame.select(list(_DAG_SCHEMA)) + return pl.DataFrame(schema=DAG_SCHEMA) + frame = pl.DataFrame(rows, schema_overrides=DAG_SCHEMA) + return frame.select(list(DAG_SCHEMA)) def _execution_rows( self, execution: ConnectExecution, unknown_names: set[str] diff --git a/sparkparse/eventlog.py b/sparkparse/eventlog.py new file mode 100644 index 0000000..cf17eed --- /dev/null +++ b/sparkparse/eventlog.py @@ -0,0 +1,296 @@ +"""Discovery and streaming reads of Spark event-log sources. + +A *logical* event log is one application attempt. On disk it is either a single +file (``[_][.][.inprogress]``) or, when +``spark.eventLog.rolling.enabled`` is set, a directory +``eventlog_v2_[_]/`` holding an ``appstatus_*`` marker and +numbered ``events__[.][.inprogress]`` segments. + +Both shapes are read the same way: as one ordered stream of lines, never as a +whole-file string. + +Supported codecs +---------------- +``none``, ``zstd`` (needs the ``zstandard`` package) and ``gz`` (stdlib). +Spark's ``lz4``, ``lzf`` and ``snappy`` event logs use Java-specific block +framing that no Python codec reads; those raise an explicit error naming the +codec rather than failing as corrupt JSON. +""" + +from __future__ import annotations + +import logging +import re +from collections.abc import Iterator +from pathlib import Path +from typing import IO, Any + +from sparkparse.models import ( + READABLE_CODECS, + EventLogCodec, + EventLogSegment, + EventLogSource, +) +from sparkparse.storage import ( + get_path_name, + is_cloud_path, + join_path, + list_files, + open_file, +) + +logger = logging.getLogger(__name__) + +ROLLING_DIR_PREFIX = "eventlog_v2_" +ROLLING_STATUS_PREFIX = "appstatus_" +ROLLING_EVENTS_PATTERN = re.compile(r"^events_(\d+)_(.+)$") +IN_PROGRESS_SUFFIX = ".inprogress" + +# Files a Spark log directory accumulates that are not event logs. +IGNORED_NAMES = {".DS_Store", "_SUCCESS", "_started", "_committed"} +IGNORED_SUFFIXES = (".crc", ".tmp", ".lock") + + +class UnsupportedCodecError(RuntimeError): + """Raised for an event log written with a codec Python cannot read.""" + + +class EventLogNotFoundError(FileNotFoundError): + """Raised when a log directory holds no readable event-log source.""" + + +def _is_ignored(name: str) -> bool: + if name in IGNORED_NAMES or name.startswith("."): + return True + return name.endswith(IGNORED_SUFFIXES) + + +def _split_codec(stem: str) -> tuple[str, EventLogCodec]: + """Split a trailing ``.`` off an event-log file name.""" + if "." not in stem: + return stem, EventLogCodec.none + base, _, ext = stem.rpartition(".") + normalized = {"gzip": "gz"}.get(ext.lower(), ext.lower()) + try: + codec = EventLogCodec(normalized) + except ValueError: + return stem, EventLogCodec.none + if codec is EventLogCodec.none: + return base, EventLogCodec.none + return base, codec + + +def _parse_file_name(name: str) -> tuple[str, EventLogCodec, bool]: + """Return (identity, codec, in_progress) for an event-log file name.""" + in_progress = name.endswith(IN_PROGRESS_SUFFIX) + stem = name.removesuffix(IN_PROGRESS_SUFFIX) if in_progress else name + identity, codec = _split_codec(stem) + return identity, codec, in_progress + + +def _modified_time(uri: str) -> float | None: + if is_cloud_path(uri): + return None + try: + return Path(uri).stat().st_mtime + except OSError: + return None + + +def _rolling_source(dir_uri: str) -> EventLogSource | None: + """Build a source from an ``eventlog_v2_*`` rolling-log directory.""" + dir_name = get_path_name(dir_uri) + app_identity = dir_name.removeprefix(ROLLING_DIR_PREFIX) + + segments: list[EventLogSegment] = [] + status_in_progress = False + saw_status = False + for entry in list_files(dir_uri): + entry_name = get_path_name(entry) + if _is_ignored(entry_name): + continue + if entry_name.startswith(ROLLING_STATUS_PREFIX): + saw_status = True + status_in_progress = entry_name.endswith(IN_PROGRESS_SUFFIX) + continue + identity, codec, in_progress = _parse_file_name(entry_name) + match = ROLLING_EVENTS_PATTERN.match(identity) + if not match: + logger.debug("Ignoring non-segment file in rolling log dir: %s", entry) + continue + segments.append( + EventLogSegment( + uri=entry, + index=int(match.group(1)), + codec=codec, + in_progress=in_progress, + ) + ) + + if not segments: + return None + + segments.sort(key=lambda segment: (segment.index or 0, segment.uri)) + modified = [ + mtime + for mtime in (_modified_time(s.uri) for s in segments) + if mtime is not None + ] + return EventLogSource( + name=dir_name, + application_id=app_identity, + segments=segments, + rolling=True, + # An in-progress status marker, or no marker at all, means the + # application never wrote its completion record. + complete=saw_status and not status_in_progress, + modified=max(modified) if modified else None, + root_uri=dir_uri, + ) + + +def single_file_source(uri: str) -> EventLogSource: + name = get_path_name(uri) + identity, codec, in_progress = _parse_file_name(name) + return EventLogSource( + name=identity, + application_id=identity, + segments=[EventLogSegment(uri=uri, codec=codec, in_progress=in_progress)], + rolling=False, + complete=not in_progress, + modified=_modified_time(uri), + root_uri=uri, + ) + + +def _is_dir(uri: str) -> bool: + if is_cloud_path(uri): + # Cloud object stores have no directories; a rolling log shows up as a + # prefix, which list_files enumerates. + return bool(get_path_name(uri).startswith(ROLLING_DIR_PREFIX)) + return Path(uri).is_dir() + + +def discover_sources(log_dir: str | Path) -> list[EventLogSource]: + """Return every logical event log under ``log_dir``. + + Marker files, checksums and hidden files are ignored. Rolling-log + directories collapse into one source with ordered segments. + """ + log_dir_str = str(log_dir) + sources: list[EventLogSource] = [] + for entry in list_files(log_dir_str): + entry_name = get_path_name(entry) + if _is_ignored(entry_name): + continue + if entry_name.startswith(ROLLING_DIR_PREFIX) or _is_dir(entry): + if not entry_name.startswith(ROLLING_DIR_PREFIX): + logger.debug("Ignoring non-event-log directory: %s", entry) + continue + source = _rolling_source(entry) + if source is not None: + sources.append(source) + continue + sources.append(single_file_source(entry)) + return sources + + +def _sort_key(source: EventLogSource) -> tuple[float, str]: + # Newest policy: modification time when the filesystem reports one, + # otherwise the name. Ties fall back to the name so the order is stable. + return ( + source.modified if source.modified is not None else float("-inf"), + source.name, + ) + + +def newest_source(sources: list[EventLogSource]) -> EventLogSource: + """Return the most recently modified source (name order breaks ties).""" + return max(sources, key=_sort_key) + + +def resolve_source(log_dir: str | Path, log_file: str | None = None) -> EventLogSource: + """Resolve one event-log source from ``log_dir``. + + ``log_file`` selects a source explicitly, by file name, rolling-directory + name, application id, or full path. Without it the newest source is used: + most recent modification time, falling back to name order when the + filesystem reports no mtime (cloud object listings). + """ + log_dir_str = str(log_dir) + if log_file is not None: + candidate = log_file if is_cloud_path(log_file) else str(log_file) + direct = candidate if "/" in candidate else join_path(log_dir_str, candidate) + if _is_dir(direct) or get_path_name(direct).startswith(ROLLING_DIR_PREFIX): + source = _rolling_source(direct) + if source is None: + raise EventLogNotFoundError( + f"No event-log segments found in rolling log dir: {direct}" + ) + return source + matches = [ + source + for source in discover_sources(log_dir_str) + if log_file in (source.name, get_path_name(source.root_uri or "")) + or source.application_id == log_file + or source.root_uri == log_file + ] + if len(matches) == 1: + return matches[0] + if len(matches) > 1: + raise ValueError( + f"'{log_file}' matches {len(matches)} event logs in {log_dir_str}: " + f"{[m.root_uri for m in matches]}" + ) + return single_file_source(direct) + + sources = discover_sources(log_dir_str) + if not sources: + raise EventLogNotFoundError(f"No event log files found in: {log_dir_str}") + return newest_source(sources) + + +def _open_segment(segment: EventLogSegment) -> IO[Any]: + """Open one segment as a text stream, decompressing as needed.""" + if segment.codec not in READABLE_CODECS: + raise UnsupportedCodecError( + f"Event log {segment.uri} is compressed with '{segment.codec}', which " + "uses Java-specific framing that Python cannot decode. Re-run with " + "spark.eventLog.compression.codec=zstd (or none), or decompress the " + "log with Spark's history server tooling first. " + f"Readable codecs: {sorted(c.value for c in READABLE_CODECS)}." + ) + + if segment.codec is EventLogCodec.none: + return open_file(segment.uri, "r") + + raw = open_file(segment.uri, "rb") + if segment.codec is EventLogCodec.gz: + import gzip + import io + + return io.TextIOWrapper(gzip.GzipFile(fileobj=raw), encoding="utf-8") + + try: + import zstandard + except ImportError as exc: # pragma: no cover - exercised only without zstandard + raise UnsupportedCodecError( + f"Reading {segment.uri} needs the 'zstandard' package: " + "install it with `uv add zstandard` or the sparkparse[zstd] extra." + ) from exc + import io + + reader = zstandard.ZstdDecompressor().stream_reader(raw) + return io.TextIOWrapper(reader, encoding="utf-8") + + +def iter_lines(source: EventLogSource) -> Iterator[tuple[str, int, str]]: + """Yield ``(uri, line_number, line)`` across the source's segments. + + Segments are read in order and one line at a time, so peak memory does not + scale with log size. + """ + for segment in source.segments: + with _open_segment(segment) as handle: + for line_number, line in enumerate(handle, start=1): + yield segment.uri, line_number, line diff --git a/sparkparse/history.py b/sparkparse/history.py index e2de02f..ea1fb8a 100644 --- a/sparkparse/history.py +++ b/sparkparse/history.py @@ -22,7 +22,12 @@ import polars as pl -from sparkparse.analyze import find_cartesian_joins, find_largest_scans +from sparkparse.analyze import ( + OUTPUT_ACCOUNTING_COLUMNS, + find_cartesian_joins, + find_largest_scans, + retained_outputs, +) from sparkparse.models import ( CapabilityStatus, CaptureResult, @@ -110,10 +115,16 @@ def record_from_dfs( else: duration_s = None + kept = retained_outputs(combined) + def total(column: str) -> int | None: - if column not in combined.columns or combined.height == 0: + # Output totals count retained outputs only; a recomputed or raced + # partition would otherwise report its bytes twice. Spill and time + # count every attempt. + frame = kept if column in OUTPUT_ACCOUNTING_COLUMNS else combined + if column not in frame.columns or frame.height == 0: return None - values = combined[column].drop_nulls() + values = frame[column].drop_nulls() if len(values) == 0: return None return int(values.sum() or 0) diff --git a/sparkparse/models.py b/sparkparse/models.py index 63253e7..7fba6fd 100644 --- a/sparkparse/models.py +++ b/sparkparse/models.py @@ -30,10 +30,29 @@ class Job(BaseModel): stages: list[int] | None = Field(alias="Stage IDs", default=None) +class StageStatus(StrEnum): + """Outcome of one stage attempt. + + ``running`` means the attempt was submitted but no completion event was + observed — an incomplete log, not a stage that took zero time. + """ + + succeeded = "succeeded" + failed = "failed" + running = "running" + + class Stage(BaseModel): + """One stage *attempt*. ``stage_id`` alone is not a unique key: a retried + stage emits a second Submitted/Completed pair with a higher attempt id.""" + stage_id: int + stage_attempt_id: int = 0 event_type: EventType - stage_timestamp: int + stage_timestamp: int | None = None + num_tasks: int | None = None + status: StageStatus = StageStatus.running + failure_reason: str | None = None class Metric(BaseModel): @@ -1433,10 +1452,20 @@ class Metrics(BaseModel): output_metrics: OutputMetrics +class TaskStatus(StrEnum): + """Outcome of one task attempt, taken from ``Task End Reason``.""" + + success = "success" + failed = "failed" + killed = "killed" + + class Task(BaseModel): task_id: int = Field(alias="Task ID") stage_id: int = Field(alias="Stage ID") + stage_attempt_id: int = Field(alias="Stage Attempt ID", default=0) index: int = Field(alias="Index") + partition_id: int | None = Field(alias="Partition ID", default=None) attempt: int = Field(alias="Attempt") task_start_time: int = Field(alias="Launch Time") task_finish_time: int = Field(alias="Finish Time") @@ -1447,8 +1476,12 @@ class Task(BaseModel): speculative: bool = Field(alias="Speculative") failed: bool = Field(alias="Failed") killed: bool = Field(alias="Killed") - metrics: Metrics - accumulators: list[Accumulator] + status: TaskStatus = TaskStatus.success + failure_reason: str | None = None + # A task that died before its metrics were serialized has no metrics at + # all. None says "not measured"; zeros would claim the task did no work. + metrics: Metrics | None = None + accumulators: list[Accumulator] = Field(default_factory=list) class DriverAccumUpdates(BaseModel): @@ -1457,6 +1490,68 @@ class DriverAccumUpdates(BaseModel): update: int +class EventLogCodec(StrEnum): + """Compression codecs recognized on event-log file names. + + Only ``none``, ``zstd`` and ``gz`` can actually be read: Spark's ``lz4``, + ``lzf`` and ``snappy`` event logs use Java-specific block framing + (``LZ4BlockOutputStream``, Xerial Snappy) that no Python codec reads. Those + are recognized so the failure names the codec instead of surfacing as a + JSON decode error. + """ + + none = "none" + zstd = "zstd" + gz = "gz" + lz4 = "lz4" + lzf = "lzf" + snappy = "snappy" + + +READABLE_CODECS: frozenset[EventLogCodec] = frozenset( + {EventLogCodec.none, EventLogCodec.zstd, EventLogCodec.gz} +) + + +class EventLogSegment(BaseModel): + """One physical file belonging to a logical event log.""" + + uri: str + index: int | None = None + codec: EventLogCodec = EventLogCodec.none + in_progress: bool = False + + +class EventLogSource(BaseModel): + """A logical event log: one application, one or more ordered segments.""" + + name: str + application_id: str | None = None + attempt_id: str | None = None + segments: list[EventLogSegment] = Field(default_factory=list) + rolling: bool = False + complete: bool = True + modified: float | None = None + root_uri: str | None = None + + @property + def uris(self) -> list[str]: + return [segment.uri for segment in self.segments] + + @property + def codecs(self) -> list[EventLogCodec]: + return sorted({segment.codec for segment in self.segments}) + + +class LogDiagnostic(BaseModel): + """One recoverable problem encountered while reading an event log.""" + + code: str + message: str + uri: str | None = None + line: int | None = None + + class ParsedLog(BaseModel): name: str jobs: list[Job] @@ -1465,6 +1560,13 @@ class ParsedLog(BaseModel): queries: list[PhysicalPlan] query_times: list[QueryEvent] driver_accum_updates: list[DriverAccumUpdates] + source: EventLogSource | None = None + diagnostics: list[LogDiagnostic] = Field(default_factory=list) + spark_version: str | None = None + application_id: str | None = None + application_name: str | None = None + unknown_events: dict[str, int] = Field(default_factory=dict) + events_read: int = 0 class OutputFormat(StrEnum): @@ -1475,9 +1577,20 @@ class OutputFormat(StrEnum): class ParsedLogDataFrames(BaseModel): + """Parsed output frames. + + ``combined`` holds one row per task *attempt*. ``job_stage`` and + ``query_stage`` are association tables: a stage can belong to several jobs + and serve several queries, and joining those relations into ``combined`` + would duplicate task rows and double-count their metrics. + """ + model_config = ConfigDict(arbitrary_types_allowed=True) combined: pl.DataFrame dag: pl.DataFrame + job_stage: pl.DataFrame | None = None + query_stage: pl.DataFrame | None = None + diagnostics: list[LogDiagnostic] = Field(default_factory=list) class CaptureStatus(StrEnum): diff --git a/sparkparse/pages/home.py b/sparkparse/pages/home.py index 6ebe819..cc46490 100644 --- a/sparkparse/pages/home.py +++ b/sparkparse/pages/home.py @@ -10,7 +10,9 @@ from pydantic import BaseModel from sparkparse.common import resolve_dir, timeit -from sparkparse.parse import check_if_log_has_queries +from sparkparse.eventlog import discover_sources, iter_lines +from sparkparse.models import EventLogSource +from sparkparse.parse import source_has_queries @callback( @@ -20,15 +22,15 @@ @timeit @lru_cache def get_available_logs(_) -> list[str]: - log_path = Path(resolve_dir(get_app().server.config["LOG_DIR"])) - log_files = tuple(log_path.glob("*")) + log_path = resolve_dir(get_app().server.config["LOG_DIR"]) + # discover_sources collapses rolled segments into one logical log and skips + # marker files, so the table lists applications rather than files. + sources = discover_sources(log_path) - # filter out logs that don't have queries with ThreadPoolExecutor(max_workers=8) as executor: - results = list(executor.map(check_if_log_has_queries, log_files)) - log_files_filtered = [x.as_posix() for x, y in zip(log_files, results) if y] + results = list(executor.map(source_has_queries, sources)) - return sorted(log_files_filtered) + return sorted(source.name for source, has in zip(sources, results) if has) class LogDuration(BaseModel): @@ -48,30 +50,25 @@ class RawLogDetails(BaseModel): duration_formatted: str -def get_log_duration(log: Path) -> LogDuration: - with log.open("r") as f: - lines = f.readlines() - +def get_log_duration(source: EventLogSource) -> LogDuration: start_timestamp: datetime.datetime | None = None end_timestamp: datetime.datetime | None = None - # iterate over first few lines until first timestamp found - for line in lines: + # One streaming pass: keep the first timestamp seen and overwrite the last. + # Reading the whole log into a list costs memory proportional to log size. + for _, _, line in iter_lines(source): if "Timestamp" not in line: continue - - entry = json.loads(line) - start_timestamp = datetime.datetime.fromtimestamp(entry["Timestamp"] / 1000) - break - - # iterate backwards until last timestamp found - for line in reversed(lines): - if "Timestamp" not in line: + try: + entry = json.loads(line) + except json.JSONDecodeError: continue - - entry = json.loads(line) - end_timestamp = datetime.datetime.fromtimestamp(entry["Timestamp"] / 1000) - break + timestamp = entry.get("Timestamp") + if timestamp is None: + continue + end_timestamp = datetime.datetime.fromtimestamp(timestamp / 1000) + if start_timestamp is None: + start_timestamp = end_timestamp if start_timestamp is None or end_timestamp is None: raise ValueError("Could not find start and end timestamps in log") @@ -109,16 +106,25 @@ def get_log_table(available_logs: list[Path], dark_mode: bool): log_items = [] theme = "ag-theme-alpine-dark" if dark_mode else "ag-theme-alpine" + log_dir = resolve_dir(get_app().server.config["LOG_DIR"]) + sources = {source.name: source for source in discover_sources(log_dir)} + for log_str in available_logs: - log = Path(log_str) - duration = get_log_duration(log) + source = sources.get(str(log_str)) + if source is None: + continue + duration = get_log_duration(source) + size_bytes = sum( + Path(uri).stat().st_size for uri in source.uris if Path(uri).exists() + ) + modified = source.modified or 0 log_items.append( RawLogDetails( - name=log.stem, - modified=datetime.datetime.fromtimestamp(log.stat().st_mtime).strftime( + name=source.name, + modified=datetime.datetime.fromtimestamp(modified).strftime( "%Y-%m-%d %H:%M:%S" ), - size_mb=round(log.stat().st_size / 1024 / 1024, 2), + size_mb=round(size_bytes / 1024 / 1024, 2), start_time=duration.start_time, end_time=duration.end_time, duration_seconds=duration.duration_seconds, diff --git a/sparkparse/parse.py b/sparkparse/parse.py index 7ce779f..adc25b9 100644 --- a/sparkparse/parse.py +++ b/sparkparse/parse.py @@ -2,23 +2,41 @@ import json import logging import re +from collections.abc import Iterator from pathlib import Path from typing import Any, cast import polars as pl -from sparkparse.clean import log_to_combined_df, log_to_dag_df, write_parsed_log +from sparkparse.clean import ( + job_stage_associations, + log_to_combined_df, + log_to_dag_df, + query_stage_associations, + write_parsed_log, +) from sparkparse.common import resolve_dir, timeit +from sparkparse.eventlog import ( + IGNORED_NAMES, + EventLogNotFoundError, + UnsupportedCodecError, + discover_sources, + iter_lines, + resolve_source, + single_file_source, +) from sparkparse.models import ( NODE_ID_PATTERN, NODE_TYPE_DETAIL_MAP, NODE_TYPE_PATTERN, Accumulator, DriverAccumUpdates, + EventLogSource, EventType, ExecutorMetrics, InputMetrics, Job, + LogDiagnostic, Metrics, NodeType, OutputFormat, @@ -38,18 +56,14 @@ ShuffleReadMetrics, ShuffleWriteMetrics, Stage, + StageStatus, Task, TaskMetrics, + TaskStatus, deserialize_insert_into_hadoop_fs_relation_command_detail, deserialize_scan_detail, ) -from sparkparse.storage import ( - get_path_stem, - is_cloud_path, - join_path, - list_files, - open_file, -) +from sparkparse.storage import get_path_name, is_cloud_path logger = logging.getLogger(__name__) @@ -69,44 +83,91 @@ def parse_job(line_dict: dict) -> Job: def parse_stage(line_dict: dict) -> Stage: + stage_info = line_dict["Stage Info"] + failure_reason = stage_info.get("Failure Reason") if line_dict["Event"].endswith("Submitted"): event_type = EventType.start - timestamp = line_dict["Stage Info"]["Submission Time"] + timestamp = stage_info.get("Submission Time") + status = StageStatus.running else: event_type = EventType.end - timestamp = line_dict["Stage Info"]["Completion Time"] + timestamp = stage_info.get("Completion Time") + status = StageStatus.failed if failure_reason else StageStatus.succeeded return Stage( - stage_id=line_dict["Stage Info"]["Stage ID"], + stage_id=stage_info["Stage ID"], + # A retried stage reuses its stage id, so the attempt id is part of + # the key. Older logs spell it "Attempt ID". + stage_attempt_id=stage_info.get( + "Stage Attempt ID", stage_info.get("Attempt ID", 0) + ), event_type=event_type, stage_timestamp=timestamp, + num_tasks=stage_info.get("Number of Tasks"), + status=status, + failure_reason=failure_reason, ) +def _task_end_reason(line_dict: dict) -> tuple[TaskStatus, str | None]: + """Map ``Task End Reason`` onto a status and a human-readable reason.""" + reason = line_dict.get("Task End Reason") or {} + kind = reason.get("Reason", "Success") + if kind == "Success": + return TaskStatus.success, None + # Prefer the short summary Spark already composed; the stack trace is a + # last resort because the reason is stored, not just logged. + class_name = reason.get("Class Name") + description = reason.get("Description") + if class_name and description: + detail = f"{class_name}: {description}" + else: + detail = ( + class_name + or description + or reason.get("Kill Reason") + or reason.get("Full Stack Trace") + or kind + ) + status = TaskStatus.killed if kind == "TaskKilled" else TaskStatus.failed + return status, str(detail)[:500] + + def parse_task(line_dict: dict) -> Task: task_info = line_dict["Task Info"] task_info["Stage ID"] = line_dict["Stage ID"] + task_info["Stage Attempt ID"] = line_dict.get("Stage Attempt ID", 0) task_info["Task Type"] = line_dict["Task Type"] task_id = task_info["Task ID"] - task_metrics = line_dict["Task Metrics"] - metrics = Metrics( - task_metrics=TaskMetrics(**task_metrics), - executor_metrics=ExecutorMetrics(**line_dict["Task Executor Metrics"]), - shuffle_read_metrics=ShuffleReadMetrics(**task_metrics["Shuffle Read Metrics"]), - shuffle_write_metrics=ShuffleWriteMetrics( - **task_metrics["Shuffle Write Metrics"] - ), - input_metrics=InputMetrics(**task_metrics["Input Metrics"]), - output_metrics=OutputMetrics(**task_metrics["Output Metrics"]), - ) + task_metrics = line_dict.get("Task Metrics") + # A task that failed before its metrics were serialized reports none. That + # is missing data, not a task that consumed nothing. + if task_metrics is None: + metrics = None + else: + metrics = Metrics( + task_metrics=TaskMetrics(**task_metrics), + executor_metrics=ExecutorMetrics(**line_dict["Task Executor Metrics"]), + shuffle_read_metrics=ShuffleReadMetrics( + **task_metrics["Shuffle Read Metrics"] + ), + shuffle_write_metrics=ShuffleWriteMetrics( + **task_metrics["Shuffle Write Metrics"] + ), + input_metrics=InputMetrics(**task_metrics["Input Metrics"]), + output_metrics=OutputMetrics(**task_metrics["Output Metrics"]), + ) accumulators = [ - Accumulator(task_id=task_id, **i) for i in task_info["Accumulables"] + Accumulator(task_id=task_id, **i) for i in task_info.get("Accumulables", []) ] + status, failure_reason = _task_end_reason(line_dict) return Task( metrics=metrics, accumulators=accumulators, - **line_dict["Task Info"], + status=status, + failure_reason=failure_reason, + **task_info, ) @@ -509,12 +570,25 @@ def parse_physical_plan(line_dict: dict, strict: bool = False) -> PhysicalPlan: ) -def get_parsed_log_name(parsed_plan: PhysicalPlan, out_name: str | None) -> str: +def get_parsed_log_name( + parsed_plan: PhysicalPlan | None, + out_name: str | None, + source: EventLogSource | None = None, +) -> str: + """Derive the output identity for a parsed log. + + Query ids restart at zero in every application, so the name carries the + timestamp and, when no plan paths are available, the application identity. + """ name_len_limit = 100 today = datetime.datetime.now().strftime("%Y-%m-%dT%H-%M-%S") if out_name is not None: return out_name[:name_len_limit] + if parsed_plan is None: + fallback = (source.application_id or source.name) if source else "no_queries" + return f"{today}__{fallback}"[: name_len_limit + 21] + source_model_strings = [] target_model_strings = [] @@ -528,8 +602,8 @@ def get_parsed_log_name(parsed_plan: PhysicalPlan, out_name: str | None) -> str: target_model_strings.append(node.details) sources = [] - for source in source_model_strings: - locations = deserialize_scan_detail(source).location.location + for scan_detail in source_model_strings: + locations = deserialize_scan_detail(scan_detail).location.location sources.extend(locations) targets = [] @@ -565,87 +639,195 @@ def parse_driver_accum_update(line_dict: dict) -> list[DriverAccumUpdates]: def check_if_log_has_queries(log_path: str | Path) -> bool: + """Return True if the log contains at least one SQL execution. + + Reads incrementally and stops at the first match, so a multi-gigabyte log + is not copied into memory to answer a yes/no question. + """ log_path_str = str(log_path) - if ".DS_Store" in log_path_str: + if get_path_name(log_path_str) in IGNORED_NAMES: + return False + + return source_has_queries(single_file_source(log_path_str)) + + +def source_has_queries(source: EventLogSource) -> bool: + """Return True if ``source`` contains at least one SQL execution.""" + try: + for _, _, line in iter_lines(source): + if "SparkListenerSQLExecutionStart" in line: + return True + except (UnsupportedCodecError, OSError) as exc: + logger.warning("Could not read %s: %s", source.root_uri, exc) return False + return False - with open_file(log_path_str, "r") as f: - all_contents = f.read() - return "SparkListenerSQLExecutionStart" in all_contents +# Plans arrive strongest-last: the initial plan is available at +# SQLExecutionStart, adaptive updates refine it, and the final adaptive plan is +# authoritative. A query that never reaches AQE keeps its initial plan. +PLAN_RANK_INITIAL = 0 +PLAN_RANK_ADAPTIVE = 1 +PLAN_RANK_FINAL = 2 + + +def _is_final_adaptive_plan(line_dict: dict) -> bool: + simple_string = line_dict.get("sparkPlanInfo", {}).get("simpleString", "") + return simple_string.split("isFinalPlan=")[-1] == "true" + + +def iter_events( + source: EventLogSource, + strict: bool = False, + diagnostics: list[LogDiagnostic] | None = None, +) -> Iterator[tuple[str, int, dict]]: + """Yield decoded events from ``source``, one line at a time. + + A line that fails to decode is corruption when more lines follow it and a + truncated tail when it is the last line of the last segment — a log that + was still being written. Tolerant mode records both as diagnostics; strict + mode raises with the source URI and line number. + """ + sink = diagnostics if diagnostics is not None else [] + pending: tuple[str, int, str] | None = None + + for uri, line_number, line in iter_lines(source): + if pending is not None: + bad_uri, bad_line, bad_msg = pending + pending = None + message = f"Corrupt event at {bad_uri}:{bad_line} ({bad_msg})" + if strict: + raise ValueError(message) + logger.warning(message) + sink.append( + LogDiagnostic( + code="corrupt_line", message=message, uri=bad_uri, line=bad_line + ) + ) + if not line.strip(): + continue + try: + yield uri, line_number, json.loads(line) + except json.JSONDecodeError as exc: + pending = (uri, line_number, str(exc)) + + if pending is not None: + bad_uri, bad_line, bad_msg = pending + message = ( + f"Truncated final event at {bad_uri}:{bad_line} ({bad_msg}); " + "the log was still being written or was copied mid-flush" + ) + if strict: + raise ValueError(message) + logger.warning(message) + sink.append( + LogDiagnostic( + code="truncated_tail", message=message, uri=bad_uri, line=bad_line + ) + ) @timeit -def parse_log( - log_path: str | Path, out_name: str | None = None, strict: bool = False +def parse_source( + source: EventLogSource, out_name: str | None = None, strict: bool = False ) -> ParsedLog: - log_path_str = str(log_path) - logger.debug(f"Starting to parse log file: {log_path_str}") - with open_file(log_path_str, "r") as f: - all_contents = f.readlines() - - start_point = "SparkListenerApplicationStart" - start_index: int | None = None - for i, line in enumerate(all_contents): - line_dict = json.loads(line) - if line_dict["Event"] == start_point: - start_index = i + 1 - break - - if start_index is None: - raise ValueError( - f"Could not find '{start_point}' event in log file: {log_path_str}" - ) + """Parse one logical event log into a :class:`ParsedLog`. - contents_to_parse = all_contents[start_index:] - - jobs = [] - stages = [] - tasks = [] - queries = [] - query_times = [] - driver_accum_updates = [] - for i, line in enumerate(contents_to_parse, start_index): - logger.debug("-" * 40) - logger.debug(f"[line {i:04d}] parse start") - line_dict = json.loads(line) - event_type = line_dict["Event"] - if event_type.startswith("SparkListenerJob"): + Neither ``SparkListenerApplicationStart`` nor an adaptive execution update + is required: a log missing either is incomplete, not invalid. + """ + logger.debug("Starting to parse event log source: %s", source.uris) + + diagnostics: list[LogDiagnostic] = [] + jobs: list[Job] = [] + stages: list[Stage] = [] + tasks: list[Task] = [] + query_times: list[QueryEvent] = [] + driver_accum_updates: list[DriverAccumUpdates] = [] + unknown_events: dict[str, int] = {} + + # query_id -> (rank, event) so a later, stronger plan replaces a weaker one + # without keeping every intermediate plan in memory. + plans: dict[int, tuple[int, dict]] = {} + + spark_version: str | None = None + application_id: str | None = None + application_name: str | None = None + events_read = 0 + tasks_missing_metrics = 0 + + for uri, line_number, line_dict in iter_events(source, strict, diagnostics): + events_read += 1 + event_type = line_dict.get("Event") + if event_type is None: + message = f"Event with no 'Event' field at {uri}:{line_number}" + if strict: + raise ValueError(message) + diagnostics.append( + LogDiagnostic( + code="malformed_event", + message=message, + uri=uri, + line=line_number, + ) + ) + continue + + if event_type == "SparkListenerLogStart": + spark_version = line_dict.get("Spark Version") + elif event_type == "SparkListenerApplicationStart": + application_id = line_dict.get("App ID") + application_name = line_dict.get("App Name") + elif event_type.startswith("SparkListenerJob"): job = parse_job(line_dict) jobs.append(job) + # %-style so the message is only built when debug logging is on; + # this runs once per event. logger.debug( - f"[line {i:04d}] parse finish - job#{job.job_id} type:{job.event_type}" + "[%s:%d] job#%d %s", uri, line_number, job.job_id, job.event_type ) elif event_type.startswith("SparkListenerStage"): stage = parse_stage(line_dict) stages.append(stage) logger.debug( - f"[line {i:04d}] parse finish - stage#{stage.stage_id} type:{stage.event_type}" + "[%s:%d] stage#%d.%d %s", + uri, + line_number, + stage.stage_id, + stage.stage_attempt_id, + stage.event_type, ) elif event_type == "SparkListenerTaskEnd": task = parse_task(line_dict) + if task.metrics is None: + tasks_missing_metrics += 1 tasks.append(task) logger.debug( - f"[line {i:04d}] parse finish - task#{task.task_id} stage#{task.stage_id}" + "[%s:%d] task#%d stage#%d.%d %s", + uri, + line_number, + task.task_id, + task.stage_id, + task.stage_attempt_id, + task.status, ) elif event_type.endswith("SparkListenerSQLAdaptiveExecutionUpdate"): - is_final_plan = ( - line_dict["sparkPlanInfo"]["simpleString"].split("isFinalPlan=")[-1] - == "true" - ) - if is_final_plan: - logger.debug(f"Found final plan at line {i}") - queries.append(line_dict) - logger.debug( - f"[line {i:04d}] parse skip - unhandled event type {event_type}" + rank = ( + PLAN_RANK_FINAL + if _is_final_adaptive_plan(line_dict) + else PLAN_RANK_ADAPTIVE ) + _record_plan(plans, line_dict, rank) elif event_type.endswith("SparkListenerSQLExecutionStart"): + # The initial physical plan ships with the start event. Without it + # a non-AQE workload would have no plan at all. + if "physicalPlanDescription" in line_dict: + _record_plan(plans, line_dict, PLAN_RANK_INITIAL) + description = line_dict.get("description", "") try: - query_function = QueryFunction(line_dict["description"].split(" ")[0]) + query_function = QueryFunction(description.split(" ")[0]) except ValueError: - msg = ( - f"Unknown query function: {line_dict['description'].split(' ')[0]}" - ) + msg = f"Unknown query function: {description.split(' ')[0]}" if strict: raise ValueError(msg) logger.warning(msg) @@ -668,16 +850,92 @@ def parse_log( ) elif event_type.endswith("DriverAccumUpdates"): driver_accum_updates.extend(parse_driver_accum_update(line_dict)) + else: + unknown_events[event_type] = unknown_events.get(event_type, 0) + 1 + + if application_id is None and source.application_id is not None: + # No ApplicationStart record: fall back to the identity in the path. + application_id = source.application_id + diagnostics.append( + LogDiagnostic( + code="missing_application_start", + message=( + "No SparkListenerApplicationStart event; application identity " + "was taken from the log file name." + ), + uri=source.uris[0] if source.uris else None, + ) + ) - if len(queries) == 0: - raise ValueError("No queries found in log file") + if tasks_missing_metrics: + diagnostics.append( + LogDiagnostic( + code="tasks_missing_metrics", + message=( + f"{tasks_missing_metrics} task(s) reported no Task Metrics; " + "their resource usage is unknown, not zero." + ), + ) + ) + + parsed_queries: list[PhysicalPlan] = [] + for query_id, (rank, plan_event) in sorted(plans.items()): + try: + parsed_queries.append(parse_physical_plan(plan_event, strict=strict)) + except Exception as exc: + message = f"Could not parse physical plan for query {query_id}: {exc}" + if strict: + raise + logger.warning(message) + diagnostics.append(LogDiagnostic(code="plan_parse_failed", message=message)) + continue + if rank < PLAN_RANK_FINAL: + diagnostics.append( + LogDiagnostic( + code="non_final_plan", + message=( + f"Query {query_id} has no final adaptive plan; using the " + f"{'initial' if rank == PLAN_RANK_INITIAL else 'in-progress adaptive'}" + " plan. Node metrics may be incomplete." + ), + ) + ) + + if not parsed_queries: + diagnostics.append( + LogDiagnostic( + code="no_queries", + message=( + "No SQL executions found in the event log. Non-SQL " + "applications produce jobs and tasks but no query plans." + ), + uri=source.uris[0] if source.uris else None, + ) + ) + + if not source.complete: + diagnostics.append( + LogDiagnostic( + code="incomplete_source", + message=( + "The event log is still in progress; results cover only the " + "events written so far." + ), + uri=source.uris[-1] if source.uris else None, + ) + ) - parsed_queries = [parse_physical_plan(query, strict=strict) for query in queries] logger.debug( - f"Finished parsing log [n={len(jobs)} jobs | n={len(stages)} stages | n={len(tasks)} tasks | n={len(parsed_queries)} queries]" + "Finished parsing log [n=%d jobs | n=%d stages | n=%d tasks | n=%d queries]", + len(jobs), + len(stages), + len(tasks), + len(parsed_queries), ) - parsed_log_name = get_parsed_log_name(parsed_queries[0], out_name) + parsed_log_name = get_parsed_log_name( + _plan_for_naming(parsed_queries), out_name, source + ) return ParsedLog( name=parsed_log_name, @@ -687,9 +945,52 @@ def parse_log( queries=parsed_queries, query_times=query_times, driver_accum_updates=driver_accum_updates, + source=source, + diagnostics=diagnostics, + spark_version=spark_version, + application_id=application_id, + application_name=application_name, + unknown_events=unknown_events, + events_read=events_read, ) +_NAMING_NODE_TYPES = (NodeType.Scan, NodeType.InsertIntoHadoopFsRelationCommand) + + +def _plan_for_naming(queries: list[PhysicalPlan]) -> PhysicalPlan | None: + """Pick the query whose plan names the data the run touched. + + The first query is often a schema probe with no scan detail; naming the + output after it loses the identity the paths would have given. + """ + for query in queries: + if any( + node.node_type in _NAMING_NODE_TYPES and node.details is not None + for node in query.nodes + ): + return query + return queries[0] if queries else None + + +def _record_plan( + plans: dict[int, tuple[int, dict]], line_dict: dict, rank: int +) -> None: + """Keep the strongest plan seen for a query; later ties win.""" + query_id = line_dict["executionId"] + existing = plans.get(query_id) + if existing is None or rank >= existing[0]: + plans[query_id] = (rank, line_dict) + + +def parse_log( + log_path: str | Path, out_name: str | None = None, strict: bool = False +) -> ParsedLog: + """Parse a single event-log file. See :func:`parse_source` for directories + of rolled segments.""" + return parse_source(single_file_source(str(log_path)), out_name, strict=strict) + + def get_parsed_metrics( log_dir: str | Path = "data/logs/raw", log_file: str | None = None, @@ -699,9 +1000,45 @@ def get_parsed_metrics( verbose: bool = False, strict: bool = False, ) -> ParsedLogDataFrames: - log_dir_str = str(log_dir) - cloud = is_cloud_path(log_dir_str) + """Parse one event-log source from ``log_dir`` into DataFrames. + + ``log_file`` selects a source explicitly (file name, rolling-log directory + name, application id, or full path). Without it the newest source is used. + Use :func:`get_all_parsed_metrics` to parse every application in a + directory. + """ + _configure_logging(verbose) + source = resolve_source(_resolve_log_dir(log_dir), log_file) + return _parse_and_write(source, out_dir, out_name, out_format, strict) + + +def get_all_parsed_metrics( + log_dir: str | Path = "data/logs/raw", + out_dir: str | None = "data/logs/parsed", + out_format: OutputFormat | None = OutputFormat.csv, + verbose: bool = False, + strict: bool = False, +) -> dict[str, ParsedLogDataFrames]: + """Parse every application under ``log_dir``, keyed by source name. + + Query ids restart at zero in each application, so results are kept per + source rather than concatenated. + """ + _configure_logging(verbose) + resolved_dir = _resolve_log_dir(log_dir) + sources = discover_sources(resolved_dir) + if not sources: + raise EventLogNotFoundError(f"No event log files found in: {resolved_dir}") + + results: dict[str, ParsedLogDataFrames] = {} + for source in sorted(sources, key=lambda item: item.name): + results[source.name] = _parse_and_write( + source, out_dir, None, out_format, strict, disambiguate=True + ) + return results + +def _configure_logging(verbose: bool) -> None: if verbose: logging.basicConfig( level=logging.DEBUG, @@ -711,35 +1048,48 @@ def get_parsed_metrics( else: logging.basicConfig(level=logging.INFO, format="%(message)s") - if cloud: - if log_file is None: - candidates = sorted(list_files(log_dir_str)) - if not candidates: - raise ValueError(f"No log files found in cloud path: {log_dir_str}") - log_to_parse = candidates[-1] - else: - log_to_parse = join_path(log_dir_str, log_file) - log_stem = get_path_stem(log_to_parse) - else: - log_dir_path = Path(resolve_dir(log_dir)) - if log_file is None: - log_to_parse = sorted(log_dir_path.glob("*"))[-1] - else: - log_to_parse = log_dir_path / log_file - log_stem = log_to_parse.stem - logging.info(f"Reading log file: {log_to_parse}") +def _resolve_log_dir(log_dir: str | Path) -> str: + if is_cloud_path(str(log_dir)): + return str(log_dir) + return str(resolve_dir(log_dir)) - result = parse_log(log_to_parse, out_name, strict=strict) - dag_df = log_to_dag_df(result) - combined_df = log_to_combined_df(result, dag_df, log_stem) - output = ParsedLogDataFrames(combined=combined_df, dag=dag_df) +def _parse_and_write( + source: EventLogSource, + out_dir: str | None, + out_name: str | None, + out_format: OutputFormat | None, + strict: bool, + disambiguate: bool = False, +) -> ParsedLogDataFrames: + logging.info(f"Reading event log: {source.root_uri or source.uris}") + + result = parse_source(source, out_name, strict=strict) + for diagnostic in result.diagnostics: + logging.info(f"[{diagnostic.code}] {diagnostic.message}") + + dag_df = log_to_dag_df(result) + combined_df = log_to_combined_df(result, dag_df, source.name) + + output = ParsedLogDataFrames( + combined=combined_df, + dag=dag_df, + job_stage=job_stage_associations(result), + query_stage=query_stage_associations(dag_df), + diagnostics=result.diagnostics, + ) if out_dir is None or out_format is None: logging.info("Skipping writing parsed log") return output + parsed_name = result.name + if disambiguate and out_name is None: + # Two applications reading the same paths in the same second derive the + # same name; the application identity keeps their outputs apart. + parsed_name = f"{parsed_name}__{source.name}"[:160] + if out_format == OutputFormat.csv: dag_df = ( dag_df.explode("accumulators") @@ -766,7 +1116,7 @@ def get_parsed_metrics( df=dag_df, out_dir=out_dir, out_format=out_format, - parsed_name=result.name, + parsed_name=parsed_name, suffix="_dag", ) @@ -774,7 +1124,7 @@ def get_parsed_metrics( df=combined_df, out_dir=out_dir, out_format=out_format, - parsed_name=result.name, + parsed_name=parsed_name, suffix="_combined", ) diff --git a/sparkparse/schemas.py b/sparkparse/schemas.py new file mode 100644 index 0000000..f8387fa --- /dev/null +++ b/sparkparse/schemas.py @@ -0,0 +1,81 @@ +"""Canonical output frame schemas shared by every capture backend. + +An empty result still has to be typed: downstream analysis distinguishes "no +rows" from "column missing", and an untyped empty frame collapses that +difference. +""" + +from __future__ import annotations + +import polars as pl + +from sparkparse.models import ParsedLogDataFrames + +ACCUM_TOTALS_STRUCT = pl.Struct( + { + "metric_name": pl.Utf8, + "metric_type": pl.Utf8, + "value": pl.Float64, + # Exact integer counterpart of ``value``; null when the metric is + # fractional or was rescaled. ``value`` is float64 and cannot hold a + # count above 2**53 without rounding it. + "value_exact": pl.Int64, + "readable_value": pl.Float64, + "readable_unit": pl.Utf8, + "readable_str": pl.Utf8, + } +) + +DAG_SCHEMA: dict[str, pl.PolarsDataType] = { + "log_name": pl.Utf8, + "parsed_log_name": pl.Utf8, + "query_id": pl.Int64, + "query_function": pl.Utf8, + "query_header": pl.Utf8, + "query_start_timestamp": pl.Utf8, + "query_end_timestamp": pl.Utf8, + "query_duration_seconds": pl.Float64, + "source_execution_id": pl.Utf8, + "node_id": pl.Int64, + "node_type": pl.Utf8, + "node_name": pl.Utf8, + "child_nodes": pl.Utf8, + "whole_stage_codegen_id": pl.Int64, + "details": pl.Utf8, + "accumulator_totals": pl.List(ACCUM_TOTALS_STRUCT), + "n_accumulator_totals": pl.Int64, + "node_duration_minutes": pl.Float64, + "n_accumulators": pl.Int64, + "node_id_adj": pl.Int64, +} + +COMBINED_SCHEMA: dict[str, pl.PolarsDataType] = { + "log_name": pl.Utf8, + "parsed_log_name": pl.Utf8, + "query_id": pl.Int64, + "stage_id": pl.Int64, + "task_id": pl.Int64, + "task_duration_seconds": pl.Float64, + "bytes_read": pl.Int64, + "records_read": pl.Int64, + "bytes_written": pl.Int64, + "records_written": pl.Int64, + "memory_bytes_spilled": pl.Int64, + "disk_bytes_spilled": pl.Int64, + "shuffle_bytes_read": pl.Int64, + "shuffle_bytes_written": pl.Int64, + "shuffle_remote_bytes_read": pl.Int64, + "shuffle_local_bytes_read": pl.Int64, + "executor_run_time_seconds": pl.Float64, + "jvm_gc_time_seconds": pl.Float64, + "executor_id": pl.Utf8, + "nodes": pl.List(pl.Utf8), +} + + +def empty_capture_dataframes() -> ParsedLogDataFrames: + """Return the canonical typed empty frames used by capture backends.""" + return ParsedLogDataFrames( + dag=pl.DataFrame(schema=DAG_SCHEMA), + combined=pl.DataFrame(schema=COMBINED_SCHEMA), + ) diff --git a/tests/benchmark_ingestion.py b/tests/benchmark_ingestion.py new file mode 100644 index 0000000..7517303 --- /dev/null +++ b/tests/benchmark_ingestion.py @@ -0,0 +1,69 @@ +"""Measure event-log ingestion cost: elapsed time and process peak RSS. + +Run as a script so the reported peak RSS belongs to one parse only: +``ru_maxrss`` is a process high-water mark that never falls, so measuring +several sizes inside one interpreter would report the largest run for all of +them. + + python -m tests.benchmark_ingestion [log_file] + +Prints one JSON object describing the run. ``tests/test_eventlog.py`` drives it +across increasing fixtures; it is also usable by hand against a real log. +""" + +from __future__ import annotations + +import json +import resource +import sys +import time + + +def _peak_rss_bytes() -> int: + """Peak resident set size of this process, in bytes. + + ``ru_maxrss`` is bytes on macOS and kilobytes on Linux. + """ + peak = resource.getrusage(resource.RUSAGE_SELF).ru_maxrss + return peak if sys.platform == "darwin" else peak * 1024 + + +def measure(log_dir: str, log_file: str | None = None) -> dict: + # Imported here so the interpreter/library startup cost is paid before the + # baseline RSS reading below. + from sparkparse.eventlog import resolve_source + from sparkparse.parse import parse_source + + source = resolve_source(log_dir, log_file) + baseline_rss = _peak_rss_bytes() + + start = time.perf_counter() + parsed = parse_source(source) + elapsed = time.perf_counter() - start + + peak_rss = _peak_rss_bytes() + return { + "log": source.root_uri, + "events_read": parsed.events_read, + "tasks": len(parsed.tasks), + "stages": len(parsed.stages), + "queries": len(parsed.queries), + "elapsed_s": elapsed, + "baseline_rss_bytes": baseline_rss, + "peak_rss_bytes": peak_rss, + "retained_rss_bytes": max(peak_rss - baseline_rss, 0), + } + + +def main(argv: list[str]) -> int: + if not argv: + print(__doc__, file=sys.stderr) + return 2 + log_dir = argv[0] + log_file = argv[1] if len(argv) > 1 else None + print(json.dumps(measure(log_dir, log_file))) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv[1:])) diff --git a/tests/data/test_full_parsing/expected_nested_final_plans.json b/tests/data/test_full_parsing/expected_nested_final_plans.json index b9fc9ab..3502084 100644 --- a/tests/data/test_full_parsing/expected_nested_final_plans.json +++ b/tests/data/test_full_parsing/expected_nested_final_plans.json @@ -95,7 +95,7 @@ "accumulators": [ { "stage_id": 2, - "task_id": 9, + "task_id": 2, "accumulator_id": 132, "metric_name": "scan time", "metric_type": "timing", @@ -108,20 +108,20 @@ }, { "stage_id": 2, - "task_id": 2, + "task_id": 3, "accumulator_id": 132, "metric_name": "scan time", "metric_type": "timing", - "value": 798.0, - "value_exact": 798, + "value": 797.0, + "value_exact": 797, "unit": "ms", - "readable_value": 798.0, + "readable_value": 797.0, "readable_unit": "ms", - "readable_str": "798.0 ms" + "readable_str": "797.0 ms" }, { "stage_id": 2, - "task_id": 8, + "task_id": 4, "accumulator_id": 132, "metric_name": "scan time", "metric_type": "timing", @@ -134,20 +134,20 @@ }, { "stage_id": 2, - "task_id": 3, + "task_id": 5, "accumulator_id": 132, "metric_name": "scan time", "metric_type": "timing", - "value": 797.0, - "value_exact": 797, + "value": 795.0, + "value_exact": 795, "unit": "ms", - "readable_value": 797.0, + "readable_value": 795.0, "readable_unit": "ms", - "readable_str": "797.0 ms" + "readable_str": "795.0 ms" }, { "stage_id": 2, - "task_id": 10, + "task_id": 6, "accumulator_id": 132, "metric_name": "scan time", "metric_type": "timing", @@ -160,20 +160,20 @@ }, { "stage_id": 2, - "task_id": 11, + "task_id": 7, "accumulator_id": 132, "metric_name": "scan time", "metric_type": "timing", - "value": 798.0, - "value_exact": 798, + "value": 796.0, + "value_exact": 796, "unit": "ms", - "readable_value": 798.0, + "readable_value": 796.0, "readable_unit": "ms", - "readable_str": "798.0 ms" + "readable_str": "796.0 ms" }, { "stage_id": 2, - "task_id": 4, + "task_id": 8, "accumulator_id": 132, "metric_name": "scan time", "metric_type": "timing", @@ -186,42 +186,42 @@ }, { "stage_id": 2, - "task_id": 5, + "task_id": 9, "accumulator_id": 132, "metric_name": "scan time", "metric_type": "timing", - "value": 795.0, - "value_exact": 795, + "value": 798.0, + "value_exact": 798, "unit": "ms", - "readable_value": 795.0, + "readable_value": 798.0, "readable_unit": "ms", - "readable_str": "795.0 ms" + "readable_str": "798.0 ms" }, { "stage_id": 2, - "task_id": 7, + "task_id": 10, "accumulator_id": 132, "metric_name": "scan time", "metric_type": "timing", - "value": 796.0, - "value_exact": 796, + "value": 797.0, + "value_exact": 797, "unit": "ms", - "readable_value": 796.0, + "readable_value": 797.0, "readable_unit": "ms", - "readable_str": "796.0 ms" + "readable_str": "797.0 ms" }, { "stage_id": 2, - "task_id": 6, + "task_id": 11, "accumulator_id": 132, "metric_name": "scan time", "metric_type": "timing", - "value": 797.0, - "value_exact": 797, + "value": 798.0, + "value_exact": 798, "unit": "ms", - "readable_value": 797.0, + "readable_value": 798.0, "readable_unit": "ms", - "readable_str": "797.0 ms" + "readable_str": "798.0 ms" }, { "stage_id": 2, @@ -442,7 +442,7 @@ "accumulators": [ { "stage_id": 2, - "task_id": 6, + "task_id": 2, "accumulator_id": 221, "metric_name": "number of input batches", "metric_type": "sum", @@ -455,7 +455,7 @@ }, { "stage_id": 2, - "task_id": 2, + "task_id": 3, "accumulator_id": 221, "metric_name": "number of input batches", "metric_type": "sum", @@ -468,7 +468,7 @@ }, { "stage_id": 2, - "task_id": 3, + "task_id": 4, "accumulator_id": 221, "metric_name": "number of input batches", "metric_type": "sum", @@ -481,7 +481,7 @@ }, { "stage_id": 2, - "task_id": 4, + "task_id": 5, "accumulator_id": 221, "metric_name": "number of input batches", "metric_type": "sum", @@ -494,7 +494,7 @@ }, { "stage_id": 2, - "task_id": 5, + "task_id": 6, "accumulator_id": 221, "metric_name": "number of input batches", "metric_type": "sum", @@ -572,7 +572,7 @@ }, { "stage_id": 2, - "task_id": 9, + "task_id": 2, "accumulator_id": 220, "metric_name": "number of output rows", "metric_type": "sum", @@ -585,7 +585,7 @@ }, { "stage_id": 2, - "task_id": 2, + "task_id": 3, "accumulator_id": 220, "metric_name": "number of output rows", "metric_type": "sum", @@ -598,7 +598,7 @@ }, { "stage_id": 2, - "task_id": 3, + "task_id": 4, "accumulator_id": 220, "metric_name": "number of output rows", "metric_type": "sum", @@ -611,7 +611,7 @@ }, { "stage_id": 2, - "task_id": 4, + "task_id": 5, "accumulator_id": 220, "metric_name": "number of output rows", "metric_type": "sum", @@ -624,7 +624,7 @@ }, { "stage_id": 2, - "task_id": 5, + "task_id": 6, "accumulator_id": 220, "metric_name": "number of output rows", "metric_type": "sum", @@ -637,7 +637,7 @@ }, { "stage_id": 2, - "task_id": 6, + "task_id": 7, "accumulator_id": 220, "metric_name": "number of output rows", "metric_type": "sum", @@ -650,7 +650,7 @@ }, { "stage_id": 2, - "task_id": 7, + "task_id": 8, "accumulator_id": 220, "metric_name": "number of output rows", "metric_type": "sum", @@ -663,7 +663,7 @@ }, { "stage_id": 2, - "task_id": 8, + "task_id": 9, "accumulator_id": 220, "metric_name": "number of output rows", "metric_type": "sum", @@ -796,68 +796,68 @@ }, { "stage_id": 2, - "task_id": 6, + "task_id": 2, "accumulator_id": 218, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 7.078793, + "value": 7.109666, "value_exact": null, "unit": "ms", "readable_value": 7.1, "readable_unit": "ms", - "readable_str": "7.08 ms" + "readable_str": "7.11 ms" }, { "stage_id": 2, - "task_id": 11, + "task_id": 3, "accumulator_id": 218, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 5.42842, + "value": 6.600916, "value_exact": null, "unit": "ms", - "readable_value": 5.4, + "readable_value": 6.6, "readable_unit": "ms", - "readable_str": "5.43 ms" + "readable_str": "6.6 ms" }, { "stage_id": 2, - "task_id": 10, + "task_id": 4, "accumulator_id": 218, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 4.9475, + "value": 5.071706, "value_exact": null, "unit": "ms", - "readable_value": 4.9, + "readable_value": 5.1, "readable_unit": "ms", - "readable_str": "4.95 ms" + "readable_str": "5.07 ms" }, { "stage_id": 2, - "task_id": 9, + "task_id": 5, "accumulator_id": 218, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 6.893539, + "value": 6.927581, "value_exact": null, "unit": "ms", "readable_value": 6.9, "readable_unit": "ms", - "readable_str": "6.89 ms" + "readable_str": "6.93 ms" }, { "stage_id": 2, - "task_id": 8, + "task_id": 6, "accumulator_id": 218, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 6.705041, + "value": 7.078793, "value_exact": null, "unit": "ms", - "readable_value": 6.7, + "readable_value": 7.1, "readable_unit": "ms", - "readable_str": "6.71 ms" + "readable_str": "7.08 ms" }, { "stage_id": 2, @@ -874,59 +874,59 @@ }, { "stage_id": 2, - "task_id": 2, + "task_id": 8, "accumulator_id": 218, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 7.109666, + "value": 6.705041, "value_exact": null, "unit": "ms", - "readable_value": 7.1, + "readable_value": 6.7, "readable_unit": "ms", - "readable_str": "7.11 ms" + "readable_str": "6.71 ms" }, { "stage_id": 2, - "task_id": 3, + "task_id": 9, "accumulator_id": 218, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 6.600916, + "value": 6.893539, "value_exact": null, "unit": "ms", - "readable_value": 6.6, + "readable_value": 6.9, "readable_unit": "ms", - "readable_str": "6.6 ms" + "readable_str": "6.89 ms" }, { "stage_id": 2, - "task_id": 4, + "task_id": 10, "accumulator_id": 218, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 5.071706, + "value": 4.9475, "value_exact": null, "unit": "ms", - "readable_value": 5.1, + "readable_value": 4.9, "readable_unit": "ms", - "readable_str": "5.07 ms" + "readable_str": "4.95 ms" }, { "stage_id": 2, - "task_id": 5, + "task_id": 11, "accumulator_id": 218, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 6.927581, + "value": 5.42842, "value_exact": null, "unit": "ms", - "readable_value": 6.9, + "readable_value": 5.4, "readable_unit": "ms", - "readable_str": "6.93 ms" + "readable_str": "5.43 ms" }, { "stage_id": 2, - "task_id": 4, + "task_id": 2, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -939,7 +939,7 @@ }, { "stage_id": 2, - "task_id": 11, + "task_id": 3, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -952,7 +952,7 @@ }, { "stage_id": 2, - "task_id": 10, + "task_id": 4, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -965,7 +965,7 @@ }, { "stage_id": 2, - "task_id": 9, + "task_id": 5, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -978,7 +978,7 @@ }, { "stage_id": 2, - "task_id": 8, + "task_id": 6, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -1004,7 +1004,7 @@ }, { "stage_id": 2, - "task_id": 6, + "task_id": 8, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -1017,7 +1017,7 @@ }, { "stage_id": 2, - "task_id": 5, + "task_id": 9, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -1030,7 +1030,7 @@ }, { "stage_id": 2, - "task_id": 3, + "task_id": 10, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -1043,7 +1043,7 @@ }, { "stage_id": 2, - "task_id": 2, + "task_id": 11, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -1080,45 +1080,6 @@ "readable_unit": "KiB", "readable_str": "484.47 KiB" }, - { - "stage_id": 2, - "task_id": 11, - "accumulator_id": 216, - "metric_name": "shuffle bytes written", - "metric_type": "size", - "value": 49486.0, - "value_exact": 49486, - "unit": "B", - "readable_value": 48.3, - "readable_unit": "KiB", - "readable_str": "48.33 KiB" - }, - { - "stage_id": 2, - "task_id": 10, - "accumulator_id": 216, - "metric_name": "shuffle bytes written", - "metric_type": "size", - "value": 49459.0, - "value_exact": 49459, - "unit": "B", - "readable_value": 48.3, - "readable_unit": "KiB", - "readable_str": "48.3 KiB" - }, - { - "stage_id": 2, - "task_id": 9, - "accumulator_id": 216, - "metric_name": "shuffle bytes written", - "metric_type": "size", - "value": 49719.0, - "value_exact": 49719, - "unit": "B", - "readable_value": 48.6, - "readable_unit": "KiB", - "readable_str": "48.55 KiB" - }, { "stage_id": 2, "task_id": 2, @@ -1211,34 +1172,73 @@ "readable_str": "48.31 KiB" }, { - "stage_id": 4, - "task_id": 13, - "accumulator_id": 200, - "metric_name": "local blocks read", - "metric_type": "sum", - "value": 10.0, - "value_exact": 10, - "unit": "", - "readable_value": 10.0, - "readable_unit": "", - "readable_str": "10.0" - }, - { - "stage_id": 4, - "task_id": 12, - "accumulator_id": 200, - "metric_name": "local blocks read", - "metric_type": "sum", - "value": 10.0, - "value_exact": 10, - "unit": "", - "readable_value": 10.0, - "readable_unit": "", - "readable_str": "10.0" + "stage_id": 2, + "task_id": 9, + "accumulator_id": 216, + "metric_name": "shuffle bytes written", + "metric_type": "size", + "value": 49719.0, + "value_exact": 49719, + "unit": "B", + "readable_value": 48.6, + "readable_unit": "KiB", + "readable_str": "48.55 KiB" }, { - "stage_id": 4, - "task_id": 12, + "stage_id": 2, + "task_id": 10, + "accumulator_id": 216, + "metric_name": "shuffle bytes written", + "metric_type": "size", + "value": 49459.0, + "value_exact": 49459, + "unit": "B", + "readable_value": 48.3, + "readable_unit": "KiB", + "readable_str": "48.3 KiB" + }, + { + "stage_id": 2, + "task_id": 11, + "accumulator_id": 216, + "metric_name": "shuffle bytes written", + "metric_type": "size", + "value": 49486.0, + "value_exact": 49486, + "unit": "B", + "readable_value": 48.3, + "readable_unit": "KiB", + "readable_str": "48.33 KiB" + }, + { + "stage_id": 4, + "task_id": 12, + "accumulator_id": 200, + "metric_name": "local blocks read", + "metric_type": "sum", + "value": 10.0, + "value_exact": 10, + "unit": "", + "readable_value": 10.0, + "readable_unit": "", + "readable_str": "10.0" + }, + { + "stage_id": 4, + "task_id": 13, + "accumulator_id": 200, + "metric_name": "local blocks read", + "metric_type": "sum", + "value": 10.0, + "value_exact": 10, + "unit": "", + "readable_value": 10.0, + "readable_unit": "", + "readable_str": "10.0" + }, + { + "stage_id": 4, + "task_id": 12, "accumulator_id": 205, "metric_name": "records read", "metric_type": "sum", @@ -1264,7 +1264,7 @@ }, { "stage_id": 2, - "task_id": 11, + "task_id": 2, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1277,7 +1277,7 @@ }, { "stage_id": 2, - "task_id": 6, + "task_id": 3, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1290,7 +1290,7 @@ }, { "stage_id": 2, - "task_id": 10, + "task_id": 4, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1303,7 +1303,7 @@ }, { "stage_id": 2, - "task_id": 9, + "task_id": 5, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1316,7 +1316,7 @@ }, { "stage_id": 2, - "task_id": 8, + "task_id": 6, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1342,7 +1342,7 @@ }, { "stage_id": 2, - "task_id": 2, + "task_id": 8, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1355,7 +1355,7 @@ }, { "stage_id": 2, - "task_id": 3, + "task_id": 9, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1368,7 +1368,7 @@ }, { "stage_id": 2, - "task_id": 4, + "task_id": 10, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1381,7 +1381,7 @@ }, { "stage_id": 2, - "task_id": 5, + "task_id": 11, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1719,7 +1719,7 @@ }, { "stage_id": 6, - "task_id": 15, + "task_id": 14, "accumulator_id": 323, "metric_name": "number of output rows", "metric_type": "sum", @@ -1732,7 +1732,7 @@ }, { "stage_id": 6, - "task_id": 14, + "task_id": 15, "accumulator_id": 323, "metric_name": "number of output rows", "metric_type": "sum", @@ -1871,7 +1871,7 @@ }, { "stage_id": 6, - "task_id": 15, + "task_id": 14, "accumulator_id": 319, "metric_name": "shuffle bytes written", "metric_type": "size", @@ -1884,7 +1884,7 @@ }, { "stage_id": 6, - "task_id": 14, + "task_id": 15, "accumulator_id": 319, "metric_name": "shuffle bytes written", "metric_type": "size", @@ -1923,7 +1923,7 @@ }, { "stage_id": 6, - "task_id": 15, + "task_id": 14, "accumulator_id": 320, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1936,7 +1936,7 @@ }, { "stage_id": 6, - "task_id": 14, + "task_id": 15, "accumulator_id": 320, "metric_name": "shuffle records written", "metric_type": "sum", @@ -2182,16 +2182,16 @@ "accumulators": [ { "stage_id": 2, - "task_id": 11, + "task_id": 2, "accumulator_id": 219, "metric_name": "duration", "metric_type": "timing", - "value": 825.0, - "value_exact": 825, + "value": 823.0, + "value_exact": 823, "unit": "ms", - "readable_value": 825.0, + "readable_value": 823.0, "readable_unit": "ms", - "readable_str": "825.0 ms" + "readable_str": "823.0 ms" }, { "stage_id": 2, @@ -2208,42 +2208,42 @@ }, { "stage_id": 2, - "task_id": 9, + "task_id": 4, "accumulator_id": 219, "metric_name": "duration", "metric_type": "timing", - "value": 822.0, - "value_exact": 822, + "value": 825.0, + "value_exact": 825, "unit": "ms", - "readable_value": 822.0, + "readable_value": 825.0, "readable_unit": "ms", - "readable_str": "822.0 ms" + "readable_str": "825.0 ms" }, { "stage_id": 2, - "task_id": 10, + "task_id": 5, "accumulator_id": 219, "metric_name": "duration", "metric_type": "timing", - "value": 825.0, - "value_exact": 825, + "value": 822.0, + "value_exact": 822, "unit": "ms", - "readable_value": 825.0, + "readable_value": 822.0, "readable_unit": "ms", - "readable_str": "825.0 ms" + "readable_str": "822.0 ms" }, { "stage_id": 2, - "task_id": 8, + "task_id": 6, "accumulator_id": 219, "metric_name": "duration", "metric_type": "timing", - "value": 823.0, - "value_exact": 823, + "value": 822.0, + "value_exact": 822, "unit": "ms", - "readable_value": 823.0, + "readable_value": 822.0, "readable_unit": "ms", - "readable_str": "823.0 ms" + "readable_str": "822.0 ms" }, { "stage_id": 2, @@ -2260,20 +2260,20 @@ }, { "stage_id": 2, - "task_id": 6, + "task_id": 8, "accumulator_id": 219, "metric_name": "duration", "metric_type": "timing", - "value": 822.0, - "value_exact": 822, + "value": 823.0, + "value_exact": 823, "unit": "ms", - "readable_value": 822.0, + "readable_value": 823.0, "readable_unit": "ms", - "readable_str": "822.0 ms" + "readable_str": "823.0 ms" }, { "stage_id": 2, - "task_id": 5, + "task_id": 9, "accumulator_id": 219, "metric_name": "duration", "metric_type": "timing", @@ -2286,20 +2286,20 @@ }, { "stage_id": 2, - "task_id": 2, + "task_id": 10, "accumulator_id": 219, "metric_name": "duration", "metric_type": "timing", - "value": 823.0, - "value_exact": 823, + "value": 825.0, + "value_exact": 825, "unit": "ms", - "readable_value": 823.0, + "readable_value": 825.0, "readable_unit": "ms", - "readable_str": "823.0 ms" + "readable_str": "825.0 ms" }, { "stage_id": 2, - "task_id": 4, + "task_id": 11, "accumulator_id": 219, "metric_name": "duration", "metric_type": "timing", @@ -2477,141 +2477,282 @@ "node_name": "[4] WholeStageCodegen" }, { - "query_id": 2, + "query_id": 1, "query_start_timestamp": "2025-03-22T01:19:40", "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 0.31, + "query_duration_seconds": 0.01, "whole_stage_codegen_id": null, "node_id": 1, - "node_type": "InMemoryTableScan", + "node_type": "Unknown", "child_nodes": "2", - "details": "{\"node_id\":1,\"node_type\":\"InMemoryTableScan\",\"detail\":{\"output\":[\"id1#0\",\"v3#8\"]}}", - "query_function": "count", - "query_header": "2 - count [0.01 min]", - "accumulators": [ - { - "stage_id": 11, - "task_id": 17, - "accumulator_id": 474, - "metric_name": "number of output rows", - "metric_type": "sum", - "value": 1000.0, - "value_exact": 1000, - "unit": "", - "readable_value": 1000.0, - "readable_unit": "", - "readable_str": "1000.0" - }, - { - "stage_id": 11, - "task_id": 18, - "accumulator_id": 474, - "metric_name": "number of output rows", - "metric_type": "sum", - "value": 1000.0, - "value_exact": 1000, - "unit": "", - "readable_value": 1000.0, - "readable_unit": "", - "readable_str": "1000.0" - } - ], - "accumulator_totals": [ - { - "stage_id": 11, - "task_id": 17, - "accumulator_id": 474, - "metric_name": "number of output rows", - "metric_type": "sum", - "value": 2000.0, - "value_exact": 2000, - "unit": "", - "readable_value": 2000.0, - "readable_unit": "", - "readable_str": "2000.0" - } - ], - "n_accumulators": 2, - "n_accumulator_totals": 1, + "details": "{\"node_id\":1,\"node_type\":\"Unknown\",\"detail\":{\"raw\":\"(1) Execute CreateViewCommand\\nOutput: []\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "1 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, "node_duration_minutes": null, "node_id_adj": 1, - "node_name": "[1] InMemoryTableScan" + "node_name": "[1] Unknown" }, { - "query_id": 2, + "query_id": 1, "query_start_timestamp": "2025-03-22T01:19:40", "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 0.31, + "query_duration_seconds": 0.01, "whole_stage_codegen_id": null, "node_id": 2, - "node_type": "InMemoryRelation", - "child_nodes": "21", - "details": "{\"node_id\":2,\"node_type\":\"InMemoryRelation\",\"detail\":{\"arguments\":\"[id1#0, id2#1, id3#2, id4#3L, id5#4L, id6#5L, v1#6L, v2#7L, v3#8], CachedRDDBuilder(org.apache.spark.sql.execution.columnar.DefaultCachedBatchSerializer@d36a649,StorageLevel(disk, memory, deserialized, 1 replicas),AdaptiveSparkPlan isFinalPlan=true\"}}", - "query_function": "count", - "query_header": "2 - count [0.01 min]", + "node_type": "Unknown", + "child_nodes": "9", + "details": "{\"node_id\":2,\"node_type\":\"Unknown\",\"detail\":{\"raw\":\"(2) CreateViewCommand\\nArguments: `cached_union`, false, true, LocalTempView, true\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "1 - createOrReplaceTempView [0.0 min]", "accumulators": null, "accumulator_totals": null, "n_accumulators": null, "n_accumulator_totals": null, "node_duration_minutes": null, "node_id_adj": 2, - "node_name": "[2] InMemoryRelation" + "node_name": "[2] Unknown" }, { - "query_id": 2, + "query_id": 1, "query_start_timestamp": "2025-03-22T01:19:40", "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 0.31, + "query_duration_seconds": 0.01, "whole_stage_codegen_id": null, "node_id": 3, - "node_type": "Scan", + "node_type": "Unknown", "child_nodes": "", - "details": "{\"node_id\":3,\"node_type\":\"Scan\",\"detail\":{\"output\":[\"id1#0\",\"id2#1\",\"id3#2\",\"id4#3L\",\"id5#4L\",\"id6#5L\",\"v1#6L\",\"v2#7L\",\"v3#8\"],\"batched\":true,\"location\":{\"location_type\":\"InMemoryFileIndex\",\"location\":[\"sparkparse/data/raw/G1_1e7_1e7_100_0.parquet\"]},\"read_schema\":\"struct\"}}", - "query_function": "count", - "query_header": "2 - count [0.01 min]", - "accumulators": [ - { - "stage_id": 2, - "task_id": 2, - "accumulator_id": 132, - "metric_name": "scan time", - "metric_type": "timing", - "value": 798.0, - "value_exact": 798, - "unit": "ms", - "readable_value": 798.0, - "readable_unit": "ms", - "readable_str": "798.0 ms" - }, - { - "stage_id": 2, - "task_id": 11, - "accumulator_id": 132, - "metric_name": "scan time", - "metric_type": "timing", - "value": 798.0, - "value_exact": 798, - "unit": "ms", - "readable_value": 798.0, - "readable_unit": "ms", - "readable_str": "798.0 ms" - }, - { - "stage_id": 2, - "task_id": 10, - "accumulator_id": 132, - "metric_name": "scan time", - "metric_type": "timing", - "value": 797.0, - "value_exact": 797, - "unit": "ms", - "readable_value": 797.0, - "readable_unit": "ms", - "readable_str": "797.0 ms" + "details": "{\"node_id\":3,\"node_type\":\"Unknown\",\"detail\":{\"raw\":\"(3) LogicalRelation\\nArguments: parquet, [id1#0, id2#1, id3#2, id4#3L, id5#4L, id6#5L, v1#6L, v2#7L, v3#8], false\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "1 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 3, + "node_name": "[3] Unknown" + }, + { + "query_id": 1, + "query_start_timestamp": "2025-03-22T01:19:40", + "query_end_timestamp": "2025-03-22T01:19:40", + "query_duration_seconds": 0.01, + "whole_stage_codegen_id": null, + "node_id": 4, + "node_type": "LocalLimit", + "child_nodes": "3", + "details": "{\"node_id\":4,\"node_type\":\"LocalLimit\",\"detail\":{\"raw\":\"(4) LocalLimit\\nArguments: 1000\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "1 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 4, + "node_name": "[4] LocalLimit" + }, + { + "query_id": 1, + "query_start_timestamp": "2025-03-22T01:19:40", + "query_end_timestamp": "2025-03-22T01:19:40", + "query_duration_seconds": 0.01, + "whole_stage_codegen_id": null, + "node_id": 5, + "node_type": "GlobalLimit", + "child_nodes": "4", + "details": "{\"node_id\":5,\"node_type\":\"GlobalLimit\",\"detail\":{\"raw\":\"(5) GlobalLimit\\nArguments: 1000\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "1 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 5, + "node_name": "[5] GlobalLimit" + }, + { + "query_id": 1, + "query_start_timestamp": "2025-03-22T01:19:40", + "query_end_timestamp": "2025-03-22T01:19:40", + "query_duration_seconds": 0.01, + "whole_stage_codegen_id": null, + "node_id": 6, + "node_type": "Unknown", + "child_nodes": "", + "details": "{\"node_id\":6,\"node_type\":\"Unknown\",\"detail\":{\"raw\":\"(6) LogicalRelation\\nArguments: parquet, [id1#18, id2#19, id3#20, id4#21L, id5#22L, id6#23L, v1#24L, v2#25L, v3#26], false\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "1 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 6, + "node_name": "[6] Unknown" + }, + { + "query_id": 1, + "query_start_timestamp": "2025-03-22T01:19:40", + "query_end_timestamp": "2025-03-22T01:19:40", + "query_duration_seconds": 0.01, + "whole_stage_codegen_id": null, + "node_id": 7, + "node_type": "LocalLimit", + "child_nodes": "6", + "details": "{\"node_id\":7,\"node_type\":\"LocalLimit\",\"detail\":{\"raw\":\"(7) LocalLimit\\nArguments: 1000\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "1 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 7, + "node_name": "[7] LocalLimit" + }, + { + "query_id": 1, + "query_start_timestamp": "2025-03-22T01:19:40", + "query_end_timestamp": "2025-03-22T01:19:40", + "query_duration_seconds": 0.01, + "whole_stage_codegen_id": null, + "node_id": 8, + "node_type": "GlobalLimit", + "child_nodes": "7", + "details": "{\"node_id\":8,\"node_type\":\"GlobalLimit\",\"detail\":{\"raw\":\"(8) GlobalLimit\\nArguments: 1000\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "1 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 8, + "node_name": "[8] GlobalLimit" + }, + { + "query_id": 1, + "query_start_timestamp": "2025-03-22T01:19:40", + "query_end_timestamp": "2025-03-22T01:19:40", + "query_duration_seconds": 0.01, + "whole_stage_codegen_id": null, + "node_id": 9, + "node_type": "Union", + "child_nodes": "5, 8", + "details": "{\"node_id\":9,\"node_type\":\"Union\",\"detail\":null}", + "query_function": "createOrReplaceTempView", + "query_header": "1 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 9, + "node_name": "[9] Union" + }, + { + "query_id": 2, + "query_start_timestamp": "2025-03-22T01:19:40", + "query_end_timestamp": "2025-03-22T01:19:40", + "query_duration_seconds": 0.31, + "whole_stage_codegen_id": null, + "node_id": 1, + "node_type": "InMemoryTableScan", + "child_nodes": "2", + "details": "{\"node_id\":1,\"node_type\":\"InMemoryTableScan\",\"detail\":{\"output\":[\"id1#0\",\"v3#8\"]}}", + "query_function": "count", + "query_header": "2 - count [0.01 min]", + "accumulators": [ + { + "stage_id": 11, + "task_id": 17, + "accumulator_id": 474, + "metric_name": "number of output rows", + "metric_type": "sum", + "value": 1000.0, + "value_exact": 1000, + "unit": "", + "readable_value": 1000.0, + "readable_unit": "", + "readable_str": "1000.0" }, + { + "stage_id": 11, + "task_id": 18, + "accumulator_id": 474, + "metric_name": "number of output rows", + "metric_type": "sum", + "value": 1000.0, + "value_exact": 1000, + "unit": "", + "readable_value": 1000.0, + "readable_unit": "", + "readable_str": "1000.0" + } + ], + "accumulator_totals": [ + { + "stage_id": 11, + "task_id": 17, + "accumulator_id": 474, + "metric_name": "number of output rows", + "metric_type": "sum", + "value": 2000.0, + "value_exact": 2000, + "unit": "", + "readable_value": 2000.0, + "readable_unit": "", + "readable_str": "2000.0" + } + ], + "n_accumulators": 2, + "n_accumulator_totals": 1, + "node_duration_minutes": null, + "node_id_adj": 1, + "node_name": "[1] InMemoryTableScan" + }, + { + "query_id": 2, + "query_start_timestamp": "2025-03-22T01:19:40", + "query_end_timestamp": "2025-03-22T01:19:40", + "query_duration_seconds": 0.31, + "whole_stage_codegen_id": null, + "node_id": 2, + "node_type": "InMemoryRelation", + "child_nodes": "21", + "details": "{\"node_id\":2,\"node_type\":\"InMemoryRelation\",\"detail\":{\"arguments\":\"[id1#0, id2#1, id3#2, id4#3L, id5#4L, id6#5L, v1#6L, v2#7L, v3#8], CachedRDDBuilder(org.apache.spark.sql.execution.columnar.DefaultCachedBatchSerializer@d36a649,StorageLevel(disk, memory, deserialized, 1 replicas),AdaptiveSparkPlan isFinalPlan=true\"}}", + "query_function": "count", + "query_header": "2 - count [0.01 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 2, + "node_name": "[2] InMemoryRelation" + }, + { + "query_id": 2, + "query_start_timestamp": "2025-03-22T01:19:40", + "query_end_timestamp": "2025-03-22T01:19:40", + "query_duration_seconds": 0.31, + "whole_stage_codegen_id": null, + "node_id": 3, + "node_type": "Scan", + "child_nodes": "", + "details": "{\"node_id\":3,\"node_type\":\"Scan\",\"detail\":{\"output\":[\"id1#0\",\"id2#1\",\"id3#2\",\"id4#3L\",\"id5#4L\",\"id6#5L\",\"v1#6L\",\"v2#7L\",\"v3#8\"],\"batched\":true,\"location\":{\"location_type\":\"InMemoryFileIndex\",\"location\":[\"sparkparse/data/raw/G1_1e7_1e7_100_0.parquet\"]},\"read_schema\":\"struct\"}}", + "query_function": "count", + "query_header": "2 - count [0.01 min]", + "accumulators": [ { "stage_id": 2, - "task_id": 9, + "task_id": 2, "accumulator_id": 132, "metric_name": "scan time", "metric_type": "timing", @@ -2624,7 +2765,7 @@ }, { "stage_id": 2, - "task_id": 8, + "task_id": 3, "accumulator_id": 132, "metric_name": "scan time", "metric_type": "timing", @@ -2637,16 +2778,29 @@ }, { "stage_id": 2, - "task_id": 7, + "task_id": 4, "accumulator_id": 132, "metric_name": "scan time", "metric_type": "timing", - "value": 796.0, - "value_exact": 796, + "value": 797.0, + "value_exact": 797, "unit": "ms", - "readable_value": 796.0, + "readable_value": 797.0, "readable_unit": "ms", - "readable_str": "796.0 ms" + "readable_str": "797.0 ms" + }, + { + "stage_id": 2, + "task_id": 5, + "accumulator_id": 132, + "metric_name": "scan time", + "metric_type": "timing", + "value": 795.0, + "value_exact": 795, + "unit": "ms", + "readable_value": 795.0, + "readable_unit": "ms", + "readable_str": "795.0 ms" }, { "stage_id": 2, @@ -2663,20 +2817,20 @@ }, { "stage_id": 2, - "task_id": 5, + "task_id": 7, "accumulator_id": 132, "metric_name": "scan time", "metric_type": "timing", - "value": 795.0, - "value_exact": 795, + "value": 796.0, + "value_exact": 796, "unit": "ms", - "readable_value": 795.0, + "readable_value": 796.0, "readable_unit": "ms", - "readable_str": "795.0 ms" + "readable_str": "796.0 ms" }, { "stage_id": 2, - "task_id": 4, + "task_id": 8, "accumulator_id": 132, "metric_name": "scan time", "metric_type": "timing", @@ -2689,7 +2843,20 @@ }, { "stage_id": 2, - "task_id": 3, + "task_id": 9, + "accumulator_id": 132, + "metric_name": "scan time", + "metric_type": "timing", + "value": 798.0, + "value_exact": 798, + "unit": "ms", + "readable_value": 798.0, + "readable_unit": "ms", + "readable_str": "798.0 ms" + }, + { + "stage_id": 2, + "task_id": 10, "accumulator_id": 132, "metric_name": "scan time", "metric_type": "timing", @@ -2700,6 +2867,19 @@ "readable_unit": "ms", "readable_str": "797.0 ms" }, + { + "stage_id": 2, + "task_id": 11, + "accumulator_id": 132, + "metric_name": "scan time", + "metric_type": "timing", + "value": 798.0, + "value_exact": 798, + "unit": "ms", + "readable_value": 798.0, + "readable_unit": "ms", + "readable_str": "798.0 ms" + }, { "stage_id": 2, "task_id": 2, @@ -2741,7 +2921,7 @@ }, { "stage_id": 2, - "task_id": 11, + "task_id": 5, "accumulator_id": 131, "metric_name": "number of output rows", "metric_type": "sum", @@ -2754,7 +2934,7 @@ }, { "stage_id": 2, - "task_id": 10, + "task_id": 6, "accumulator_id": 131, "metric_name": "number of output rows", "metric_type": "sum", @@ -2767,7 +2947,7 @@ }, { "stage_id": 2, - "task_id": 9, + "task_id": 7, "accumulator_id": 131, "metric_name": "number of output rows", "metric_type": "sum", @@ -2793,7 +2973,7 @@ }, { "stage_id": 2, - "task_id": 7, + "task_id": 9, "accumulator_id": 131, "metric_name": "number of output rows", "metric_type": "sum", @@ -2806,7 +2986,7 @@ }, { "stage_id": 2, - "task_id": 6, + "task_id": 10, "accumulator_id": 131, "metric_name": "number of output rows", "metric_type": "sum", @@ -2819,7 +2999,7 @@ }, { "stage_id": 2, - "task_id": 5, + "task_id": 11, "accumulator_id": 131, "metric_name": "number of output rows", "metric_type": "sum", @@ -2880,7 +3060,7 @@ "accumulators": [ { "stage_id": 2, - "task_id": 9, + "task_id": 2, "accumulator_id": 221, "metric_name": "number of input batches", "metric_type": "sum", @@ -2893,7 +3073,7 @@ }, { "stage_id": 2, - "task_id": 11, + "task_id": 3, "accumulator_id": 221, "metric_name": "number of input batches", "metric_type": "sum", @@ -2906,7 +3086,7 @@ }, { "stage_id": 2, - "task_id": 2, + "task_id": 4, "accumulator_id": 221, "metric_name": "number of input batches", "metric_type": "sum", @@ -2919,7 +3099,7 @@ }, { "stage_id": 2, - "task_id": 3, + "task_id": 5, "accumulator_id": 221, "metric_name": "number of input batches", "metric_type": "sum", @@ -2932,7 +3112,7 @@ }, { "stage_id": 2, - "task_id": 4, + "task_id": 6, "accumulator_id": 221, "metric_name": "number of input batches", "metric_type": "sum", @@ -2945,7 +3125,7 @@ }, { "stage_id": 2, - "task_id": 5, + "task_id": 7, "accumulator_id": 221, "metric_name": "number of input batches", "metric_type": "sum", @@ -2958,7 +3138,7 @@ }, { "stage_id": 2, - "task_id": 6, + "task_id": 8, "accumulator_id": 221, "metric_name": "number of input batches", "metric_type": "sum", @@ -2971,7 +3151,7 @@ }, { "stage_id": 2, - "task_id": 7, + "task_id": 9, "accumulator_id": 221, "metric_name": "number of input batches", "metric_type": "sum", @@ -2984,7 +3164,7 @@ }, { "stage_id": 2, - "task_id": 8, + "task_id": 10, "accumulator_id": 221, "metric_name": "number of input batches", "metric_type": "sum", @@ -2997,7 +3177,7 @@ }, { "stage_id": 2, - "task_id": 10, + "task_id": 11, "accumulator_id": 221, "metric_name": "number of input batches", "metric_type": "sum", @@ -3062,7 +3242,7 @@ }, { "stage_id": 2, - "task_id": 11, + "task_id": 6, "accumulator_id": 220, "metric_name": "number of output rows", "metric_type": "sum", @@ -3075,7 +3255,7 @@ }, { "stage_id": 2, - "task_id": 10, + "task_id": 7, "accumulator_id": 220, "metric_name": "number of output rows", "metric_type": "sum", @@ -3088,7 +3268,7 @@ }, { "stage_id": 2, - "task_id": 6, + "task_id": 8, "accumulator_id": 220, "metric_name": "number of output rows", "metric_type": "sum", @@ -3114,7 +3294,7 @@ }, { "stage_id": 2, - "task_id": 8, + "task_id": 10, "accumulator_id": 220, "metric_name": "number of output rows", "metric_type": "sum", @@ -3127,7 +3307,7 @@ }, { "stage_id": 2, - "task_id": 7, + "task_id": 11, "accumulator_id": 220, "metric_name": "number of output rows", "metric_type": "sum", @@ -3364,7 +3544,7 @@ }, { "stage_id": 2, - "task_id": 9, + "task_id": 2, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -3377,7 +3557,7 @@ }, { "stage_id": 2, - "task_id": 7, + "task_id": 3, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -3390,7 +3570,7 @@ }, { "stage_id": 2, - "task_id": 8, + "task_id": 4, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -3403,7 +3583,7 @@ }, { "stage_id": 2, - "task_id": 10, + "task_id": 5, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -3416,7 +3596,7 @@ }, { "stage_id": 2, - "task_id": 11, + "task_id": 6, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -3429,7 +3609,7 @@ }, { "stage_id": 2, - "task_id": 2, + "task_id": 7, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -3442,7 +3622,7 @@ }, { "stage_id": 2, - "task_id": 6, + "task_id": 8, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -3455,7 +3635,7 @@ }, { "stage_id": 2, - "task_id": 5, + "task_id": 9, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -3468,7 +3648,7 @@ }, { "stage_id": 2, - "task_id": 3, + "task_id": 10, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -3481,7 +3661,7 @@ }, { "stage_id": 2, - "task_id": 4, + "task_id": 11, "accumulator_id": 197, "metric_name": "data size", "metric_type": "size", @@ -3494,7 +3674,7 @@ }, { "stage_id": 4, - "task_id": 13, + "task_id": 12, "accumulator_id": 203, "metric_name": "local bytes read", "metric_type": "size", @@ -3507,7 +3687,7 @@ }, { "stage_id": 4, - "task_id": 12, + "task_id": 13, "accumulator_id": 203, "metric_name": "local bytes read", "metric_type": "size", @@ -3520,133 +3700,133 @@ }, { "stage_id": 2, - "task_id": 3, + "task_id": 2, "accumulator_id": 216, "metric_name": "shuffle bytes written", "metric_type": "size", - "value": 49709.0, - "value_exact": 49709, + "value": 49684.0, + "value_exact": 49684, "unit": "B", "readable_value": 48.5, "readable_unit": "KiB", - "readable_str": "48.54 KiB" + "readable_str": "48.52 KiB" }, { "stage_id": 2, - "task_id": 8, + "task_id": 3, "accumulator_id": 216, "metric_name": "shuffle bytes written", "metric_type": "size", - "value": 49468.0, - "value_exact": 49468, + "value": 49709.0, + "value_exact": 49709, "unit": "B", - "readable_value": 48.3, + "readable_value": 48.5, "readable_unit": "KiB", - "readable_str": "48.31 KiB" + "readable_str": "48.54 KiB" }, { "stage_id": 2, - "task_id": 2, + "task_id": 4, "accumulator_id": 216, "metric_name": "shuffle bytes written", "metric_type": "size", - "value": 49684.0, - "value_exact": 49684, + "value": 49535.0, + "value_exact": 49535, "unit": "B", - "readable_value": 48.5, + "readable_value": 48.4, "readable_unit": "KiB", - "readable_str": "48.52 KiB" + "readable_str": "48.37 KiB" }, { "stage_id": 2, - "task_id": 11, + "task_id": 5, "accumulator_id": 216, "metric_name": "shuffle bytes written", "metric_type": "size", - "value": 49486.0, - "value_exact": 49486, + "value": 49789.0, + "value_exact": 49789, "unit": "B", - "readable_value": 48.3, + "readable_value": 48.6, "readable_unit": "KiB", - "readable_str": "48.33 KiB" + "readable_str": "48.62 KiB" }, { "stage_id": 2, - "task_id": 10, + "task_id": 6, "accumulator_id": 216, "metric_name": "shuffle bytes written", "metric_type": "size", - "value": 49459.0, - "value_exact": 49459, + "value": 49669.0, + "value_exact": 49669, "unit": "B", - "readable_value": 48.3, + "readable_value": 48.5, "readable_unit": "KiB", - "readable_str": "48.3 KiB" + "readable_str": "48.5 KiB" }, { "stage_id": 2, - "task_id": 4, + "task_id": 7, "accumulator_id": 216, "metric_name": "shuffle bytes written", "metric_type": "size", - "value": 49535.0, - "value_exact": 49535, + "value": 49581.0, + "value_exact": 49581, "unit": "B", "readable_value": 48.4, "readable_unit": "KiB", - "readable_str": "48.37 KiB" + "readable_str": "48.42 KiB" }, { "stage_id": 2, - "task_id": 9, + "task_id": 8, "accumulator_id": 216, "metric_name": "shuffle bytes written", "metric_type": "size", - "value": 49719.0, - "value_exact": 49719, + "value": 49468.0, + "value_exact": 49468, "unit": "B", - "readable_value": 48.6, + "readable_value": 48.3, "readable_unit": "KiB", - "readable_str": "48.55 KiB" + "readable_str": "48.31 KiB" }, { "stage_id": 2, - "task_id": 5, + "task_id": 9, "accumulator_id": 216, "metric_name": "shuffle bytes written", "metric_type": "size", - "value": 49789.0, - "value_exact": 49789, + "value": 49719.0, + "value_exact": 49719, "unit": "B", "readable_value": 48.6, "readable_unit": "KiB", - "readable_str": "48.62 KiB" + "readable_str": "48.55 KiB" }, { "stage_id": 2, - "task_id": 7, + "task_id": 10, "accumulator_id": 216, "metric_name": "shuffle bytes written", "metric_type": "size", - "value": 49581.0, - "value_exact": 49581, + "value": 49459.0, + "value_exact": 49459, "unit": "B", - "readable_value": 48.4, + "readable_value": 48.3, "readable_unit": "KiB", - "readable_str": "48.42 KiB" + "readable_str": "48.3 KiB" }, { "stage_id": 2, - "task_id": 6, + "task_id": 11, "accumulator_id": 216, "metric_name": "shuffle bytes written", "metric_type": "size", - "value": 49669.0, - "value_exact": 49669, + "value": 49486.0, + "value_exact": 49486, "unit": "B", - "readable_value": 48.5, + "readable_value": 48.3, "readable_unit": "KiB", - "readable_str": "48.5 KiB" + "readable_str": "48.33 KiB" }, { "stage_id": 4, @@ -3702,7 +3882,7 @@ }, { "stage_id": 2, - "task_id": 10, + "task_id": 2, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -3715,7 +3895,7 @@ }, { "stage_id": 2, - "task_id": 11, + "task_id": 3, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -3728,7 +3908,7 @@ }, { "stage_id": 2, - "task_id": 9, + "task_id": 4, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -3741,7 +3921,7 @@ }, { "stage_id": 2, - "task_id": 8, + "task_id": 5, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -3754,7 +3934,7 @@ }, { "stage_id": 2, - "task_id": 7, + "task_id": 6, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -3767,7 +3947,7 @@ }, { "stage_id": 2, - "task_id": 6, + "task_id": 7, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -3780,7 +3960,7 @@ }, { "stage_id": 2, - "task_id": 5, + "task_id": 8, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -3793,7 +3973,7 @@ }, { "stage_id": 2, - "task_id": 4, + "task_id": 9, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -3806,7 +3986,7 @@ }, { "stage_id": 2, - "task_id": 3, + "task_id": 10, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -3819,7 +3999,7 @@ }, { "stage_id": 2, - "task_id": 2, + "task_id": 11, "accumulator_id": 217, "metric_name": "shuffle records written", "metric_type": "sum", @@ -4118,7 +4298,7 @@ "accumulators": [ { "stage_id": 11, - "task_id": 18, + "task_id": 17, "accumulator_id": 609, "metric_name": "time in aggregation build", "metric_type": "timing", @@ -4131,7 +4311,7 @@ }, { "stage_id": 11, - "task_id": 17, + "task_id": 18, "accumulator_id": 609, "metric_name": "time in aggregation build", "metric_type": "timing", @@ -4296,7 +4476,7 @@ }, { "stage_id": 11, - "task_id": 18, + "task_id": 17, "accumulator_id": 583, "metric_name": "data size", "metric_type": "size", @@ -4309,7 +4489,7 @@ }, { "stage_id": 11, - "task_id": 17, + "task_id": 18, "accumulator_id": 583, "metric_name": "data size", "metric_type": "size", @@ -4335,7 +4515,7 @@ }, { "stage_id": 11, - "task_id": 18, + "task_id": 17, "accumulator_id": 602, "metric_name": "shuffle bytes written", "metric_type": "size", @@ -4348,7 +4528,7 @@ }, { "stage_id": 11, - "task_id": 17, + "task_id": 18, "accumulator_id": 602, "metric_name": "shuffle bytes written", "metric_type": "size", @@ -5143,46 +5323,33 @@ "accumulators": [ { "stage_id": 2, - "task_id": 11, - "accumulator_id": 219, - "metric_name": "duration", - "metric_type": "timing", - "value": 825.0, - "value_exact": 825, - "unit": "ms", - "readable_value": 825.0, - "readable_unit": "ms", - "readable_str": "825.0 ms" - }, - { - "stage_id": 2, - "task_id": 10, + "task_id": 2, "accumulator_id": 219, "metric_name": "duration", "metric_type": "timing", - "value": 825.0, - "value_exact": 825, + "value": 823.0, + "value_exact": 823, "unit": "ms", - "readable_value": 825.0, + "readable_value": 823.0, "readable_unit": "ms", - "readable_str": "825.0 ms" + "readable_str": "823.0 ms" }, { "stage_id": 2, - "task_id": 9, + "task_id": 3, "accumulator_id": 219, "metric_name": "duration", "metric_type": "timing", - "value": 822.0, - "value_exact": 822, + "value": 823.0, + "value_exact": 823, "unit": "ms", - "readable_value": 822.0, + "readable_value": 823.0, "readable_unit": "ms", - "readable_str": "822.0 ms" + "readable_str": "823.0 ms" }, { "stage_id": 2, - "task_id": 7, + "task_id": 4, "accumulator_id": 219, "metric_name": "duration", "metric_type": "timing", @@ -5195,7 +5362,7 @@ }, { "stage_id": 2, - "task_id": 6, + "task_id": 5, "accumulator_id": 219, "metric_name": "duration", "metric_type": "timing", @@ -5208,7 +5375,7 @@ }, { "stage_id": 2, - "task_id": 5, + "task_id": 6, "accumulator_id": 219, "metric_name": "duration", "metric_type": "timing", @@ -5221,7 +5388,7 @@ }, { "stage_id": 2, - "task_id": 4, + "task_id": 7, "accumulator_id": 219, "metric_name": "duration", "metric_type": "timing", @@ -5247,41 +5414,54 @@ }, { "stage_id": 2, - "task_id": 3, + "task_id": 9, "accumulator_id": 219, "metric_name": "duration", "metric_type": "timing", - "value": 823.0, - "value_exact": 823, + "value": 822.0, + "value_exact": 822, "unit": "ms", - "readable_value": 823.0, + "readable_value": 822.0, "readable_unit": "ms", - "readable_str": "823.0 ms" + "readable_str": "822.0 ms" }, { "stage_id": 2, - "task_id": 2, + "task_id": 10, "accumulator_id": 219, "metric_name": "duration", "metric_type": "timing", - "value": 823.0, - "value_exact": 823, + "value": 825.0, + "value_exact": 825, "unit": "ms", - "readable_value": 823.0, + "readable_value": 825.0, "readable_unit": "ms", - "readable_str": "823.0 ms" - } - ], - "accumulator_totals": [ + "readable_str": "825.0 ms" + }, { "stage_id": 2, "task_id": 11, "accumulator_id": 219, "metric_name": "duration", "metric_type": "timing", - "value": 8235.0, - "value_exact": 8235, - "unit": "ms", + "value": 825.0, + "value_exact": 825, + "unit": "ms", + "readable_value": 825.0, + "readable_unit": "ms", + "readable_str": "825.0 ms" + } + ], + "accumulator_totals": [ + { + "stage_id": 2, + "task_id": 11, + "accumulator_id": 219, + "metric_name": "duration", + "metric_type": "timing", + "value": 8235.0, + "value_exact": 8235, + "unit": "ms", "readable_value": 8.2, "readable_unit": "s", "readable_str": "8.24 s" @@ -5328,1240 +5508,142 @@ "metric_name": "duration", "metric_type": "timing", "value": 6.0, - "value_exact": 6, - "unit": "ms", - "readable_value": 6.0, - "readable_unit": "ms", - "readable_str": "6.0 ms" - } - ], - "n_accumulators": 1, - "n_accumulator_totals": 1, - "node_duration_minutes": 0.0001, - "node_id_adj": 2, - "node_name": "[2] WholeStageCodegen" - }, - { - "query_id": 2, - "query_start_timestamp": "2025-03-22T01:19:40", - "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 0.31, - "whole_stage_codegen_id": 3, - "node_id": 100003, - "node_type": "WholeStageCodegen", - "child_nodes": null, - "details": null, - "query_function": "count", - "query_header": "2 - count [0.01 min]", - "accumulators": [ - { - "stage_id": 4, - "task_id": 12, - "accumulator_id": 257, - "metric_name": "duration", - "metric_type": "timing", - "value": 77.0, - "value_exact": 77, - "unit": "ms", - "readable_value": 77.0, - "readable_unit": "ms", - "readable_str": "77.0 ms" - } - ], - "accumulator_totals": [ - { - "stage_id": 4, - "task_id": 12, - "accumulator_id": 257, - "metric_name": "duration", - "metric_type": "timing", - "value": 77.0, - "value_exact": 77, - "unit": "ms", - "readable_value": 77.0, - "readable_unit": "ms", - "readable_str": "77.0 ms" - } - ], - "n_accumulators": 1, - "n_accumulator_totals": 1, - "node_duration_minutes": 0.0012833333333333334, - "node_id_adj": 3, - "node_name": "[3] WholeStageCodegen" - }, - { - "query_id": 2, - "query_start_timestamp": "2025-03-22T01:19:40", - "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 0.31, - "whole_stage_codegen_id": 4, - "node_id": 100004, - "node_type": "WholeStageCodegen", - "child_nodes": null, - "details": null, - "query_function": "count", - "query_header": "2 - count [0.01 min]", - "accumulators": [ - { - "stage_id": 4, - "task_id": 13, - "accumulator_id": 258, - "metric_name": "duration", - "metric_type": "timing", - "value": 77.0, - "value_exact": 77, - "unit": "ms", - "readable_value": 77.0, - "readable_unit": "ms", - "readable_str": "77.0 ms" - } - ], - "accumulator_totals": [ - { - "stage_id": 4, - "task_id": 13, - "accumulator_id": 258, - "metric_name": "duration", - "metric_type": "timing", - "value": 77.0, - "value_exact": 77, - "unit": "ms", - "readable_value": 77.0, - "readable_unit": "ms", - "readable_str": "77.0 ms" - } - ], - "n_accumulators": 1, - "n_accumulator_totals": 1, - "node_duration_minutes": 0.0012833333333333334, - "node_id_adj": 4, - "node_name": "[4] WholeStageCodegen" - }, - { - "log_name": "nested_final_plans", - "query_id": 0, - "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:38", - "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 1.48, - "job_id": 2, - "stage_id": 2, - "job_start_timestamp": "2025-03-22T01:19:38.938000000", - "job_end_timestamp": "2025-03-22T01:19:39.884000000", - "job_duration_seconds": 0.9460000000000001, - "stage_start_timestamp": "2025-03-22T01:19:38.941000000", - "stage_end_timestamp": "2025-03-22T01:19:39.881000000", - "stage_duration_seconds": 0.9400000000000001, - "task_id": 2, - "task_start_timestamp": "2025-03-22T01:19:38.959000000", - "task_end_timestamp": "2025-03-22T01:19:39.878000000", - "task_duration_seconds": 0.919, - "nodes": [ - "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", - "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange" - ], - "executor_run_time_seconds": 0.000869, - "executor_cpu_time_seconds": 0.469563, - "executor_deserialize_time_seconds": 3.7999999999999995e-05, - "executor_deserialize_cpu_time_seconds": 0.018882, - "result_size_bytes": 2078, - "jvm_gc_time_seconds": 6.3e-05, - "result_serialization_time_seconds": 4e-06, - "memory_bytes_spilled": 0, - "disk_bytes_spilled": 0, - "peak_execution_memory_bytes": 0, - "bytes_read": 26807898, - "records_read": 4096, - "bytes_written": 0, - "records_written": 0, - "shuffle_remote_blocks_fetched": 0, - "shuffle_local_blocks_fetched": 0, - "shuffle_fetch_wait_time_seconds": 0.0, - "shuffle_remote_bytes_read": 0, - "shuffle_remote_bytes_read_to_disk": 0, - "shuffle_local_bytes_read": 0, - "shuffle_records_read": 0, - "shuffle_remote_requests_duration": 0, - "shuffle_bytes_read": 0, - "shuffle_bytes_written": 49684, - "shuffle_write_time_seconds": 0.007109666000000001, - "shuffle_records_written": 1000, - "merged_corrupt_block_chunks": 0, - "merged_fetch_fallback_count": 0, - "merged_remote_blocks_fetched": 0, - "merged_local_blocks_fetched": 0, - "merged_remote_chunks_fetched": 0, - "merged_local_chunks_fetched": 0, - "merged_remote_bytes_read": 0, - "merged_local_bytes_read": 0, - "merged_remote_requests_duration": 0, - "executor_id": "driver", - "host": "localhost", - "index": 0, - "attempt": 0, - "failed": false, - "killed": false, - "speculative": false, - "task_loc": "PROCESS_LOCAL", - "task_type": "ShuffleMapTask" - }, - { - "log_name": "nested_final_plans", - "query_id": 0, - "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:38", - "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 1.48, - "job_id": 2, - "stage_id": 2, - "job_start_timestamp": "2025-03-22T01:19:38.938000000", - "job_end_timestamp": "2025-03-22T01:19:39.884000000", - "job_duration_seconds": 0.9460000000000001, - "stage_start_timestamp": "2025-03-22T01:19:38.941000000", - "stage_end_timestamp": "2025-03-22T01:19:39.881000000", - "stage_duration_seconds": 0.9400000000000001, - "task_id": 3, - "task_start_timestamp": "2025-03-22T01:19:38.961000000", - "task_end_timestamp": "2025-03-22T01:19:39.877000000", - "task_duration_seconds": 0.916, - "nodes": [ - "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", - "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange" - ], - "executor_run_time_seconds": 0.0008619999999999999, - "executor_cpu_time_seconds": 0.41519700000000004, - "executor_deserialize_time_seconds": 3.7e-05, - "executor_deserialize_cpu_time_seconds": 0.017843, - "result_size_bytes": 2078, - "jvm_gc_time_seconds": 6.3e-05, - "result_serialization_time_seconds": 9e-06, - "memory_bytes_spilled": 0, - "disk_bytes_spilled": 0, - "peak_execution_memory_bytes": 0, - "bytes_read": 26800959, - "records_read": 4096, - "bytes_written": 0, - "records_written": 0, - "shuffle_remote_blocks_fetched": 0, - "shuffle_local_blocks_fetched": 0, - "shuffle_fetch_wait_time_seconds": 0.0, - "shuffle_remote_bytes_read": 0, - "shuffle_remote_bytes_read_to_disk": 0, - "shuffle_local_bytes_read": 0, - "shuffle_records_read": 0, - "shuffle_remote_requests_duration": 0, - "shuffle_bytes_read": 0, - "shuffle_bytes_written": 49709, - "shuffle_write_time_seconds": 0.006600916, - "shuffle_records_written": 1000, - "merged_corrupt_block_chunks": 0, - "merged_fetch_fallback_count": 0, - "merged_remote_blocks_fetched": 0, - "merged_local_blocks_fetched": 0, - "merged_remote_chunks_fetched": 0, - "merged_local_chunks_fetched": 0, - "merged_remote_bytes_read": 0, - "merged_local_bytes_read": 0, - "merged_remote_requests_duration": 0, - "executor_id": "driver", - "host": "localhost", - "index": 1, - "attempt": 0, - "failed": false, - "killed": false, - "speculative": false, - "task_loc": "PROCESS_LOCAL", - "task_type": "ShuffleMapTask" - }, - { - "log_name": "nested_final_plans", - "query_id": 0, - "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:38", - "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 1.48, - "job_id": 2, - "stage_id": 2, - "job_start_timestamp": "2025-03-22T01:19:38.938000000", - "job_end_timestamp": "2025-03-22T01:19:39.884000000", - "job_duration_seconds": 0.9460000000000001, - "stage_start_timestamp": "2025-03-22T01:19:38.941000000", - "stage_end_timestamp": "2025-03-22T01:19:39.881000000", - "stage_duration_seconds": 0.9400000000000001, - "task_id": 4, - "task_start_timestamp": "2025-03-22T01:19:38.961000000", - "task_end_timestamp": "2025-03-22T01:19:39.879000000", - "task_duration_seconds": 0.918, - "nodes": [ - "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", - "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange" - ], - "executor_run_time_seconds": 0.000868, - "executor_cpu_time_seconds": 0.428864, - "executor_deserialize_time_seconds": 3.7999999999999995e-05, - "executor_deserialize_cpu_time_seconds": 0.019114000000000003, - "result_size_bytes": 2078, - "jvm_gc_time_seconds": 6.3e-05, - "result_serialization_time_seconds": 3e-06, - "memory_bytes_spilled": 0, - "disk_bytes_spilled": 0, - "peak_execution_memory_bytes": 0, - "bytes_read": 26813082, - "records_read": 4096, - "bytes_written": 0, - "records_written": 0, - "shuffle_remote_blocks_fetched": 0, - "shuffle_local_blocks_fetched": 0, - "shuffle_fetch_wait_time_seconds": 0.0, - "shuffle_remote_bytes_read": 0, - "shuffle_remote_bytes_read_to_disk": 0, - "shuffle_local_bytes_read": 0, - "shuffle_records_read": 0, - "shuffle_remote_requests_duration": 0, - "shuffle_bytes_read": 0, - "shuffle_bytes_written": 49535, - "shuffle_write_time_seconds": 0.0050717060000000005, - "shuffle_records_written": 1000, - "merged_corrupt_block_chunks": 0, - "merged_fetch_fallback_count": 0, - "merged_remote_blocks_fetched": 0, - "merged_local_blocks_fetched": 0, - "merged_remote_chunks_fetched": 0, - "merged_local_chunks_fetched": 0, - "merged_remote_bytes_read": 0, - "merged_local_bytes_read": 0, - "merged_remote_requests_duration": 0, - "executor_id": "driver", - "host": "localhost", - "index": 2, - "attempt": 0, - "failed": false, - "killed": false, - "speculative": false, - "task_loc": "PROCESS_LOCAL", - "task_type": "ShuffleMapTask" - }, - { - "log_name": "nested_final_plans", - "query_id": 0, - "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:38", - "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 1.48, - "job_id": 2, - "stage_id": 2, - "job_start_timestamp": "2025-03-22T01:19:38.938000000", - "job_end_timestamp": "2025-03-22T01:19:39.884000000", - "job_duration_seconds": 0.9460000000000001, - "stage_start_timestamp": "2025-03-22T01:19:38.941000000", - "stage_end_timestamp": "2025-03-22T01:19:39.881000000", - "stage_duration_seconds": 0.9400000000000001, - "task_id": 5, - "task_start_timestamp": "2025-03-22T01:19:38.961000000", - "task_end_timestamp": "2025-03-22T01:19:39.878000000", - "task_duration_seconds": 0.917, - "nodes": [ - "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", - "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange" - ], - "executor_run_time_seconds": 0.000868, - "executor_cpu_time_seconds": 0.43382800000000005, - "executor_deserialize_time_seconds": 3.7999999999999995e-05, - "executor_deserialize_cpu_time_seconds": 0.018223, - "result_size_bytes": 2078, - "jvm_gc_time_seconds": 6.3e-05, - "result_serialization_time_seconds": 3e-06, - "memory_bytes_spilled": 0, - "disk_bytes_spilled": 0, - "peak_execution_memory_bytes": 0, - "bytes_read": 26794482, - "records_read": 4096, - "bytes_written": 0, - "records_written": 0, - "shuffle_remote_blocks_fetched": 0, - "shuffle_local_blocks_fetched": 0, - "shuffle_fetch_wait_time_seconds": 0.0, - "shuffle_remote_bytes_read": 0, - "shuffle_remote_bytes_read_to_disk": 0, - "shuffle_local_bytes_read": 0, - "shuffle_records_read": 0, - "shuffle_remote_requests_duration": 0, - "shuffle_bytes_read": 0, - "shuffle_bytes_written": 49789, - "shuffle_write_time_seconds": 0.006927581, - "shuffle_records_written": 1000, - "merged_corrupt_block_chunks": 0, - "merged_fetch_fallback_count": 0, - "merged_remote_blocks_fetched": 0, - "merged_local_blocks_fetched": 0, - "merged_remote_chunks_fetched": 0, - "merged_local_chunks_fetched": 0, - "merged_remote_bytes_read": 0, - "merged_local_bytes_read": 0, - "merged_remote_requests_duration": 0, - "executor_id": "driver", - "host": "localhost", - "index": 3, - "attempt": 0, - "failed": false, - "killed": false, - "speculative": false, - "task_loc": "PROCESS_LOCAL", - "task_type": "ShuffleMapTask" - }, - { - "log_name": "nested_final_plans", - "query_id": 0, - "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:38", - "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 1.48, - "job_id": 2, - "stage_id": 2, - "job_start_timestamp": "2025-03-22T01:19:38.938000000", - "job_end_timestamp": "2025-03-22T01:19:39.884000000", - "job_duration_seconds": 0.9460000000000001, - "stage_start_timestamp": "2025-03-22T01:19:38.941000000", - "stage_end_timestamp": "2025-03-22T01:19:39.881000000", - "stage_duration_seconds": 0.9400000000000001, - "task_id": 6, - "task_start_timestamp": "2025-03-22T01:19:38.962000000", - "task_end_timestamp": "2025-03-22T01:19:39.877000000", - "task_duration_seconds": 0.915, - "nodes": [ - "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", - "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange" - ], - "executor_run_time_seconds": 0.000868, - "executor_cpu_time_seconds": 0.445184, - "executor_deserialize_time_seconds": 3.7999999999999995e-05, - "executor_deserialize_cpu_time_seconds": 0.018531000000000002, - "result_size_bytes": 2078, - "jvm_gc_time_seconds": 6.3e-05, - "result_serialization_time_seconds": 3e-06, - "memory_bytes_spilled": 0, - "disk_bytes_spilled": 0, - "peak_execution_memory_bytes": 0, - "bytes_read": 20608787, - "records_read": 4096, - "bytes_written": 0, - "records_written": 0, - "shuffle_remote_blocks_fetched": 0, - "shuffle_local_blocks_fetched": 0, - "shuffle_fetch_wait_time_seconds": 0.0, - "shuffle_remote_bytes_read": 0, - "shuffle_remote_bytes_read_to_disk": 0, - "shuffle_local_bytes_read": 0, - "shuffle_records_read": 0, - "shuffle_remote_requests_duration": 0, - "shuffle_bytes_read": 0, - "shuffle_bytes_written": 49669, - "shuffle_write_time_seconds": 0.007078793000000001, - "shuffle_records_written": 1000, - "merged_corrupt_block_chunks": 0, - "merged_fetch_fallback_count": 0, - "merged_remote_blocks_fetched": 0, - "merged_local_blocks_fetched": 0, - "merged_remote_chunks_fetched": 0, - "merged_local_chunks_fetched": 0, - "merged_remote_bytes_read": 0, - "merged_local_bytes_read": 0, - "merged_remote_requests_duration": 0, - "executor_id": "driver", - "host": "localhost", - "index": 4, - "attempt": 0, - "failed": false, - "killed": false, - "speculative": false, - "task_loc": "PROCESS_LOCAL", - "task_type": "ShuffleMapTask" - }, - { - "log_name": "nested_final_plans", - "query_id": 0, - "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:38", - "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 1.48, - "job_id": 2, - "stage_id": 2, - "job_start_timestamp": "2025-03-22T01:19:38.938000000", - "job_end_timestamp": "2025-03-22T01:19:39.884000000", - "job_duration_seconds": 0.9460000000000001, - "stage_start_timestamp": "2025-03-22T01:19:38.941000000", - "stage_end_timestamp": "2025-03-22T01:19:39.881000000", - "stage_duration_seconds": 0.9400000000000001, - "task_id": 7, - "task_start_timestamp": "2025-03-22T01:19:38.962000000", - "task_end_timestamp": "2025-03-22T01:19:39.878000000", - "task_duration_seconds": 0.916, - "nodes": [ - "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", - "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange" - ], - "executor_run_time_seconds": 0.000868, - "executor_cpu_time_seconds": 0.425117, - "executor_deserialize_time_seconds": 3.7e-05, - "executor_deserialize_cpu_time_seconds": 0.017276, - "result_size_bytes": 2078, - "jvm_gc_time_seconds": 6.3e-05, - "result_serialization_time_seconds": 2e-06, - "memory_bytes_spilled": 0, - "disk_bytes_spilled": 0, - "peak_execution_memory_bytes": 0, - "bytes_read": 26793376, - "records_read": 4096, - "bytes_written": 0, - "records_written": 0, - "shuffle_remote_blocks_fetched": 0, - "shuffle_local_blocks_fetched": 0, - "shuffle_fetch_wait_time_seconds": 0.0, - "shuffle_remote_bytes_read": 0, - "shuffle_remote_bytes_read_to_disk": 0, - "shuffle_local_bytes_read": 0, - "shuffle_records_read": 0, - "shuffle_remote_requests_duration": 0, - "shuffle_bytes_read": 0, - "shuffle_bytes_written": 49581, - "shuffle_write_time_seconds": 0.004944462, - "shuffle_records_written": 1000, - "merged_corrupt_block_chunks": 0, - "merged_fetch_fallback_count": 0, - "merged_remote_blocks_fetched": 0, - "merged_local_blocks_fetched": 0, - "merged_remote_chunks_fetched": 0, - "merged_local_chunks_fetched": 0, - "merged_remote_bytes_read": 0, - "merged_local_bytes_read": 0, - "merged_remote_requests_duration": 0, - "executor_id": "driver", - "host": "localhost", - "index": 5, - "attempt": 0, - "failed": false, - "killed": false, - "speculative": false, - "task_loc": "PROCESS_LOCAL", - "task_type": "ShuffleMapTask" - }, - { - "log_name": "nested_final_plans", - "query_id": 0, - "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:38", - "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 1.48, - "job_id": 2, - "stage_id": 2, - "job_start_timestamp": "2025-03-22T01:19:38.938000000", - "job_end_timestamp": "2025-03-22T01:19:39.884000000", - "job_duration_seconds": 0.9460000000000001, - "stage_start_timestamp": "2025-03-22T01:19:38.941000000", - "stage_end_timestamp": "2025-03-22T01:19:39.881000000", - "stage_duration_seconds": 0.9400000000000001, - "task_id": 8, - "task_start_timestamp": "2025-03-22T01:19:38.962000000", - "task_end_timestamp": "2025-03-22T01:19:39.878000000", - "task_duration_seconds": 0.916, - "nodes": [ - "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", - "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange" - ], - "executor_run_time_seconds": 0.000869, - "executor_cpu_time_seconds": 0.43211900000000003, - "executor_deserialize_time_seconds": 3.6e-05, - "executor_deserialize_cpu_time_seconds": 0.018000000000000002, - "result_size_bytes": 2078, - "jvm_gc_time_seconds": 6.3e-05, - "result_serialization_time_seconds": 4e-06, - "memory_bytes_spilled": 0, - "disk_bytes_spilled": 0, - "peak_execution_memory_bytes": 0, - "bytes_read": 26792617, - "records_read": 4096, - "bytes_written": 0, - "records_written": 0, - "shuffle_remote_blocks_fetched": 0, - "shuffle_local_blocks_fetched": 0, - "shuffle_fetch_wait_time_seconds": 0.0, - "shuffle_remote_bytes_read": 0, - "shuffle_remote_bytes_read_to_disk": 0, - "shuffle_local_bytes_read": 0, - "shuffle_records_read": 0, - "shuffle_remote_requests_duration": 0, - "shuffle_bytes_read": 0, - "shuffle_bytes_written": 49468, - "shuffle_write_time_seconds": 0.006705041, - "shuffle_records_written": 1000, - "merged_corrupt_block_chunks": 0, - "merged_fetch_fallback_count": 0, - "merged_remote_blocks_fetched": 0, - "merged_local_blocks_fetched": 0, - "merged_remote_chunks_fetched": 0, - "merged_local_chunks_fetched": 0, - "merged_remote_bytes_read": 0, - "merged_local_bytes_read": 0, - "merged_remote_requests_duration": 0, - "executor_id": "driver", - "host": "localhost", - "index": 6, - "attempt": 0, - "failed": false, - "killed": false, - "speculative": false, - "task_loc": "PROCESS_LOCAL", - "task_type": "ShuffleMapTask" - }, - { - "log_name": "nested_final_plans", - "query_id": 0, - "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:38", - "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 1.48, - "job_id": 2, - "stage_id": 2, - "job_start_timestamp": "2025-03-22T01:19:38.938000000", - "job_end_timestamp": "2025-03-22T01:19:39.884000000", - "job_duration_seconds": 0.9460000000000001, - "stage_start_timestamp": "2025-03-22T01:19:38.941000000", - "stage_end_timestamp": "2025-03-22T01:19:39.881000000", - "stage_duration_seconds": 0.9400000000000001, - "task_id": 9, - "task_start_timestamp": "2025-03-22T01:19:38.962000000", - "task_end_timestamp": "2025-03-22T01:19:39.877000000", - "task_duration_seconds": 0.915, - "nodes": [ - "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", - "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange" - ], - "executor_run_time_seconds": 0.000861, - "executor_cpu_time_seconds": 0.433779, - "executor_deserialize_time_seconds": 3.7e-05, - "executor_deserialize_cpu_time_seconds": 0.017854000000000002, - "result_size_bytes": 2078, - "jvm_gc_time_seconds": 6.3e-05, - "result_serialization_time_seconds": 9e-06, - "memory_bytes_spilled": 0, - "disk_bytes_spilled": 0, - "peak_execution_memory_bytes": 0, - "bytes_read": 26802658, - "records_read": 4096, - "bytes_written": 0, - "records_written": 0, - "shuffle_remote_blocks_fetched": 0, - "shuffle_local_blocks_fetched": 0, - "shuffle_fetch_wait_time_seconds": 0.0, - "shuffle_remote_bytes_read": 0, - "shuffle_remote_bytes_read_to_disk": 0, - "shuffle_local_bytes_read": 0, - "shuffle_records_read": 0, - "shuffle_remote_requests_duration": 0, - "shuffle_bytes_read": 0, - "shuffle_bytes_written": 49719, - "shuffle_write_time_seconds": 0.006893539000000001, - "shuffle_records_written": 1000, - "merged_corrupt_block_chunks": 0, - "merged_fetch_fallback_count": 0, - "merged_remote_blocks_fetched": 0, - "merged_local_blocks_fetched": 0, - "merged_remote_chunks_fetched": 0, - "merged_local_chunks_fetched": 0, - "merged_remote_bytes_read": 0, - "merged_local_bytes_read": 0, - "merged_remote_requests_duration": 0, - "executor_id": "driver", - "host": "localhost", - "index": 7, - "attempt": 0, - "failed": false, - "killed": false, - "speculative": false, - "task_loc": "PROCESS_LOCAL", - "task_type": "ShuffleMapTask" - }, - { - "log_name": "nested_final_plans", - "query_id": 0, - "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:38", - "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 1.48, - "job_id": 2, - "stage_id": 2, - "job_start_timestamp": "2025-03-22T01:19:38.938000000", - "job_end_timestamp": "2025-03-22T01:19:39.884000000", - "job_duration_seconds": 0.9460000000000001, - "stage_start_timestamp": "2025-03-22T01:19:38.941000000", - "stage_end_timestamp": "2025-03-22T01:19:39.881000000", - "stage_duration_seconds": 0.9400000000000001, - "task_id": 10, - "task_start_timestamp": "2025-03-22T01:19:38.962000000", - "task_end_timestamp": "2025-03-22T01:19:39.877000000", - "task_duration_seconds": 0.915, - "nodes": [ - "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", - "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange" - ], - "executor_run_time_seconds": 0.000868, - "executor_cpu_time_seconds": 0.415075, - "executor_deserialize_time_seconds": 3.7e-05, - "executor_deserialize_cpu_time_seconds": 0.016955, - "result_size_bytes": 2078, - "jvm_gc_time_seconds": 6.3e-05, - "result_serialization_time_seconds": 2e-06, - "memory_bytes_spilled": 0, - "disk_bytes_spilled": 0, - "peak_execution_memory_bytes": 0, - "bytes_read": 26800017, - "records_read": 4096, - "bytes_written": 0, - "records_written": 0, - "shuffle_remote_blocks_fetched": 0, - "shuffle_local_blocks_fetched": 0, - "shuffle_fetch_wait_time_seconds": 0.0, - "shuffle_remote_bytes_read": 0, - "shuffle_remote_bytes_read_to_disk": 0, - "shuffle_local_bytes_read": 0, - "shuffle_records_read": 0, - "shuffle_remote_requests_duration": 0, - "shuffle_bytes_read": 0, - "shuffle_bytes_written": 49459, - "shuffle_write_time_seconds": 0.0049475000000000005, - "shuffle_records_written": 1000, - "merged_corrupt_block_chunks": 0, - "merged_fetch_fallback_count": 0, - "merged_remote_blocks_fetched": 0, - "merged_local_blocks_fetched": 0, - "merged_remote_chunks_fetched": 0, - "merged_local_chunks_fetched": 0, - "merged_remote_bytes_read": 0, - "merged_local_bytes_read": 0, - "merged_remote_requests_duration": 0, - "executor_id": "driver", - "host": "localhost", - "index": 8, - "attempt": 0, - "failed": false, - "killed": false, - "speculative": false, - "task_loc": "PROCESS_LOCAL", - "task_type": "ShuffleMapTask" - }, - { - "log_name": "nested_final_plans", - "query_id": 0, - "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:38", - "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 1.48, - "job_id": 2, - "stage_id": 2, - "job_start_timestamp": "2025-03-22T01:19:38.938000000", - "job_end_timestamp": "2025-03-22T01:19:39.884000000", - "job_duration_seconds": 0.9460000000000001, - "stage_start_timestamp": "2025-03-22T01:19:38.941000000", - "stage_end_timestamp": "2025-03-22T01:19:39.881000000", - "stage_duration_seconds": 0.9400000000000001, - "task_id": 11, - "task_start_timestamp": "2025-03-22T01:19:38.962000000", - "task_end_timestamp": "2025-03-22T01:19:39.879000000", - "task_duration_seconds": 0.917, - "nodes": [ - "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", - "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange" - ], - "executor_run_time_seconds": 0.0008619999999999999, - "executor_cpu_time_seconds": 0.42130100000000004, - "executor_deserialize_time_seconds": 3.6e-05, - "executor_deserialize_cpu_time_seconds": 0.016753, - "result_size_bytes": 2078, - "jvm_gc_time_seconds": 6.3e-05, - "result_serialization_time_seconds": 9e-06, - "memory_bytes_spilled": 0, - "disk_bytes_spilled": 0, - "peak_execution_memory_bytes": 0, - "bytes_read": 20611073, - "records_read": 4096, - "bytes_written": 0, - "records_written": 0, - "shuffle_remote_blocks_fetched": 0, - "shuffle_local_blocks_fetched": 0, - "shuffle_fetch_wait_time_seconds": 0.0, - "shuffle_remote_bytes_read": 0, - "shuffle_remote_bytes_read_to_disk": 0, - "shuffle_local_bytes_read": 0, - "shuffle_records_read": 0, - "shuffle_remote_requests_duration": 0, - "shuffle_bytes_read": 0, - "shuffle_bytes_written": 49486, - "shuffle_write_time_seconds": 0.005428420000000001, - "shuffle_records_written": 1000, - "merged_corrupt_block_chunks": 0, - "merged_fetch_fallback_count": 0, - "merged_remote_blocks_fetched": 0, - "merged_local_blocks_fetched": 0, - "merged_remote_chunks_fetched": 0, - "merged_local_chunks_fetched": 0, - "merged_remote_bytes_read": 0, - "merged_local_bytes_read": 0, - "merged_remote_requests_duration": 0, - "executor_id": "driver", - "host": "localhost", - "index": 9, - "attempt": 0, - "failed": false, - "killed": false, - "speculative": false, - "task_loc": "PROCESS_LOCAL", - "task_type": "ShuffleMapTask" - }, - { - "log_name": "nested_final_plans", - "query_id": 0, - "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:38", - "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 1.48, - "job_id": 3, - "stage_id": 4, - "job_start_timestamp": "2025-03-22T01:19:39.938000000", - "job_end_timestamp": "2025-03-22T01:19:40.035000000", - "job_duration_seconds": 0.097, - "stage_start_timestamp": "2025-03-22T01:19:39.942000000", - "stage_end_timestamp": "2025-03-22T01:19:40.035000000", - "stage_duration_seconds": 0.093, - "task_id": 12, - "task_start_timestamp": "2025-03-22T01:19:39.951000000", - "task_end_timestamp": "2025-03-22T01:19:40.034000000", - "task_duration_seconds": 0.083, - "nodes": [ - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange" - ], - "executor_run_time_seconds": 7.4e-05, - "executor_cpu_time_seconds": 0.053965000000000006, - "executor_deserialize_time_seconds": 4.9999999999999996e-06, - "executor_deserialize_cpu_time_seconds": 0.004205, - "result_size_bytes": 3008, - "jvm_gc_time_seconds": 0.0, - "result_serialization_time_seconds": 0.0, - "memory_bytes_spilled": 0, - "disk_bytes_spilled": 0, - "peak_execution_memory_bytes": 0, - "bytes_read": 0, - "records_read": 0, - "bytes_written": 0, - "records_written": 0, - "shuffle_remote_blocks_fetched": 0, - "shuffle_local_blocks_fetched": 0, - "shuffle_fetch_wait_time_seconds": 0.0, - "shuffle_remote_bytes_read": 0, - "shuffle_remote_bytes_read_to_disk": 0, - "shuffle_local_bytes_read": 0, - "shuffle_records_read": 0, - "shuffle_remote_requests_duration": 0, - "shuffle_bytes_read": 0, - "shuffle_bytes_written": 0, - "shuffle_write_time_seconds": 0.0, - "shuffle_records_written": 0, - "merged_corrupt_block_chunks": 0, - "merged_fetch_fallback_count": 0, - "merged_remote_blocks_fetched": 0, - "merged_local_blocks_fetched": 0, - "merged_remote_chunks_fetched": 0, - "merged_local_chunks_fetched": 0, - "merged_remote_bytes_read": 0, - "merged_local_bytes_read": 0, - "merged_remote_requests_duration": 0, - "executor_id": "driver", - "host": "localhost", - "index": 0, - "attempt": 0, - "failed": false, - "killed": false, - "speculative": false, - "task_loc": "NODE_LOCAL", - "task_type": "ResultTask" - }, - { - "log_name": "nested_final_plans", - "query_id": 0, - "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:38", - "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 1.48, - "job_id": 3, - "stage_id": 4, - "job_start_timestamp": "2025-03-22T01:19:39.938000000", - "job_end_timestamp": "2025-03-22T01:19:40.035000000", - "job_duration_seconds": 0.097, - "stage_start_timestamp": "2025-03-22T01:19:39.942000000", - "stage_end_timestamp": "2025-03-22T01:19:40.035000000", - "stage_duration_seconds": 0.093, - "task_id": 13, - "task_start_timestamp": "2025-03-22T01:19:39.953000000", - "task_end_timestamp": "2025-03-22T01:19:40.034000000", - "task_duration_seconds": 0.081, - "nodes": [ - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange" - ], - "executor_run_time_seconds": 7.4e-05, - "executor_cpu_time_seconds": 0.058588, - "executor_deserialize_time_seconds": 4e-06, - "executor_deserialize_cpu_time_seconds": 0.004350000000000001, - "result_size_bytes": 3008, - "jvm_gc_time_seconds": 0.0, - "result_serialization_time_seconds": 0.0, - "memory_bytes_spilled": 0, - "disk_bytes_spilled": 0, - "peak_execution_memory_bytes": 0, - "bytes_read": 0, - "records_read": 0, - "bytes_written": 0, - "records_written": 0, - "shuffle_remote_blocks_fetched": 0, - "shuffle_local_blocks_fetched": 0, - "shuffle_fetch_wait_time_seconds": 0.0, - "shuffle_remote_bytes_read": 0, - "shuffle_remote_bytes_read_to_disk": 0, - "shuffle_local_bytes_read": 0, - "shuffle_records_read": 0, - "shuffle_remote_requests_duration": 0, - "shuffle_bytes_read": 0, - "shuffle_bytes_written": 0, - "shuffle_write_time_seconds": 0.0, - "shuffle_records_written": 0, - "merged_corrupt_block_chunks": 0, - "merged_fetch_fallback_count": 0, - "merged_remote_blocks_fetched": 0, - "merged_local_blocks_fetched": 0, - "merged_remote_chunks_fetched": 0, - "merged_local_chunks_fetched": 0, - "merged_remote_bytes_read": 0, - "merged_local_bytes_read": 0, - "merged_remote_requests_duration": 0, - "executor_id": "driver", - "host": "localhost", - "index": 1, - "attempt": 0, - "failed": false, - "killed": false, - "speculative": false, - "task_loc": "NODE_LOCAL", - "task_type": "ResultTask" - }, - { - "log_name": "nested_final_plans", - "query_id": 0, - "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:38", - "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 1.48, - "job_id": 4, - "stage_id": 6, - "job_start_timestamp": "2025-03-22T01:19:40.070000000", - "job_end_timestamp": "2025-03-22T01:19:40.115000000", - "job_duration_seconds": 0.045, - "stage_start_timestamp": "2025-03-22T01:19:40.071000000", - "stage_end_timestamp": "2025-03-22T01:19:40.114000000", - "stage_duration_seconds": 0.043000000000000003, - "task_id": 14, - "task_start_timestamp": "2025-03-22T01:19:40.077000000", - "task_end_timestamp": "2025-03-22T01:19:40.114000000", - "task_duration_seconds": 0.037, - "nodes": [ - "[1] InMemoryTableScan", - "[23] HashAggregate", - "[23] HashAggregate", - "[24] Exchange", - "[24] Exchange", - "[24] Exchange", - "[24] Exchange" - ], - "executor_run_time_seconds": 2.4999999999999998e-05, - "executor_cpu_time_seconds": 0.016722, - "executor_deserialize_time_seconds": 9e-06, - "executor_deserialize_cpu_time_seconds": 0.0053560000000000005, - "result_size_bytes": 3788, - "jvm_gc_time_seconds": 0.0, - "result_serialization_time_seconds": 0.0, - "memory_bytes_spilled": 0, - "disk_bytes_spilled": 0, - "peak_execution_memory_bytes": 0, - "bytes_read": 51280, - "records_read": 1, - "bytes_written": 0, - "records_written": 0, - "shuffle_remote_blocks_fetched": 0, - "shuffle_local_blocks_fetched": 0, - "shuffle_fetch_wait_time_seconds": 0.0, - "shuffle_remote_bytes_read": 0, - "shuffle_remote_bytes_read_to_disk": 0, - "shuffle_local_bytes_read": 0, - "shuffle_records_read": 0, - "shuffle_remote_requests_duration": 0, - "shuffle_bytes_read": 0, - "shuffle_bytes_written": 59, - "shuffle_write_time_seconds": 0.000598084, - "shuffle_records_written": 1, - "merged_corrupt_block_chunks": 0, - "merged_fetch_fallback_count": 0, - "merged_remote_blocks_fetched": 0, - "merged_local_blocks_fetched": 0, - "merged_remote_chunks_fetched": 0, - "merged_local_chunks_fetched": 0, - "merged_remote_bytes_read": 0, - "merged_local_bytes_read": 0, - "merged_remote_requests_duration": 0, - "executor_id": "driver", - "host": "localhost", - "index": 0, - "attempt": 0, - "failed": false, - "killed": false, - "speculative": false, - "task_loc": "PROCESS_LOCAL", - "task_type": "ShuffleMapTask" + "value_exact": 6, + "unit": "ms", + "readable_value": 6.0, + "readable_unit": "ms", + "readable_str": "6.0 ms" + } + ], + "n_accumulators": 1, + "n_accumulator_totals": 1, + "node_duration_minutes": 0.0001, + "node_id_adj": 2, + "node_name": "[2] WholeStageCodegen" }, { - "log_name": "nested_final_plans", - "query_id": 0, - "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:38", + "query_id": 2, + "query_start_timestamp": "2025-03-22T01:19:40", "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 1.48, - "job_id": 4, - "stage_id": 6, - "job_start_timestamp": "2025-03-22T01:19:40.070000000", - "job_end_timestamp": "2025-03-22T01:19:40.115000000", - "job_duration_seconds": 0.045, - "stage_start_timestamp": "2025-03-22T01:19:40.071000000", - "stage_end_timestamp": "2025-03-22T01:19:40.114000000", - "stage_duration_seconds": 0.043000000000000003, - "task_id": 15, - "task_start_timestamp": "2025-03-22T01:19:40.078000000", - "task_end_timestamp": "2025-03-22T01:19:40.114000000", - "task_duration_seconds": 0.036000000000000004, - "nodes": [ - "[1] InMemoryTableScan", - "[23] HashAggregate", - "[23] HashAggregate", - "[24] Exchange", - "[24] Exchange", - "[24] Exchange", - "[24] Exchange" + "query_duration_seconds": 0.31, + "whole_stage_codegen_id": 3, + "node_id": 100003, + "node_type": "WholeStageCodegen", + "child_nodes": null, + "details": null, + "query_function": "count", + "query_header": "2 - count [0.01 min]", + "accumulators": [ + { + "stage_id": 4, + "task_id": 12, + "accumulator_id": 257, + "metric_name": "duration", + "metric_type": "timing", + "value": 77.0, + "value_exact": 77, + "unit": "ms", + "readable_value": 77.0, + "readable_unit": "ms", + "readable_str": "77.0 ms" + } ], - "executor_run_time_seconds": 2.4999999999999998e-05, - "executor_cpu_time_seconds": 0.010671, - "executor_deserialize_time_seconds": 8e-06, - "executor_deserialize_cpu_time_seconds": 0.004172, - "result_size_bytes": 3788, - "jvm_gc_time_seconds": 0.0, - "result_serialization_time_seconds": 0.0, - "memory_bytes_spilled": 0, - "disk_bytes_spilled": 0, - "peak_execution_memory_bytes": 0, - "bytes_read": 51280, - "records_read": 1, - "bytes_written": 0, - "records_written": 0, - "shuffle_remote_blocks_fetched": 0, - "shuffle_local_blocks_fetched": 0, - "shuffle_fetch_wait_time_seconds": 0.0, - "shuffle_remote_bytes_read": 0, - "shuffle_remote_bytes_read_to_disk": 0, - "shuffle_local_bytes_read": 0, - "shuffle_records_read": 0, - "shuffle_remote_requests_duration": 0, - "shuffle_bytes_read": 0, - "shuffle_bytes_written": 59, - "shuffle_write_time_seconds": 0.000658334, - "shuffle_records_written": 1, - "merged_corrupt_block_chunks": 0, - "merged_fetch_fallback_count": 0, - "merged_remote_blocks_fetched": 0, - "merged_local_blocks_fetched": 0, - "merged_remote_chunks_fetched": 0, - "merged_local_chunks_fetched": 0, - "merged_remote_bytes_read": 0, - "merged_local_bytes_read": 0, - "merged_remote_requests_duration": 0, - "executor_id": "driver", - "host": "localhost", - "index": 1, - "attempt": 0, - "failed": false, - "killed": false, - "speculative": false, - "task_loc": "PROCESS_LOCAL", - "task_type": "ShuffleMapTask" + "accumulator_totals": [ + { + "stage_id": 4, + "task_id": 12, + "accumulator_id": 257, + "metric_name": "duration", + "metric_type": "timing", + "value": 77.0, + "value_exact": 77, + "unit": "ms", + "readable_value": 77.0, + "readable_unit": "ms", + "readable_str": "77.0 ms" + } + ], + "n_accumulators": 1, + "n_accumulator_totals": 1, + "node_duration_minutes": 0.0012833333333333334, + "node_id_adj": 3, + "node_name": "[3] WholeStageCodegen" }, { - "log_name": "nested_final_plans", - "query_id": 0, - "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:38", + "query_id": 2, + "query_start_timestamp": "2025-03-22T01:19:40", "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 1.48, - "job_id": 5, - "stage_id": 9, - "job_start_timestamp": "2025-03-22T01:19:40.134000000", - "job_end_timestamp": "2025-03-22T01:19:40.153000000", - "job_duration_seconds": 0.019, - "stage_start_timestamp": "2025-03-22T01:19:40.135000000", - "stage_end_timestamp": "2025-03-22T01:19:40.152000000", - "stage_duration_seconds": 0.017, - "task_id": 16, - "task_start_timestamp": "2025-03-22T01:19:40.137000000", - "task_end_timestamp": "2025-03-22T01:19:40.152000000", - "task_duration_seconds": 0.015, - "nodes": [ - "[24] Exchange", - "[24] Exchange", - "[24] Exchange", - "[24] Exchange", - "[26] HashAggregate", - "[26] HashAggregate" + "query_duration_seconds": 0.31, + "whole_stage_codegen_id": 4, + "node_id": 100004, + "node_type": "WholeStageCodegen", + "child_nodes": null, + "details": null, + "query_function": "count", + "query_header": "2 - count [0.01 min]", + "accumulators": [ + { + "stage_id": 4, + "task_id": 13, + "accumulator_id": 258, + "metric_name": "duration", + "metric_type": "timing", + "value": 77.0, + "value_exact": 77, + "unit": "ms", + "readable_value": 77.0, + "readable_unit": "ms", + "readable_str": "77.0 ms" + } ], - "executor_run_time_seconds": 6e-06, - "executor_cpu_time_seconds": 0.0066170000000000005, - "executor_deserialize_time_seconds": 4.9999999999999996e-06, - "executor_deserialize_cpu_time_seconds": 0.004958000000000001, - "result_size_bytes": 4038, - "jvm_gc_time_seconds": 0.0, - "result_serialization_time_seconds": 1e-06, - "memory_bytes_spilled": 0, - "disk_bytes_spilled": 0, - "peak_execution_memory_bytes": 0, - "bytes_read": 0, - "records_read": 0, - "bytes_written": 0, - "records_written": 0, - "shuffle_remote_blocks_fetched": 0, - "shuffle_local_blocks_fetched": 2, - "shuffle_fetch_wait_time_seconds": 0.0, - "shuffle_remote_bytes_read": 0, - "shuffle_remote_bytes_read_to_disk": 0, - "shuffle_local_bytes_read": 118, - "shuffle_records_read": 2, - "shuffle_remote_requests_duration": 0, - "shuffle_bytes_read": 118, - "shuffle_bytes_written": 0, - "shuffle_write_time_seconds": 0.0, - "shuffle_records_written": 0, - "merged_corrupt_block_chunks": 0, - "merged_fetch_fallback_count": 0, - "merged_remote_blocks_fetched": 0, - "merged_local_blocks_fetched": 0, - "merged_remote_chunks_fetched": 0, - "merged_local_chunks_fetched": 0, - "merged_remote_bytes_read": 0, - "merged_local_bytes_read": 0, - "merged_remote_requests_duration": 0, - "executor_id": "driver", - "host": "localhost", - "index": 0, - "attempt": 0, - "failed": false, - "killed": false, - "speculative": false, - "task_loc": "NODE_LOCAL", - "task_type": "ResultTask" + "accumulator_totals": [ + { + "stage_id": 4, + "task_id": 13, + "accumulator_id": 258, + "metric_name": "duration", + "metric_type": "timing", + "value": 77.0, + "value_exact": 77, + "unit": "ms", + "readable_value": 77.0, + "readable_unit": "ms", + "readable_str": "77.0 ms" + } + ], + "n_accumulators": 1, + "n_accumulator_totals": 1, + "node_duration_minutes": 0.0012833333333333334, + "node_id_adj": 4, + "node_name": "[4] WholeStageCodegen" }, { "log_name": "nested_final_plans", - "query_id": 2, + "query_id": 0, "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:40", + "query_start_timestamp": "2025-03-22T01:19:38", "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 0.31, + "query_duration_seconds": 1.48, + "query_count": 2, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-22T01:19:38.938000000", "job_end_timestamp": "2025-03-22T01:19:39.884000000", "job_duration_seconds": 0.9460000000000001, "stage_start_timestamp": "2025-03-22T01:19:38.941000000", "stage_end_timestamp": "2025-03-22T01:19:39.881000000", "stage_duration_seconds": 0.9400000000000001, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 2, "task_start_timestamp": "2025-03-22T01:19:38.959000000", "task_end_timestamp": "2025-03-22T01:19:39.878000000", "task_duration_seconds": 0.919, "nodes": [ "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", "[6] Exchange" ], "executor_run_time_seconds": 0.000869, @@ -6602,7 +5684,11 @@ "executor_id": "driver", "host": "localhost", "index": 0, + "partition_id": 0, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -6611,31 +5697,31 @@ }, { "log_name": "nested_final_plans", - "query_id": 2, + "query_id": 0, "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:40", + "query_start_timestamp": "2025-03-22T01:19:38", "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 0.31, + "query_duration_seconds": 1.48, + "query_count": 2, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-22T01:19:38.938000000", "job_end_timestamp": "2025-03-22T01:19:39.884000000", "job_duration_seconds": 0.9460000000000001, "stage_start_timestamp": "2025-03-22T01:19:38.941000000", "stage_end_timestamp": "2025-03-22T01:19:39.881000000", "stage_duration_seconds": 0.9400000000000001, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 3, "task_start_timestamp": "2025-03-22T01:19:38.961000000", "task_end_timestamp": "2025-03-22T01:19:39.877000000", "task_duration_seconds": 0.916, "nodes": [ "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", "[6] Exchange" ], "executor_run_time_seconds": 0.0008619999999999999, @@ -6676,7 +5762,11 @@ "executor_id": "driver", "host": "localhost", "index": 1, + "partition_id": 1, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -6685,31 +5775,31 @@ }, { "log_name": "nested_final_plans", - "query_id": 2, + "query_id": 0, "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:40", + "query_start_timestamp": "2025-03-22T01:19:38", "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 0.31, + "query_duration_seconds": 1.48, + "query_count": 2, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-22T01:19:38.938000000", "job_end_timestamp": "2025-03-22T01:19:39.884000000", "job_duration_seconds": 0.9460000000000001, "stage_start_timestamp": "2025-03-22T01:19:38.941000000", "stage_end_timestamp": "2025-03-22T01:19:39.881000000", "stage_duration_seconds": 0.9400000000000001, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 4, "task_start_timestamp": "2025-03-22T01:19:38.961000000", "task_end_timestamp": "2025-03-22T01:19:39.879000000", "task_duration_seconds": 0.918, "nodes": [ "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", "[6] Exchange" ], "executor_run_time_seconds": 0.000868, @@ -6750,7 +5840,11 @@ "executor_id": "driver", "host": "localhost", "index": 2, + "partition_id": 2, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -6759,31 +5853,31 @@ }, { "log_name": "nested_final_plans", - "query_id": 2, + "query_id": 0, "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:40", + "query_start_timestamp": "2025-03-22T01:19:38", "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 0.31, + "query_duration_seconds": 1.48, + "query_count": 2, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-22T01:19:38.938000000", "job_end_timestamp": "2025-03-22T01:19:39.884000000", "job_duration_seconds": 0.9460000000000001, "stage_start_timestamp": "2025-03-22T01:19:38.941000000", "stage_end_timestamp": "2025-03-22T01:19:39.881000000", "stage_duration_seconds": 0.9400000000000001, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 5, "task_start_timestamp": "2025-03-22T01:19:38.961000000", "task_end_timestamp": "2025-03-22T01:19:39.878000000", "task_duration_seconds": 0.917, "nodes": [ "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", "[6] Exchange" ], "executor_run_time_seconds": 0.000868, @@ -6824,7 +5918,11 @@ "executor_id": "driver", "host": "localhost", "index": 3, + "partition_id": 3, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -6833,31 +5931,31 @@ }, { "log_name": "nested_final_plans", - "query_id": 2, + "query_id": 0, "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:40", + "query_start_timestamp": "2025-03-22T01:19:38", "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 0.31, + "query_duration_seconds": 1.48, + "query_count": 2, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-22T01:19:38.938000000", "job_end_timestamp": "2025-03-22T01:19:39.884000000", "job_duration_seconds": 0.9460000000000001, "stage_start_timestamp": "2025-03-22T01:19:38.941000000", "stage_end_timestamp": "2025-03-22T01:19:39.881000000", "stage_duration_seconds": 0.9400000000000001, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 6, "task_start_timestamp": "2025-03-22T01:19:38.962000000", "task_end_timestamp": "2025-03-22T01:19:39.877000000", "task_duration_seconds": 0.915, "nodes": [ "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", "[6] Exchange" ], "executor_run_time_seconds": 0.000868, @@ -6898,7 +5996,11 @@ "executor_id": "driver", "host": "localhost", "index": 4, + "partition_id": 4, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -6907,31 +6009,31 @@ }, { "log_name": "nested_final_plans", - "query_id": 2, + "query_id": 0, "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:40", + "query_start_timestamp": "2025-03-22T01:19:38", "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 0.31, + "query_duration_seconds": 1.48, + "query_count": 2, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-22T01:19:38.938000000", "job_end_timestamp": "2025-03-22T01:19:39.884000000", "job_duration_seconds": 0.9460000000000001, "stage_start_timestamp": "2025-03-22T01:19:38.941000000", "stage_end_timestamp": "2025-03-22T01:19:39.881000000", "stage_duration_seconds": 0.9400000000000001, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 7, "task_start_timestamp": "2025-03-22T01:19:38.962000000", "task_end_timestamp": "2025-03-22T01:19:39.878000000", "task_duration_seconds": 0.916, "nodes": [ "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", "[6] Exchange" ], "executor_run_time_seconds": 0.000868, @@ -6972,7 +6074,11 @@ "executor_id": "driver", "host": "localhost", "index": 5, + "partition_id": 5, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -6981,31 +6087,31 @@ }, { "log_name": "nested_final_plans", - "query_id": 2, + "query_id": 0, "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:40", + "query_start_timestamp": "2025-03-22T01:19:38", "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 0.31, + "query_duration_seconds": 1.48, + "query_count": 2, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-22T01:19:38.938000000", "job_end_timestamp": "2025-03-22T01:19:39.884000000", "job_duration_seconds": 0.9460000000000001, "stage_start_timestamp": "2025-03-22T01:19:38.941000000", "stage_end_timestamp": "2025-03-22T01:19:39.881000000", "stage_duration_seconds": 0.9400000000000001, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 8, "task_start_timestamp": "2025-03-22T01:19:38.962000000", "task_end_timestamp": "2025-03-22T01:19:39.878000000", "task_duration_seconds": 0.916, "nodes": [ "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", "[6] Exchange" ], "executor_run_time_seconds": 0.000869, @@ -7046,7 +6152,11 @@ "executor_id": "driver", "host": "localhost", "index": 6, + "partition_id": 6, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -7055,31 +6165,31 @@ }, { "log_name": "nested_final_plans", - "query_id": 2, + "query_id": 0, "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:40", + "query_start_timestamp": "2025-03-22T01:19:38", "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 0.31, + "query_duration_seconds": 1.48, + "query_count": 2, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-22T01:19:38.938000000", "job_end_timestamp": "2025-03-22T01:19:39.884000000", "job_duration_seconds": 0.9460000000000001, "stage_start_timestamp": "2025-03-22T01:19:38.941000000", "stage_end_timestamp": "2025-03-22T01:19:39.881000000", "stage_duration_seconds": 0.9400000000000001, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 9, "task_start_timestamp": "2025-03-22T01:19:38.962000000", "task_end_timestamp": "2025-03-22T01:19:39.877000000", "task_duration_seconds": 0.915, "nodes": [ "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", "[6] Exchange" ], "executor_run_time_seconds": 0.000861, @@ -7120,7 +6230,11 @@ "executor_id": "driver", "host": "localhost", "index": 7, + "partition_id": 7, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -7129,31 +6243,31 @@ }, { "log_name": "nested_final_plans", - "query_id": 2, + "query_id": 0, "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:40", + "query_start_timestamp": "2025-03-22T01:19:38", "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 0.31, + "query_duration_seconds": 1.48, + "query_count": 2, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-22T01:19:38.938000000", "job_end_timestamp": "2025-03-22T01:19:39.884000000", "job_duration_seconds": 0.9460000000000001, "stage_start_timestamp": "2025-03-22T01:19:38.941000000", "stage_end_timestamp": "2025-03-22T01:19:39.881000000", "stage_duration_seconds": 0.9400000000000001, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 10, "task_start_timestamp": "2025-03-22T01:19:38.962000000", "task_end_timestamp": "2025-03-22T01:19:39.877000000", "task_duration_seconds": 0.915, "nodes": [ "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", "[6] Exchange" ], "executor_run_time_seconds": 0.000868, @@ -7194,7 +6308,11 @@ "executor_id": "driver", "host": "localhost", "index": 8, + "partition_id": 8, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -7203,45 +6321,275 @@ }, { "log_name": "nested_final_plans", - "query_id": 2, + "query_id": 0, "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:40", + "query_start_timestamp": "2025-03-22T01:19:38", "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 0.31, + "query_duration_seconds": 1.48, + "query_count": 2, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-22T01:19:38.938000000", "job_end_timestamp": "2025-03-22T01:19:39.884000000", "job_duration_seconds": 0.9460000000000001, "stage_start_timestamp": "2025-03-22T01:19:38.941000000", "stage_end_timestamp": "2025-03-22T01:19:39.881000000", "stage_duration_seconds": 0.9400000000000001, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 11, "task_start_timestamp": "2025-03-22T01:19:38.962000000", "task_end_timestamp": "2025-03-22T01:19:39.879000000", "task_duration_seconds": 0.917, "nodes": [ "[3] Scan", - "[3] Scan", - "[4] ColumnarToRow", "[4] ColumnarToRow", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", "[6] Exchange" ], - "executor_run_time_seconds": 0.0008619999999999999, - "executor_cpu_time_seconds": 0.42130100000000004, - "executor_deserialize_time_seconds": 3.6e-05, - "executor_deserialize_cpu_time_seconds": 0.016753, - "result_size_bytes": 2078, - "jvm_gc_time_seconds": 6.3e-05, - "result_serialization_time_seconds": 9e-06, + "executor_run_time_seconds": 0.0008619999999999999, + "executor_cpu_time_seconds": 0.42130100000000004, + "executor_deserialize_time_seconds": 3.6e-05, + "executor_deserialize_cpu_time_seconds": 0.016753, + "result_size_bytes": 2078, + "jvm_gc_time_seconds": 6.3e-05, + "result_serialization_time_seconds": 9e-06, + "memory_bytes_spilled": 0, + "disk_bytes_spilled": 0, + "peak_execution_memory_bytes": 0, + "bytes_read": 20611073, + "records_read": 4096, + "bytes_written": 0, + "records_written": 0, + "shuffle_remote_blocks_fetched": 0, + "shuffle_local_blocks_fetched": 0, + "shuffle_fetch_wait_time_seconds": 0.0, + "shuffle_remote_bytes_read": 0, + "shuffle_remote_bytes_read_to_disk": 0, + "shuffle_local_bytes_read": 0, + "shuffle_records_read": 0, + "shuffle_remote_requests_duration": 0, + "shuffle_bytes_read": 0, + "shuffle_bytes_written": 49486, + "shuffle_write_time_seconds": 0.005428420000000001, + "shuffle_records_written": 1000, + "merged_corrupt_block_chunks": 0, + "merged_fetch_fallback_count": 0, + "merged_remote_blocks_fetched": 0, + "merged_local_blocks_fetched": 0, + "merged_remote_chunks_fetched": 0, + "merged_local_chunks_fetched": 0, + "merged_remote_bytes_read": 0, + "merged_local_bytes_read": 0, + "merged_remote_requests_duration": 0, + "executor_id": "driver", + "host": "localhost", + "index": 9, + "partition_id": 9, + "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, + "failed": false, + "killed": false, + "speculative": false, + "task_loc": "PROCESS_LOCAL", + "task_type": "ShuffleMapTask" + }, + { + "log_name": "nested_final_plans", + "query_id": 0, + "query_function": "count", + "query_start_timestamp": "2025-03-22T01:19:38", + "query_end_timestamp": "2025-03-22T01:19:40", + "query_duration_seconds": 1.48, + "query_count": 2, + "job_id": 3, + "stage_id": 4, + "stage_attempt_id": 0, + "job_start_timestamp": "2025-03-22T01:19:39.938000000", + "job_end_timestamp": "2025-03-22T01:19:40.035000000", + "job_duration_seconds": 0.097, + "stage_start_timestamp": "2025-03-22T01:19:39.942000000", + "stage_end_timestamp": "2025-03-22T01:19:40.035000000", + "stage_duration_seconds": 0.093, + "stage_num_tasks": 2, + "stage_status": "succeeded", + "stage_failure_reason": null, + "task_id": 12, + "task_start_timestamp": "2025-03-22T01:19:39.951000000", + "task_end_timestamp": "2025-03-22T01:19:40.034000000", + "task_duration_seconds": 0.083, + "nodes": [ + "[6] Exchange" + ], + "executor_run_time_seconds": 7.4e-05, + "executor_cpu_time_seconds": 0.053965000000000006, + "executor_deserialize_time_seconds": 4.9999999999999996e-06, + "executor_deserialize_cpu_time_seconds": 0.004205, + "result_size_bytes": 3008, + "jvm_gc_time_seconds": 0.0, + "result_serialization_time_seconds": 0.0, + "memory_bytes_spilled": 0, + "disk_bytes_spilled": 0, + "peak_execution_memory_bytes": 0, + "bytes_read": 0, + "records_read": 0, + "bytes_written": 0, + "records_written": 0, + "shuffle_remote_blocks_fetched": 0, + "shuffle_local_blocks_fetched": 0, + "shuffle_fetch_wait_time_seconds": 0.0, + "shuffle_remote_bytes_read": 0, + "shuffle_remote_bytes_read_to_disk": 0, + "shuffle_local_bytes_read": 0, + "shuffle_records_read": 0, + "shuffle_remote_requests_duration": 0, + "shuffle_bytes_read": 0, + "shuffle_bytes_written": 0, + "shuffle_write_time_seconds": 0.0, + "shuffle_records_written": 0, + "merged_corrupt_block_chunks": 0, + "merged_fetch_fallback_count": 0, + "merged_remote_blocks_fetched": 0, + "merged_local_blocks_fetched": 0, + "merged_remote_chunks_fetched": 0, + "merged_local_chunks_fetched": 0, + "merged_remote_bytes_read": 0, + "merged_local_bytes_read": 0, + "merged_remote_requests_duration": 0, + "executor_id": "driver", + "host": "localhost", + "index": 0, + "partition_id": 0, + "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, + "failed": false, + "killed": false, + "speculative": false, + "task_loc": "NODE_LOCAL", + "task_type": "ResultTask" + }, + { + "log_name": "nested_final_plans", + "query_id": 0, + "query_function": "count", + "query_start_timestamp": "2025-03-22T01:19:38", + "query_end_timestamp": "2025-03-22T01:19:40", + "query_duration_seconds": 1.48, + "query_count": 2, + "job_id": 3, + "stage_id": 4, + "stage_attempt_id": 0, + "job_start_timestamp": "2025-03-22T01:19:39.938000000", + "job_end_timestamp": "2025-03-22T01:19:40.035000000", + "job_duration_seconds": 0.097, + "stage_start_timestamp": "2025-03-22T01:19:39.942000000", + "stage_end_timestamp": "2025-03-22T01:19:40.035000000", + "stage_duration_seconds": 0.093, + "stage_num_tasks": 2, + "stage_status": "succeeded", + "stage_failure_reason": null, + "task_id": 13, + "task_start_timestamp": "2025-03-22T01:19:39.953000000", + "task_end_timestamp": "2025-03-22T01:19:40.034000000", + "task_duration_seconds": 0.081, + "nodes": [ + "[6] Exchange" + ], + "executor_run_time_seconds": 7.4e-05, + "executor_cpu_time_seconds": 0.058588, + "executor_deserialize_time_seconds": 4e-06, + "executor_deserialize_cpu_time_seconds": 0.004350000000000001, + "result_size_bytes": 3008, + "jvm_gc_time_seconds": 0.0, + "result_serialization_time_seconds": 0.0, + "memory_bytes_spilled": 0, + "disk_bytes_spilled": 0, + "peak_execution_memory_bytes": 0, + "bytes_read": 0, + "records_read": 0, + "bytes_written": 0, + "records_written": 0, + "shuffle_remote_blocks_fetched": 0, + "shuffle_local_blocks_fetched": 0, + "shuffle_fetch_wait_time_seconds": 0.0, + "shuffle_remote_bytes_read": 0, + "shuffle_remote_bytes_read_to_disk": 0, + "shuffle_local_bytes_read": 0, + "shuffle_records_read": 0, + "shuffle_remote_requests_duration": 0, + "shuffle_bytes_read": 0, + "shuffle_bytes_written": 0, + "shuffle_write_time_seconds": 0.0, + "shuffle_records_written": 0, + "merged_corrupt_block_chunks": 0, + "merged_fetch_fallback_count": 0, + "merged_remote_blocks_fetched": 0, + "merged_local_blocks_fetched": 0, + "merged_remote_chunks_fetched": 0, + "merged_local_chunks_fetched": 0, + "merged_remote_bytes_read": 0, + "merged_local_bytes_read": 0, + "merged_remote_requests_duration": 0, + "executor_id": "driver", + "host": "localhost", + "index": 1, + "partition_id": 1, + "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, + "failed": false, + "killed": false, + "speculative": false, + "task_loc": "NODE_LOCAL", + "task_type": "ResultTask" + }, + { + "log_name": "nested_final_plans", + "query_id": 0, + "query_function": "count", + "query_start_timestamp": "2025-03-22T01:19:38", + "query_end_timestamp": "2025-03-22T01:19:40", + "query_duration_seconds": 1.48, + "query_count": 1, + "job_id": 4, + "stage_id": 6, + "stage_attempt_id": 0, + "job_start_timestamp": "2025-03-22T01:19:40.070000000", + "job_end_timestamp": "2025-03-22T01:19:40.115000000", + "job_duration_seconds": 0.045, + "stage_start_timestamp": "2025-03-22T01:19:40.071000000", + "stage_end_timestamp": "2025-03-22T01:19:40.114000000", + "stage_duration_seconds": 0.043000000000000003, + "stage_num_tasks": 2, + "stage_status": "succeeded", + "stage_failure_reason": null, + "task_id": 14, + "task_start_timestamp": "2025-03-22T01:19:40.077000000", + "task_end_timestamp": "2025-03-22T01:19:40.114000000", + "task_duration_seconds": 0.037, + "nodes": [ + "[1] InMemoryTableScan", + "[23] HashAggregate", + "[24] Exchange" + ], + "executor_run_time_seconds": 2.4999999999999998e-05, + "executor_cpu_time_seconds": 0.016722, + "executor_deserialize_time_seconds": 9e-06, + "executor_deserialize_cpu_time_seconds": 0.0053560000000000005, + "result_size_bytes": 3788, + "jvm_gc_time_seconds": 0.0, + "result_serialization_time_seconds": 0.0, "memory_bytes_spilled": 0, "disk_bytes_spilled": 0, "peak_execution_memory_bytes": 0, - "bytes_read": 20611073, - "records_read": 4096, + "bytes_read": 51280, + "records_read": 1, "bytes_written": 0, "records_written": 0, "shuffle_remote_blocks_fetched": 0, @@ -7253,9 +6601,9 @@ "shuffle_records_read": 0, "shuffle_remote_requests_duration": 0, "shuffle_bytes_read": 0, - "shuffle_bytes_written": 49486, - "shuffle_write_time_seconds": 0.005428420000000001, - "shuffle_records_written": 1000, + "shuffle_bytes_written": 59, + "shuffle_write_time_seconds": 0.000598084, + "shuffle_records_written": 1, "merged_corrupt_block_chunks": 0, "merged_fetch_fallback_count": 0, "merged_remote_blocks_fetched": 0, @@ -7267,8 +6615,12 @@ "merged_remote_requests_duration": 0, "executor_id": "driver", "host": "localhost", - "index": 9, + "index": 0, + "partition_id": 0, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -7277,41 +6629,45 @@ }, { "log_name": "nested_final_plans", - "query_id": 2, + "query_id": 0, "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:40", + "query_start_timestamp": "2025-03-22T01:19:38", "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 0.31, - "job_id": 3, - "stage_id": 4, - "job_start_timestamp": "2025-03-22T01:19:39.938000000", - "job_end_timestamp": "2025-03-22T01:19:40.035000000", - "job_duration_seconds": 0.097, - "stage_start_timestamp": "2025-03-22T01:19:39.942000000", - "stage_end_timestamp": "2025-03-22T01:19:40.035000000", - "stage_duration_seconds": 0.093, - "task_id": 12, - "task_start_timestamp": "2025-03-22T01:19:39.951000000", - "task_end_timestamp": "2025-03-22T01:19:40.034000000", - "task_duration_seconds": 0.083, + "query_duration_seconds": 1.48, + "query_count": 1, + "job_id": 4, + "stage_id": 6, + "stage_attempt_id": 0, + "job_start_timestamp": "2025-03-22T01:19:40.070000000", + "job_end_timestamp": "2025-03-22T01:19:40.115000000", + "job_duration_seconds": 0.045, + "stage_start_timestamp": "2025-03-22T01:19:40.071000000", + "stage_end_timestamp": "2025-03-22T01:19:40.114000000", + "stage_duration_seconds": 0.043000000000000003, + "stage_num_tasks": 2, + "stage_status": "succeeded", + "stage_failure_reason": null, + "task_id": 15, + "task_start_timestamp": "2025-03-22T01:19:40.078000000", + "task_end_timestamp": "2025-03-22T01:19:40.114000000", + "task_duration_seconds": 0.036000000000000004, "nodes": [ - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange" + "[1] InMemoryTableScan", + "[23] HashAggregate", + "[24] Exchange" ], - "executor_run_time_seconds": 7.4e-05, - "executor_cpu_time_seconds": 0.053965000000000006, - "executor_deserialize_time_seconds": 4.9999999999999996e-06, - "executor_deserialize_cpu_time_seconds": 0.004205, - "result_size_bytes": 3008, + "executor_run_time_seconds": 2.4999999999999998e-05, + "executor_cpu_time_seconds": 0.010671, + "executor_deserialize_time_seconds": 8e-06, + "executor_deserialize_cpu_time_seconds": 0.004172, + "result_size_bytes": 3788, "jvm_gc_time_seconds": 0.0, "result_serialization_time_seconds": 0.0, "memory_bytes_spilled": 0, "disk_bytes_spilled": 0, "peak_execution_memory_bytes": 0, - "bytes_read": 0, - "records_read": 0, + "bytes_read": 51280, + "records_read": 1, "bytes_written": 0, "records_written": 0, "shuffle_remote_blocks_fetched": 0, @@ -7323,9 +6679,9 @@ "shuffle_records_read": 0, "shuffle_remote_requests_duration": 0, "shuffle_bytes_read": 0, - "shuffle_bytes_written": 0, - "shuffle_write_time_seconds": 0.0, - "shuffle_records_written": 0, + "shuffle_bytes_written": 59, + "shuffle_write_time_seconds": 0.000658334, + "shuffle_records_written": 1, "merged_corrupt_block_chunks": 0, "merged_fetch_fallback_count": 0, "merged_remote_blocks_fetched": 0, @@ -7337,46 +6693,53 @@ "merged_remote_requests_duration": 0, "executor_id": "driver", "host": "localhost", - "index": 0, + "index": 1, + "partition_id": 1, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, - "task_loc": "NODE_LOCAL", - "task_type": "ResultTask" + "task_loc": "PROCESS_LOCAL", + "task_type": "ShuffleMapTask" }, { "log_name": "nested_final_plans", - "query_id": 2, + "query_id": 0, "query_function": "count", - "query_start_timestamp": "2025-03-22T01:19:40", + "query_start_timestamp": "2025-03-22T01:19:38", "query_end_timestamp": "2025-03-22T01:19:40", - "query_duration_seconds": 0.31, - "job_id": 3, - "stage_id": 4, - "job_start_timestamp": "2025-03-22T01:19:39.938000000", - "job_end_timestamp": "2025-03-22T01:19:40.035000000", - "job_duration_seconds": 0.097, - "stage_start_timestamp": "2025-03-22T01:19:39.942000000", - "stage_end_timestamp": "2025-03-22T01:19:40.035000000", - "stage_duration_seconds": 0.093, - "task_id": 13, - "task_start_timestamp": "2025-03-22T01:19:39.953000000", - "task_end_timestamp": "2025-03-22T01:19:40.034000000", - "task_duration_seconds": 0.081, + "query_duration_seconds": 1.48, + "query_count": 1, + "job_id": 5, + "stage_id": 9, + "stage_attempt_id": 0, + "job_start_timestamp": "2025-03-22T01:19:40.134000000", + "job_end_timestamp": "2025-03-22T01:19:40.153000000", + "job_duration_seconds": 0.019, + "stage_start_timestamp": "2025-03-22T01:19:40.135000000", + "stage_end_timestamp": "2025-03-22T01:19:40.152000000", + "stage_duration_seconds": 0.017, + "stage_num_tasks": 1, + "stage_status": "succeeded", + "stage_failure_reason": null, + "task_id": 16, + "task_start_timestamp": "2025-03-22T01:19:40.137000000", + "task_end_timestamp": "2025-03-22T01:19:40.152000000", + "task_duration_seconds": 0.015, "nodes": [ - "[6] Exchange", - "[6] Exchange", - "[6] Exchange", - "[6] Exchange" + "[24] Exchange", + "[26] HashAggregate" ], - "executor_run_time_seconds": 7.4e-05, - "executor_cpu_time_seconds": 0.058588, - "executor_deserialize_time_seconds": 4e-06, - "executor_deserialize_cpu_time_seconds": 0.004350000000000001, - "result_size_bytes": 3008, + "executor_run_time_seconds": 6e-06, + "executor_cpu_time_seconds": 0.0066170000000000005, + "executor_deserialize_time_seconds": 4.9999999999999996e-06, + "executor_deserialize_cpu_time_seconds": 0.004958000000000001, + "result_size_bytes": 4038, "jvm_gc_time_seconds": 0.0, - "result_serialization_time_seconds": 0.0, + "result_serialization_time_seconds": 1e-06, "memory_bytes_spilled": 0, "disk_bytes_spilled": 0, "peak_execution_memory_bytes": 0, @@ -7385,14 +6748,14 @@ "bytes_written": 0, "records_written": 0, "shuffle_remote_blocks_fetched": 0, - "shuffle_local_blocks_fetched": 0, + "shuffle_local_blocks_fetched": 2, "shuffle_fetch_wait_time_seconds": 0.0, "shuffle_remote_bytes_read": 0, "shuffle_remote_bytes_read_to_disk": 0, - "shuffle_local_bytes_read": 0, - "shuffle_records_read": 0, + "shuffle_local_bytes_read": 118, + "shuffle_records_read": 2, "shuffle_remote_requests_duration": 0, - "shuffle_bytes_read": 0, + "shuffle_bytes_read": 118, "shuffle_bytes_written": 0, "shuffle_write_time_seconds": 0.0, "shuffle_records_written": 0, @@ -7407,8 +6770,12 @@ "merged_remote_requests_duration": 0, "executor_id": "driver", "host": "localhost", - "index": 1, + "index": 0, + "partition_id": 0, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -7422,14 +6789,19 @@ "query_start_timestamp": "2025-03-22T01:19:40", "query_end_timestamp": "2025-03-22T01:19:40", "query_duration_seconds": 0.31, + "query_count": 1, "job_id": 6, "stage_id": 11, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-22T01:19:40.289000000", "job_end_timestamp": "2025-03-22T01:19:40.431000000", "job_duration_seconds": 0.14200000000000002, "stage_start_timestamp": "2025-03-22T01:19:40.290000000", "stage_end_timestamp": "2025-03-22T01:19:40.430000000", "stage_duration_seconds": 0.14, + "stage_num_tasks": 2, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 17, "task_start_timestamp": "2025-03-22T01:19:40.294000000", "task_end_timestamp": "2025-03-22T01:19:40.430000000", @@ -7437,11 +6809,6 @@ "nodes": [ "[1] InMemoryTableScan", "[23] HashAggregate", - "[23] HashAggregate", - "[23] HashAggregate", - "[24] Exchange", - "[24] Exchange", - "[24] Exchange", "[24] Exchange" ], "executor_run_time_seconds": 0.00013099999999999999, @@ -7482,7 +6849,11 @@ "executor_id": "driver", "host": "localhost", "index": 0, + "partition_id": 0, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -7496,14 +6867,19 @@ "query_start_timestamp": "2025-03-22T01:19:40", "query_end_timestamp": "2025-03-22T01:19:40", "query_duration_seconds": 0.31, + "query_count": 1, "job_id": 6, "stage_id": 11, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-22T01:19:40.289000000", "job_end_timestamp": "2025-03-22T01:19:40.431000000", "job_duration_seconds": 0.14200000000000002, "stage_start_timestamp": "2025-03-22T01:19:40.290000000", "stage_end_timestamp": "2025-03-22T01:19:40.430000000", "stage_duration_seconds": 0.14, + "stage_num_tasks": 2, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 18, "task_start_timestamp": "2025-03-22T01:19:40.294000000", "task_end_timestamp": "2025-03-22T01:19:40.430000000", @@ -7511,11 +6887,6 @@ "nodes": [ "[1] InMemoryTableScan", "[23] HashAggregate", - "[23] HashAggregate", - "[23] HashAggregate", - "[24] Exchange", - "[24] Exchange", - "[24] Exchange", "[24] Exchange" ], "executor_run_time_seconds": 0.00013, @@ -7556,7 +6927,11 @@ "executor_id": "driver", "host": "localhost", "index": 1, + "partition_id": 1, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -7570,31 +6945,27 @@ "query_start_timestamp": "2025-03-22T01:19:40", "query_end_timestamp": "2025-03-22T01:19:40", "query_duration_seconds": 0.31, + "query_count": 1, "job_id": 7, "stage_id": 14, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-22T01:19:40.474000000", "job_end_timestamp": "2025-03-22T01:19:40.510000000", "job_duration_seconds": 0.036000000000000004, "stage_start_timestamp": "2025-03-22T01:19:40.478000000", "stage_end_timestamp": "2025-03-22T01:19:40.510000000", "stage_duration_seconds": 0.032, + "stage_num_tasks": 1, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 19, "task_start_timestamp": "2025-03-22T01:19:40.482000000", "task_end_timestamp": "2025-03-22T01:19:40.509000000", "task_duration_seconds": 0.027, "nodes": [ "[24] Exchange", - "[24] Exchange", - "[24] Exchange", - "[24] Exchange", - "[27] HashAggregate", "[27] HashAggregate", - "[27] HashAggregate", - "[28] HashAggregate", "[28] HashAggregate", - "[29] Exchange", - "[29] Exchange", - "[29] Exchange", "[29] Exchange" ], "executor_run_time_seconds": 1.9999999999999998e-05, @@ -7635,7 +7006,11 @@ "executor_id": "driver", "host": "localhost", "index": 0, + "partition_id": 0, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -7649,24 +7024,25 @@ "query_start_timestamp": "2025-03-22T01:19:40", "query_end_timestamp": "2025-03-22T01:19:40", "query_duration_seconds": 0.31, + "query_count": 1, "job_id": 8, "stage_id": 18, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-22T01:19:40.523000000", "job_end_timestamp": "2025-03-22T01:19:40.534000000", "job_duration_seconds": 0.011, "stage_start_timestamp": "2025-03-22T01:19:40.524000000", "stage_end_timestamp": "2025-03-22T01:19:40.533000000", "stage_duration_seconds": 0.009000000000000001, + "stage_num_tasks": 1, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 20, "task_start_timestamp": "2025-03-22T01:19:40.526000000", "task_end_timestamp": "2025-03-22T01:19:40.533000000", "task_duration_seconds": 0.007, "nodes": [ "[29] Exchange", - "[29] Exchange", - "[29] Exchange", - "[29] Exchange", - "[31] HashAggregate", "[31] HashAggregate" ], "executor_run_time_seconds": 4e-06, @@ -7707,11 +7083,163 @@ "executor_id": "driver", "host": "localhost", "index": 0, + "partition_id": 0, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, "task_loc": "NODE_LOCAL", "task_type": "ResultTask" + }, + { + "log_name": "nested_final_plans", + "query_id": null, + "query_function": null, + "query_start_timestamp": null, + "query_end_timestamp": null, + "query_duration_seconds": null, + "query_count": 0, + "job_id": 0, + "stage_id": 0, + "stage_attempt_id": 0, + "job_start_timestamp": "2025-03-22T01:19:37.539000000", + "job_end_timestamp": "2025-03-22T01:19:37.855000000", + "job_duration_seconds": 0.316, + "stage_start_timestamp": "2025-03-22T01:19:37.547000000", + "stage_end_timestamp": "2025-03-22T01:19:37.853000000", + "stage_duration_seconds": 0.306, + "stage_num_tasks": 1, + "stage_status": "succeeded", + "stage_failure_reason": null, + "task_id": 0, + "task_start_timestamp": "2025-03-22T01:19:37.635000000", + "task_end_timestamp": "2025-03-22T01:19:37.850000000", + "task_duration_seconds": 0.215, + "nodes": [], + "executor_run_time_seconds": 0.000146, + "executor_cpu_time_seconds": 0.021813000000000003, + "executor_deserialize_time_seconds": 3.7e-05, + "executor_deserialize_cpu_time_seconds": 0.033016000000000004, + "result_size_bytes": 1942, + "jvm_gc_time_seconds": 0.0, + "result_serialization_time_seconds": 2e-06, + "memory_bytes_spilled": 0, + "disk_bytes_spilled": 0, + "peak_execution_memory_bytes": 0, + "bytes_read": 0, + "records_read": 0, + "bytes_written": 0, + "records_written": 0, + "shuffle_remote_blocks_fetched": 0, + "shuffle_local_blocks_fetched": 0, + "shuffle_fetch_wait_time_seconds": 0.0, + "shuffle_remote_bytes_read": 0, + "shuffle_remote_bytes_read_to_disk": 0, + "shuffle_local_bytes_read": 0, + "shuffle_records_read": 0, + "shuffle_remote_requests_duration": 0, + "shuffle_bytes_read": 0, + "shuffle_bytes_written": 0, + "shuffle_write_time_seconds": 0.0, + "shuffle_records_written": 0, + "merged_corrupt_block_chunks": 0, + "merged_fetch_fallback_count": 0, + "merged_remote_blocks_fetched": 0, + "merged_local_blocks_fetched": 0, + "merged_remote_chunks_fetched": 0, + "merged_local_chunks_fetched": 0, + "merged_remote_bytes_read": 0, + "merged_local_bytes_read": 0, + "merged_remote_requests_duration": 0, + "executor_id": "driver", + "host": "localhost", + "index": 0, + "partition_id": 0, + "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, + "failed": false, + "killed": false, + "speculative": false, + "task_loc": "PROCESS_LOCAL", + "task_type": "ResultTask" + }, + { + "log_name": "nested_final_plans", + "query_id": null, + "query_function": null, + "query_start_timestamp": null, + "query_end_timestamp": null, + "query_duration_seconds": null, + "query_count": 0, + "job_id": 1, + "stage_id": 1, + "stage_attempt_id": 0, + "job_start_timestamp": "2025-03-22T01:19:38.277000000", + "job_end_timestamp": "2025-03-22T01:19:38.298000000", + "job_duration_seconds": 0.021, + "stage_start_timestamp": "2025-03-22T01:19:38.278000000", + "stage_end_timestamp": "2025-03-22T01:19:38.298000000", + "stage_duration_seconds": 0.02, + "stage_num_tasks": 1, + "stage_status": "succeeded", + "stage_failure_reason": null, + "task_id": 1, + "task_start_timestamp": "2025-03-22T01:19:38.287000000", + "task_end_timestamp": "2025-03-22T01:19:38.298000000", + "task_duration_seconds": 0.011, + "nodes": [], + "executor_run_time_seconds": 3e-06, + "executor_cpu_time_seconds": 0.0015290000000000002, + "executor_deserialize_time_seconds": 4e-06, + "executor_deserialize_cpu_time_seconds": 0.004817, + "result_size_bytes": 1899, + "jvm_gc_time_seconds": 0.0, + "result_serialization_time_seconds": 0.0, + "memory_bytes_spilled": 0, + "disk_bytes_spilled": 0, + "peak_execution_memory_bytes": 0, + "bytes_read": 0, + "records_read": 0, + "bytes_written": 0, + "records_written": 0, + "shuffle_remote_blocks_fetched": 0, + "shuffle_local_blocks_fetched": 0, + "shuffle_fetch_wait_time_seconds": 0.0, + "shuffle_remote_bytes_read": 0, + "shuffle_remote_bytes_read_to_disk": 0, + "shuffle_local_bytes_read": 0, + "shuffle_records_read": 0, + "shuffle_remote_requests_duration": 0, + "shuffle_bytes_read": 0, + "shuffle_bytes_written": 0, + "shuffle_write_time_seconds": 0.0, + "shuffle_records_written": 0, + "merged_corrupt_block_chunks": 0, + "merged_fetch_fallback_count": 0, + "merged_remote_blocks_fetched": 0, + "merged_local_blocks_fetched": 0, + "merged_remote_chunks_fetched": 0, + "merged_local_chunks_fetched": 0, + "merged_remote_bytes_read": 0, + "merged_local_bytes_read": 0, + "merged_remote_requests_duration": 0, + "executor_id": "driver", + "host": "localhost", + "index": 0, + "partition_id": 0, + "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, + "failed": false, + "killed": false, + "speculative": false, + "task_loc": "PROCESS_LOCAL", + "task_type": "ResultTask" } ] \ No newline at end of file diff --git a/tests/data/test_full_parsing/expected_nested_loop_join.json b/tests/data/test_full_parsing/expected_nested_loop_join.json index dfc3b58..65b1454 100644 --- a/tests/data/test_full_parsing/expected_nested_loop_join.json +++ b/tests/data/test_full_parsing/expected_nested_loop_join.json @@ -1,4 +1,424 @@ [ + { + "query_id": 0, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.01, + "whole_stage_codegen_id": null, + "node_id": 1, + "node_type": "Unknown", + "child_nodes": "2", + "details": "{\"node_id\":1,\"node_type\":\"Unknown\",\"detail\":{\"raw\":\"(1) Execute CreateViewCommand\\nOutput: []\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "0 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 1, + "node_name": "[1] Unknown" + }, + { + "query_id": 0, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.01, + "whole_stage_codegen_id": null, + "node_id": 2, + "node_type": "Unknown", + "child_nodes": "5", + "details": "{\"node_id\":2,\"node_type\":\"Unknown\",\"detail\":{\"raw\":\"(2) CreateViewCommand\\nArguments: `df`, false, true, LocalTempView, true\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "0 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 2, + "node_name": "[2] Unknown" + }, + { + "query_id": 0, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.01, + "whole_stage_codegen_id": null, + "node_id": 3, + "node_type": "Unknown", + "child_nodes": "", + "details": "{\"node_id\":3,\"node_type\":\"Unknown\",\"detail\":{\"raw\":\"(3) LogicalRelation\\nArguments: parquet, [id1#0, id2#1, id3#2, id4#3L, id5#4L, id6#5L, v1#6L, v2#7L, v3#8], false\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "0 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 3, + "node_name": "[3] Unknown" + }, + { + "query_id": 0, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.01, + "whole_stage_codegen_id": null, + "node_id": 4, + "node_type": "LocalLimit", + "child_nodes": "3", + "details": "{\"node_id\":4,\"node_type\":\"LocalLimit\",\"detail\":{\"raw\":\"(4) LocalLimit\\nArguments: 1000\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "0 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 4, + "node_name": "[4] LocalLimit" + }, + { + "query_id": 0, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.01, + "whole_stage_codegen_id": null, + "node_id": 5, + "node_type": "GlobalLimit", + "child_nodes": "4", + "details": "{\"node_id\":5,\"node_type\":\"GlobalLimit\",\"detail\":{\"raw\":\"(5) GlobalLimit\\nArguments: 1000\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "0 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 5, + "node_name": "[5] GlobalLimit" + }, + { + "query_id": 1, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.0, + "whole_stage_codegen_id": null, + "node_id": 1, + "node_type": "Unknown", + "child_nodes": "2", + "details": "{\"node_id\":1,\"node_type\":\"Unknown\",\"detail\":{\"raw\":\"(1) Execute CreateViewCommand\\nOutput: []\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "1 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 1, + "node_name": "[1] Unknown" + }, + { + "query_id": 1, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.0, + "whole_stage_codegen_id": null, + "node_id": 2, + "node_type": "Unknown", + "child_nodes": "8", + "details": "{\"node_id\":2,\"node_type\":\"Unknown\",\"detail\":{\"raw\":\"(2) CreateViewCommand\\nArguments: `df2`, false, true, LocalTempView, true\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "1 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 2, + "node_name": "[2] Unknown" + }, + { + "query_id": 1, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.0, + "whole_stage_codegen_id": null, + "node_id": 3, + "node_type": "Unknown", + "child_nodes": "", + "details": "{\"node_id\":3,\"node_type\":\"Unknown\",\"detail\":{\"raw\":\"(3) LogicalRelation\\nArguments: parquet, [id1#0, id2#1, id3#2, id4#3L, id5#4L, id6#5L, v1#6L, v2#7L, v3#8], false\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "1 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 3, + "node_name": "[3] Unknown" + }, + { + "query_id": 1, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.0, + "whole_stage_codegen_id": null, + "node_id": 4, + "node_type": "LocalLimit", + "child_nodes": "3", + "details": "{\"node_id\":4,\"node_type\":\"LocalLimit\",\"detail\":{\"raw\":\"(4) LocalLimit\\nArguments: 1000\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "1 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 4, + "node_name": "[4] LocalLimit" + }, + { + "query_id": 1, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.0, + "whole_stage_codegen_id": null, + "node_id": 5, + "node_type": "GlobalLimit", + "child_nodes": "4", + "details": "{\"node_id\":5,\"node_type\":\"GlobalLimit\",\"detail\":{\"raw\":\"(5) GlobalLimit\\nArguments: 1000\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "1 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 5, + "node_name": "[5] GlobalLimit" + }, + { + "query_id": 1, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.0, + "whole_stage_codegen_id": null, + "node_id": 6, + "node_type": "Unknown", + "child_nodes": "5", + "details": "{\"node_id\":6,\"node_type\":\"Unknown\",\"detail\":{\"raw\":\"(6) View\\nArguments: `df`, true\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "1 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 6, + "node_name": "[6] Unknown" + }, + { + "query_id": 1, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.0, + "whole_stage_codegen_id": null, + "node_id": 7, + "node_type": "Unknown", + "child_nodes": "6", + "details": "{\"node_id\":7,\"node_type\":\"Unknown\",\"detail\":{\"raw\":\"(7) SubqueryAlias\\nArguments: df\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "1 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 7, + "node_name": "[7] Unknown" + }, + { + "query_id": 1, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.0, + "whole_stage_codegen_id": null, + "node_id": 8, + "node_type": "Project", + "child_nodes": "7", + "details": "{\"node_id\":8,\"node_type\":\"Project\",\"detail\":{\"raw\":\"(8) Project\\nArguments: [id1#0, id2#1, id3#2, id4#3L, id5#4L, id6#5L, v1#6L, v2#7L, v3#8]\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "1 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 8, + "node_name": "[8] Project" + }, + { + "query_id": 2, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.0, + "whole_stage_codegen_id": null, + "node_id": 1, + "node_type": "Unknown", + "child_nodes": "2", + "details": "{\"node_id\":1,\"node_type\":\"Unknown\",\"detail\":{\"raw\":\"(1) Execute CreateViewCommand\\nOutput: []\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "2 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 1, + "node_name": "[1] Unknown" + }, + { + "query_id": 2, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.0, + "whole_stage_codegen_id": null, + "node_id": 2, + "node_type": "Unknown", + "child_nodes": "8", + "details": "{\"node_id\":2,\"node_type\":\"Unknown\",\"detail\":{\"raw\":\"(2) CreateViewCommand\\nArguments: `df3`, false, true, LocalTempView, true\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "2 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 2, + "node_name": "[2] Unknown" + }, + { + "query_id": 2, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.0, + "whole_stage_codegen_id": null, + "node_id": 3, + "node_type": "Unknown", + "child_nodes": "", + "details": "{\"node_id\":3,\"node_type\":\"Unknown\",\"detail\":{\"raw\":\"(3) LogicalRelation\\nArguments: parquet, [id1#0, id2#1, id3#2, id4#3L, id5#4L, id6#5L, v1#6L, v2#7L, v3#8], false\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "2 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 3, + "node_name": "[3] Unknown" + }, + { + "query_id": 2, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.0, + "whole_stage_codegen_id": null, + "node_id": 4, + "node_type": "LocalLimit", + "child_nodes": "3", + "details": "{\"node_id\":4,\"node_type\":\"LocalLimit\",\"detail\":{\"raw\":\"(4) LocalLimit\\nArguments: 1000\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "2 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 4, + "node_name": "[4] LocalLimit" + }, + { + "query_id": 2, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.0, + "whole_stage_codegen_id": null, + "node_id": 5, + "node_type": "GlobalLimit", + "child_nodes": "4", + "details": "{\"node_id\":5,\"node_type\":\"GlobalLimit\",\"detail\":{\"raw\":\"(5) GlobalLimit\\nArguments: 1000\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "2 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 5, + "node_name": "[5] GlobalLimit" + }, + { + "query_id": 2, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.0, + "whole_stage_codegen_id": null, + "node_id": 6, + "node_type": "Unknown", + "child_nodes": "5", + "details": "{\"node_id\":6,\"node_type\":\"Unknown\",\"detail\":{\"raw\":\"(6) View\\nArguments: `df`, true\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "2 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 6, + "node_name": "[6] Unknown" + }, + { + "query_id": 2, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.0, + "whole_stage_codegen_id": null, + "node_id": 7, + "node_type": "Unknown", + "child_nodes": "6", + "details": "{\"node_id\":7,\"node_type\":\"Unknown\",\"detail\":{\"raw\":\"(7) SubqueryAlias\\nArguments: df\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "2 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 7, + "node_name": "[7] Unknown" + }, + { + "query_id": 2, + "query_start_timestamp": "2025-03-20T18:57:58", + "query_end_timestamp": "2025-03-20T18:57:58", + "query_duration_seconds": 0.0, + "whole_stage_codegen_id": null, + "node_id": 8, + "node_type": "Project", + "child_nodes": "7", + "details": "{\"node_id\":8,\"node_type\":\"Project\",\"detail\":{\"raw\":\"(8) Project\\nArguments: [id1#0, id2#1, id3#2, id4#3L, id5#4L, id6#5L, v1#6L, v2#7L, v3#8]\"}}", + "query_function": "createOrReplaceTempView", + "query_header": "2 - createOrReplaceTempView [0.0 min]", + "accumulators": null, + "accumulator_totals": null, + "n_accumulators": null, + "n_accumulator_totals": null, + "node_duration_minutes": null, + "node_id_adj": 8, + "node_name": "[8] Project" + }, { "query_id": 3, "query_start_timestamp": "2025-03-20T18:57:59", @@ -14,7 +434,7 @@ "accumulators": [ { "stage_id": 1, - "task_id": 8, + "task_id": 1, "accumulator_id": 73, "metric_name": "scan time", "metric_type": "timing", @@ -92,7 +512,7 @@ }, { "stage_id": 1, - "task_id": 1, + "task_id": 7, "accumulator_id": 73, "metric_name": "scan time", "metric_type": "timing", @@ -105,7 +525,7 @@ }, { "stage_id": 1, - "task_id": 7, + "task_id": 8, "accumulator_id": 73, "metric_name": "scan time", "metric_type": "timing", @@ -118,7 +538,7 @@ }, { "stage_id": 1, - "task_id": 10, + "task_id": 9, "accumulator_id": 73, "metric_name": "scan time", "metric_type": "timing", @@ -131,7 +551,7 @@ }, { "stage_id": 1, - "task_id": 9, + "task_id": 10, "accumulator_id": 73, "metric_name": "scan time", "metric_type": "timing", @@ -209,7 +629,7 @@ }, { "stage_id": 1, - "task_id": 10, + "task_id": 6, "accumulator_id": 72, "metric_name": "number of output rows", "metric_type": "sum", @@ -222,7 +642,7 @@ }, { "stage_id": 1, - "task_id": 9, + "task_id": 7, "accumulator_id": 72, "metric_name": "number of output rows", "metric_type": "sum", @@ -248,7 +668,7 @@ }, { "stage_id": 1, - "task_id": 7, + "task_id": 9, "accumulator_id": 72, "metric_name": "number of output rows", "metric_type": "sum", @@ -261,7 +681,7 @@ }, { "stage_id": 1, - "task_id": 6, + "task_id": 10, "accumulator_id": 72, "metric_name": "number of output rows", "metric_type": "sum", @@ -361,7 +781,7 @@ "accumulators": [ { "stage_id": 1, - "task_id": 4, + "task_id": 1, "accumulator_id": 180, "metric_name": "number of input batches", "metric_type": "sum", @@ -387,7 +807,7 @@ }, { "stage_id": 1, - "task_id": 1, + "task_id": 3, "accumulator_id": 180, "metric_name": "number of input batches", "metric_type": "sum", @@ -400,7 +820,7 @@ }, { "stage_id": 1, - "task_id": 9, + "task_id": 4, "accumulator_id": 180, "metric_name": "number of input batches", "metric_type": "sum", @@ -426,7 +846,7 @@ }, { "stage_id": 1, - "task_id": 10, + "task_id": 6, "accumulator_id": 180, "metric_name": "number of input batches", "metric_type": "sum", @@ -439,7 +859,7 @@ }, { "stage_id": 1, - "task_id": 3, + "task_id": 7, "accumulator_id": 180, "metric_name": "number of input batches", "metric_type": "sum", @@ -465,7 +885,7 @@ }, { "stage_id": 1, - "task_id": 7, + "task_id": 9, "accumulator_id": 180, "metric_name": "number of input batches", "metric_type": "sum", @@ -478,7 +898,7 @@ }, { "stage_id": 1, - "task_id": 6, + "task_id": 10, "accumulator_id": 180, "metric_name": "number of input batches", "metric_type": "sum", @@ -491,7 +911,7 @@ }, { "stage_id": 1, - "task_id": 10, + "task_id": 1, "accumulator_id": 179, "metric_name": "number of output rows", "metric_type": "sum", @@ -504,7 +924,7 @@ }, { "stage_id": 1, - "task_id": 9, + "task_id": 2, "accumulator_id": 179, "metric_name": "number of output rows", "metric_type": "sum", @@ -517,7 +937,7 @@ }, { "stage_id": 1, - "task_id": 8, + "task_id": 3, "accumulator_id": 179, "metric_name": "number of output rows", "metric_type": "sum", @@ -530,7 +950,7 @@ }, { "stage_id": 1, - "task_id": 7, + "task_id": 4, "accumulator_id": 179, "metric_name": "number of output rows", "metric_type": "sum", @@ -543,7 +963,7 @@ }, { "stage_id": 1, - "task_id": 6, + "task_id": 5, "accumulator_id": 179, "metric_name": "number of output rows", "metric_type": "sum", @@ -556,7 +976,7 @@ }, { "stage_id": 1, - "task_id": 5, + "task_id": 6, "accumulator_id": 179, "metric_name": "number of output rows", "metric_type": "sum", @@ -569,7 +989,7 @@ }, { "stage_id": 1, - "task_id": 4, + "task_id": 7, "accumulator_id": 179, "metric_name": "number of output rows", "metric_type": "sum", @@ -582,7 +1002,7 @@ }, { "stage_id": 1, - "task_id": 3, + "task_id": 8, "accumulator_id": 179, "metric_name": "number of output rows", "metric_type": "sum", @@ -595,7 +1015,7 @@ }, { "stage_id": 1, - "task_id": 2, + "task_id": 9, "accumulator_id": 179, "metric_name": "number of output rows", "metric_type": "sum", @@ -608,7 +1028,7 @@ }, { "stage_id": 1, - "task_id": 1, + "task_id": 10, "accumulator_id": 179, "metric_name": "number of output rows", "metric_type": "sum", @@ -688,8 +1108,8 @@ "query_header": "3 - count [0.01 min]", "accumulators": [ { - "stage_id": 8, - "task_id": 23, + "stage_id": 6, + "task_id": 22, "accumulator_id": 163, "metric_name": "fetch wait time", "metric_type": "timing", @@ -701,8 +1121,8 @@ "readable_str": "0.0 ms" }, { - "stage_id": 6, - "task_id": 22, + "stage_id": 8, + "task_id": 23, "accumulator_id": 163, "metric_name": "fetch wait time", "metric_type": "timing", @@ -713,19 +1133,6 @@ "readable_unit": "ms", "readable_str": "0.0 ms" }, - { - "stage_id": 1, - "task_id": 6, - "accumulator_id": 177, - "metric_name": "shuffle write time", - "metric_type": "timing", - "value": 8.35754, - "value_exact": null, - "unit": "ms", - "readable_value": 8.4, - "readable_unit": "ms", - "readable_str": "8.36 ms" - }, { "stage_id": 1, "task_id": 1, @@ -791,6 +1198,19 @@ "readable_unit": "ms", "readable_str": "8.58 ms" }, + { + "stage_id": 1, + "task_id": 6, + "accumulator_id": 177, + "metric_name": "shuffle write time", + "metric_type": "timing", + "value": 8.35754, + "value_exact": null, + "unit": "ms", + "readable_value": 8.4, + "readable_unit": "ms", + "readable_str": "8.36 ms" + }, { "stage_id": 1, "task_id": 7, @@ -858,7 +1278,7 @@ }, { "stage_id": 1, - "task_id": 10, + "task_id": 2, "accumulator_id": 156, "metric_name": "data size", "metric_type": "size", @@ -871,7 +1291,7 @@ }, { "stage_id": 1, - "task_id": 2, + "task_id": 3, "accumulator_id": 156, "metric_name": "data size", "metric_type": "size", @@ -884,7 +1304,7 @@ }, { "stage_id": 1, - "task_id": 3, + "task_id": 4, "accumulator_id": 156, "metric_name": "data size", "metric_type": "size", @@ -897,7 +1317,7 @@ }, { "stage_id": 1, - "task_id": 4, + "task_id": 5, "accumulator_id": 156, "metric_name": "data size", "metric_type": "size", @@ -910,7 +1330,7 @@ }, { "stage_id": 1, - "task_id": 5, + "task_id": 6, "accumulator_id": 156, "metric_name": "data size", "metric_type": "size", @@ -923,7 +1343,7 @@ }, { "stage_id": 1, - "task_id": 6, + "task_id": 7, "accumulator_id": 156, "metric_name": "data size", "metric_type": "size", @@ -936,7 +1356,7 @@ }, { "stage_id": 1, - "task_id": 7, + "task_id": 8, "accumulator_id": 156, "metric_name": "data size", "metric_type": "size", @@ -949,7 +1369,7 @@ }, { "stage_id": 1, - "task_id": 8, + "task_id": 9, "accumulator_id": 156, "metric_name": "data size", "metric_type": "size", @@ -962,7 +1382,7 @@ }, { "stage_id": 1, - "task_id": 9, + "task_id": 10, "accumulator_id": 156, "metric_name": "data size", "metric_type": "size", @@ -974,8 +1394,8 @@ "readable_str": "23.44 KiB" }, { - "stage_id": 8, - "task_id": 23, + "stage_id": 6, + "task_id": 22, "accumulator_id": 162, "metric_name": "local bytes read", "metric_type": "size", @@ -987,8 +1407,8 @@ "readable_str": "40.44 KiB" }, { - "stage_id": 6, - "task_id": 22, + "stage_id": 8, + "task_id": 23, "accumulator_id": 162, "metric_name": "local bytes read", "metric_type": "size", @@ -1001,20 +1421,7 @@ }, { "stage_id": 1, - "task_id": 4, - "accumulator_id": 175, - "metric_name": "shuffle bytes written", - "metric_type": "size", - "value": 4132.0, - "value_exact": 4132, - "unit": "B", - "readable_value": 4.0, - "readable_unit": "KiB", - "readable_str": "4.04 KiB" - }, - { - "stage_id": 1, - "task_id": 10, + "task_id": 1, "accumulator_id": 175, "metric_name": "shuffle bytes written", "metric_type": "size", @@ -1027,16 +1434,16 @@ }, { "stage_id": 1, - "task_id": 5, + "task_id": 2, "accumulator_id": 175, "metric_name": "shuffle bytes written", "metric_type": "size", - "value": 4135.0, - "value_exact": 4135, + "value": 4150.0, + "value_exact": 4150, "unit": "B", - "readable_value": 4.0, + "readable_value": 4.1, "readable_unit": "KiB", - "readable_str": "4.04 KiB" + "readable_str": "4.05 KiB" }, { "stage_id": 1, @@ -1053,29 +1460,29 @@ }, { "stage_id": 1, - "task_id": 2, + "task_id": 4, "accumulator_id": 175, "metric_name": "shuffle bytes written", "metric_type": "size", - "value": 4150.0, - "value_exact": 4150, + "value": 4132.0, + "value_exact": 4132, "unit": "B", - "readable_value": 4.1, + "readable_value": 4.0, "readable_unit": "KiB", - "readable_str": "4.05 KiB" + "readable_str": "4.04 KiB" }, { "stage_id": 1, - "task_id": 1, + "task_id": 5, "accumulator_id": 175, "metric_name": "shuffle bytes written", "metric_type": "size", - "value": 4144.0, - "value_exact": 4144, + "value": 4135.0, + "value_exact": 4135, "unit": "B", "readable_value": 4.0, "readable_unit": "KiB", - "readable_str": "4.05 KiB" + "readable_str": "4.04 KiB" }, { "stage_id": 1, @@ -1130,8 +1537,21 @@ "readable_str": "4.06 KiB" }, { - "stage_id": 8, - "task_id": 23, + "stage_id": 1, + "task_id": 10, + "accumulator_id": 175, + "metric_name": "shuffle bytes written", + "metric_type": "size", + "value": 4144.0, + "value_exact": 4144, + "unit": "B", + "readable_value": 4.0, + "readable_unit": "KiB", + "readable_str": "4.05 KiB" + }, + { + "stage_id": 6, + "task_id": 22, "accumulator_id": 159, "metric_name": "local blocks read", "metric_type": "sum", @@ -1143,8 +1563,8 @@ "readable_str": "10.0" }, { - "stage_id": 6, - "task_id": 22, + "stage_id": 8, + "task_id": 23, "accumulator_id": 159, "metric_name": "local blocks read", "metric_type": "sum", @@ -1156,8 +1576,8 @@ "readable_str": "10.0" }, { - "stage_id": 8, - "task_id": 23, + "stage_id": 6, + "task_id": 22, "accumulator_id": 164, "metric_name": "records read", "metric_type": "sum", @@ -1169,8 +1589,8 @@ "readable_str": "1000.0" }, { - "stage_id": 6, - "task_id": 22, + "stage_id": 8, + "task_id": 23, "accumulator_id": 164, "metric_name": "records read", "metric_type": "sum", @@ -1183,7 +1603,7 @@ }, { "stage_id": 1, - "task_id": 9, + "task_id": 1, "accumulator_id": 176, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1196,7 +1616,7 @@ }, { "stage_id": 1, - "task_id": 1, + "task_id": 2, "accumulator_id": 176, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1209,7 +1629,7 @@ }, { "stage_id": 1, - "task_id": 10, + "task_id": 3, "accumulator_id": 176, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1222,7 +1642,7 @@ }, { "stage_id": 1, - "task_id": 8, + "task_id": 4, "accumulator_id": 176, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1235,7 +1655,7 @@ }, { "stage_id": 1, - "task_id": 7, + "task_id": 5, "accumulator_id": 176, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1261,7 +1681,7 @@ }, { "stage_id": 1, - "task_id": 5, + "task_id": 7, "accumulator_id": 176, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1274,7 +1694,7 @@ }, { "stage_id": 1, - "task_id": 4, + "task_id": 8, "accumulator_id": 176, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1287,7 +1707,7 @@ }, { "stage_id": 1, - "task_id": 3, + "task_id": 9, "accumulator_id": 176, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1300,7 +1720,7 @@ }, { "stage_id": 1, - "task_id": 2, + "task_id": 10, "accumulator_id": 176, "metric_name": "shuffle records written", "metric_type": "sum", @@ -1740,44 +2160,18 @@ "node_name": "[13] BroadcastHashJoin" }, { - "query_id": 3, - "query_start_timestamp": "2025-03-20T18:57:59", - "query_end_timestamp": "2025-03-20T18:57:59", - "query_duration_seconds": 0.87, - "whole_stage_codegen_id": null, - "node_id": 14, - "node_type": "Scan", - "child_nodes": "", - "details": "{\"node_id\":14,\"node_type\":\"Scan\",\"detail\":{\"output\":null,\"batched\":true,\"location\":{\"location_type\":\"InMemoryFileIndex\",\"location\":[\"sparkparse/data/raw/G1_1e7_1e7_100_0.parquet\"]},\"read_schema\":\"struct<>\"}}", - "query_function": "count", - "query_header": "3 - count [0.01 min]", - "accumulators": [ - { - "stage_id": 2, - "task_id": 15, - "accumulator_id": 138, - "metric_name": "scan time", - "metric_type": "timing", - "value": 12.0, - "value_exact": 12, - "unit": "ms", - "readable_value": 12.0, - "readable_unit": "ms", - "readable_str": "12.0 ms" - }, - { - "stage_id": 2, - "task_id": 18, - "accumulator_id": 138, - "metric_name": "scan time", - "metric_type": "timing", - "value": 10.0, - "value_exact": 10, - "unit": "ms", - "readable_value": 10.0, - "readable_unit": "ms", - "readable_str": "10.0 ms" - }, + "query_id": 3, + "query_start_timestamp": "2025-03-20T18:57:59", + "query_end_timestamp": "2025-03-20T18:57:59", + "query_duration_seconds": 0.87, + "whole_stage_codegen_id": null, + "node_id": 14, + "node_type": "Scan", + "child_nodes": "", + "details": "{\"node_id\":14,\"node_type\":\"Scan\",\"detail\":{\"output\":null,\"batched\":true,\"location\":{\"location_type\":\"InMemoryFileIndex\",\"location\":[\"sparkparse/data/raw/G1_1e7_1e7_100_0.parquet\"]},\"read_schema\":\"struct<>\"}}", + "query_function": "count", + "query_header": "3 - count [0.01 min]", + "accumulators": [ { "stage_id": 2, "task_id": 11, @@ -1832,29 +2226,29 @@ }, { "stage_id": 2, - "task_id": 20, + "task_id": 15, "accumulator_id": 138, "metric_name": "scan time", "metric_type": "timing", - "value": 13.0, - "value_exact": 13, + "value": 12.0, + "value_exact": 12, "unit": "ms", - "readable_value": 13.0, + "readable_value": 12.0, "readable_unit": "ms", - "readable_str": "13.0 ms" + "readable_str": "12.0 ms" }, { "stage_id": 2, - "task_id": 19, + "task_id": 16, "accumulator_id": 138, "metric_name": "scan time", "metric_type": "timing", - "value": 14.0, - "value_exact": 14, + "value": 12.0, + "value_exact": 12, "unit": "ms", - "readable_value": 14.0, + "readable_value": 12.0, "readable_unit": "ms", - "readable_str": "14.0 ms" + "readable_str": "12.0 ms" }, { "stage_id": 2, @@ -1871,20 +2265,46 @@ }, { "stage_id": 2, - "task_id": 16, + "task_id": 18, "accumulator_id": 138, "metric_name": "scan time", "metric_type": "timing", - "value": 12.0, - "value_exact": 12, + "value": 10.0, + "value_exact": 10, "unit": "ms", - "readable_value": 12.0, + "readable_value": 10.0, "readable_unit": "ms", - "readable_str": "12.0 ms" + "readable_str": "10.0 ms" }, { "stage_id": 2, - "task_id": 18, + "task_id": 19, + "accumulator_id": 138, + "metric_name": "scan time", + "metric_type": "timing", + "value": 14.0, + "value_exact": 14, + "unit": "ms", + "readable_value": 14.0, + "readable_unit": "ms", + "readable_str": "14.0 ms" + }, + { + "stage_id": 2, + "task_id": 20, + "accumulator_id": 138, + "metric_name": "scan time", + "metric_type": "timing", + "value": 13.0, + "value_exact": 13, + "unit": "ms", + "readable_value": 13.0, + "readable_unit": "ms", + "readable_str": "13.0 ms" + }, + { + "stage_id": 2, + "task_id": 11, "accumulator_id": 137, "metric_name": "number of output rows", "metric_type": "sum", @@ -1910,7 +2330,7 @@ }, { "stage_id": 2, - "task_id": 11, + "task_id": 13, "accumulator_id": 137, "metric_name": "number of output rows", "metric_type": "sum", @@ -1923,7 +2343,7 @@ }, { "stage_id": 2, - "task_id": 20, + "task_id": 14, "accumulator_id": 137, "metric_name": "number of output rows", "metric_type": "sum", @@ -1936,7 +2356,7 @@ }, { "stage_id": 2, - "task_id": 19, + "task_id": 15, "accumulator_id": 137, "metric_name": "number of output rows", "metric_type": "sum", @@ -1949,7 +2369,7 @@ }, { "stage_id": 2, - "task_id": 17, + "task_id": 16, "accumulator_id": 137, "metric_name": "number of output rows", "metric_type": "sum", @@ -1962,7 +2382,7 @@ }, { "stage_id": 2, - "task_id": 16, + "task_id": 17, "accumulator_id": 137, "metric_name": "number of output rows", "metric_type": "sum", @@ -1975,7 +2395,7 @@ }, { "stage_id": 2, - "task_id": 15, + "task_id": 18, "accumulator_id": 137, "metric_name": "number of output rows", "metric_type": "sum", @@ -1988,7 +2408,7 @@ }, { "stage_id": 2, - "task_id": 14, + "task_id": 19, "accumulator_id": 137, "metric_name": "number of output rows", "metric_type": "sum", @@ -2001,7 +2421,7 @@ }, { "stage_id": 2, - "task_id": 13, + "task_id": 20, "accumulator_id": 137, "metric_name": "number of output rows", "metric_type": "sum", @@ -2101,7 +2521,7 @@ "accumulators": [ { "stage_id": 2, - "task_id": 16, + "task_id": 11, "accumulator_id": 216, "metric_name": "number of input batches", "metric_type": "sum", @@ -2114,7 +2534,7 @@ }, { "stage_id": 2, - "task_id": 17, + "task_id": 12, "accumulator_id": 216, "metric_name": "number of input batches", "metric_type": "sum", @@ -2127,7 +2547,7 @@ }, { "stage_id": 2, - "task_id": 18, + "task_id": 13, "accumulator_id": 216, "metric_name": "number of input batches", "metric_type": "sum", @@ -2140,7 +2560,7 @@ }, { "stage_id": 2, - "task_id": 19, + "task_id": 14, "accumulator_id": 216, "metric_name": "number of input batches", "metric_type": "sum", @@ -2153,7 +2573,7 @@ }, { "stage_id": 2, - "task_id": 20, + "task_id": 15, "accumulator_id": 216, "metric_name": "number of input batches", "metric_type": "sum", @@ -2166,7 +2586,7 @@ }, { "stage_id": 2, - "task_id": 13, + "task_id": 16, "accumulator_id": 216, "metric_name": "number of input batches", "metric_type": "sum", @@ -2179,7 +2599,7 @@ }, { "stage_id": 2, - "task_id": 14, + "task_id": 17, "accumulator_id": 216, "metric_name": "number of input batches", "metric_type": "sum", @@ -2192,7 +2612,7 @@ }, { "stage_id": 2, - "task_id": 15, + "task_id": 18, "accumulator_id": 216, "metric_name": "number of input batches", "metric_type": "sum", @@ -2205,7 +2625,7 @@ }, { "stage_id": 2, - "task_id": 12, + "task_id": 19, "accumulator_id": 216, "metric_name": "number of input batches", "metric_type": "sum", @@ -2218,7 +2638,7 @@ }, { "stage_id": 2, - "task_id": 11, + "task_id": 20, "accumulator_id": 216, "metric_name": "number of input batches", "metric_type": "sum", @@ -2231,7 +2651,7 @@ }, { "stage_id": 2, - "task_id": 17, + "task_id": 11, "accumulator_id": 215, "metric_name": "number of output rows", "metric_type": "sum", @@ -2244,7 +2664,7 @@ }, { "stage_id": 2, - "task_id": 18, + "task_id": 12, "accumulator_id": 215, "metric_name": "number of output rows", "metric_type": "sum", @@ -2257,7 +2677,7 @@ }, { "stage_id": 2, - "task_id": 19, + "task_id": 13, "accumulator_id": 215, "metric_name": "number of output rows", "metric_type": "sum", @@ -2270,7 +2690,7 @@ }, { "stage_id": 2, - "task_id": 20, + "task_id": 14, "accumulator_id": 215, "metric_name": "number of output rows", "metric_type": "sum", @@ -2283,7 +2703,7 @@ }, { "stage_id": 2, - "task_id": 12, + "task_id": 15, "accumulator_id": 215, "metric_name": "number of output rows", "metric_type": "sum", @@ -2296,7 +2716,7 @@ }, { "stage_id": 2, - "task_id": 13, + "task_id": 16, "accumulator_id": 215, "metric_name": "number of output rows", "metric_type": "sum", @@ -2309,7 +2729,7 @@ }, { "stage_id": 2, - "task_id": 14, + "task_id": 17, "accumulator_id": 215, "metric_name": "number of output rows", "metric_type": "sum", @@ -2322,7 +2742,7 @@ }, { "stage_id": 2, - "task_id": 15, + "task_id": 18, "accumulator_id": 215, "metric_name": "number of output rows", "metric_type": "sum", @@ -2335,7 +2755,7 @@ }, { "stage_id": 2, - "task_id": 11, + "task_id": 19, "accumulator_id": 215, "metric_name": "number of output rows", "metric_type": "sum", @@ -2348,7 +2768,7 @@ }, { "stage_id": 2, - "task_id": 16, + "task_id": 20, "accumulator_id": 215, "metric_name": "number of output rows", "metric_type": "sum", @@ -2455,124 +2875,124 @@ }, { "stage_id": 2, - "task_id": 20, + "task_id": 12, "accumulator_id": 213, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 3.431834, + "value": 2.1829989999999997, "value_exact": null, "unit": "ms", - "readable_value": 3.4, + "readable_value": 2.2, "readable_unit": "ms", - "readable_str": "3.43 ms" + "readable_str": "2.18 ms" }, { "stage_id": 2, - "task_id": 19, + "task_id": 13, "accumulator_id": 213, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 2.5622089999999997, + "value": 0.335542, "value_exact": null, "unit": "ms", - "readable_value": 2.6, + "readable_value": 0.3, "readable_unit": "ms", - "readable_str": "2.56 ms" + "readable_str": "0.34 ms" }, { "stage_id": 2, - "task_id": 12, + "task_id": 14, "accumulator_id": 213, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 2.1829989999999997, + "value": 0.37258399999999997, "value_exact": null, "unit": "ms", - "readable_value": 2.2, + "readable_value": 0.4, "readable_unit": "ms", - "readable_str": "2.18 ms" + "readable_str": "0.37 ms" }, { "stage_id": 2, - "task_id": 13, + "task_id": 15, "accumulator_id": 213, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 0.335542, + "value": 0.316959, "value_exact": null, "unit": "ms", "readable_value": 0.3, "readable_unit": "ms", - "readable_str": "0.34 ms" + "readable_str": "0.32 ms" }, { "stage_id": 2, - "task_id": 18, + "task_id": 16, "accumulator_id": 213, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 0.521707, + "value": 0.35362699999999997, "value_exact": null, "unit": "ms", - "readable_value": 0.5, + "readable_value": 0.4, "readable_unit": "ms", - "readable_str": "0.52 ms" + "readable_str": "0.35 ms" }, { "stage_id": 2, - "task_id": 14, + "task_id": 17, "accumulator_id": 213, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 0.37258399999999997, + "value": 0.338293, "value_exact": null, "unit": "ms", - "readable_value": 0.4, + "readable_value": 0.3, "readable_unit": "ms", - "readable_str": "0.37 ms" + "readable_str": "0.34 ms" }, { "stage_id": 2, - "task_id": 15, + "task_id": 18, "accumulator_id": 213, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 0.316959, + "value": 0.521707, "value_exact": null, "unit": "ms", - "readable_value": 0.3, + "readable_value": 0.5, "readable_unit": "ms", - "readable_str": "0.32 ms" + "readable_str": "0.52 ms" }, { "stage_id": 2, - "task_id": 16, + "task_id": 19, "accumulator_id": 213, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 0.35362699999999997, + "value": 2.5622089999999997, "value_exact": null, "unit": "ms", - "readable_value": 0.4, + "readable_value": 2.6, "readable_unit": "ms", - "readable_str": "0.35 ms" + "readable_str": "2.56 ms" }, { "stage_id": 2, - "task_id": 17, + "task_id": 20, "accumulator_id": 213, "metric_name": "shuffle write time", "metric_type": "timing", - "value": 0.338293, + "value": 3.431834, "value_exact": null, "unit": "ms", - "readable_value": 0.3, + "readable_value": 3.4, "readable_unit": "ms", - "readable_str": "0.34 ms" + "readable_str": "3.43 ms" }, { "stage_id": 2, - "task_id": 16, + "task_id": 11, "accumulator_id": 192, "metric_name": "data size", "metric_type": "size", @@ -2585,7 +3005,7 @@ }, { "stage_id": 2, - "task_id": 20, + "task_id": 12, "accumulator_id": 192, "metric_name": "data size", "metric_type": "size", @@ -2598,7 +3018,7 @@ }, { "stage_id": 2, - "task_id": 19, + "task_id": 13, "accumulator_id": 192, "metric_name": "data size", "metric_type": "size", @@ -2611,7 +3031,7 @@ }, { "stage_id": 2, - "task_id": 18, + "task_id": 14, "accumulator_id": 192, "metric_name": "data size", "metric_type": "size", @@ -2624,7 +3044,7 @@ }, { "stage_id": 2, - "task_id": 17, + "task_id": 15, "accumulator_id": 192, "metric_name": "data size", "metric_type": "size", @@ -2637,7 +3057,7 @@ }, { "stage_id": 2, - "task_id": 15, + "task_id": 16, "accumulator_id": 192, "metric_name": "data size", "metric_type": "size", @@ -2650,7 +3070,7 @@ }, { "stage_id": 2, - "task_id": 11, + "task_id": 17, "accumulator_id": 192, "metric_name": "data size", "metric_type": "size", @@ -2663,7 +3083,7 @@ }, { "stage_id": 2, - "task_id": 14, + "task_id": 18, "accumulator_id": 192, "metric_name": "data size", "metric_type": "size", @@ -2676,7 +3096,7 @@ }, { "stage_id": 2, - "task_id": 13, + "task_id": 19, "accumulator_id": 192, "metric_name": "data size", "metric_type": "size", @@ -2689,7 +3109,7 @@ }, { "stage_id": 2, - "task_id": 12, + "task_id": 20, "accumulator_id": 192, "metric_name": "data size", "metric_type": "size", @@ -2715,7 +3135,7 @@ }, { "stage_id": 2, - "task_id": 14, + "task_id": 11, "accumulator_id": 211, "metric_name": "shuffle bytes written", "metric_type": "size", @@ -2728,7 +3148,7 @@ }, { "stage_id": 2, - "task_id": 13, + "task_id": 12, "accumulator_id": 211, "metric_name": "shuffle bytes written", "metric_type": "size", @@ -2741,7 +3161,7 @@ }, { "stage_id": 2, - "task_id": 12, + "task_id": 13, "accumulator_id": 211, "metric_name": "shuffle bytes written", "metric_type": "size", @@ -2754,7 +3174,7 @@ }, { "stage_id": 2, - "task_id": 11, + "task_id": 14, "accumulator_id": 211, "metric_name": "shuffle bytes written", "metric_type": "size", @@ -2767,7 +3187,7 @@ }, { "stage_id": 2, - "task_id": 20, + "task_id": 15, "accumulator_id": 211, "metric_name": "shuffle bytes written", "metric_type": "size", @@ -2780,7 +3200,7 @@ }, { "stage_id": 2, - "task_id": 19, + "task_id": 16, "accumulator_id": 211, "metric_name": "shuffle bytes written", "metric_type": "size", @@ -2793,7 +3213,7 @@ }, { "stage_id": 2, - "task_id": 18, + "task_id": 17, "accumulator_id": 211, "metric_name": "shuffle bytes written", "metric_type": "size", @@ -2806,7 +3226,7 @@ }, { "stage_id": 2, - "task_id": 15, + "task_id": 18, "accumulator_id": 211, "metric_name": "shuffle bytes written", "metric_type": "size", @@ -2819,7 +3239,7 @@ }, { "stage_id": 2, - "task_id": 17, + "task_id": 19, "accumulator_id": 211, "metric_name": "shuffle bytes written", "metric_type": "size", @@ -2832,7 +3252,7 @@ }, { "stage_id": 2, - "task_id": 16, + "task_id": 20, "accumulator_id": 211, "metric_name": "shuffle bytes written", "metric_type": "size", @@ -2871,7 +3291,7 @@ }, { "stage_id": 2, - "task_id": 14, + "task_id": 11, "accumulator_id": 212, "metric_name": "shuffle records written", "metric_type": "sum", @@ -2884,7 +3304,7 @@ }, { "stage_id": 2, - "task_id": 15, + "task_id": 12, "accumulator_id": 212, "metric_name": "shuffle records written", "metric_type": "sum", @@ -2897,7 +3317,7 @@ }, { "stage_id": 2, - "task_id": 16, + "task_id": 13, "accumulator_id": 212, "metric_name": "shuffle records written", "metric_type": "sum", @@ -2910,7 +3330,7 @@ }, { "stage_id": 2, - "task_id": 17, + "task_id": 14, "accumulator_id": 212, "metric_name": "shuffle records written", "metric_type": "sum", @@ -2923,7 +3343,7 @@ }, { "stage_id": 2, - "task_id": 18, + "task_id": 15, "accumulator_id": 212, "metric_name": "shuffle records written", "metric_type": "sum", @@ -2936,7 +3356,7 @@ }, { "stage_id": 2, - "task_id": 19, + "task_id": 16, "accumulator_id": 212, "metric_name": "shuffle records written", "metric_type": "sum", @@ -2949,7 +3369,7 @@ }, { "stage_id": 2, - "task_id": 20, + "task_id": 17, "accumulator_id": 212, "metric_name": "shuffle records written", "metric_type": "sum", @@ -2962,7 +3382,7 @@ }, { "stage_id": 2, - "task_id": 11, + "task_id": 18, "accumulator_id": 212, "metric_name": "shuffle records written", "metric_type": "sum", @@ -2975,7 +3395,7 @@ }, { "stage_id": 2, - "task_id": 12, + "task_id": 19, "accumulator_id": 212, "metric_name": "shuffle records written", "metric_type": "sum", @@ -2988,7 +3408,7 @@ }, { "stage_id": 2, - "task_id": 13, + "task_id": 20, "accumulator_id": 212, "metric_name": "shuffle records written", "metric_type": "sum", @@ -3502,7 +3922,7 @@ "accumulators": [ { "stage_id": 1, - "task_id": 10, + "task_id": 1, "accumulator_id": 178, "metric_name": "duration", "metric_type": "timing", @@ -3515,20 +3935,20 @@ }, { "stage_id": 1, - "task_id": 9, + "task_id": 2, "accumulator_id": 178, "metric_name": "duration", "metric_type": "timing", - "value": 197.0, - "value_exact": 197, + "value": 196.0, + "value_exact": 196, "unit": "ms", - "readable_value": 197.0, + "readable_value": 196.0, "readable_unit": "ms", - "readable_str": "197.0 ms" + "readable_str": "196.0 ms" }, { "stage_id": 1, - "task_id": 8, + "task_id": 3, "accumulator_id": 178, "metric_name": "duration", "metric_type": "timing", @@ -3541,72 +3961,72 @@ }, { "stage_id": 1, - "task_id": 7, + "task_id": 4, "accumulator_id": 178, "metric_name": "duration", "metric_type": "timing", - "value": 197.0, - "value_exact": 197, + "value": 196.0, + "value_exact": 196, "unit": "ms", - "readable_value": 197.0, + "readable_value": 196.0, "readable_unit": "ms", - "readable_str": "197.0 ms" + "readable_str": "196.0 ms" }, { "stage_id": 1, - "task_id": 6, + "task_id": 5, "accumulator_id": 178, "metric_name": "duration", "metric_type": "timing", - "value": 197.0, - "value_exact": 197, + "value": 196.0, + "value_exact": 196, "unit": "ms", - "readable_value": 197.0, + "readable_value": 196.0, "readable_unit": "ms", - "readable_str": "197.0 ms" + "readable_str": "196.0 ms" }, { "stage_id": 1, - "task_id": 5, + "task_id": 6, "accumulator_id": 178, "metric_name": "duration", "metric_type": "timing", - "value": 196.0, - "value_exact": 196, + "value": 197.0, + "value_exact": 197, "unit": "ms", - "readable_value": 196.0, + "readable_value": 197.0, "readable_unit": "ms", - "readable_str": "196.0 ms" + "readable_str": "197.0 ms" }, { "stage_id": 1, - "task_id": 1, + "task_id": 7, "accumulator_id": 178, "metric_name": "duration", "metric_type": "timing", - "value": 196.0, - "value_exact": 196, + "value": 197.0, + "value_exact": 197, "unit": "ms", - "readable_value": 196.0, + "readable_value": 197.0, "readable_unit": "ms", - "readable_str": "196.0 ms" + "readable_str": "197.0 ms" }, { "stage_id": 1, - "task_id": 4, + "task_id": 8, "accumulator_id": 178, "metric_name": "duration", "metric_type": "timing", - "value": 196.0, - "value_exact": 196, + "value": 197.0, + "value_exact": 197, "unit": "ms", - "readable_value": 196.0, + "readable_value": 197.0, "readable_unit": "ms", - "readable_str": "196.0 ms" + "readable_str": "197.0 ms" }, { "stage_id": 1, - "task_id": 3, + "task_id": 9, "accumulator_id": 178, "metric_name": "duration", "metric_type": "timing", @@ -3619,7 +4039,7 @@ }, { "stage_id": 1, - "task_id": 2, + "task_id": 10, "accumulator_id": 178, "metric_name": "duration", "metric_type": "timing", @@ -3667,7 +4087,20 @@ "accumulators": [ { "stage_id": 2, - "task_id": 19, + "task_id": 11, + "accumulator_id": 214, + "metric_name": "duration", + "metric_type": "timing", + "value": 9.0, + "value_exact": 9, + "unit": "ms", + "readable_value": 9.0, + "readable_unit": "ms", + "readable_str": "9.0 ms" + }, + { + "stage_id": 2, + "task_id": 12, "accumulator_id": 214, "metric_name": "duration", "metric_type": "timing", @@ -3680,20 +4113,20 @@ }, { "stage_id": 2, - "task_id": 17, + "task_id": 13, "accumulator_id": 214, "metric_name": "duration", "metric_type": "timing", - "value": 14.0, - "value_exact": 14, + "value": 10.0, + "value_exact": 10, "unit": "ms", - "readable_value": 14.0, + "readable_value": 10.0, "readable_unit": "ms", - "readable_str": "14.0 ms" + "readable_str": "10.0 ms" }, { "stage_id": 2, - "task_id": 11, + "task_id": 14, "accumulator_id": 214, "metric_name": "duration", "metric_type": "timing", @@ -3719,42 +4152,29 @@ }, { "stage_id": 2, - "task_id": 12, - "accumulator_id": 214, - "metric_name": "duration", - "metric_type": "timing", - "value": 18.0, - "value_exact": 18, - "unit": "ms", - "readable_value": 18.0, - "readable_unit": "ms", - "readable_str": "18.0 ms" - }, - { - "stage_id": 2, - "task_id": 13, + "task_id": 16, "accumulator_id": 214, "metric_name": "duration", "metric_type": "timing", - "value": 10.0, - "value_exact": 10, + "value": 12.0, + "value_exact": 12, "unit": "ms", - "readable_value": 10.0, + "readable_value": 12.0, "readable_unit": "ms", - "readable_str": "10.0 ms" + "readable_str": "12.0 ms" }, { "stage_id": 2, - "task_id": 14, + "task_id": 17, "accumulator_id": 214, "metric_name": "duration", "metric_type": "timing", - "value": 9.0, - "value_exact": 9, + "value": 14.0, + "value_exact": 14, "unit": "ms", - "readable_value": 9.0, + "readable_value": 14.0, "readable_unit": "ms", - "readable_str": "9.0 ms" + "readable_str": "14.0 ms" }, { "stage_id": 2, @@ -3771,16 +4191,16 @@ }, { "stage_id": 2, - "task_id": 16, + "task_id": 19, "accumulator_id": 214, "metric_name": "duration", "metric_type": "timing", - "value": 12.0, - "value_exact": 12, + "value": 18.0, + "value_exact": 18, "unit": "ms", - "readable_value": 12.0, + "readable_value": 18.0, "readable_unit": "ms", - "readable_str": "12.0 ms" + "readable_str": "18.0 ms" }, { "stage_id": 2, @@ -3968,26 +4388,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 1, "stage_id": 1, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.250000000", "job_end_timestamp": "2025-03-20T18:57:59.576000000", "job_duration_seconds": 0.326, "stage_start_timestamp": "2025-03-20T18:57:59.253000000", "stage_end_timestamp": "2025-03-20T18:57:59.563000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 1, "task_start_timestamp": "2025-03-20T18:57:59.274000000", "task_end_timestamp": "2025-03-20T18:57:59.557000000", "task_duration_seconds": 0.28300000000000003, "nodes": [ "[1] Scan", - "[1] Scan", - "[2] ColumnarToRow", "[2] ColumnarToRow", - "[4] Exchange", - "[4] Exchange", - "[4] Exchange", "[4] Exchange" ], "executor_run_time_seconds": 0.00022999999999999998, @@ -4028,7 +4448,11 @@ "executor_id": "driver", "host": "localhost", "index": 0, + "partition_id": 0, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -4042,26 +4466,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 1, "stage_id": 1, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.250000000", "job_end_timestamp": "2025-03-20T18:57:59.576000000", "job_duration_seconds": 0.326, "stage_start_timestamp": "2025-03-20T18:57:59.253000000", "stage_end_timestamp": "2025-03-20T18:57:59.563000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 2, "task_start_timestamp": "2025-03-20T18:57:59.276000000", "task_end_timestamp": "2025-03-20T18:57:59.561000000", "task_duration_seconds": 0.28500000000000003, "nodes": [ - "[1] Scan", "[1] Scan", "[2] ColumnarToRow", - "[2] ColumnarToRow", - "[4] Exchange", - "[4] Exchange", - "[4] Exchange", "[4] Exchange" ], "executor_run_time_seconds": 0.00022999999999999998, @@ -4102,7 +4526,11 @@ "executor_id": "driver", "host": "localhost", "index": 1, + "partition_id": 1, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -4116,26 +4544,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 1, "stage_id": 1, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.250000000", "job_end_timestamp": "2025-03-20T18:57:59.576000000", "job_duration_seconds": 0.326, "stage_start_timestamp": "2025-03-20T18:57:59.253000000", "stage_end_timestamp": "2025-03-20T18:57:59.563000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 3, "task_start_timestamp": "2025-03-20T18:57:59.276000000", "task_end_timestamp": "2025-03-20T18:57:59.558000000", "task_duration_seconds": 0.28200000000000003, "nodes": [ "[1] Scan", - "[1] Scan", - "[2] ColumnarToRow", "[2] ColumnarToRow", - "[4] Exchange", - "[4] Exchange", - "[4] Exchange", "[4] Exchange" ], "executor_run_time_seconds": 0.00022999999999999998, @@ -4176,7 +4604,11 @@ "executor_id": "driver", "host": "localhost", "index": 2, + "partition_id": 2, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -4190,26 +4622,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 1, "stage_id": 1, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.250000000", "job_end_timestamp": "2025-03-20T18:57:59.576000000", "job_duration_seconds": 0.326, "stage_start_timestamp": "2025-03-20T18:57:59.253000000", "stage_end_timestamp": "2025-03-20T18:57:59.563000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 4, "task_start_timestamp": "2025-03-20T18:57:59.277000000", "task_end_timestamp": "2025-03-20T18:57:59.562000000", "task_duration_seconds": 0.28500000000000003, "nodes": [ - "[1] Scan", "[1] Scan", "[2] ColumnarToRow", - "[2] ColumnarToRow", - "[4] Exchange", - "[4] Exchange", - "[4] Exchange", "[4] Exchange" ], "executor_run_time_seconds": 0.00022999999999999998, @@ -4250,7 +4682,11 @@ "executor_id": "driver", "host": "localhost", "index": 3, + "partition_id": 3, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -4264,26 +4700,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 1, "stage_id": 1, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.250000000", "job_end_timestamp": "2025-03-20T18:57:59.576000000", "job_duration_seconds": 0.326, "stage_start_timestamp": "2025-03-20T18:57:59.253000000", "stage_end_timestamp": "2025-03-20T18:57:59.563000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 5, "task_start_timestamp": "2025-03-20T18:57:59.277000000", "task_end_timestamp": "2025-03-20T18:57:59.553000000", "task_duration_seconds": 0.276, "nodes": [ "[1] Scan", - "[1] Scan", - "[2] ColumnarToRow", "[2] ColumnarToRow", - "[4] Exchange", - "[4] Exchange", - "[4] Exchange", "[4] Exchange" ], "executor_run_time_seconds": 0.00023099999999999998, @@ -4324,7 +4760,11 @@ "executor_id": "driver", "host": "localhost", "index": 4, + "partition_id": 4, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -4338,26 +4778,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 1, "stage_id": 1, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.250000000", "job_end_timestamp": "2025-03-20T18:57:59.576000000", "job_duration_seconds": 0.326, "stage_start_timestamp": "2025-03-20T18:57:59.253000000", "stage_end_timestamp": "2025-03-20T18:57:59.563000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 6, "task_start_timestamp": "2025-03-20T18:57:59.277000000", "task_end_timestamp": "2025-03-20T18:57:59.558000000", "task_duration_seconds": 0.281, "nodes": [ - "[1] Scan", "[1] Scan", "[2] ColumnarToRow", - "[2] ColumnarToRow", - "[4] Exchange", - "[4] Exchange", - "[4] Exchange", "[4] Exchange" ], "executor_run_time_seconds": 0.00022999999999999998, @@ -4398,7 +4838,11 @@ "executor_id": "driver", "host": "localhost", "index": 5, + "partition_id": 5, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -4412,26 +4856,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 1, "stage_id": 1, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.250000000", "job_end_timestamp": "2025-03-20T18:57:59.576000000", "job_duration_seconds": 0.326, "stage_start_timestamp": "2025-03-20T18:57:59.253000000", "stage_end_timestamp": "2025-03-20T18:57:59.563000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 7, "task_start_timestamp": "2025-03-20T18:57:59.277000000", "task_end_timestamp": "2025-03-20T18:57:59.557000000", "task_duration_seconds": 0.28, "nodes": [ "[1] Scan", - "[1] Scan", - "[2] ColumnarToRow", "[2] ColumnarToRow", - "[4] Exchange", - "[4] Exchange", - "[4] Exchange", "[4] Exchange" ], "executor_run_time_seconds": 0.000233, @@ -4472,7 +4916,11 @@ "executor_id": "driver", "host": "localhost", "index": 6, + "partition_id": 6, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -4486,26 +4934,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 1, "stage_id": 1, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.250000000", "job_end_timestamp": "2025-03-20T18:57:59.576000000", "job_duration_seconds": 0.326, "stage_start_timestamp": "2025-03-20T18:57:59.253000000", "stage_end_timestamp": "2025-03-20T18:57:59.563000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 8, "task_start_timestamp": "2025-03-20T18:57:59.278000000", "task_end_timestamp": "2025-03-20T18:57:59.554000000", "task_duration_seconds": 0.276, "nodes": [ - "[1] Scan", "[1] Scan", "[2] ColumnarToRow", - "[2] ColumnarToRow", - "[4] Exchange", - "[4] Exchange", - "[4] Exchange", "[4] Exchange" ], "executor_run_time_seconds": 0.00022899999999999998, @@ -4546,7 +4994,11 @@ "executor_id": "driver", "host": "localhost", "index": 7, + "partition_id": 7, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -4560,26 +5012,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 1, "stage_id": 1, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.250000000", "job_end_timestamp": "2025-03-20T18:57:59.576000000", "job_duration_seconds": 0.326, "stage_start_timestamp": "2025-03-20T18:57:59.253000000", "stage_end_timestamp": "2025-03-20T18:57:59.563000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 9, "task_start_timestamp": "2025-03-20T18:57:59.278000000", "task_end_timestamp": "2025-03-20T18:57:59.558000000", "task_duration_seconds": 0.28, "nodes": [ "[1] Scan", - "[1] Scan", - "[2] ColumnarToRow", "[2] ColumnarToRow", - "[4] Exchange", - "[4] Exchange", - "[4] Exchange", "[4] Exchange" ], "executor_run_time_seconds": 0.00022899999999999998, @@ -4620,7 +5072,11 @@ "executor_id": "driver", "host": "localhost", "index": 8, + "partition_id": 8, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -4634,26 +5090,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 1, "stage_id": 1, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.250000000", "job_end_timestamp": "2025-03-20T18:57:59.576000000", "job_duration_seconds": 0.326, "stage_start_timestamp": "2025-03-20T18:57:59.253000000", "stage_end_timestamp": "2025-03-20T18:57:59.563000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 10, "task_start_timestamp": "2025-03-20T18:57:59.278000000", "task_end_timestamp": "2025-03-20T18:57:59.552000000", "task_duration_seconds": 0.274, "nodes": [ - "[1] Scan", "[1] Scan", "[2] ColumnarToRow", - "[2] ColumnarToRow", - "[4] Exchange", - "[4] Exchange", - "[4] Exchange", "[4] Exchange" ], "executor_run_time_seconds": 0.00022899999999999998, @@ -4694,7 +5150,11 @@ "executor_id": "driver", "host": "localhost", "index": 9, + "partition_id": 9, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -4708,26 +5168,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.275000000", "job_end_timestamp": "2025-03-20T18:57:59.586000000", "job_duration_seconds": 0.311, "stage_start_timestamp": "2025-03-20T18:57:59.276000000", "stage_end_timestamp": "2025-03-20T18:57:59.586000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 11, "task_start_timestamp": "2025-03-20T18:57:59.549000000", "task_end_timestamp": "2025-03-20T18:57:59.581000000", "task_duration_seconds": 0.032, "nodes": [ "[14] Scan", - "[14] Scan", - "[15] ColumnarToRow", "[15] ColumnarToRow", - "[17] Exchange", - "[17] Exchange", - "[17] Exchange", "[17] Exchange" ], "executor_run_time_seconds": 2.6e-05, @@ -4768,7 +5228,11 @@ "executor_id": "driver", "host": "localhost", "index": 0, + "partition_id": 0, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -4782,26 +5246,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.275000000", "job_end_timestamp": "2025-03-20T18:57:59.586000000", "job_duration_seconds": 0.311, "stage_start_timestamp": "2025-03-20T18:57:59.276000000", "stage_end_timestamp": "2025-03-20T18:57:59.586000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 12, "task_start_timestamp": "2025-03-20T18:57:59.552000000", "task_end_timestamp": "2025-03-20T18:57:59.585000000", "task_duration_seconds": 0.033, "nodes": [ "[14] Scan", - "[14] Scan", - "[15] ColumnarToRow", "[15] ColumnarToRow", - "[17] Exchange", - "[17] Exchange", - "[17] Exchange", "[17] Exchange" ], "executor_run_time_seconds": 2.8e-05, @@ -4842,7 +5306,11 @@ "executor_id": "driver", "host": "localhost", "index": 1, + "partition_id": 1, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -4856,26 +5324,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.275000000", "job_end_timestamp": "2025-03-20T18:57:59.586000000", "job_duration_seconds": 0.311, "stage_start_timestamp": "2025-03-20T18:57:59.276000000", "stage_end_timestamp": "2025-03-20T18:57:59.586000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 13, "task_start_timestamp": "2025-03-20T18:57:59.553000000", "task_end_timestamp": "2025-03-20T18:57:59.581000000", "task_duration_seconds": 0.028, "nodes": [ "[14] Scan", - "[14] Scan", - "[15] ColumnarToRow", "[15] ColumnarToRow", - "[17] Exchange", - "[17] Exchange", - "[17] Exchange", "[17] Exchange" ], "executor_run_time_seconds": 2.3e-05, @@ -4916,7 +5384,11 @@ "executor_id": "driver", "host": "localhost", "index": 2, + "partition_id": 2, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -4930,26 +5402,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.275000000", "job_end_timestamp": "2025-03-20T18:57:59.586000000", "job_duration_seconds": 0.311, "stage_start_timestamp": "2025-03-20T18:57:59.276000000", "stage_end_timestamp": "2025-03-20T18:57:59.586000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 14, "task_start_timestamp": "2025-03-20T18:57:59.554000000", "task_end_timestamp": "2025-03-20T18:57:59.581000000", "task_duration_seconds": 0.027, "nodes": [ "[14] Scan", - "[14] Scan", - "[15] ColumnarToRow", "[15] ColumnarToRow", - "[17] Exchange", - "[17] Exchange", - "[17] Exchange", "[17] Exchange" ], "executor_run_time_seconds": 2.1e-05, @@ -4990,7 +5462,11 @@ "executor_id": "driver", "host": "localhost", "index": 3, + "partition_id": 3, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -5004,26 +5480,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.275000000", "job_end_timestamp": "2025-03-20T18:57:59.586000000", "job_duration_seconds": 0.311, "stage_start_timestamp": "2025-03-20T18:57:59.276000000", "stage_end_timestamp": "2025-03-20T18:57:59.586000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 15, "task_start_timestamp": "2025-03-20T18:57:59.554000000", "task_end_timestamp": "2025-03-20T18:57:59.579000000", "task_duration_seconds": 0.025, "nodes": [ "[14] Scan", - "[14] Scan", - "[15] ColumnarToRow", "[15] ColumnarToRow", - "[17] Exchange", - "[17] Exchange", - "[17] Exchange", "[17] Exchange" ], "executor_run_time_seconds": 1.8999999999999998e-05, @@ -5064,7 +5540,11 @@ "executor_id": "driver", "host": "localhost", "index": 4, + "partition_id": 4, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -5078,26 +5558,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.275000000", "job_end_timestamp": "2025-03-20T18:57:59.586000000", "job_duration_seconds": 0.311, "stage_start_timestamp": "2025-03-20T18:57:59.276000000", "stage_end_timestamp": "2025-03-20T18:57:59.586000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 16, "task_start_timestamp": "2025-03-20T18:57:59.555000000", "task_end_timestamp": "2025-03-20T18:57:59.578000000", "task_duration_seconds": 0.023, "nodes": [ "[14] Scan", - "[14] Scan", - "[15] ColumnarToRow", "[15] ColumnarToRow", - "[17] Exchange", - "[17] Exchange", - "[17] Exchange", "[17] Exchange" ], "executor_run_time_seconds": 1.6e-05, @@ -5138,7 +5618,11 @@ "executor_id": "driver", "host": "localhost", "index": 5, + "partition_id": 5, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -5152,26 +5636,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.275000000", "job_end_timestamp": "2025-03-20T18:57:59.586000000", "job_duration_seconds": 0.311, "stage_start_timestamp": "2025-03-20T18:57:59.276000000", "stage_end_timestamp": "2025-03-20T18:57:59.586000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 17, "task_start_timestamp": "2025-03-20T18:57:59.556000000", "task_end_timestamp": "2025-03-20T18:57:59.580000000", "task_duration_seconds": 0.024, "nodes": [ "[14] Scan", - "[14] Scan", - "[15] ColumnarToRow", "[15] ColumnarToRow", - "[17] Exchange", - "[17] Exchange", - "[17] Exchange", "[17] Exchange" ], "executor_run_time_seconds": 1.6e-05, @@ -5212,7 +5696,11 @@ "executor_id": "driver", "host": "localhost", "index": 6, + "partition_id": 6, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -5226,26 +5714,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.275000000", "job_end_timestamp": "2025-03-20T18:57:59.586000000", "job_duration_seconds": 0.311, "stage_start_timestamp": "2025-03-20T18:57:59.276000000", "stage_end_timestamp": "2025-03-20T18:57:59.586000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 18, "task_start_timestamp": "2025-03-20T18:57:59.557000000", "task_end_timestamp": "2025-03-20T18:57:59.577000000", "task_duration_seconds": 0.02, "nodes": [ "[14] Scan", - "[14] Scan", - "[15] ColumnarToRow", "[15] ColumnarToRow", - "[17] Exchange", - "[17] Exchange", - "[17] Exchange", "[17] Exchange" ], "executor_run_time_seconds": 1.4999999999999999e-05, @@ -5286,7 +5774,11 @@ "executor_id": "driver", "host": "localhost", "index": 7, + "partition_id": 7, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -5300,26 +5792,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.275000000", "job_end_timestamp": "2025-03-20T18:57:59.586000000", "job_duration_seconds": 0.311, "stage_start_timestamp": "2025-03-20T18:57:59.276000000", "stage_end_timestamp": "2025-03-20T18:57:59.586000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 19, "task_start_timestamp": "2025-03-20T18:57:59.557000000", "task_end_timestamp": "2025-03-20T18:57:59.585000000", "task_duration_seconds": 0.028, "nodes": [ "[14] Scan", - "[14] Scan", - "[15] ColumnarToRow", "[15] ColumnarToRow", - "[17] Exchange", - "[17] Exchange", - "[17] Exchange", "[17] Exchange" ], "executor_run_time_seconds": 2.3e-05, @@ -5360,7 +5852,11 @@ "executor_id": "driver", "host": "localhost", "index": 8, + "partition_id": 8, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -5374,26 +5870,26 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 2, "stage_id": 2, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.275000000", "job_end_timestamp": "2025-03-20T18:57:59.586000000", "job_duration_seconds": 0.311, "stage_start_timestamp": "2025-03-20T18:57:59.276000000", "stage_end_timestamp": "2025-03-20T18:57:59.586000000", "stage_duration_seconds": 0.31, + "stage_num_tasks": 10, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 20, "task_start_timestamp": "2025-03-20T18:57:59.560000000", "task_end_timestamp": "2025-03-20T18:57:59.585000000", "task_duration_seconds": 0.025, "nodes": [ "[14] Scan", - "[14] Scan", - "[15] ColumnarToRow", "[15] ColumnarToRow", - "[17] Exchange", - "[17] Exchange", - "[17] Exchange", "[17] Exchange" ], "executor_run_time_seconds": 2.1e-05, @@ -5434,7 +5930,11 @@ "executor_id": "driver", "host": "localhost", "index": 9, + "partition_id": 9, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -5448,22 +5948,24 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 3, "stage_id": 4, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.623000000", "job_end_timestamp": "2025-03-20T18:57:59.675000000", "job_duration_seconds": 0.052000000000000005, "stage_start_timestamp": "2025-03-20T18:57:59.626000000", "stage_end_timestamp": "2025-03-20T18:57:59.675000000", "stage_duration_seconds": 0.049, + "stage_num_tasks": 1, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 21, "task_start_timestamp": "2025-03-20T18:57:59.631000000", "task_end_timestamp": "2025-03-20T18:57:59.674000000", "task_duration_seconds": 0.043000000000000003, "nodes": [ - "[17] Exchange", - "[17] Exchange", - "[17] Exchange", "[17] Exchange" ], "executor_run_time_seconds": 3.2999999999999996e-05, @@ -5504,7 +6006,11 @@ "executor_id": "driver", "host": "localhost", "index": 0, + "partition_id": 0, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -5518,22 +6024,24 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 4, "stage_id": 6, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.631000000", "job_end_timestamp": "2025-03-20T18:57:59.676000000", "job_duration_seconds": 0.045, "stage_start_timestamp": "2025-03-20T18:57:59.632000000", "stage_end_timestamp": "2025-03-20T18:57:59.675000000", "stage_duration_seconds": 0.043000000000000003, + "stage_num_tasks": 1, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 22, "task_start_timestamp": "2025-03-20T18:57:59.635000000", "task_end_timestamp": "2025-03-20T18:57:59.675000000", "task_duration_seconds": 0.04, "nodes": [ - "[4] Exchange", - "[4] Exchange", - "[4] Exchange", "[4] Exchange", "[10] Filter" ], @@ -5575,7 +6083,11 @@ "executor_id": "driver", "host": "localhost", "index": 0, + "partition_id": 0, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, @@ -5589,28 +6101,28 @@ "query_start_timestamp": "2025-03-20T18:57:59", "query_end_timestamp": "2025-03-20T18:57:59", "query_duration_seconds": 0.87, + "query_count": 1, "job_id": 5, "stage_id": 8, + "stage_attempt_id": 0, "job_start_timestamp": "2025-03-20T18:57:59.754000000", "job_end_timestamp": "2025-03-20T18:57:59.909000000", "job_duration_seconds": 0.155, "stage_start_timestamp": "2025-03-20T18:57:59.755000000", "stage_end_timestamp": "2025-03-20T18:57:59.909000000", "stage_duration_seconds": 0.154, + "stage_num_tasks": 1, + "stage_status": "succeeded", + "stage_failure_reason": null, "task_id": 23, "task_start_timestamp": "2025-03-20T18:57:59.758000000", "task_end_timestamp": "2025-03-20T18:57:59.909000000", "task_duration_seconds": 0.151, "nodes": [ - "[4] Exchange", - "[4] Exchange", - "[4] Exchange", "[4] Exchange", "[13] BroadcastHashJoin", "[22] BroadcastNestedLoopJoin", "[24] HashAggregate", - "[24] HashAggregate", - "[25] HashAggregate", "[25] HashAggregate" ], "executor_run_time_seconds": 0.000147, @@ -5651,11 +6163,89 @@ "executor_id": "driver", "host": "localhost", "index": 0, + "partition_id": 0, "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, "failed": false, "killed": false, "speculative": false, "task_loc": "NODE_LOCAL", "task_type": "ResultTask" + }, + { + "log_name": "nested_loop_join", + "query_id": null, + "query_function": null, + "query_start_timestamp": null, + "query_end_timestamp": null, + "query_duration_seconds": null, + "query_count": 0, + "job_id": 0, + "stage_id": 0, + "stage_attempt_id": 0, + "job_start_timestamp": "2025-03-20T18:57:57.857000000", + "job_end_timestamp": "2025-03-20T18:57:58.179000000", + "job_duration_seconds": 0.322, + "stage_start_timestamp": "2025-03-20T18:57:57.868000000", + "stage_end_timestamp": "2025-03-20T18:57:58.177000000", + "stage_duration_seconds": 0.309, + "stage_num_tasks": 1, + "stage_status": "succeeded", + "stage_failure_reason": null, + "task_id": 0, + "task_start_timestamp": "2025-03-20T18:57:57.961000000", + "task_end_timestamp": "2025-03-20T18:57:58.172000000", + "task_duration_seconds": 0.211, + "nodes": [], + "executor_run_time_seconds": 0.000139, + "executor_cpu_time_seconds": 0.021457, + "executor_deserialize_time_seconds": 3.9e-05, + "executor_deserialize_cpu_time_seconds": 0.033703000000000004, + "result_size_bytes": 1942, + "jvm_gc_time_seconds": 0.0, + "result_serialization_time_seconds": 2e-06, + "memory_bytes_spilled": 0, + "disk_bytes_spilled": 0, + "peak_execution_memory_bytes": 0, + "bytes_read": 0, + "records_read": 0, + "bytes_written": 0, + "records_written": 0, + "shuffle_remote_blocks_fetched": 0, + "shuffle_local_blocks_fetched": 0, + "shuffle_fetch_wait_time_seconds": 0.0, + "shuffle_remote_bytes_read": 0, + "shuffle_remote_bytes_read_to_disk": 0, + "shuffle_local_bytes_read": 0, + "shuffle_records_read": 0, + "shuffle_remote_requests_duration": 0, + "shuffle_bytes_read": 0, + "shuffle_bytes_written": 0, + "shuffle_write_time_seconds": 0.0, + "shuffle_records_written": 0, + "merged_corrupt_block_chunks": 0, + "merged_fetch_fallback_count": 0, + "merged_remote_blocks_fetched": 0, + "merged_local_blocks_fetched": 0, + "merged_remote_chunks_fetched": 0, + "merged_local_chunks_fetched": 0, + "merged_remote_bytes_read": 0, + "merged_local_bytes_read": 0, + "merged_remote_requests_duration": 0, + "executor_id": "driver", + "host": "localhost", + "index": 0, + "partition_id": 0, + "attempt": 0, + "task_status": "success", + "task_succeeded": true, + "task_failure_reason": null, + "failed": false, + "killed": false, + "speculative": false, + "task_loc": "PROCESS_LOCAL", + "task_type": "ResultTask" } ] \ No newline at end of file diff --git a/tests/eventlog_fixtures.py b/tests/eventlog_fixtures.py new file mode 100644 index 0000000..58eb7bb --- /dev/null +++ b/tests/eventlog_fixtures.py @@ -0,0 +1,168 @@ +"""Builders for adversarial event logs. + +Real Spark cannot be asked to produce a truncated file, a corrupt line or a +stage retry on demand, so these helpers derive such logs from the recorded +fixtures by slicing and mutating their events. +""" + +from __future__ import annotations + +import copy +import json +from collections.abc import Callable, Iterable +from pathlib import Path + +FULL_LOGS = Path(__file__).parent / "data" / "full_logs" + + +def log_lines(name: str = "nested_loop_join") -> list[str]: + return FULL_LOGS.joinpath(name).read_text().splitlines() + + +def events(name: str = "nested_loop_join") -> list[dict]: + return [json.loads(line) for line in log_lines(name)] + + +def write_log( + path: Path, lines: Iterable[str | dict], newline_at_end: bool = True +) -> Path: + rendered = [line if isinstance(line, str) else json.dumps(line) for line in lines] + text = "\n".join(rendered) + if newline_at_end: + text += "\n" + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(text) + return path + + +def drop_events(lines: list[str], predicate: Callable[[dict], bool]) -> list[str]: + """Return ``lines`` without the events matching ``predicate``.""" + kept = [] + for line in lines: + event = json.loads(line) + if predicate(event): + continue + kept.append(line) + return kept + + +def map_events(lines: list[str], fn: Callable[[dict], dict | None]) -> list[str]: + out = [] + for line in lines: + result = fn(json.loads(line)) + if result is not None: + out.append(json.dumps(result)) + return out + + +def is_adaptive_update(event: dict) -> bool: + return event["Event"].endswith("SparkListenerSQLAdaptiveExecutionUpdate") + + +def stage_attempt(event: dict, attempt: int) -> dict: + """Return a copy of a stage event re-stamped as a later attempt.""" + out = copy.deepcopy(event) + out["Stage Info"]["Stage Attempt ID"] = attempt + return out + + +def task_attempt( + event: dict, + *, + task_id: int, + stage_attempt_id: int | None = None, + attempt: int | None = None, + failed: bool = False, + killed: bool = False, + speculative: bool = False, + drop_metrics: bool = False, +) -> dict: + """Return a copy of a TaskEnd event describing another attempt.""" + out = copy.deepcopy(event) + out["Task Info"]["Task ID"] = task_id + if stage_attempt_id is not None: + out["Stage Attempt ID"] = stage_attempt_id + if attempt is not None: + out["Task Info"]["Attempt"] = attempt + out["Task Info"]["Failed"] = failed + out["Task Info"]["Killed"] = killed + out["Task Info"]["Speculative"] = speculative + if failed: + out["Task End Reason"] = { + "Reason": "ExceptionFailure", + "Class Name": "java.lang.OutOfMemoryError", + "Description": "Java heap space", + } + if killed: + out["Task End Reason"] = { + "Reason": "TaskKilled", + "Kill Reason": "another attempt succeeded", + } + if drop_metrics: + out.pop("Task Metrics", None) + return out + + +def rolling_log_dir( + root: Path, + lines: list[str], + app_id: str = "app-rolled-0001", + segments: int = 2, + in_progress: bool = False, +) -> Path: + """Write ``lines`` split across a Spark rolling-log directory layout.""" + log_dir = root / f"eventlog_v2_{app_id}" + log_dir.mkdir(parents=True, exist_ok=True) + status = log_dir / f"appstatus_{app_id}" + status.write_text("") + if in_progress: + status.rename(log_dir / f"appstatus_{app_id}.inprogress") + + chunk = max(1, len(lines) // segments + 1) + for index in range(segments): + piece = lines[index * chunk : (index + 1) * chunk] + write_log(log_dir / f"events_{index + 1}_{app_id}", piece) + return log_dir + + +def scale_task_volume(lines: list[str], copies: int) -> list[str]: + """Return the log with each TaskEnd repeated ``copies`` extra times. + + Each copy is a distinct attempt of the same partition — unique task id, + higher attempt number, later finish time — so the log stays a valid + sequence of events while the volume the parser must retain grows. + """ + if copies <= 0: + return list(lines) + + parsed = [json.loads(line) for line in lines] + next_id = ( + max( + ( + event["Task Info"]["Task ID"] + for event in parsed + if event["Event"] == "SparkListenerTaskEnd" + ), + default=0, + ) + + 1 + ) + + out: list[str] = [] + for event in parsed: + out.append(json.dumps(event)) + if event["Event"] != "SparkListenerTaskEnd": + continue + for copy_index in range(copies): + duplicate = task_attempt( + event, + task_id=next_id, + attempt=event["Task Info"]["Attempt"] + copy_index + 1, + speculative=True, + ) + duplicate["Task Info"]["Finish Time"] = ( + event["Task Info"]["Finish Time"] + copy_index + 1 + ) + out.append(json.dumps(duplicate)) + next_id += 1 + return out diff --git a/tests/test_capture_contract.py b/tests/test_capture_contract.py index 59189a6..9a39de9 100644 --- a/tests/test_capture_contract.py +++ b/tests/test_capture_contract.py @@ -176,7 +176,9 @@ def test_selects_application_log_not_last_filename(tmp_path): cap = SparkparseCapture("get", spark=_BorrowedSpark(tmp_path)) cap._log_dir = str(tmp_path) cap._metadata = cap._metadata.model_copy(update={"source_application_id": "app-1"}) - assert cap._select_log() == "app-1" + # The selection is a resolvable source path, so rolled and compressed + # logs (a directory, or a name with a codec suffix) can be selected too. + assert cap._select_log() == str(tmp_path / "app-1") cap._metadata = cap._metadata.model_copy( update={"source_application_id": "missing"} ) diff --git a/tests/test_eventlog.py b/tests/test_eventlog.py new file mode 100644 index 0000000..d75b9f3 --- /dev/null +++ b/tests/test_eventlog.py @@ -0,0 +1,357 @@ +"""Increment 3 of brief 04: source discovery, rolled segments, streaming. + +The streaming contract is checked directly — the reader must never call +``read()``/``readlines()`` on a log file — because a memory assertion alone +cannot distinguish a whole-file copy from ordinary retained state. +""" + +from __future__ import annotations + +import builtins +import gzip +import importlib.util +import json +import subprocess +import sys +import time +from pathlib import Path + +import pytest + +from sparkparse.eventlog import ( + EventLogNotFoundError, + UnsupportedCodecError, + discover_sources, + iter_lines, + resolve_source, +) +from sparkparse.models import EventLogCodec +from sparkparse.parse import get_parsed_metrics, parse_source +from tests.eventlog_fixtures import ( + log_lines, + rolling_log_dir, + scale_task_volume, + write_log, +) + +HAS_ZSTD = importlib.util.find_spec("zstandard") is not None + + +def _dag_records(dfs) -> list[dict]: + return dfs.dag.sort("query_id", "node_id").to_dicts() + + +# --------------------------------------------------------------------------- +# discovery +# --------------------------------------------------------------------------- + + +def test_marker_files_and_checksums_are_not_event_logs(tmp_path): + write_log(tmp_path / "app-0001", log_lines()) + (tmp_path / ".DS_Store").write_text("junk") + (tmp_path / "_SUCCESS").write_text("") + (tmp_path / ".app-0001.crc").write_text("") + (tmp_path / "notes.tmp").write_text("") + + sources = discover_sources(tmp_path) + + assert [source.name for source in sources] == ["app-0001"] + + +def test_rolling_log_directory_is_one_source_with_ordered_segments(tmp_path): + rolling_log_dir(tmp_path, log_lines(), app_id="app-rolled", segments=3) + + sources = discover_sources(tmp_path) + + assert len(sources) == 1 + source = sources[0] + assert source.rolling is True + assert source.complete is True + assert [segment.index for segment in source.segments] == [1, 2, 3] + assert source.application_id == "app-rolled" + + +def test_in_progress_logs_are_marked_incomplete(tmp_path): + write_log(tmp_path / "app-live.inprogress", log_lines()) + rolling_log_dir(tmp_path, log_lines(), app_id="app-live-rolled", in_progress=True) + + sources = {source.name: source for source in discover_sources(tmp_path)} + + assert sources["app-live"].complete is False + assert sources["app-live"].segments[0].in_progress is True + assert sources["eventlog_v2_app-live-rolled"].complete is False + + +def test_codec_suffix_is_recognized_without_being_read(tmp_path): + (tmp_path / "app-0001.lz4").write_bytes(b"\x04\x22\x4d\x18binary") + + source = discover_sources(tmp_path)[0] + + assert source.application_id == "app-0001" + assert source.segments[0].codec is EventLogCodec.lz4 + + +def test_newest_source_wins_by_modification_time_not_name(tmp_path): + write_log(tmp_path / "zzz-old", log_lines()) + time.sleep(0.01) + write_log(tmp_path / "aaa-new", log_lines()) + + assert resolve_source(tmp_path).name == "aaa-new" + + +def test_explicit_selection_accepts_name_app_id_and_path(tmp_path): + write_log(tmp_path / "app-a", log_lines()) + write_log(tmp_path / "app-b", log_lines()) + + for selector in ("app-a", str(tmp_path / "app-a")): + assert resolve_source(tmp_path, selector).name == "app-a" + + +def test_explicit_selection_of_a_rolling_directory(tmp_path): + rolling_log_dir(tmp_path, log_lines(), app_id="app-rolled") + + source = resolve_source(tmp_path, "eventlog_v2_app-rolled") + + assert source.rolling is True + assert len(source.segments) == 2 + + +def test_empty_directory_raises_a_typed_error(tmp_path): + with pytest.raises(EventLogNotFoundError): + resolve_source(tmp_path) + + +# --------------------------------------------------------------------------- +# reading +# --------------------------------------------------------------------------- + + +def test_rolled_segments_parse_identically_to_one_file(tmp_path): + lines = log_lines() + write_log(tmp_path / "single" / "app-single", lines) + rolling_log_dir(tmp_path / "rolled", lines, app_id="app-single", segments=4) + + single = get_parsed_metrics( + log_dir=tmp_path / "single", out_dir=None, out_format=None + ) + rolled = get_parsed_metrics( + log_dir=tmp_path / "rolled", out_dir=None, out_format=None + ) + + assert _dag_records(rolled) == _dag_records(single) + assert rolled.combined.drop("log_name", "parsed_log_name").equals( + single.combined.drop("log_name", "parsed_log_name") + ) + + +def test_segment_boundaries_do_not_split_events(tmp_path): + """Each segment holds whole lines; the reader must not glue them wrongly.""" + lines = log_lines() + rolling_log_dir(tmp_path, lines, app_id="app-rolled", segments=5) + + source = resolve_source(tmp_path) + read_back = [line.rstrip("\n") for _, _, line in iter_lines(source)] + + assert read_back == lines + assert all(json.loads(line) for line in read_back) + + +def test_gzip_compressed_log_reads_like_an_uncompressed_one(tmp_path): + lines = log_lines() + write_log(tmp_path / "plain" / "app-0001", lines) + (tmp_path / "gz").mkdir() + with gzip.open(tmp_path / "gz" / "app-0001.gz", "wt") as handle: + handle.write("\n".join(lines) + "\n") + + plain = get_parsed_metrics( + log_dir=tmp_path / "plain", out_dir=None, out_format=None + ) + compressed = get_parsed_metrics( + log_dir=tmp_path / "gz", out_dir=None, out_format=None + ) + + assert _dag_records(compressed) == _dag_records(plain) + + +@pytest.mark.skipif(not HAS_ZSTD, reason="zstandard is not installed") +def test_zstd_compressed_log_reads_like_an_uncompressed_one(tmp_path): + import zstandard + + lines = log_lines() + write_log(tmp_path / "plain" / "app-0001", lines) + (tmp_path / "zstd").mkdir() + payload = ("\n".join(lines) + "\n").encode() + (tmp_path / "zstd" / "app-0001.zstd").write_bytes( + zstandard.ZstdCompressor().compress(payload) + ) + + plain = get_parsed_metrics( + log_dir=tmp_path / "plain", out_dir=None, out_format=None + ) + compressed = get_parsed_metrics( + log_dir=tmp_path / "zstd", out_dir=None, out_format=None + ) + + assert _dag_records(compressed) == _dag_records(plain) + + +@pytest.mark.parametrize("codec", ["lz4", "snappy", "lzf"]) +def test_java_framed_codecs_fail_by_name_not_as_corrupt_json(tmp_path, codec): + (tmp_path / f"app-0001.{codec}").write_bytes(b"not really compressed") + + with pytest.raises(UnsupportedCodecError, match=codec): + get_parsed_metrics(log_dir=tmp_path, out_dir=None, out_format=None) + + +def test_reading_never_copies_the_whole_file(tmp_path, monkeypatch): + """The parser must stream: no read()/readlines() on the log handle.""" + path = write_log(tmp_path / "app-0001", log_lines()) + real_open = builtins.open + whole_file_reads: list[str] = [] + + class _WatchedFile: + def __init__(self, handle, name): + self._handle = handle + self._name = name + + def read(self, *args, **kwargs): + whole_file_reads.append(f"read:{self._name}") + return self._handle.read(*args, **kwargs) + + def readlines(self, *args, **kwargs): + whole_file_reads.append(f"readlines:{self._name}") + return self._handle.readlines(*args, **kwargs) + + def __iter__(self): + return iter(self._handle) + + def __enter__(self): + self._handle.__enter__() + return self + + def __exit__(self, *args): + return self._handle.__exit__(*args) + + def __getattr__(self, item): + return getattr(self._handle, item) + + def watched_open(file, *args, **kwargs): + handle = real_open(file, *args, **kwargs) + if str(file) == str(path): + return _WatchedFile(handle, str(file)) + return handle + + monkeypatch.setattr(builtins, "open", watched_open) + parsed = parse_source(resolve_source(tmp_path)) + + assert parsed.queries + assert whole_file_reads == [] + + +# Measured on the recorded fixtures: ~36 KB of retained model state per task. +# The bound is generous headroom over that, tight enough to catch a regression +# that starts holding raw event dicts or file text. +_MAX_RETAINED_BYTES_PER_TASK = 200_000 + + +def _benchmark(log_dir: Path, log_file: str) -> dict: + """Run one ingestion measurement in a fresh interpreter.""" + completed = subprocess.run( + [sys.executable, "-m", "tests.benchmark_ingestion", str(log_dir), log_file], + capture_output=True, + text=True, + check=True, + cwd=Path(__file__).parents[1], + ) + return json.loads(completed.stdout.strip().splitlines()[-1]) + + +def test_ingestion_cost_scales_with_retained_state_not_file_size(tmp_path): + """Benchmark: elapsed time and process peak RSS on increasing fixtures. + + Each size is parsed end to end (``parse_source``, models and all) in its own + process, because ``ru_maxrss`` is a high-water mark that would otherwise + report the largest run for every size. + + Peak RSS grows because every task, stage and plan node is retained as a + model until the frames are built — that growth is expected and is what the + per-task bound below documents. What must not grow is a copy of the log + text, which ``test_reading_never_copies_the_whole_file`` pins down directly. + """ + base = log_lines() + measurements = [] + for multiplier in (1, 5, 20): + name = f"app-{multiplier}x" + path = write_log(tmp_path / name, scale_task_volume(base, multiplier - 1)) + result = _benchmark(tmp_path, name) + result["size_bytes"] = path.stat().st_size + measurements.append(result) + + for result in measurements: + print( + f"{Path(result['log']).name}: bytes={result['size_bytes']} " + f"events={result['events_read']} tasks={result['tasks']} " + f"elapsed={result['elapsed_s']:.4f}s " + f"peak_rss={result['peak_rss_bytes']} " + f"retained_rss={result['retained_rss_bytes']}" + ) + + small, large = measurements[0], measurements[-1] + + # The larger fixtures are valid logs, not padding: same plan, more tasks. + assert {result["queries"] for result in measurements} == {small["queries"]} + assert large["tasks"] == small["tasks"] * 20 + + # Retained state is proportional to the tasks kept, not to the bytes read. + assert large["retained_rss_bytes"] < _MAX_RETAINED_BYTES_PER_TASK * large["tasks"] + + # Sub-quadratic: 20x the tasks must not cost anywhere near 400x the time. + task_ratio = large["tasks"] / small["tasks"] + assert large["elapsed_s"] < small["elapsed_s"] * task_ratio * 3 + + +def test_cloud_style_source_is_discovered_and_streamed(monkeypatch): + """A fake object store exercises the fsspec path without real credentials.""" + fsspec = pytest.importorskip("fsspec") + import sparkparse.storage as storage + + memory_fs = fsspec.filesystem("memory") + for existing in memory_fs.ls("/", detail=False): + memory_fs.rm(existing, recursive=True) + + monkeypatch.setattr( + storage, "CLOUD_PREFIXES", (*storage.CLOUD_PREFIXES, "memory://") + ) + + lines = log_lines() + with fsspec.open("memory://logs/app-cloud", "w") as handle: + handle.write("\n".join(lines) + "\n") + with fsspec.open("memory://logs/_SUCCESS", "w") as handle: + handle.write("") + + sources = discover_sources("memory://logs") + assert [source.name for source in sources] == ["app-cloud"] + + cloud = get_parsed_metrics(log_dir="memory://logs", out_dir=None, out_format=None) + local = get_parsed_metrics( + log_dir=Path("tests/data/full_logs"), + log_file="nested_loop_join", + out_dir=None, + out_format=None, + ) + + assert _dag_records(cloud) == _dag_records(local) + + +def test_dashboard_log_duration_streams_a_rolling_source(tmp_path): + """The home page lists logical sources, including rolled ones.""" + from sparkparse.pages.home import get_log_duration + + rolling_log_dir(tmp_path, log_lines(), app_id="app-rolled", segments=3) + source = resolve_source(tmp_path) + + duration = get_log_duration(source) + + assert duration.start_time <= duration.end_time + assert duration.duration_seconds >= 0 + assert duration.duration_formatted diff --git a/tests/test_parse_partial.py b/tests/test_parse_partial.py new file mode 100644 index 0000000..41d3d0c --- /dev/null +++ b/tests/test_parse_partial.py @@ -0,0 +1,601 @@ +"""Increment 1 and 2 of brief 04: partial logs, plan fallback, attempts. + +Every case here is an event log Spark really produces but the fixtures do not +contain: a non-adaptive query, a log copied mid-flush, a retried stage. +""" + +from __future__ import annotations + +import json + +import polars as pl +import pytest + +from sparkparse.eventlog import EventLogNotFoundError +from sparkparse.models import TaskStatus +from sparkparse.parse import get_all_parsed_metrics, get_parsed_metrics, parse_log +from tests.eventlog_fixtures import ( + drop_events, + is_adaptive_update, + log_lines, + map_events, + stage_attempt, + task_attempt, + write_log, +) + + +def _parse(tmp_path, lines, name="app-test-0001", **kwargs): + write_log(tmp_path / name, lines) + return parse_log(tmp_path / name, **kwargs) + + +# --------------------------------------------------------------------------- +# increment 1 — plan fallback and partial parsing +# --------------------------------------------------------------------------- + + +def test_non_adaptive_log_falls_back_to_the_start_event_plan(tmp_path): + """Without AQE the only plan is the one on SQLExecutionStart.""" + lines = drop_events(log_lines("nested_loop_join"), is_adaptive_update) + + parsed = _parse(tmp_path, lines) + + assert [query.query_id for query in parsed.queries] == [0, 1, 2, 3] + assert all(query.nodes for query in parsed.queries) + assert {d.code for d in parsed.diagnostics} == {"non_final_plan"} + + +def test_final_adaptive_plan_still_wins_over_the_initial_plan(tmp_path): + """The initial plan is a fallback, not a replacement.""" + full = _parse(tmp_path, log_lines("nested_loop_join"), name="full") + initial_only = _parse( + tmp_path, + drop_events(log_lines("nested_loop_join"), is_adaptive_update), + name="initial", + ) + + final_query = next(q for q in full.queries if q.query_id == 3) + initial_query = next(q for q in initial_only.queries if q.query_id == 3) + + assert len(final_query.nodes) != len(initial_query.nodes) + assert not [ + d for d in full.diagnostics if d.code == "non_final_plan" and " 3 " in d.message + ] + + +def test_missing_application_start_is_not_fatal(tmp_path): + lines = drop_events( + log_lines("nested_loop_join"), + lambda event: event["Event"] == "SparkListenerApplicationStart", + ) + + parsed = _parse(tmp_path, lines, name="app-no-start") + + assert parsed.queries + assert parsed.application_id == "app-no-start" + assert any(d.code == "missing_application_start" for d in parsed.diagnostics) + + +def test_log_without_sql_executions_yields_typed_empty_frames(tmp_path): + lines = drop_events( + log_lines("nested_loop_join"), + lambda event: "SQL" in event["Event"], + ) + write_log(tmp_path / "app-no-sql", lines) + + dfs = get_parsed_metrics( + log_dir=tmp_path, log_file="app-no-sql", out_dir=None, out_format=None + ) + + assert dfs.dag.is_empty() + # Typed, not shapeless: downstream code checks columns, not just row counts. + assert "query_id" in dfs.dag.columns + # The application still ran tasks; they are unattributed, not absent. + assert dfs.combined.height > 0 + assert dfs.combined["query_id"].null_count() == dfs.combined.height + assert any(d.code == "no_queries" for d in dfs.diagnostics) + + +def test_query_without_end_event_keeps_its_plan_and_reports_no_duration(tmp_path): + lines = drop_events( + log_lines("nested_loop_join"), + lambda event: event["Event"].endswith("SQLExecutionEnd") + and event["executionId"] == 3, + ) + write_log(tmp_path / "app-unfinished", lines) + + dfs = get_parsed_metrics( + log_dir=tmp_path, log_file="app-unfinished", out_dir=None, out_format=None + ) + + query_3 = dfs.dag.filter(pl.col("query_id") == 3) + assert query_3.height > 0 + assert query_3["query_end_timestamp"].null_count() == query_3.height + assert query_3["query_duration_seconds"].null_count() == query_3.height + # The label still identifies the query instead of going null with it. + assert query_3["query_header"][0] == "3 - count [unknown min]" + + +def test_all_queries_unfinished_still_produces_a_frame(tmp_path): + """A log cut off before any SQLExecutionEnd has no 'end' column to pivot.""" + lines = drop_events( + log_lines("nested_loop_join"), + lambda event: event["Event"].endswith("SQLExecutionEnd"), + ) + write_log(tmp_path / "app-cut-short", lines) + + dfs = get_parsed_metrics( + log_dir=tmp_path, log_file="app-cut-short", out_dir=None, out_format=None + ) + + assert dfs.dag.height > 0 + assert dfs.dag["query_end_timestamp"].null_count() == dfs.dag.height + + +def test_log_cut_off_mid_job_keeps_the_tasks_it_recorded(tmp_path): + """Completed tasks but no JobEnd: the job duration is unknown, not zero.""" + lines = log_lines("nested_loop_join") + cut = ( + next( + index + for index, line in enumerate(lines) + if json.loads(line)["Event"] == "SparkListenerTaskEnd" + ) + + 1 + ) + write_log(tmp_path / "app-mid-job", lines[:cut]) + + dfs = get_parsed_metrics( + log_dir=tmp_path, log_file="app-mid-job", out_dir=None, out_format=None + ) + + assert dfs.combined.height == 1 + assert dfs.combined["job_end_timestamp"].null_count() == 1 + assert dfs.combined["job_duration_seconds"].null_count() == 1 + assert dfs.combined["stage_end_timestamp"].null_count() == 1 + assert dfs.combined["stage_status"][0] == "running" + + +def test_truncated_tail_is_reported_as_truncation_not_corruption(tmp_path): + lines = log_lines("nested_loop_join") + lines.append('{"Event":"SparkListenerApplicationEnd","Timestamp":17424') + + parsed = _parse(tmp_path, lines, name="app-truncated") + + codes = {d.code for d in parsed.diagnostics} + assert "truncated_tail" in codes + assert "corrupt_line" not in codes + assert parsed.queries + + +def test_corruption_mid_file_is_distinguished_from_a_truncated_tail(tmp_path): + lines = log_lines("nested_loop_join") + corrupt_index = len(lines) // 2 + lines.insert(corrupt_index, '{"Event":"SparkListenerTaskEnd", "Stage I') + + parsed = _parse(tmp_path, lines, name="app-corrupt") + + corrupt = [d for d in parsed.diagnostics if d.code == "corrupt_line"] + assert len(corrupt) == 1 + assert corrupt[0].line == corrupt_index + 1 + assert not [d for d in parsed.diagnostics if d.code == "truncated_tail"] + + +@pytest.mark.parametrize("position", ["tail", "middle"]) +def test_strict_mode_fails_with_source_context(tmp_path, position): + lines = log_lines("nested_loop_join") + broken = '{"Event":"SparkListenerTaskEnd", "Stage I' + if position == "tail": + lines.append(broken) + else: + lines.insert(len(lines) // 2, broken) + + with pytest.raises(ValueError, match=r"app-strict:\d+"): + _parse(tmp_path, lines, name="app-strict", strict=True) + + +def test_task_without_metrics_is_kept_with_unknown_usage(tmp_path): + seen = {"done": False} + + def strip_first_task_metrics(event: dict) -> dict: + if event["Event"] == "SparkListenerTaskEnd" and not seen["done"]: + seen["done"] = True + event.pop("Task Metrics") + return event + + lines = map_events(log_lines("nested_loop_join"), strip_first_task_metrics) + parsed = _parse(tmp_path, lines, name="app-no-task-metrics") + + without_metrics = [task for task in parsed.tasks if task.metrics is None] + assert len(without_metrics) == 1 + assert any(d.code == "tasks_missing_metrics" for d in parsed.diagnostics) + + +def test_every_task_missing_metrics_still_builds_a_typed_frame(tmp_path): + def strip_metrics(event: dict) -> dict: + if event["Event"] == "SparkListenerTaskEnd": + event.pop("Task Metrics", None) + return event + + lines = map_events(log_lines("nested_loop_join"), strip_metrics) + write_log(tmp_path / "app-metricless", lines) + + dfs = get_parsed_metrics( + log_dir=tmp_path, log_file="app-metricless", out_dir=None, out_format=None + ) + + # Missing, not zero: a null total says the metric was never reported. + assert dfs.combined.height > 0 + assert dfs.combined["bytes_read"].null_count() == dfs.combined.height + + +def test_unknown_events_are_counted_not_fatal(tmp_path): + lines = log_lines("nested_loop_join") + lines.insert(5, json.dumps({"Event": "SparkListenerSomethingFromSpark5"})) + + parsed = _parse(tmp_path, lines, name="app-unknown-event") + + assert parsed.unknown_events["SparkListenerSomethingFromSpark5"] == 1 + assert parsed.queries + + +def test_empty_log_file_parses_to_nothing(tmp_path): + (tmp_path / "app-empty").write_text("") + + dfs = get_parsed_metrics( + log_dir=tmp_path, log_file="app-empty", out_dir=None, out_format=None + ) + + assert dfs.dag.is_empty() + assert dfs.combined.is_empty() + assert any(d.code == "no_queries" for d in dfs.diagnostics) + + +def test_empty_directory_reports_no_sources(tmp_path): + with pytest.raises(EventLogNotFoundError): + get_parsed_metrics(log_dir=tmp_path, out_dir=None, out_format=None) + + +# --------------------------------------------------------------------------- +# increment 2 — identity and attempts +# --------------------------------------------------------------------------- + + +def _with_stage_retry(lines: list[str], stage_id: int = 1) -> tuple[list[str], int]: + """Append a second attempt of ``stage_id`` with one task of its own.""" + parsed = [json.loads(line) for line in lines] + submitted = next( + event + for event in parsed + if event["Event"] == "SparkListenerStageSubmitted" + and event["Stage Info"]["Stage ID"] == stage_id + ) + completed = next( + event + for event in parsed + if event["Event"] == "SparkListenerStageCompleted" + and event["Stage Info"]["Stage ID"] == stage_id + ) + task = next( + event + for event in parsed + if event["Event"] == "SparkListenerTaskEnd" and event["Stage ID"] == stage_id + ) + retry_task_id = ( + max( + event["Task Info"]["Task ID"] + for event in parsed + if event["Event"] == "SparkListenerTaskEnd" + ) + + 1 + ) + + extra = [ + stage_attempt(submitted, 1), + task_attempt(task, task_id=retry_task_id, stage_attempt_id=1, attempt=1), + stage_attempt(completed, 1), + ] + end_index = next( + i + for i, event in enumerate(parsed) + if event["Event"] == "SparkListenerApplicationEnd" + ) + merged = parsed[:end_index] + extra + parsed[end_index:] + return [json.dumps(event) for event in merged], retry_task_id + + +def test_stage_retry_is_kept_as_a_separate_attempt(tmp_path): + lines, retry_task_id = _with_stage_retry(log_lines("nested_loop_join")) + write_log(tmp_path / "app-retry", lines) + + dfs = get_parsed_metrics( + log_dir=tmp_path, log_file="app-retry", out_dir=None, out_format=None + ) + + attempts = dfs.combined.filter(pl.col("stage_id") == 1)["stage_attempt_id"] + assert set(attempts.to_list()) == {0, 1} + # One row per task attempt: the retry must not duplicate the original rows. + assert ( + dfs.combined.height + == dfs.combined.select( + pl.struct("stage_id", "stage_attempt_id", "task_id") + ).n_unique() + ) + assert retry_task_id in dfs.combined["task_id"].to_list() + + +def test_failed_and_speculative_attempts_are_kept_out_of_output_totals(tmp_path): + parsed = [json.loads(line) for line in log_lines("nested_loop_join")] + template = next( + event for event in parsed if event["Event"] == "SparkListenerTaskEnd" + ) + next_id = ( + max( + event["Task Info"]["Task ID"] + for event in parsed + if event["Event"] == "SparkListenerTaskEnd" + ) + + 1 + ) + extras = [ + task_attempt(template, task_id=next_id, attempt=1, failed=True), + task_attempt( + template, task_id=next_id + 1, attempt=2, killed=True, speculative=True + ), + ] + end_index = next( + i + for i, event in enumerate(parsed) + if event["Event"] == "SparkListenerApplicationEnd" + ) + merged = parsed[:end_index] + extras + parsed[end_index:] + write_log(tmp_path / "app-attempts", [json.dumps(e) for e in merged]) + + dfs = get_parsed_metrics( + log_dir=tmp_path, log_file="app-attempts", out_dir=None, out_format=None + ) + baseline = get_parsed_metrics( + log_dir="tests/data/full_logs", + log_file="nested_loop_join", + out_dir=None, + out_format=None, + ) + + statuses = set(dfs.combined["task_status"].to_list()) + assert statuses == { + TaskStatus.success.value, + TaskStatus.failed.value, + TaskStatus.killed.value, + } + assert dfs.combined.height == baseline.combined.height + 2 + + from sparkparse.analyze import retained_outputs + + kept = retained_outputs(dfs.combined) + assert kept.height == baseline.combined.height + # Output accounting ignores the discarded attempts... + assert kept["bytes_read"].sum() == baseline.combined["bytes_read"].sum() + # ...while the resources they burned are still on the ledger. + assert ( + dfs.combined["executor_run_time_seconds"].sum() + > baseline.combined["executor_run_time_seconds"].sum() + ) + + +def test_reused_stage_does_not_multiply_task_rows(tmp_path): + """A stage listed by two jobs is one physical stage, not two.""" + parsed = [json.loads(line) for line in log_lines("nested_loop_join")] + job_start = next( + event for event in parsed if event["Event"] == "SparkListenerJobStart" + ) + shared_stage = job_start["Stage IDs"][0] + later_job = next( + event + for event in parsed + if event["Event"] == "SparkListenerJobStart" + and shared_stage not in event["Stage IDs"] + ) + later_job["Stage IDs"] = [*later_job["Stage IDs"], shared_stage] + write_log(tmp_path / "app-reuse", [json.dumps(e) for e in parsed]) + + dfs = get_parsed_metrics( + log_dir=tmp_path, log_file="app-reuse", out_dir=None, out_format=None + ) + baseline = get_parsed_metrics( + log_dir="tests/data/full_logs", + log_file="nested_loop_join", + out_dir=None, + out_format=None, + ) + + assert dfs.combined.height == baseline.combined.height + assert dfs.combined["bytes_read"].sum() == baseline.combined["bytes_read"].sum() + + # The relation itself is preserved, just not joined into the task frame. + assert dfs.job_stage is not None + jobs_for_stage = dfs.job_stage.filter(pl.col("stage_id") == shared_stage) + assert jobs_for_stage.height == 2 + + +def test_task_shared_by_two_queries_appears_once_with_the_relation_kept(): + dfs = get_parsed_metrics( + log_dir="tests/data/full_logs", + log_file="nested_final_plans", + out_dir=None, + out_format=None, + ) + + keys = dfs.combined.select("stage_id", "task_id") + assert keys.n_unique() == dfs.combined.height + assert max(dfs.combined["query_count"].to_list()) > 1 + + assert dfs.query_stage is not None + shared = ( + dfs.query_stage.group_by("stage_id") + .agg(pl.col("query_id").n_unique().alias("queries")) + .filter(pl.col("queries") > 1) + ) + assert shared.height > 0 + + +def test_repeated_query_ids_across_applications_stay_separate(tmp_path): + for name in ("app-a", "app-b"): + write_log(tmp_path / name, log_lines("nested_loop_join")) + + results = get_all_parsed_metrics(log_dir=tmp_path, out_dir=None, out_format=None) + + assert set(results) == {"app-a", "app-b"} + for dfs in results.values(): + assert set(dfs.dag["query_id"].unique().to_list()) == {0, 1, 2, 3} + + +def test_summary_records_which_attempts_each_total_covers(tmp_path): + from sparkparse.analyze import to_plan_summary + + dfs = get_parsed_metrics( + log_dir="tests/data/full_logs", + log_file="nested_loop_join", + out_dir=None, + out_format=None, + ) + + summary = to_plan_summary(dfs, "nested_loop_join") + + assert summary["total_basis"]["bytes_read"] == "retained_outputs" + assert summary["total_basis"]["memory_bytes_spilled"] == "all_attempts" + assert summary["total_basis"]["executor_run_time_seconds"] == "all_attempts" + + +def _append_task_events(lines: list[str], build) -> tuple[list[str], list[dict]]: + """Insert task events built by ``build(template, next_task_id)`` before ApplicationEnd.""" + parsed = [json.loads(line) for line in lines] + # The task that read the most: duplicating a task with no input metrics + # would make the "counted twice" assertions vacuous. + template = max( + (event for event in parsed if event["Event"] == "SparkListenerTaskEnd"), + key=lambda event: event["Task Metrics"]["Input Metrics"]["Bytes Read"], + ) + next_id = ( + max( + event["Task Info"]["Task ID"] + for event in parsed + if event["Event"] == "SparkListenerTaskEnd" + ) + + 1 + ) + extras = build(template, next_id) + end_index = next( + index + for index, event in enumerate(parsed) + if event["Event"] == "SparkListenerApplicationEnd" + ) + merged = parsed[:end_index] + extras + parsed[end_index:] + return [json.dumps(event) for event in merged], extras + + +def _baseline(): + return get_parsed_metrics( + log_dir="tests/data/full_logs", + log_file="nested_loop_join", + out_dir=None, + out_format=None, + ) + + +def test_losing_speculative_copy_does_not_count_its_output_twice(tmp_path): + """Both copies succeed; only the one that committed produced output.""" + + def build(template, next_id): + loser = task_attempt(template, task_id=next_id, attempt=1, speculative=True) + # Finished after the original, so the commit was already awarded. + loser["Task Info"]["Finish Time"] = template["Task Info"]["Finish Time"] + 5_000 + return [loser] + + lines, _ = _append_task_events(log_lines("nested_loop_join"), build) + write_log(tmp_path / "app-speculative", lines) + + dfs = get_parsed_metrics( + log_dir=tmp_path, log_file="app-speculative", out_dir=None, out_format=None + ) + baseline = _baseline() + + from sparkparse.analyze import retained_outputs + + # Both attempts are on the resource ledger... + assert dfs.combined.height == baseline.combined.height + 1 + assert dfs.combined["task_status"].to_list().count("success") == ( + baseline.combined.height + 1 + ) + # ...but only one of them counts as output. + kept = retained_outputs(dfs.combined) + assert kept.height == baseline.combined.height + assert kept["bytes_read"].sum() == baseline.combined["bytes_read"].sum() + assert ( + dfs.combined["executor_run_time_seconds"].sum() + > baseline.combined["executor_run_time_seconds"].sum() + ) + + +def test_partition_recomputed_in_a_later_stage_attempt_counts_once(tmp_path): + """A fetch failure recomputes a partition; the newer output supersedes.""" + + def build(template, next_id): + recomputed = task_attempt( + template, task_id=next_id, stage_attempt_id=1, attempt=1 + ) + recomputed["Task Info"]["Finish Time"] = ( + template["Task Info"]["Finish Time"] + 10_000 + ) + return [recomputed] + + lines, extras = _append_task_events(log_lines("nested_loop_join"), build) + write_log(tmp_path / "app-recomputed", lines) + + dfs = get_parsed_metrics( + log_dir=tmp_path, log_file="app-recomputed", out_dir=None, out_format=None + ) + baseline = _baseline() + + from sparkparse.analyze import retained_outputs + + kept = retained_outputs(dfs.combined) + assert dfs.combined.height == baseline.combined.height + 1 + assert kept.height == baseline.combined.height + # The surviving row is the recomputed one, not the superseded attempt. + recomputed_id = extras[0]["Task Info"]["Task ID"] + partition = extras[0]["Task Info"]["Index"] + survivor = kept.filter( + (pl.col("stage_id") == extras[0]["Stage ID"]) & (pl.col("index") == partition) + ) + assert survivor["task_id"].to_list() == [recomputed_id] + assert survivor["stage_attempt_id"].to_list() == [1] + + +def test_duplicate_successful_outputs_do_not_inflate_summary_or_history(tmp_path): + def build(template, next_id): + duplicate = task_attempt(template, task_id=next_id, attempt=1, speculative=True) + duplicate["Task Info"]["Finish Time"] = ( + template["Task Info"]["Finish Time"] + 5_000 + ) + return [duplicate] + + lines, _ = _append_task_events(log_lines("nested_loop_join"), build) + write_log(tmp_path / "app-dup", lines) + + from sparkparse.analyze import to_plan_summary + from sparkparse.history import record_from_dfs + + dfs = get_parsed_metrics( + log_dir=tmp_path, log_file="app-dup", out_dir=None, out_format=None + ) + baseline = _baseline() + + assert ( + to_plan_summary(dfs, "dup")["totals"]["bytes_read"] + == to_plan_summary(baseline, "baseline")["totals"]["bytes_read"] + ) + assert ( + record_from_dfs(dfs, "dup").bytes_read + == record_from_dfs(baseline, "baseline").bytes_read + ) diff --git a/uv.lock b/uv.lock index c79bb40..a5e40e0 100644 --- a/uv.lock +++ b/uv.lock @@ -2395,6 +2395,9 @@ viz = [ { name = "pandas" }, { name = "plotly" }, ] +zstd = [ + { name = "zstandard" }, +] [package.dev-dependencies] dev = [ @@ -2411,6 +2414,7 @@ dev = [ { name = "pyspark" }, { name = "pytest" }, { name = "ruff" }, + { name = "zstandard" }, ] [package.metadata] @@ -2430,8 +2434,9 @@ requires-dist = [ { name = "pyspark", marker = "extra == 'spark'", specifier = ">=3.5.5" }, { name = "s3fs", marker = "extra == 'cloud'", specifier = ">=2024.1.0" }, { name = "s3fs", marker = "extra == 's3'", specifier = ">=2024.1.0" }, + { name = "zstandard", marker = "extra == 'zstd'", specifier = ">=0.22" }, ] -provides-extras = ["viz", "spark", "s3", "azure", "gcs", "cloud", "delta"] +provides-extras = ["viz", "spark", "s3", "azure", "gcs", "cloud", "delta", "zstd"] [package.metadata.requires-dev] dev = [ @@ -2448,6 +2453,7 @@ dev = [ { name = "pyspark", specifier = ">=3.5.5" }, { name = "pytest", specifier = ">=8.3.5" }, { name = "ruff", specifier = ">=0.9.9" }, + { name = "zstandard", specifier = ">=0.25.0" }, ] [[package]] @@ -2736,3 +2742,77 @@ sdist = { url = "https://files.pythonhosted.org/packages/3f/50/bad581df71744867e wheels = [ { url = "https://files.pythonhosted.org/packages/b7/1a/7e4798e9339adc931158c9d69ecc34f5e6791489d469f5e50ec15e35f458/zipp-3.21.0-py3-none-any.whl", hash = "sha256:ac1bbe05fd2991f160ebce24ffbac5f6d11d83dc90891255885223d42b3cd931", size = 9630, upload-time = "2024-11-10T15:05:19.275Z" }, ] + +[[package]] +name = "zstandard" +version = "0.25.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/fd/aa/3e0508d5a5dd96529cdc5a97011299056e14c6505b678fd58938792794b1/zstandard-0.25.0.tar.gz", hash = "sha256:7713e1179d162cf5c7906da876ec2ccb9c3a9dcbdffef0cc7f70c3667a205f0b", size = 711513, upload-time = "2025-09-14T22:15:54.002Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/2a/83/c3ca27c363d104980f1c9cee1101cc8ba724ac8c28a033ede6aab89585b1/zstandard-0.25.0-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:933b65d7680ea337180733cf9e87293cc5500cc0eb3fc8769f4d3c88d724ec5c", size = 795254, upload-time = "2025-09-14T22:16:26.137Z" }, + { url = "https://files.pythonhosted.org/packages/ac/4d/e66465c5411a7cf4866aeadc7d108081d8ceba9bc7abe6b14aa21c671ec3/zstandard-0.25.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:a3f79487c687b1fc69f19e487cd949bf3aae653d181dfb5fde3bf6d18894706f", size = 640559, upload-time = "2025-09-14T22:16:27.973Z" }, + { url = "https://files.pythonhosted.org/packages/12/56/354fe655905f290d3b147b33fe946b0f27e791e4b50a5f004c802cb3eb7b/zstandard-0.25.0-cp311-cp311-manylinux2010_i686.manylinux2014_i686.manylinux_2_12_i686.manylinux_2_17_i686.whl", hash = "sha256:0bbc9a0c65ce0eea3c34a691e3c4b6889f5f3909ba4822ab385fab9057099431", size = 5348020, upload-time = "2025-09-14T22:16:29.523Z" }, + { url = "https://files.pythonhosted.org/packages/3b/13/2b7ed68bd85e69a2069bcc72141d378f22cae5a0f3b353a2c8f50ef30c1b/zstandard-0.25.0-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:01582723b3ccd6939ab7b3a78622c573799d5d8737b534b86d0e06ac18dbde4a", size = 5058126, upload-time = "2025-09-14T22:16:31.811Z" }, + { url = "https://files.pythonhosted.org/packages/c9/dd/fdaf0674f4b10d92cb120ccff58bbb6626bf8368f00ebfd2a41ba4a0dc99/zstandard-0.25.0-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.whl", hash = "sha256:5f1ad7bf88535edcf30038f6919abe087f606f62c00a87d7e33e7fc57cb69fcc", size = 5405390, upload-time = "2025-09-14T22:16:33.486Z" }, + { url = "https://files.pythonhosted.org/packages/0f/67/354d1555575bc2490435f90d67ca4dd65238ff2f119f30f72d5cde09c2ad/zstandard-0.25.0-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.whl", hash = "sha256:06acb75eebeedb77b69048031282737717a63e71e4ae3f77cc0c3b9508320df6", size = 5452914, upload-time = "2025-09-14T22:16:35.277Z" }, + { url = "https://files.pythonhosted.org/packages/bb/1f/e9cfd801a3f9190bf3e759c422bbfd2247db9d7f3d54a56ecde70137791a/zstandard-0.25.0-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:9300d02ea7c6506f00e627e287e0492a5eb0371ec1670ae852fefffa6164b072", size = 5559635, upload-time = "2025-09-14T22:16:37.141Z" }, + { url = "https://files.pythonhosted.org/packages/21/88/5ba550f797ca953a52d708c8e4f380959e7e3280af029e38fbf47b55916e/zstandard-0.25.0-cp311-cp311-musllinux_1_1_aarch64.whl", hash = "sha256:bfd06b1c5584b657a2892a6014c2f4c20e0db0208c159148fa78c65f7e0b0277", size = 5048277, upload-time = "2025-09-14T22:16:38.807Z" }, + { url = "https://files.pythonhosted.org/packages/46/c0/ca3e533b4fa03112facbe7fbe7779cb1ebec215688e5df576fe5429172e0/zstandard-0.25.0-cp311-cp311-musllinux_1_1_x86_64.whl", hash = "sha256:f373da2c1757bb7f1acaf09369cdc1d51d84131e50d5fa9863982fd626466313", size = 5574377, upload-time = "2025-09-14T22:16:40.523Z" }, + { url = "https://files.pythonhosted.org/packages/12/9b/3fb626390113f272abd0799fd677ea33d5fc3ec185e62e6be534493c4b60/zstandard-0.25.0-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:6c0e5a65158a7946e7a7affa6418878ef97ab66636f13353b8502d7ea03c8097", size = 4961493, upload-time = "2025-09-14T22:16:43.3Z" }, + { url = "https://files.pythonhosted.org/packages/cb/d3/23094a6b6a4b1343b27ae68249daa17ae0651fcfec9ed4de09d14b940285/zstandard-0.25.0-cp311-cp311-musllinux_1_2_i686.whl", hash = "sha256:c8e167d5adf59476fa3e37bee730890e389410c354771a62e3c076c86f9f7778", size = 5269018, upload-time = "2025-09-14T22:16:45.292Z" }, + { url = "https://files.pythonhosted.org/packages/8c/a7/bb5a0c1c0f3f4b5e9d5b55198e39de91e04ba7c205cc46fcb0f95f0383c1/zstandard-0.25.0-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:98750a309eb2f020da61e727de7d7ba3c57c97cf6213f6f6277bb7fb42a8e065", size = 5443672, upload-time = "2025-09-14T22:16:47.076Z" }, + { url = "https://files.pythonhosted.org/packages/27/22/503347aa08d073993f25109c36c8d9f029c7d5949198050962cb568dfa5e/zstandard-0.25.0-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:22a086cff1b6ceca18a8dd6096ec631e430e93a8e70a9ca5efa7561a00f826fa", size = 5822753, upload-time = "2025-09-14T22:16:49.316Z" }, + { url = "https://files.pythonhosted.org/packages/e2/be/94267dc6ee64f0f8ba2b2ae7c7a2df934a816baaa7291db9e1aa77394c3c/zstandard-0.25.0-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:72d35d7aa0bba323965da807a462b0966c91608ef3a48ba761678cb20ce5d8b7", size = 5366047, upload-time = "2025-09-14T22:16:51.328Z" }, + { url = "https://files.pythonhosted.org/packages/7b/a3/732893eab0a3a7aecff8b99052fecf9f605cf0fb5fb6d0290e36beee47a4/zstandard-0.25.0-cp311-cp311-win32.whl", hash = "sha256:f5aeea11ded7320a84dcdd62a3d95b5186834224a9e55b92ccae35d21a8b63d4", size = 436484, upload-time = "2025-09-14T22:16:55.005Z" }, + { url = "https://files.pythonhosted.org/packages/43/a3/c6155f5c1cce691cb80dfd38627046e50af3ee9ddc5d0b45b9b063bfb8c9/zstandard-0.25.0-cp311-cp311-win_amd64.whl", hash = "sha256:daab68faadb847063d0c56f361a289c4f268706b598afbf9ad113cbe5c38b6b2", size = 506183, upload-time = "2025-09-14T22:16:52.753Z" }, + { url = "https://files.pythonhosted.org/packages/8c/3e/8945ab86a0820cc0e0cdbf38086a92868a9172020fdab8a03ac19662b0e5/zstandard-0.25.0-cp311-cp311-win_arm64.whl", hash = "sha256:22a06c5df3751bb7dc67406f5374734ccee8ed37fc5981bf1ad7041831fa1137", size = 462533, upload-time = "2025-09-14T22:16:53.878Z" }, + { url = "https://files.pythonhosted.org/packages/82/fc/f26eb6ef91ae723a03e16eddb198abcfce2bc5a42e224d44cc8b6765e57e/zstandard-0.25.0-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:7b3c3a3ab9daa3eed242d6ecceead93aebbb8f5f84318d82cee643e019c4b73b", size = 795738, upload-time = "2025-09-14T22:16:56.237Z" }, + { url = "https://files.pythonhosted.org/packages/aa/1c/d920d64b22f8dd028a8b90e2d756e431a5d86194caa78e3819c7bf53b4b3/zstandard-0.25.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:913cbd31a400febff93b564a23e17c3ed2d56c064006f54efec210d586171c00", size = 640436, upload-time = "2025-09-14T22:16:57.774Z" }, + { url = "https://files.pythonhosted.org/packages/53/6c/288c3f0bd9fcfe9ca41e2c2fbfd17b2097f6af57b62a81161941f09afa76/zstandard-0.25.0-cp312-cp312-manylinux2010_i686.manylinux2014_i686.manylinux_2_12_i686.manylinux_2_17_i686.whl", hash = "sha256:011d388c76b11a0c165374ce660ce2c8efa8e5d87f34996aa80f9c0816698b64", size = 5343019, upload-time = "2025-09-14T22:16:59.302Z" }, + { url = "https://files.pythonhosted.org/packages/1e/15/efef5a2f204a64bdb5571e6161d49f7ef0fffdbca953a615efbec045f60f/zstandard-0.25.0-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:6dffecc361d079bb48d7caef5d673c88c8988d3d33fb74ab95b7ee6da42652ea", size = 5063012, upload-time = "2025-09-14T22:17:01.156Z" }, + { url = "https://files.pythonhosted.org/packages/b7/37/a6ce629ffdb43959e92e87ebdaeebb5ac81c944b6a75c9c47e300f85abdf/zstandard-0.25.0-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.whl", hash = "sha256:7149623bba7fdf7e7f24312953bcf73cae103db8cae49f8154dd1eadc8a29ecb", size = 5394148, upload-time = "2025-09-14T22:17:03.091Z" }, + { url = "https://files.pythonhosted.org/packages/e3/79/2bf870b3abeb5c070fe2d670a5a8d1057a8270f125ef7676d29ea900f496/zstandard-0.25.0-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.whl", hash = "sha256:6a573a35693e03cf1d67799fd01b50ff578515a8aeadd4595d2a7fa9f3ec002a", size = 5451652, upload-time = "2025-09-14T22:17:04.979Z" }, + { url = "https://files.pythonhosted.org/packages/53/60/7be26e610767316c028a2cbedb9a3beabdbe33e2182c373f71a1c0b88f36/zstandard-0.25.0-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:5a56ba0db2d244117ed744dfa8f6f5b366e14148e00de44723413b2f3938a902", size = 5546993, upload-time = "2025-09-14T22:17:06.781Z" }, + { url = "https://files.pythonhosted.org/packages/85/c7/3483ad9ff0662623f3648479b0380d2de5510abf00990468c286c6b04017/zstandard-0.25.0-cp312-cp312-musllinux_1_1_aarch64.whl", hash = "sha256:10ef2a79ab8e2974e2075fb984e5b9806c64134810fac21576f0668e7ea19f8f", size = 5046806, upload-time = "2025-09-14T22:17:08.415Z" }, + { url = "https://files.pythonhosted.org/packages/08/b3/206883dd25b8d1591a1caa44b54c2aad84badccf2f1de9e2d60a446f9a25/zstandard-0.25.0-cp312-cp312-musllinux_1_1_x86_64.whl", hash = "sha256:aaf21ba8fb76d102b696781bddaa0954b782536446083ae3fdaa6f16b25a1c4b", size = 5576659, upload-time = "2025-09-14T22:17:10.164Z" }, + { url = "https://files.pythonhosted.org/packages/9d/31/76c0779101453e6c117b0ff22565865c54f48f8bd807df2b00c2c404b8e0/zstandard-0.25.0-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:1869da9571d5e94a85a5e8d57e4e8807b175c9e4a6294e3b66fa4efb074d90f6", size = 4953933, upload-time = "2025-09-14T22:17:11.857Z" }, + { url = "https://files.pythonhosted.org/packages/18/e1/97680c664a1bf9a247a280a053d98e251424af51f1b196c6d52f117c9720/zstandard-0.25.0-cp312-cp312-musllinux_1_2_i686.whl", hash = "sha256:809c5bcb2c67cd0ed81e9229d227d4ca28f82d0f778fc5fea624a9def3963f91", size = 5268008, upload-time = "2025-09-14T22:17:13.627Z" }, + { url = "https://files.pythonhosted.org/packages/1e/73/316e4010de585ac798e154e88fd81bb16afc5c5cb1a72eeb16dd37e8024a/zstandard-0.25.0-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:f27662e4f7dbf9f9c12391cb37b4c4c3cb90ffbd3b1fb9284dadbbb8935fa708", size = 5433517, upload-time = "2025-09-14T22:17:16.103Z" }, + { url = "https://files.pythonhosted.org/packages/5b/60/dd0f8cfa8129c5a0ce3ea6b7f70be5b33d2618013a161e1ff26c2b39787c/zstandard-0.25.0-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:99c0c846e6e61718715a3c9437ccc625de26593fea60189567f0118dc9db7512", size = 5814292, upload-time = "2025-09-14T22:17:17.827Z" }, + { url = "https://files.pythonhosted.org/packages/fc/5f/75aafd4b9d11b5407b641b8e41a57864097663699f23e9ad4dbb91dc6bfe/zstandard-0.25.0-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:474d2596a2dbc241a556e965fb76002c1ce655445e4e3bf38e5477d413165ffa", size = 5360237, upload-time = "2025-09-14T22:17:19.954Z" }, + { url = "https://files.pythonhosted.org/packages/ff/8d/0309daffea4fcac7981021dbf21cdb2e3427a9e76bafbcdbdf5392ff99a4/zstandard-0.25.0-cp312-cp312-win32.whl", hash = "sha256:23ebc8f17a03133b4426bcc04aabd68f8236eb78c3760f12783385171b0fd8bd", size = 436922, upload-time = "2025-09-14T22:17:24.398Z" }, + { url = "https://files.pythonhosted.org/packages/79/3b/fa54d9015f945330510cb5d0b0501e8253c127cca7ebe8ba46a965df18c5/zstandard-0.25.0-cp312-cp312-win_amd64.whl", hash = "sha256:ffef5a74088f1e09947aecf91011136665152e0b4b359c42be3373897fb39b01", size = 506276, upload-time = "2025-09-14T22:17:21.429Z" }, + { url = "https://files.pythonhosted.org/packages/ea/6b/8b51697e5319b1f9ac71087b0af9a40d8a6288ff8025c36486e0c12abcc4/zstandard-0.25.0-cp312-cp312-win_arm64.whl", hash = "sha256:181eb40e0b6a29b3cd2849f825e0fa34397f649170673d385f3598ae17cca2e9", size = 462679, upload-time = "2025-09-14T22:17:23.147Z" }, + { url = "https://files.pythonhosted.org/packages/35/0b/8df9c4ad06af91d39e94fa96cc010a24ac4ef1378d3efab9223cc8593d40/zstandard-0.25.0-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:ec996f12524f88e151c339688c3897194821d7f03081ab35d31d1e12ec975e94", size = 795735, upload-time = "2025-09-14T22:17:26.042Z" }, + { url = "https://files.pythonhosted.org/packages/3f/06/9ae96a3e5dcfd119377ba33d4c42a7d89da1efabd5cb3e366b156c45ff4d/zstandard-0.25.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:a1a4ae2dec3993a32247995bdfe367fc3266da832d82f8438c8570f989753de1", size = 640440, upload-time = "2025-09-14T22:17:27.366Z" }, + { url = "https://files.pythonhosted.org/packages/d9/14/933d27204c2bd404229c69f445862454dcc101cd69ef8c6068f15aaec12c/zstandard-0.25.0-cp313-cp313-manylinux2010_i686.manylinux2014_i686.manylinux_2_12_i686.manylinux_2_17_i686.whl", hash = "sha256:e96594a5537722fdfb79951672a2a63aec5ebfb823e7560586f7484819f2a08f", size = 5343070, upload-time = "2025-09-14T22:17:28.896Z" }, + { url = "https://files.pythonhosted.org/packages/6d/db/ddb11011826ed7db9d0e485d13df79b58586bfdec56e5c84a928a9a78c1c/zstandard-0.25.0-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:bfc4e20784722098822e3eee42b8e576b379ed72cca4a7cb856ae733e62192ea", size = 5063001, upload-time = "2025-09-14T22:17:31.044Z" }, + { url = "https://files.pythonhosted.org/packages/db/00/87466ea3f99599d02a5238498b87bf84a6348290c19571051839ca943777/zstandard-0.25.0-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.whl", hash = "sha256:457ed498fc58cdc12fc48f7950e02740d4f7ae9493dd4ab2168a47c93c31298e", size = 5394120, upload-time = "2025-09-14T22:17:32.711Z" }, + { url = "https://files.pythonhosted.org/packages/2b/95/fc5531d9c618a679a20ff6c29e2b3ef1d1f4ad66c5e161ae6ff847d102a9/zstandard-0.25.0-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.whl", hash = "sha256:fd7a5004eb1980d3cefe26b2685bcb0b17989901a70a1040d1ac86f1d898c551", size = 5451230, upload-time = "2025-09-14T22:17:34.41Z" }, + { url = "https://files.pythonhosted.org/packages/63/4b/e3678b4e776db00f9f7b2fe58e547e8928ef32727d7a1ff01dea010f3f13/zstandard-0.25.0-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:8e735494da3db08694d26480f1493ad2cf86e99bdd53e8e9771b2752a5c0246a", size = 5547173, upload-time = "2025-09-14T22:17:36.084Z" }, + { url = "https://files.pythonhosted.org/packages/4e/d5/ba05ed95c6b8ec30bd468dfeab20589f2cf709b5c940483e31d991f2ca58/zstandard-0.25.0-cp313-cp313-musllinux_1_1_aarch64.whl", hash = "sha256:3a39c94ad7866160a4a46d772e43311a743c316942037671beb264e395bdd611", size = 5046736, upload-time = "2025-09-14T22:17:37.891Z" }, + { url = "https://files.pythonhosted.org/packages/50/d5/870aa06b3a76c73eced65c044b92286a3c4e00554005ff51962deef28e28/zstandard-0.25.0-cp313-cp313-musllinux_1_1_x86_64.whl", hash = "sha256:172de1f06947577d3a3005416977cce6168f2261284c02080e7ad0185faeced3", size = 5576368, upload-time = "2025-09-14T22:17:40.206Z" }, + { url = "https://files.pythonhosted.org/packages/5d/35/398dc2ffc89d304d59bc12f0fdd931b4ce455bddf7038a0a67733a25f550/zstandard-0.25.0-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:3c83b0188c852a47cd13ef3bf9209fb0a77fa5374958b8c53aaa699398c6bd7b", size = 4954022, upload-time = "2025-09-14T22:17:41.879Z" }, + { url = "https://files.pythonhosted.org/packages/9a/5c/36ba1e5507d56d2213202ec2b05e8541734af5f2ce378c5d1ceaf4d88dc4/zstandard-0.25.0-cp313-cp313-musllinux_1_2_i686.whl", hash = "sha256:1673b7199bbe763365b81a4f3252b8e80f44c9e323fc42940dc8843bfeaf9851", size = 5267889, upload-time = "2025-09-14T22:17:43.577Z" }, + { url = "https://files.pythonhosted.org/packages/70/e8/2ec6b6fb7358b2ec0113ae202647ca7c0e9d15b61c005ae5225ad0995df5/zstandard-0.25.0-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:0be7622c37c183406f3dbf0cba104118eb16a4ea7359eeb5752f0794882fc250", size = 5433952, upload-time = "2025-09-14T22:17:45.271Z" }, + { url = "https://files.pythonhosted.org/packages/7b/01/b5f4d4dbc59ef193e870495c6f1275f5b2928e01ff5a81fecb22a06e22fb/zstandard-0.25.0-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:5f5e4c2a23ca271c218ac025bd7d635597048b366d6f31f420aaeb715239fc98", size = 5814054, upload-time = "2025-09-14T22:17:47.08Z" }, + { url = "https://files.pythonhosted.org/packages/b2/e5/fbd822d5c6f427cf158316d012c5a12f233473c2f9c5fe5ab1ae5d21f3d8/zstandard-0.25.0-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:4f187a0bb61b35119d1926aee039524d1f93aaf38a9916b8c4b78ac8514a0aaf", size = 5360113, upload-time = "2025-09-14T22:17:48.893Z" }, + { url = "https://files.pythonhosted.org/packages/8e/e0/69a553d2047f9a2c7347caa225bb3a63b6d7704ad74610cb7823baa08ed7/zstandard-0.25.0-cp313-cp313-win32.whl", hash = "sha256:7030defa83eef3e51ff26f0b7bfb229f0204b66fe18e04359ce3474ac33cbc09", size = 436936, upload-time = "2025-09-14T22:17:52.658Z" }, + { url = "https://files.pythonhosted.org/packages/d9/82/b9c06c870f3bd8767c201f1edbdf9e8dc34be5b0fbc5682c4f80fe948475/zstandard-0.25.0-cp313-cp313-win_amd64.whl", hash = "sha256:1f830a0dac88719af0ae43b8b2d6aef487d437036468ef3c2ea59c51f9d55fd5", size = 506232, upload-time = "2025-09-14T22:17:50.402Z" }, + { url = "https://files.pythonhosted.org/packages/d4/57/60c3c01243bb81d381c9916e2a6d9e149ab8627c0c7d7abb2d73384b3c0c/zstandard-0.25.0-cp313-cp313-win_arm64.whl", hash = "sha256:85304a43f4d513f5464ceb938aa02c1e78c2943b29f44a750b48b25ac999a049", size = 462671, upload-time = "2025-09-14T22:17:51.533Z" }, + { url = "https://files.pythonhosted.org/packages/3d/5c/f8923b595b55fe49e30612987ad8bf053aef555c14f05bb659dd5dbe3e8a/zstandard-0.25.0-cp314-cp314-macosx_10_13_x86_64.whl", hash = "sha256:e29f0cf06974c899b2c188ef7f783607dbef36da4c242eb6c82dcd8b512855e3", size = 795887, upload-time = "2025-09-14T22:17:54.198Z" }, + { url = "https://files.pythonhosted.org/packages/8d/09/d0a2a14fc3439c5f874042dca72a79c70a532090b7ba0003be73fee37ae2/zstandard-0.25.0-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:05df5136bc5a011f33cd25bc9f506e7426c0c9b3f9954f056831ce68f3b6689f", size = 640658, upload-time = "2025-09-14T22:17:55.423Z" }, + { url = "https://files.pythonhosted.org/packages/5d/7c/8b6b71b1ddd517f68ffb55e10834388d4f793c49c6b83effaaa05785b0b4/zstandard-0.25.0-cp314-cp314-manylinux2010_i686.manylinux_2_12_i686.manylinux_2_28_i686.whl", hash = "sha256:f604efd28f239cc21b3adb53eb061e2a205dc164be408e553b41ba2ffe0ca15c", size = 5379849, upload-time = "2025-09-14T22:17:57.372Z" }, + { url = "https://files.pythonhosted.org/packages/a4/86/a48e56320d0a17189ab7a42645387334fba2200e904ee47fc5a26c1fd8ca/zstandard-0.25.0-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:223415140608d0f0da010499eaa8ccdb9af210a543fac54bce15babbcfc78439", size = 5058095, upload-time = "2025-09-14T22:17:59.498Z" }, + { url = "https://files.pythonhosted.org/packages/f8/ad/eb659984ee2c0a779f9d06dbfe45e2dc39d99ff40a319895df2d3d9a48e5/zstandard-0.25.0-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:2e54296a283f3ab5a26fc9b8b5d4978ea0532f37b231644f367aa588930aa043", size = 5551751, upload-time = "2025-09-14T22:18:01.618Z" }, + { url = "https://files.pythonhosted.org/packages/61/b3/b637faea43677eb7bd42ab204dfb7053bd5c4582bfe6b1baefa80ac0c47b/zstandard-0.25.0-cp314-cp314-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:ca54090275939dc8ec5dea2d2afb400e0f83444b2fc24e07df7fdef677110859", size = 6364818, upload-time = "2025-09-14T22:18:03.769Z" }, + { url = "https://files.pythonhosted.org/packages/31/dc/cc50210e11e465c975462439a492516a73300ab8caa8f5e0902544fd748b/zstandard-0.25.0-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:e09bb6252b6476d8d56100e8147b803befa9a12cea144bbe629dd508800d1ad0", size = 5560402, upload-time = "2025-09-14T22:18:05.954Z" }, + { url = "https://files.pythonhosted.org/packages/c9/ae/56523ae9c142f0c08efd5e868a6da613ae76614eca1305259c3bf6a0ed43/zstandard-0.25.0-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:a9ec8c642d1ec73287ae3e726792dd86c96f5681eb8df274a757bf62b750eae7", size = 4955108, upload-time = "2025-09-14T22:18:07.68Z" }, + { url = "https://files.pythonhosted.org/packages/98/cf/c899f2d6df0840d5e384cf4c4121458c72802e8bda19691f3b16619f51e9/zstandard-0.25.0-cp314-cp314-musllinux_1_2_i686.whl", hash = "sha256:a4089a10e598eae6393756b036e0f419e8c1d60f44a831520f9af41c14216cf2", size = 5269248, upload-time = "2025-09-14T22:18:09.753Z" }, + { url = "https://files.pythonhosted.org/packages/1b/c0/59e912a531d91e1c192d3085fc0f6fb2852753c301a812d856d857ea03c6/zstandard-0.25.0-cp314-cp314-musllinux_1_2_ppc64le.whl", hash = "sha256:f67e8f1a324a900e75b5e28ffb152bcac9fbed1cc7b43f99cd90f395c4375344", size = 5430330, upload-time = "2025-09-14T22:18:11.966Z" }, + { url = "https://files.pythonhosted.org/packages/a0/1d/7e31db1240de2df22a58e2ea9a93fc6e38cc29353e660c0272b6735d6669/zstandard-0.25.0-cp314-cp314-musllinux_1_2_s390x.whl", hash = "sha256:9654dbc012d8b06fc3d19cc825af3f7bf8ae242226df5f83936cb39f5fdc846c", size = 5811123, upload-time = "2025-09-14T22:18:13.907Z" }, + { url = "https://files.pythonhosted.org/packages/f6/49/fac46df5ad353d50535e118d6983069df68ca5908d4d65b8c466150a4ff1/zstandard-0.25.0-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:4203ce3b31aec23012d3a4cf4a2ed64d12fea5269c49aed5e4c3611b938e4088", size = 5359591, upload-time = "2025-09-14T22:18:16.465Z" }, + { url = "https://files.pythonhosted.org/packages/c2/38/f249a2050ad1eea0bb364046153942e34abba95dd5520af199aed86fbb49/zstandard-0.25.0-cp314-cp314-win32.whl", hash = "sha256:da469dc041701583e34de852d8634703550348d5822e66a0c827d39b05365b12", size = 444513, upload-time = "2025-09-14T22:18:20.61Z" }, + { url = "https://files.pythonhosted.org/packages/3a/43/241f9615bcf8ba8903b3f0432da069e857fc4fd1783bd26183db53c4804b/zstandard-0.25.0-cp314-cp314-win_amd64.whl", hash = "sha256:c19bcdd826e95671065f8692b5a4aa95c52dc7a02a4c5a0cac46deb879a017a2", size = 516118, upload-time = "2025-09-14T22:18:17.849Z" }, + { url = "https://files.pythonhosted.org/packages/f0/ef/da163ce2450ed4febf6467d77ccb4cd52c4c30ab45624bad26ca0a27260c/zstandard-0.25.0-cp314-cp314-win_arm64.whl", hash = "sha256:d7541afd73985c630bafcd6338d2518ae96060075f9463d7dc14cfb33514383d", size = 476940, upload-time = "2025-09-14T22:18:19.088Z" }, +]