Skip to content

pudl_diff.dataset_report

The dataset-level report, summarizing all the tables compared.

REPORT_FILENAME = 'pudl_diff_report.json' module-attribute

Name of the JSON report, written to the output directory.

REPORT_SCHEMA_VERSION = '1.1.0' module-attribute

Version of the JSON report format written by build_pudl_diff_report().

DatasetInfo

Bases: DatasetProvenance

One of the two compared datasets: where it is, and where it came from.

Source code in src/pudl_diff/dataset_report.py
class DatasetInfo(DatasetProvenance):
    """One of the two compared datasets: where it is, and where it came from."""

    root: str
    """The root path or URL of the dataset's Parquet files. For a dataset on the
    local filesystem, an absolute path with any symlinks resolved, so it doesn't
    depend on the directory the comparison was run from."""

root instance-attribute

The root path or URL of the dataset's Parquet files. For a dataset on the local filesystem, an absolute path with any symlinks resolved, so it doesn't depend on the directory the comparison was run from.

PudlDiffReport

Bases: ReportModel

The full comparison of two PUDL datasets: the saved JSON report.

Built by build_pudl_diff_report(). Holds everything that pertains to the comparison as a whole, plus a TableDiffReport for each table.

Source code in src/pudl_diff/dataset_report.py
class PudlDiffReport(ReportModel):
    """The full comparison of two PUDL datasets: the saved JSON report.

    Built by `build_pudl_diff_report()`. Holds everything that pertains to the
    comparison as a whole, plus a `TableDiffReport` for each table.
    """

    schema_version: str = REPORT_SCHEMA_VERSION
    """Version of this report format, in `major.minor.patch` form."""
    pudl_diff_version: str
    """Version of the `pudl_diff` package that made this report, e.g. `0.1.0`, or for
    a development build, a version that says which commit it was built from. Unlike
    `schema_version`, which only changes when the report's format does, this changes
    with every release, so it records exactly which code produced the report. It has
    no default, so a report never claims to be from a version that didn't write it."""
    created: str
    """UTC ISO-8601 timestamp of when this report was generated."""
    elapsed_seconds: float | None = None
    """Wall-clock time the whole comparison took."""
    left_dataset: DatasetInfo
    """The reference dataset, e.g. the last nightly build. Additions, removals and
    changes are all measured from it to the right dataset."""
    right_dataset: DatasetInfo
    """The dataset that was compared against the left one, e.g. a local build."""
    options: DiffOptions
    """The settings the comparison was run with."""
    tables_only_in_left: list[str]
    """Tables found only in the left dataset, which aren't compared."""
    tables_only_in_right: list[str]
    """Tables found only in the right dataset, which aren't compared."""
    summary: PudlDiffSummary
    """Totals over all the compared tables."""
    tables: dict[str, TableDiffReport]
    """Each compared table's report, keyed by its name in the left dataset."""

    error: str | None = None
    """Why the comparison as a whole failed, e.g. no tables could be listed. This
    is `None` when the only failures are of individual tables, which each
    record their own `error`."""

    @pydantic.computed_field
    @property
    def success(self) -> bool:
        """Whether the comparison completed.

        That is, `error` is `None` and so is every table's. Distinct from
        `is_identical`: a comparison can succeed and still find differences.
        """
        return self.error is None and all(t.success for t in self.tables.values())

    @pydantic.computed_field
    @property
    def is_identical(self) -> bool:
        """Whether every compared table is identical, and `success` is `True`.

        Tables found in only one dataset don't count against this.
        """
        return self.success and all(t.is_identical for t in self.tables.values())

    @property
    def exit_code(self) -> int:
        """The CLI's exit status for this report.

        `0` if everything is identical, `1` if any differ, `2` if any
        comparison failed.
        """
        if not self.success:
            return 2
        return 0 if self.is_identical else 1

created instance-attribute

UTC ISO-8601 timestamp of when this report was generated.

elapsed_seconds = None class-attribute instance-attribute

Wall-clock time the whole comparison took.

error = None class-attribute instance-attribute

Why the comparison as a whole failed, e.g. no tables could be listed. This is None when the only failures are of individual tables, which each record their own error.

exit_code property

The CLI's exit status for this report.

0 if everything is identical, 1 if any differ, 2 if any comparison failed.

is_identical property

Whether every compared table is identical, and success is True.

Tables found in only one dataset don't count against this.

left_dataset instance-attribute

The reference dataset, e.g. the last nightly build. Additions, removals and changes are all measured from it to the right dataset.

options instance-attribute

The settings the comparison was run with.

pudl_diff_version instance-attribute

Version of the pudl_diff package that made this report, e.g. 0.1.0, or for a development build, a version that says which commit it was built from. Unlike schema_version, which only changes when the report's format does, this changes with every release, so it records exactly which code produced the report. It has no default, so a report never claims to be from a version that didn't write it.

right_dataset instance-attribute

The dataset that was compared against the left one, e.g. a local build.

schema_version = REPORT_SCHEMA_VERSION class-attribute instance-attribute

Version of this report format, in major.minor.patch form.

success property

Whether the comparison completed.

That is, error is None and so is every table's. Distinct from is_identical: a comparison can succeed and still find differences.

summary instance-attribute

Totals over all the compared tables.

tables instance-attribute

Each compared table's report, keyed by its name in the left dataset.

tables_only_in_left instance-attribute

Tables found only in the left dataset, which aren't compared.

tables_only_in_right instance-attribute

Tables found only in the right dataset, which aren't compared.

PudlDiffSummary

Bases: SizeComparison

Totals over every table in a PudlDiffReport.

Saves consumers from aggregating the tables themselves.

The size fields (see SizeComparison) total only the tables whose size is known on both sides.

Source code in src/pudl_diff/dataset_report.py
class PudlDiffSummary(SizeComparison):
    """Totals over every table in a `PudlDiffReport`.

    Saves consumers from aggregating the tables themselves.

    The size fields (see `SizeComparison`) total only the tables whose size
    is known on both sides.
    """

    table_count: int
    """Number of tables compared, including any whose comparison failed."""
    identical_table_count: int
    """Tables whose comparison completed and found no differences."""
    changed_table_count: int
    """Tables whose comparison completed and found differences."""
    failed_table_count: int
    """Tables whose comparison failed to complete."""
    failed_tables: list[str]
    """The names of the tables whose comparison failed."""
    schema_changed_tables: list[str]
    """Tables with columns added or removed, or with changed dtypes."""

    left_row_count: int
    """Total rows in the left tables, over all tables that could be counted."""
    right_row_count: int
    """Total rows in the right tables, over all tables that could be counted."""
    rows_added: int
    """Rows only in the right table, summed over tables with a row-level
    comparison."""
    rows_changed: int
    """Rows with the same primary key but changed values, summed over tables
    with a row-level comparison and a primary key."""
    rows_removed: int
    """Rows only in the left table, summed over tables with a row-level
    comparison."""
    no_row_diff_table_count: int
    """Tables with no row-level comparison, whether skipped or failed. Their rows
    count towards the row totals, but not the rows added, changed or removed."""
    no_row_diff_left_row_count: int
    """Total rows in the left side of those tables."""

    columns_added: int
    """Columns only in the right table, summed over all the tables."""
    columns_changed: int
    """Shared columns whose dtype changed, summed over all the tables."""
    columns_removed: int
    """Columns only in the left table, summed over all the tables."""

    peak_rss_bytes: int | None = None
    """The highest `peak_rss_bytes` of any table."""
    peak_rss_table: str | None = None
    """The table with that peak memory use."""

    @pydantic.computed_field
    @property
    def peak_rss(self) -> str | None:
        """`peak_rss_bytes` in human-readable form, e.g. `1.2 GB`."""
        if self.peak_rss_bytes is None:
            return None
        return format_bytes(self.peak_rss_bytes)

    @classmethod
    def from_tables(cls, tables: dict[str, TableDiffReport]) -> PudlDiffSummary:
        """Total up the reports of individual tables, keyed by table name."""
        reports = list(tables.values())
        failed = [name for name, report in tables.items() if not report.success]
        changed = [
            name
            for name, report in tables.items()
            if report.success and not report.is_identical
        ]

        row_counts = {
            name: r.row_count_diff
            for name, r in tables.items()
            if r.row_count_diff is not None
        }
        counted = list(row_counts.values())
        changes = {
            name: RowChanges.from_summary(r.row_diff) for name, r in tables.items()
        }
        compared = [
            c for c in changes.values() if c.added is not None and c.removed is not None
        ]
        no_row_diff = [
            name for name, c in changes.items() if c.added is None or c.removed is None
        ]

        schemas = [r.schema_diff for r in reports if r.schema_diff is not None]
        sized = [
            r
            for r in reports
            if r.left_table_bytes is not None and r.right_table_bytes is not None
        ]
        measured = {
            name: report.peak_rss_bytes
            for name, report in tables.items()
            if report.peak_rss_bytes is not None
        }
        peak_table = max(measured, key=measured.__getitem__) if measured else None

        return cls(
            table_count=len(tables),
            identical_table_count=len(tables) - len(failed) - len(changed),
            changed_table_count=len(changed),
            failed_table_count=len(failed),
            failed_tables=failed,
            schema_changed_tables=[
                name
                for name, report in tables.items()
                if report.schema_diff is not None
                and not report.schema_diff.is_identical
            ],
            left_row_count=sum(c.left_row_count for c in counted),
            right_row_count=sum(c.right_row_count for c in counted),
            rows_added=sum(c.added or 0 for c in compared),
            rows_changed=sum(c.changed or 0 for c in compared),
            rows_removed=sum(c.removed or 0 for c in compared),
            no_row_diff_table_count=len(no_row_diff),
            no_row_diff_left_row_count=sum(
                row_counts[name].left_row_count
                for name in no_row_diff
                if name in row_counts
            ),
            columns_added=sum(len(s.columns_only_in_right) for s in schemas),
            columns_changed=sum(len(s.dtype_changes) for s in schemas),
            columns_removed=sum(len(s.columns_only_in_left) for s in schemas),
            left_table_bytes=(
                sum(r.left_table_bytes or 0 for r in sized) if sized else None
            ),
            right_table_bytes=(
                sum(r.right_table_bytes or 0 for r in sized) if sized else None
            ),
            peak_rss_bytes=measured[peak_table] if peak_table is not None else None,
            peak_rss_table=peak_table,
        )

changed_table_count instance-attribute

Tables whose comparison completed and found differences.

columns_added instance-attribute

Columns only in the right table, summed over all the tables.

columns_changed instance-attribute

Shared columns whose dtype changed, summed over all the tables.

columns_removed instance-attribute

Columns only in the left table, summed over all the tables.

failed_table_count instance-attribute

Tables whose comparison failed to complete.

failed_tables instance-attribute

The names of the tables whose comparison failed.

identical_table_count instance-attribute

Tables whose comparison completed and found no differences.

left_row_count instance-attribute

Total rows in the left tables, over all tables that could be counted.

no_row_diff_left_row_count instance-attribute

Total rows in the left side of those tables.

no_row_diff_table_count instance-attribute

Tables with no row-level comparison, whether skipped or failed. Their rows count towards the row totals, but not the rows added, changed or removed.

peak_rss property

peak_rss_bytes in human-readable form, e.g. 1.2 GB.

peak_rss_bytes = None class-attribute instance-attribute

The highest peak_rss_bytes of any table.

peak_rss_table = None class-attribute instance-attribute

The table with that peak memory use.

right_row_count instance-attribute

Total rows in the right tables, over all tables that could be counted.

rows_added instance-attribute

Rows only in the right table, summed over tables with a row-level comparison.

rows_changed instance-attribute

Rows with the same primary key but changed values, summed over tables with a row-level comparison and a primary key.

rows_removed instance-attribute

Rows only in the left table, summed over tables with a row-level comparison.

schema_changed_tables instance-attribute

Tables with columns added or removed, or with changed dtypes.

table_count instance-attribute

Number of tables compared, including any whose comparison failed.

from_tables(tables) classmethod

Total up the reports of individual tables, keyed by table name.

Source code in src/pudl_diff/dataset_report.py
@classmethod
def from_tables(cls, tables: dict[str, TableDiffReport]) -> PudlDiffSummary:
    """Total up the reports of individual tables, keyed by table name."""
    reports = list(tables.values())
    failed = [name for name, report in tables.items() if not report.success]
    changed = [
        name
        for name, report in tables.items()
        if report.success and not report.is_identical
    ]

    row_counts = {
        name: r.row_count_diff
        for name, r in tables.items()
        if r.row_count_diff is not None
    }
    counted = list(row_counts.values())
    changes = {
        name: RowChanges.from_summary(r.row_diff) for name, r in tables.items()
    }
    compared = [
        c for c in changes.values() if c.added is not None and c.removed is not None
    ]
    no_row_diff = [
        name for name, c in changes.items() if c.added is None or c.removed is None
    ]

    schemas = [r.schema_diff for r in reports if r.schema_diff is not None]
    sized = [
        r
        for r in reports
        if r.left_table_bytes is not None and r.right_table_bytes is not None
    ]
    measured = {
        name: report.peak_rss_bytes
        for name, report in tables.items()
        if report.peak_rss_bytes is not None
    }
    peak_table = max(measured, key=measured.__getitem__) if measured else None

    return cls(
        table_count=len(tables),
        identical_table_count=len(tables) - len(failed) - len(changed),
        changed_table_count=len(changed),
        failed_table_count=len(failed),
        failed_tables=failed,
        schema_changed_tables=[
            name
            for name, report in tables.items()
            if report.schema_diff is not None
            and not report.schema_diff.is_identical
        ],
        left_row_count=sum(c.left_row_count for c in counted),
        right_row_count=sum(c.right_row_count for c in counted),
        rows_added=sum(c.added or 0 for c in compared),
        rows_changed=sum(c.changed or 0 for c in compared),
        rows_removed=sum(c.removed or 0 for c in compared),
        no_row_diff_table_count=len(no_row_diff),
        no_row_diff_left_row_count=sum(
            row_counts[name].left_row_count
            for name in no_row_diff
            if name in row_counts
        ),
        columns_added=sum(len(s.columns_only_in_right) for s in schemas),
        columns_changed=sum(len(s.dtype_changes) for s in schemas),
        columns_removed=sum(len(s.columns_only_in_left) for s in schemas),
        left_table_bytes=(
            sum(r.left_table_bytes or 0 for r in sized) if sized else None
        ),
        right_table_bytes=(
            sum(r.right_table_bytes or 0 for r in sized) if sized else None
        ),
        peak_rss_bytes=measured[peak_table] if peak_table is not None else None,
        peak_rss_table=peak_table,
    )

TableOutcome dataclass

What happened when comparing one table, for display.

Source code in src/pudl_diff/dataset_report.py
@dataclass(frozen=True)
class TableOutcome:
    """What happened when comparing one table, for display."""

    table_name: str
    exit_code: int
    """`0` if identical, `1` if different, `2` if the comparison failed."""
    elapsed_seconds: float | None
    error: str | None
    rows: RowChanges
    sizes: SizeComparison
    left_rows: int | None = None
    right_rows: int | None = None
    left_columns: int | None = None
    columns_added: int | None = None
    """Columns only in the right table. `None` if the comparison failed."""
    columns_removed: int | None = None
    """Columns only in the left table."""
    dtypes_changed: int = 0
    """Number of shared columns whose dtype differs between the tables."""
    peak_rss_bytes: int | None = None

columns_added = None class-attribute instance-attribute

Columns only in the right table. None if the comparison failed.

columns_removed = None class-attribute instance-attribute

Columns only in the left table.

dtypes_changed = 0 class-attribute instance-attribute

Number of shared columns whose dtype differs between the tables.

exit_code instance-attribute

0 if identical, 1 if different, 2 if the comparison failed.

build_pudl_diff_report(left, right, tables, *, options=None, tables_only_in_left=(), tables_only_in_right=(), elapsed_seconds=None, error=None)

Assemble the report on a comparison of two whole datasets.

Parameters:

Name Type Description Default
left PudlDiffDataset

The "left" dataset that was compared.

required
right PudlDiffDataset

The "right" dataset compared against it.

required
tables dict[str, TableDiffReport]

The report on each compared table, keyed by its name in left.

required
options DiffOptions | None

The settings the comparison ran with.

None
tables_only_in_left Iterable[str]

Tables that weren't compared as they aren't in right.

()
tables_only_in_right Iterable[str]

Likewise, for tables not in left.

()
elapsed_seconds float | None

How long the whole comparison took.

None
error str | None

Why the comparison failed as a whole, if it did - e.g. because no tables could be found to compare.

None
Source code in src/pudl_diff/dataset_report.py
def build_pudl_diff_report(
    left: PudlDiffDataset,
    right: PudlDiffDataset,
    tables: dict[str, TableDiffReport],
    *,
    options: DiffOptions | None = None,
    tables_only_in_left: Iterable[str] = (),
    tables_only_in_right: Iterable[str] = (),
    elapsed_seconds: float | None = None,
    error: str | None = None,
) -> PudlDiffReport:
    """Assemble the report on a comparison of two whole datasets.

    Args:
        left: The "left" dataset that was compared.
        right: The "right" dataset compared against it.
        tables: The report on each compared table, keyed by its name in `left`.
        options: The settings the comparison ran with.
        tables_only_in_left: Tables that weren't compared as they aren't in
            `right`.
        tables_only_in_right: Likewise, for tables not in `left`.
        elapsed_seconds: How long the whole comparison took.
        error: Why the comparison failed as a whole, if it did - e.g. because
            no tables could be found to compare.
    """

    def _info(dataset: PudlDiffDataset) -> DatasetInfo:
        try:
            provenance = dataset.provenance()
        except OSError, json.JSONDecodeError:
            provenance = DatasetProvenance()
        return DatasetInfo(root=dataset.display_root, **provenance.model_dump())

    return PudlDiffReport(
        pudl_diff_version=__version__,
        created=datetime.now(UTC).isoformat(),
        elapsed_seconds=elapsed_seconds,
        left_dataset=_info(left),
        right_dataset=_info(right),
        options=options or DiffOptions(),
        tables_only_in_left=sorted(tables_only_in_left),
        tables_only_in_right=sorted(tables_only_in_right),
        summary=PudlDiffSummary.from_tables(tables),
        tables=tables,
        error=error,
    )

table_outcome(table_name, report)

Boil a table's report down to what we display.

Source code in src/pudl_diff/dataset_report.py
def table_outcome(table_name: str, report: TableDiffReport) -> TableOutcome:
    """Boil a table's report down to what we display."""
    exit_code = 0 if report.is_identical else 1
    if not report.success:
        exit_code = 2
    row_counts = report.row_count_diff
    schema = report.schema_diff
    return TableOutcome(
        table_name=table_name,
        exit_code=exit_code,
        elapsed_seconds=report.elapsed_seconds,
        error=report.error,
        rows=RowChanges.from_summary(report.row_diff),
        sizes=report,
        left_rows=row_counts.left_row_count if row_counts else None,
        right_rows=row_counts.right_row_count if row_counts else None,
        left_columns=schema.left_column_count if schema else None,
        columns_added=len(schema.columns_only_in_right) if schema else None,
        columns_removed=len(schema.columns_only_in_left) if schema else None,
        dtypes_changed=len(schema.dtype_changes) if schema else 0,
        peak_rss_bytes=report.peak_rss_bytes,
    )