Skip to content

pudl_diff.cli

CLI for comparing tables between two PUDL Parquet datasets.

main(ctx, table_names, left, right, right_table, output_path, max_compare_rows, max_output_rows, rtol, atol, from_report, color, verbose, loglevel)

Compare tables between two PUDL Parquet datasets.

Compares each of the given TABLE_NAMES. If none are given, compares every table that has a Parquet file in both datasets. Tables are compared one at a time in a single process (so PUDL is only imported once). When comparing more than one, prints a summary at the end, which for all tables also lists those present in only one dataset (they aren't compared).

Writes a single JSON report covering every table (and, for differing tables, a pair of Parquet side-output files holding the differing rows) to --output-path, then exits 0 if all the tables are functionally identical, 1 if any differ, or 2 if any comparison itself failed (e.g. a table doesn't exist in one of the datasets, or a dataset's datapackage.json couldn't be read). A table that fails doesn't stop the others being compared.

With --from-report, instead shows an existing report in the same form, and exits with the code that comparison did.

Source code in src/pudl_diff/cli.py
@click.command(context_settings={"help_option_names": ["-h", "--help"]}, epilog=_EPILOG)
@click.version_option(__version__, "--version", prog_name="pudl_diff")
@click.argument("table_names", type=str, nargs=-1)
@click.option(
    "-l",
    "--left",
    type=str,
    default=None,
    help="Root path of the 'left' dataset (local path or URL, e.g. s3://...). "
    "Defaults to PUDL's nightly build outputs on S3, the reference point most "
    "diffs are measured against.",
)
@click.option(
    "-r",
    "--right",
    type=str,
    default=None,
    help="Root path of the 'right' dataset (local path or URL). Defaults to "
    "$PUDL_OUTPUT/parquet, so the diff reads as what's changed locally since "
    "the last nightly build.",
)
@click.option(
    "--right-table",
    type=str,
    default=None,
    help="Name of the table to compare in the right dataset, if it differs "
    "from the table name given - e.g. comparing a core_* table against the "
    "out_* table built from it. Defaults to the same name. Requires exactly "
    "one TABLE_NAME.",
)
@click.option(
    "-o",
    "--output-path",
    type=click.Path(file_okay=False, path_type=Path),
    default=None,
    help="Directory to write the JSON report (pudl_diff_report.json) and Parquet "
    "outputs into. "
    "Created if it doesn't exist. Defaults to the current working directory.",
)
@click.option(
    "--max-compare-rows",
    type=click.IntRange(min=0),
    default=MAX_COMPARE_ROWS,
    show_default=True,
    help="Skip row-level comparison, keeping the cheaper schema and "
    "row-count comparisons, whenever either table has more rows than this. "
    "The largest tables take minutes to compare, hence the default; raise it, "
    "or give 0 for no limit, to compare them too.",
)
@click.option(
    "--max-output-rows",
    type=int,
    default=None,
    help="Cap the number of rows written to each Parquet side-output file. "
    "Defaults to writing every differing row.",
)
@click.option(
    "--rtol",
    type=float,
    default=1e-5,
    show_default=True,
    help="Relative tolerance for float equality, matching numpy.isclose.",
)
@click.option(
    "--atol",
    type=float,
    default=1e-8,
    show_default=True,
    help="Absolute tolerance for float equality, matching numpy.isclose.",
)
@click.option(
    "--from-report",
    type=str,
    default=None,
    help="Show a saved JSON report (or the pudl_diff_report.json in a directory), "
    "at a local path or a remote URL such as gs://... or s3://..., "
    "as the tables and summary a new comparison would print, without comparing "
    "anything. Can't be combined with the options that control a comparison.",
)
@click.option(
    "--color/--no-color",
    default=None,
    help="Colorize the output. Defaults to on if stdout is a terminal, and off "
    "otherwise (e.g. when piped to a file).",
)
@click.option(
    "--verbose/--quiet",
    "verbose",
    default=False,
    help="With --quiet (the default), the live table only lists tables that aren't "
    "identical, and leaves out the size columns. --verbose lists every table, "
    "with sizes. The JSON report and final summary are the same either way.",
)
@click.option(
    "--loglevel",
    default="ERROR",
    type=click.Choice(
        ["DEBUG", "INFO", "WARNING", "ERROR", "CRITICAL"], case_sensitive=False
    ),
    show_default=True,
    help="Only show log messages at least this severe, so that they don't "
    "interrupt the report. Skipped comparisons and errors are still recorded in "
    "the JSON report.",
)
@click.pass_context
def main(
    ctx: click.Context,
    table_names: tuple[str, ...],
    left: str | None,
    right: str | None,
    right_table: str | None,
    output_path: Path | None,
    max_compare_rows: int,
    max_output_rows: int | None,
    rtol: float,
    atol: float,
    from_report: str | None,
    color: bool | None,
    verbose: bool,
    loglevel: str,
) -> None:
    """Compare tables between two PUDL Parquet datasets.

    Compares each of the given TABLE_NAMES. If none are given, compares every
    table that has a Parquet file in both datasets. Tables are compared one at a
    time in a single process (so PUDL is only imported once). When comparing
    more than one, prints a summary at the end, which for all tables also lists
    those present in only one dataset (they aren't compared).

    Writes a single JSON report covering every table (and, for differing tables,
    a pair of Parquet side-output files holding the differing rows) to
    --output-path, then exits 0 if all the tables are functionally identical,
    1 if any differ, or 2 if any comparison itself failed (e.g. a table doesn't
    exist in one of the datasets, or a dataset's datapackage.json couldn't be
    read). A table that fails doesn't stop the others being compared.

    With --from-report, instead shows an existing report in the same form, and
    exits with the code that comparison did.
    """
    if from_report is not None:
        _reject_comparison_options(ctx)
        ctx.call_on_close(_set_log_level(loglevel))
        ctx.color = sys.stdout.isatty() if color is None else color
        ctx.exit(_show_saved_report(from_report, verbose=verbose))
    if right_table is not None and len(table_names) != 1:
        raise click.UsageError("--right-table requires exactly one TABLE_NAME.")

    ctx.call_on_close(_set_log_level(loglevel))
    # click.echo consults the context's color setting, so this covers all output.
    ctx.color = sys.stdout.isatty() if color is None else color
    try:
        left_root = left or str(nightly_root())
        right_root = right or str(default_right_root())
    except RuntimeError as e:
        raise click.UsageError(str(e)) from e
    output_path = output_path or Path.cwd()
    report_path = output_path / REPORT_FILENAME

    left_dataset = PudlDiffDataset(left_root)
    right_dataset = PudlDiffDataset(right_root)
    options = table_report.DiffOptions(
        rtol=rtol,
        atol=atol,
        max_compare_rows=max_compare_rows or None,
        max_output_rows=max_output_rows,
    )

    def write_report(report: PudlDiffReport) -> None:
        output_path.mkdir(parents=True, exist_ok=True)
        report_path.write_text(report.model_dump_json(indent=2), encoding="utf-8")

    single = len(table_names) == 1
    progress = TerminalProgress(
        left_root,
        right_root,
        explicit=bool(table_names),
        show_progress=not single,
        verbose=verbose,
    )
    report = run_dataset_diff(
        left_dataset,
        right_dataset,
        output_path,
        table_names=table_names,
        right_table=right_table,
        options=options,
        on_tables_resolved=progress.tables_resolved,
        on_table_compared=progress.table_compared,
    )
    write_report(report)
    _echo_outcome(report, progress, report_path, single=single, saved=True)
    ctx.exit(report.exit_code)