@click.command(context_settings={"help_option_names": ["-h", "--help"]}, epilog=_EPILOG)
@click.version_option(__version__, "--version", prog_name="pudl_diff")
@click.argument("table_names", type=str, nargs=-1)
@click.option(
"-l",
"--left",
type=str,
default=None,
help="Root path of the 'left' dataset (local path or URL, e.g. s3://...). "
"Defaults to PUDL's nightly build outputs on S3, the reference point most "
"diffs are measured against.",
)
@click.option(
"-r",
"--right",
type=str,
default=None,
help="Root path of the 'right' dataset (local path or URL). Defaults to "
"$PUDL_OUTPUT/parquet, so the diff reads as what's changed locally since "
"the last nightly build.",
)
@click.option(
"--right-table",
type=str,
default=None,
help="Name of the table to compare in the right dataset, if it differs "
"from the table name given - e.g. comparing a core_* table against the "
"out_* table built from it. Defaults to the same name. Requires exactly "
"one TABLE_NAME.",
)
@click.option(
"-o",
"--output-path",
type=click.Path(file_okay=False, path_type=Path),
default=None,
help="Directory to write the JSON report (pudl_diff_report.json) and Parquet "
"outputs into. "
"Created if it doesn't exist. Defaults to the current working directory.",
)
@click.option(
"--max-compare-rows",
type=click.IntRange(min=0),
default=MAX_COMPARE_ROWS,
show_default=True,
help="Skip row-level comparison, keeping the cheaper schema and "
"row-count comparisons, whenever either table has more rows than this. "
"The largest tables take minutes to compare, hence the default; raise it, "
"or give 0 for no limit, to compare them too.",
)
@click.option(
"--max-output-rows",
type=int,
default=None,
help="Cap the number of rows written to each Parquet side-output file. "
"Defaults to writing every differing row.",
)
@click.option(
"--rtol",
type=float,
default=1e-5,
show_default=True,
help="Relative tolerance for float equality, matching numpy.isclose.",
)
@click.option(
"--atol",
type=float,
default=1e-8,
show_default=True,
help="Absolute tolerance for float equality, matching numpy.isclose.",
)
@click.option(
"--from-report",
type=str,
default=None,
help="Show a saved JSON report (or the pudl_diff_report.json in a directory), "
"at a local path or a remote URL such as gs://... or s3://..., "
"as the tables and summary a new comparison would print, without comparing "
"anything. Can't be combined with the options that control a comparison.",
)
@click.option(
"--color/--no-color",
default=None,
help="Colorize the output. Defaults to on if stdout is a terminal, and off "
"otherwise (e.g. when piped to a file).",
)
@click.option(
"--verbose/--quiet",
"verbose",
default=False,
help="With --quiet (the default), the live table only lists tables that aren't "
"identical, and leaves out the size columns. --verbose lists every table, "
"with sizes. The JSON report and final summary are the same either way.",
)
@click.option(
"--loglevel",
default="ERROR",
type=click.Choice(
["DEBUG", "INFO", "WARNING", "ERROR", "CRITICAL"], case_sensitive=False
),
show_default=True,
help="Only show log messages at least this severe, so that they don't "
"interrupt the report. Skipped comparisons and errors are still recorded in "
"the JSON report.",
)
@click.pass_context
def main(
ctx: click.Context,
table_names: tuple[str, ...],
left: str | None,
right: str | None,
right_table: str | None,
output_path: Path | None,
max_compare_rows: int,
max_output_rows: int | None,
rtol: float,
atol: float,
from_report: str | None,
color: bool | None,
verbose: bool,
loglevel: str,
) -> None:
"""Compare tables between two PUDL Parquet datasets.
Compares each of the given TABLE_NAMES. If none are given, compares every
table that has a Parquet file in both datasets. Tables are compared one at a
time in a single process (so PUDL is only imported once). When comparing
more than one, prints a summary at the end, which for all tables also lists
those present in only one dataset (they aren't compared).
Writes a single JSON report covering every table (and, for differing tables,
a pair of Parquet side-output files holding the differing rows) to
--output-path, then exits 0 if all the tables are functionally identical,
1 if any differ, or 2 if any comparison itself failed (e.g. a table doesn't
exist in one of the datasets, or a dataset's datapackage.json couldn't be
read). A table that fails doesn't stop the others being compared.
With --from-report, instead shows an existing report in the same form, and
exits with the code that comparison did.
"""
if from_report is not None:
_reject_comparison_options(ctx)
ctx.call_on_close(_set_log_level(loglevel))
ctx.color = sys.stdout.isatty() if color is None else color
ctx.exit(_show_saved_report(from_report, verbose=verbose))
if right_table is not None and len(table_names) != 1:
raise click.UsageError("--right-table requires exactly one TABLE_NAME.")
ctx.call_on_close(_set_log_level(loglevel))
# click.echo consults the context's color setting, so this covers all output.
ctx.color = sys.stdout.isatty() if color is None else color
try:
left_root = left or str(nightly_root())
right_root = right or str(default_right_root())
except RuntimeError as e:
raise click.UsageError(str(e)) from e
output_path = output_path or Path.cwd()
report_path = output_path / REPORT_FILENAME
left_dataset = PudlDiffDataset(left_root)
right_dataset = PudlDiffDataset(right_root)
options = table_report.DiffOptions(
rtol=rtol,
atol=atol,
max_compare_rows=max_compare_rows or None,
max_output_rows=max_output_rows,
)
def write_report(report: PudlDiffReport) -> None:
output_path.mkdir(parents=True, exist_ok=True)
report_path.write_text(report.model_dump_json(indent=2), encoding="utf-8")
single = len(table_names) == 1
progress = TerminalProgress(
left_root,
right_root,
explicit=bool(table_names),
show_progress=not single,
verbose=verbose,
)
report = run_dataset_diff(
left_dataset,
right_dataset,
output_path,
table_names=table_names,
right_table=right_table,
options=options,
on_tables_resolved=progress.tables_resolved,
on_table_compared=progress.table_compared,
)
write_report(report)
_echo_outcome(report, progress, report_path, single=single, saved=True)
ctx.exit(report.exit_code)