diff --git a/src/pytest_regressions/dataframe_regression.py b/src/pytest_regressions/dataframe_regression.py index 487457f..85e23c1 100644 --- a/src/pytest_regressions/dataframe_regression.py +++ b/src/pytest_regressions/dataframe_regression.py @@ -1,3 +1,4 @@ +import csv import os from pathlib import Path from typing import Any @@ -93,7 +94,15 @@ def _check_data_shapes(self, obtained_column: Any, expected_column: Any) -> None ) raise AssertionError(error_msg) - def _check_fn(self, obtained_filename: Path, expected_filename: Path) -> None: + def _check_fn( + self, + obtained_filename: Path, + expected_filename: Path, + *, + column_levels: int = 1, + has_multiindex_columns: bool = False, + has_index_names: bool = False, + ) -> None: """ Check if dict contents dumped to a file match the contents in expected file. """ @@ -108,8 +117,39 @@ def _check_fn(self, obtained_filename: Path, expected_filename: Path) -> None: __tracebackhide__ = True - obtained_data = pd.read_csv(str(obtained_filename)) - expected_data = pd.read_csv(str(expected_filename)) + header = list(range(column_levels)) + read_csv_options: dict[str, Any] = {"header": header} + if has_index_names: + + def read_index_names(filename: Path) -> list[str]: + with filename.open(encoding="utf-8", newline="") as stream: + rows = csv.reader(stream) + for _ in header: + next(rows, None) + names = next(rows, None) + assert names is not None, "Could not find the row index names." + return names + + assert read_index_names(obtained_filename) == read_index_names( + expected_filename + ), "Row index names are not the same." + # MultiIndex CSVs store named row indexes in an extra metadata row. + read_csv_options["skiprows"] = [column_levels] + # The Python parser skips CSV records, including quoted newlines. + read_csv_options["engine"] = "python" + + obtained_data = pd.read_csv(str(obtained_filename), **read_csv_options) + try: + expected_data = pd.read_csv(str(expected_filename), **read_csv_options) + except pd.errors.ParserError as error: + raise AssertionError( + f"Could not parse the expected results.\n{error}\n" + "To update values, use --force-regen option.\n" + ) from error + if has_multiindex_columns and column_levels == 1: + # A one-row CSV header is otherwise parsed as a flat Index. + for frame in (obtained_data, expected_data): + frame.columns = pd.MultiIndex.from_arrays([frame.columns]) comparison_tables_dict = {} for k in obtained_data.keys(): @@ -282,13 +322,29 @@ def check( self._default_tolerance = default_tolerance dump_fn = functools.partial(self._dump_fn, data_frame) + has_multiindex_columns = isinstance(data_frame.columns, pd.MultiIndex) + has_index_names = False + if has_multiindex_columns: + # Match the index labels emitted by pandas' MultiIndex CSV writer. + if isinstance(data_frame.index, pd.MultiIndex): + index_labels = [name or "" for name in data_frame.index.names] + else: + index_labels = [ + "" if name is None else name for name in data_frame.index.names + ] + has_index_names = set(index_labels) != {""} with pd.option_context(*self._pandas_display_options): perform_regression_check( datadir=self.datadir, original_datadir=self.original_datadir, request=self.request, - check_fn=self._check_fn, + check_fn=functools.partial( + self._check_fn, + column_levels=data_frame.columns.nlevels, + has_multiindex_columns=has_multiindex_columns, + has_index_names=has_index_names, + ), dump_fn=dump_fn, extension=".csv", basename=basename, diff --git a/tests/test_dataframe_regression.py b/tests/test_dataframe_regression.py index a23e568..15aa139 100644 --- a/tests/test_dataframe_regression.py +++ b/tests/test_dataframe_regression.py @@ -49,6 +49,50 @@ def compare_arrays(obtained, expected): ) +@pytest.mark.parametrize("column_levels", [1, 2]) +@pytest.mark.parametrize("rows", [0, 1]) +def test_force_regen_after_column_header_growth(pytester, column_levels, rows): + data = [[1, 2]] if rows else [] + columns = ( + 'pd.Index(["x", "y"])' + if column_levels == 1 + else 'pd.MultiIndex.from_tuples([("a", "x"), ("a", "y")])' + ) + source = """ + import pandas as pd + + def test_frame(dataframe_regression): + frame = pd.DataFrame({data}, columns={columns}, dtype="int64") + dataframe_regression.check(frame) + """ + pytester.makepyfile(test_dataframe=source.format(data=data, columns=columns)) + result = pytester.runpytest("--regen-all") + result.assert_outcomes(passed=1) + baseline = pytester.path / "test_dataframe" / "test_frame.csv" + old_contents = baseline.read_text(encoding="utf-8") + + columns = 'pd.MultiIndex.from_tuples([("a", "x", "unit"), ("a", "y", "unit")])' + pytester.makepyfile(test_dataframe=source.format(data=data, columns=columns)) + result = pytester.runpytest() + result.assert_outcomes(failed=1) + assert baseline.read_text(encoding="utf-8") == old_contents + + result = pytester.runpytest("--force-regen") + result.assert_outcomes(failed=1) + expected_contents = ",a,a\n,x,y\n,unit,unit\n" + if rows: + expected_contents += "0,1,2\n" + assert ( + baseline.read_text(encoding="utf-8") == expected_contents + ), result.stdout.str() + result.stdout.fnmatch_lines( + ["*Files differ and --force-regen set, regenerating file at:*"] + ) + + result = pytester.runpytest() + result.assert_outcomes(passed=1) + + def test_common_cases(dataframe_regression: DataFrameRegressionFixture, no_regen): # Most common case: Data is valid, is present and should pass data1 = 1.1 * np.ones(5000) @@ -164,6 +208,258 @@ def test_different_data_types( dataframe_regression.check(pd.DataFrame.from_dict({"data1": data1})) +def test_multiindex_columns_tolerance(dataframe_regression, no_regen): + df = pd.DataFrame( + [[1.1, 2.2], [3.3, 4.4]], + columns=pd.MultiIndex.from_tuples([("a", "x"), ("a", "y")]), + ) + dataframe_regression.check(df) + + df.iloc[0, 0] += 0.01 + dataframe_regression.check(df, default_tolerance=dict(atol=0.1, rtol=1e-17)) + + df.iloc[0, 0] += 0.2 + with pytest.raises(AssertionError, match="Values are not sufficiently close"): + dataframe_regression.check(df, default_tolerance=dict(atol=0.1, rtol=1e-17)) + + +@pytest.mark.parametrize("column_levels", [1, 2, 3]) +@pytest.mark.parametrize("index_name", [None, "rows"]) +@pytest.mark.parametrize("index_levels", [1, 2]) +def test_column_header_tolerance( + dataframe_regression, tmp_path, column_levels, index_name, index_levels +): + if column_levels == 1: + columns = pd.Index(["x", "y"]) + else: + columns = pd.MultiIndex.from_tuples( + [tuple(["a", value, "unit"][:column_levels]) for value in ("x", "y")] + ) + if index_levels == 1: + index = pd.Index(["first", "second"], name=index_name) + changed_index = pd.Index(["changed", "second"], name=index_name) + else: + index_names = [None, None] if index_name is None else ["group", index_name] + index = pd.MultiIndex.from_tuples( + [("a", "first"), ("a", "second")], names=index_names + ) + changed_index = pd.MultiIndex.from_tuples( + [("a", "changed"), ("a", "second")], names=index_names + ) + df = pd.DataFrame( + [[1.1, 2.2], [3.3, 4.4]], + columns=columns, + index=index, + ) + baseline = tmp_path / "baseline.csv" + df.to_csv(baseline, float_format="%.17g") + dataframe_regression.check(df, fullpath=baseline) + + df.iloc[0, 0] += 0.01 + dataframe_regression.check( + df, fullpath=baseline, default_tolerance=dict(atol=0.1, rtol=1e-17) + ) + + df.index = changed_index + with pytest.raises(AssertionError, match="Values are not sufficiently close"): + dataframe_regression.check( + df, fullpath=baseline, default_tolerance=dict(atol=0.1, rtol=1e-17) + ) + + +@pytest.mark.parametrize("column_levels", [1, 2, 3]) +@pytest.mark.parametrize("index_name", [None, "rows"]) +def test_multiindex_column_tolerances( + dataframe_regression, tmp_path, column_levels, index_name +): + df = pd.DataFrame( + [[1.1, 2.2], [3.3, 4.4]], + columns=pd.MultiIndex.from_tuples( + [ + ( + (value,) + if column_levels == 1 + else tuple(["a", value, "unit"][:column_levels]) + ) + for value in ("x", "y") + ] + ), + index=pd.Index(["first", "second"], name=index_name), + ) + baseline = tmp_path / "baseline.csv" + df.to_csv(baseline, float_format="%.17g") + dataframe_regression.check(df, fullpath=baseline) + df.iloc[0, 0] += 0.01 + dataframe_regression.check( + df, + fullpath=baseline, + tolerances={df.columns[0]: dict(atol=0.1, rtol=1e-17)}, + default_tolerance=dict(atol=1e-17, rtol=1e-17), + ) + df.iloc[0, 1] += 0.01 + with pytest.raises(AssertionError) as excinfo: + dataframe_regression.check( + df, + fullpath=baseline, + tolerances={df.columns[0]: dict(atol=0.1, rtol=1e-17)}, + default_tolerance=dict(atol=1e-17, rtol=1e-17), + ) + assert f"{df.columns[0]}:\n" not in str(excinfo.value) + assert f"{df.columns[1]}:\n" in str(excinfo.value) + + +@pytest.mark.parametrize( + "columns", + [ + pd.MultiIndex.from_tuples([("a", "renamed"), ("a", "y")]), + pd.MultiIndex.from_tuples([("a", "x", "unit"), ("a", "y", "unit")]), + pd.Index(["x", "y"]), + ], +) +def test_multiindex_column_changes(dataframe_regression, tmp_path, columns): + df = pd.DataFrame( + [[1.1, 2.2], [3.3, 4.4]], + columns=pd.MultiIndex.from_tuples([("a", "x"), ("a", "y")]), + ) + baseline = tmp_path / "baseline.csv" + df.to_csv(baseline, float_format="%.17g") + df.columns = columns + with pytest.raises( + AssertionError, match="Could not find key|Obtained and expected data shape" + ): + dataframe_regression.check(df, fullpath=baseline) + + +@pytest.mark.parametrize("index_name", [None, "rows"]) +@pytest.mark.parametrize("first_value", [1, 2**53]) +@pytest.mark.parametrize("column_levels", [1, 2]) +def test_multiindex_integer_tolerance( + dataframe_regression, tmp_path, index_name, first_value, column_levels +): + df = pd.DataFrame( + [[first_value, 2], [first_value + 2, 4]], + columns=( + pd.MultiIndex.from_arrays([["x", "y"]]) + if column_levels == 1 + else pd.MultiIndex.from_tuples([("a", "x"), ("a", "y")]) + ), + index=pd.Index(["first", "second"], name=index_name), + ) + baseline = tmp_path / "baseline.csv" + df.to_csv(baseline) + df.iloc[0, 0] += 1 + with pytest.raises(AssertionError, match="Values are not sufficiently close"): + dataframe_regression.check( + df, fullpath=baseline, default_tolerance=dict(atol=10) + ) + + +@pytest.mark.parametrize("index_name", [None, "rows", 0, "", "row,\nlabel"]) +@pytest.mark.parametrize("index_levels", [1, 2]) +@pytest.mark.parametrize("column_levels", [1, 2]) +def test_multiindex_index_names( + dataframe_regression, tmp_path, index_name, index_levels, column_levels +): + if index_levels == 1: + index = pd.Index(["first", "second"], name=index_name) + else: + index = pd.MultiIndex.from_tuples( + [("a", "first"), ("a", "second")], names=[None, index_name] + ) + df = pd.DataFrame( + [[1.1, 2.2], [3.3, 4.4]], + columns=( + pd.MultiIndex.from_arrays([["x", "y"]], names=["group,\nlabel"]) + if column_levels == 1 + else pd.MultiIndex.from_tuples( + [("a", "x"), ("a", "y")], names=["group,\nlabel", "value"] + ) + ), + index=index, + ) + baseline = tmp_path / "baseline.csv" + df.to_csv(baseline, float_format="%.17g") + dataframe_regression.check(df, fullpath=baseline) + df.iloc[0, 0] += 0.01 + dataframe_regression.check( + df, fullpath=baseline, default_tolerance=dict(atol=0.1, rtol=1e-17) + ) + + if index_levels == 1: + df.index.name = "changed" + else: + df.index = df.index.set_names([None, "changed"]) + with pytest.raises(AssertionError): + dataframe_regression.check( + df, fullpath=baseline, default_tolerance=dict(atol=0.1, rtol=1e-17) + ) + + +def test_multiindex_column_level_names(dataframe_regression, tmp_path): + df = pd.DataFrame( + [[1.1, 2.2], [3.3, 4.4]], + columns=pd.MultiIndex.from_tuples( + [("a", "x"), ("a", "y")], names=["group", "value"] + ), + ) + baseline = tmp_path / "baseline.csv" + df.to_csv(baseline, float_format="%.17g") + df.columns.names = ["changed", "value"] + with pytest.raises(AssertionError, match="Could not find key"): + dataframe_regression.check(df, fullpath=baseline) + + +@pytest.mark.parametrize("index_name", [None, "rows"]) +def test_multiindex_first_row_na(dataframe_regression, tmp_path, index_name): + df = pd.DataFrame( + [[np.nan, np.nan], [1.1, 2.2]], + columns=pd.MultiIndex.from_tuples([("a", "x"), ("a", "y")]), + index=pd.Index(["first", "second"], name=index_name), + ) + baseline = tmp_path / "baseline.csv" + df.to_csv(baseline, float_format="%.17g") + dataframe_regression.check(df, fullpath=baseline) + df.iloc[1, 0] += 0.01 + dataframe_regression.check( + df, fullpath=baseline, default_tolerance=dict(atol=0.1, rtol=1e-17) + ) + df.index = pd.Index(["changed", "second"], name=index_name) + with pytest.raises(AssertionError, match="Values are not sufficiently close"): + dataframe_regression.check( + df, fullpath=baseline, default_tolerance=dict(atol=0.1, rtol=1e-17) + ) + + +@pytest.mark.parametrize("index_name", [None, "rows"]) +def test_multiindex_empty_numeric_frame(dataframe_regression, tmp_path, index_name): + df = pd.DataFrame( + np.empty((0, 2)), + columns=pd.MultiIndex.from_tuples([("a", "x"), ("a", "y")]), + index=pd.Index([], name=index_name), + ) + baseline = tmp_path / "baseline.csv" + df.to_csv(baseline, float_format="%.17g") + dataframe_regression.check(df, fullpath=baseline) + df.loc["first"] = [1.1, 2.2] + with pytest.raises(AssertionError): + dataframe_regression.check(df, fullpath=baseline) + + +def test_multiindex_numeric_categorical(dataframe_regression, tmp_path): + df = pd.DataFrame( + {("a", "x"): pd.Categorical([1, 3], categories=[1, 2, 3])}, + index=pd.Index(["first", "second"], name="rows"), + ) + baseline = tmp_path / "baseline.csv" + df.to_csv(baseline) + dataframe_regression.check(df, fullpath=baseline) + df.iloc[0, 0] = 2 + with pytest.raises(AssertionError, match="Values are not sufficiently close"): + dataframe_regression.check( + df, fullpath=baseline, default_tolerance=dict(atol=10) + ) + + class Foo: def __init__(self, bar): self.bar = bar diff --git a/tests/test_dataframe_regression/test_multiindex_columns_tolerance.csv b/tests/test_dataframe_regression/test_multiindex_columns_tolerance.csv new file mode 100644 index 0000000..b7c190f --- /dev/null +++ b/tests/test_dataframe_regression/test_multiindex_columns_tolerance.csv @@ -0,0 +1,4 @@ +,a,a +,x,y +0,1.1000000000000001,2.2000000000000002 +1,3.2999999999999998,4.4000000000000004