Edge Cases and Schema Changes

Testing Edge Cases and Schema Changes

Most incidents come from empty batches, NULLs, keys missing from a lookup, and upstream schema changes.

tests/test_edge_cases.py: empty, null, unknown and changed inputsPython
from decimal import Decimal
import pytest
from pyspark.errors import AnalysisException
from pyspark.testing import assertDataFrameEqual, assertSchemaEqual
from booknest.transforms import explode_lines, genre_revenue, order_size
def test_orders_without_items_produce_no_lines(make_orders):
    orders = make_orders([(1, "cancelled", [], "0.00"), (2, "cancelled", None, None)])
    assert explode_lines(orders).count() == 0
def test_unknown_book_is_reported_not_dropped(spark, make_orders):
    orders = make_orders([(1, "delivered", [(99, 1, "10.00")], "10.00")])
    books = spark.createDataFrame([(1, "Fiction")], "id BIGINT, genre STRING")
    assertDataFrameEqual(genre_revenue(explode_lines(orders), books),
                         [("Unknown", Decimal("10.00"))])
def test_null_total_is_small(make_orders):
    got = order_size(make_orders([(1, "pending", [], None)])).first()["size"]
    assert got == "small"                     # documents current behavior: NULL is not >= 50
def test_output_schema_is_stable(spark, make_orders):
    no_books = spark.createDataFrame([], "id BIGINT, genre STRING")
    got = genre_revenue(explode_lines(make_orders([])), no_books)
    expected = spark.createDataFrame([], "genre STRING NOT NULL, revenue DECIMAL(27,2)")
    assertSchemaEqual(got.schema, expected.schema)       # empty input still has the contract
def test_missing_column_fails_early(spark):
    renamed = spark.createDataFrame([(1, 1)], "id INT, cust INT")   # upstream renamed columns
    with pytest.raises(AnalysisException, match="UNRESOLVED_COLUMN"):
        explode_lines(renamed)

The schema test pins the output contract: its first version expected DECIMAL(22,2) and failed with [DIFFERENT_SCHEMA], showing DecimalType(27,2) (a sum widens the DECIMAL(17,2) amount by 10 digits) and a non-nullable genre from coalesce. The null test documents current behavior, and the last proves a renamed upstream column fails at plan time with UNRESOLVED_COLUMN rather than producing wrong numbers.