Starting a SparkSession takes seconds, so tests share one through a session-scoped fixture; a second fixture builds typed orders from compact tuples.
import pytest
from pyspark.sql import SparkSession
@pytest.fixture(scope="session")
def spark():
session = (SparkSession.builder.master("local[2]").appName("booknest-tests")
.config("spark.sql.shuffle.partitions", "2") # tiny data, few tasks
.config("spark.sql.session.timeZone", "UTC")
.config("spark.ui.enabled", "false") # no UI, no port clashes
.getOrCreate())
yield session
session.stop()
ORDER_SCHEMA = """order_id BIGINT, customer_id INT, order_ts TIMESTAMP, status STRING,
items ARRAY<STRUCT<book_id: INT, qty: INT, unit_price: DECIMAL(6,2)>>, total DECIMAL(10,2)"""
@pytest.fixture
def make_orders(spark):
"""Orders from (order_id, status, [(book_id, qty, price)], total) tuples."""
from datetime import datetime
from decimal import Decimal
def build(rows):
data = [(oid, 1, datetime(2026, 1, 1), status,
None if items is None else [(b, q, Decimal(p)) for b, q, p in items],
None if total is None else Decimal(total))
for oid, status, items, total in rows]
return spark.createDataFrame(data, ORDER_SCHEMA)
return buildConfigure the session for tiny data (two shuffle partitions, not 200), production's UTC time zone, and no UI, so parallel runs never compete for a port. Build test data with an explicit schema: inference would turn DECIMAL money into double (Schemas).