pytest SparkSession Fixtures

pytest Fixtures for a Local SparkSession

Starting a SparkSession takes seconds, so tests share one through a session-scoped fixture; a second fixture builds typed orders from compact tuples.

tests/conftest.py: one local session per test run, and a row builderPython
import pytest
from pyspark.sql import SparkSession
@pytest.fixture(scope="session")
def spark():
    session = (SparkSession.builder.master("local[2]").appName("booknest-tests")
               .config("spark.sql.shuffle.partitions", "2")      # tiny data, few tasks
               .config("spark.sql.session.timeZone", "UTC")
               .config("spark.ui.enabled", "false")              # no UI, no port clashes
               .getOrCreate())
    yield session
    session.stop()
ORDER_SCHEMA = """order_id BIGINT, customer_id INT, order_ts TIMESTAMP, status STRING,
  items ARRAY<STRUCT<book_id: INT, qty: INT, unit_price: DECIMAL(6,2)>>, total DECIMAL(10,2)"""
@pytest.fixture
def make_orders(spark):
    """Orders from (order_id, status, [(book_id, qty, price)], total) tuples."""
    from datetime import datetime
    from decimal import Decimal
    def build(rows):
        data = [(oid, 1, datetime(2026, 1, 1), status,
                 None if items is None else [(b, q, Decimal(p)) for b, q, p in items],
                 None if total is None else Decimal(total))
                for oid, status, items, total in rows]
        return spark.createDataFrame(data, ORDER_SCHEMA)
    return build

Configure the session for tiny data (two shuffle partitions, not 200), production's UTC time zone, and no UI, so parallel runs never compete for a port. Build test data with an explicit schema: inference would turn DECIMAL money into double (Schemas).