Compatibility Modes Compared

This script writes one order with a version 1 schema in all three formats and reads it back with four version 2 schemas. Protobuf classes are built at run time from a FileDescriptorProto, so no protoc step is needed.

matrix.py: four schema changes read back by Avro, Protobuf and ParquetPython
import io
import fastavro, pyarrow as pa, pyarrow.parquet as pq
from google.protobuf import descriptor_pb2 as d, descriptor_pool, message_factory
T = {"int": ("int", pa.int32(), 5), "long": ("long", pa.int64(), 3),   # Avro, Arrow, proto
     "string": ("string", pa.string(), 9), "bool": ("boolean", pa.bool_(), 8)}
V1 = [("order_id", "long"), ("customer_id", "int"), ("channel", "string")]
V2 = {"add gift_wrap": V1 + [("gift_wrap", "bool")],
      "rename channel": V1[:2] + [("sales_channel", "string")],
      "widen int to long": [V1[0], ("customer_id", "long"), V1[2]],
      "int to string": [V1[0], ("customer_id", "string"), V1[2]]}
ROW = {"order_id": 1, "customer_id": 2518, "channel": "ios"}
avro = lambda s: fastavro.parse_schema({"type": "record", "name": "Order", "fields": [
    {"name": n, "type": T[t][0]} | ({"default": False} if t == "bool" else {}) for n, t in s]})
arrow = lambda s: pa.schema([(n, T[t][1]) for n, t in s])
def proto(s):            # field numbers follow position, as in a .proto edited in place
    f = d.FileDescriptorProto(name="o.proto", package="t", syntax="proto3")
    f.message_type.add(name="Order").field.extend(
        d.FieldDescriptorProto(name=n, number=i, type=T[t][2], label=1)
        for i, (n, t) in enumerate(s, 1))
    (pool := descriptor_pool.DescriptorPool()).Add(f)
    return message_factory.GetMessageClass(pool.FindMessageTypeByName("t.Order"))
def run(read):
    try:
        return repr(read())
    except Exception as e:
        return type(e).__name__
av = io.BytesIO()                                          # version 1 writes one order
fastavro.schemaless_writer(av, avro(V1), ROW)
pb = proto(V1)(**ROW).SerializeToString()
pq.write_table(pa.Table.from_pylist([ROW], arrow(V1)), "v1.parquet")
print(f"{'change':18} {'Avro':22} {'Proto':6} Parquet")
for change, s in V2.items():                               # version 2 reads it back
    field = next(n for n, t in s if (n, t) not in V1)
    av.seek(0)
    a = run(lambda: fastavro.schemaless_reader(av, avro(V1), avro(s))[field])
    p = run(lambda: getattr(proto(s).FromString(pb), field))
    q = run(lambda: pq.read_table("v1.parquet", schema=arrow(s))[field][0].as_py())
    print(f"{change:18} {a:22} {p:6} {q}")
Output
change             Avro                   Proto  Parquet
add gift_wrap      False                  False  None
rename channel     SchemaResolutionError  'ios'  None
widen int to long  2518                   2518   2518
int to string      SchemaResolutionError  ''     '2518'

A new reader on old data tests backward compatibility; the reverse tests forward, and both make full. Avro 129 fails loudly: the rename lacks an alias, and int to string is no legal promotion. Protobuf never fails: names do not travel, so the rename is free, but the type change silently reads ''. Parquet 129 fills gaps with nulls, even for the defaulted Boolean, and casts int to string. Only widening is safe everywhere; an error stops a deploy, but a silent null needs a quality check (Lakehouses, Data Quality and Governance).