>>> from pyspark.sql import functions as F, types as T

>>> schema = T.StructType([
...     T.StructField("id", T.IntegerType(), False),
...     T.StructField(
...         "struct_col",
...         T.StructType([T.StructField("nested_field", T.StringType(), True)]),
...         True,
...     ),
... ])
>>> df = spark.createDataFrame([(1, ("x",))], schema)
>>> df.withColumn("struct_col.nested_field", F.lit("y")).collect()
[Row(id=1, struct_col=Row(nested_field='x'), struct_col.nested_field='y')]

A replaced column is a new attribute, so it no longer has the input qualifier,
while the columns that are kept retain it.

>>> df = spark.createDataFrame([(1, 10, 3), (2, 20, 1), (3, 30, 2)], "a int, b int, c int")
>>> replaced = df.alias("x").withColumn("b", F.col("b") * 10)
>>> sorted(replaced.select("x.a", "b", "x.c").collect())
[Row(a=1, b=100, c=3), Row(a=2, b=200, c=1), Row(a=3, b=300, c=2)]
>>> sorted(replaced.select(F.col("x.*")).collect())
[Row(a=1, c=3), Row(a=2, c=1), Row(a=3, c=2)]
>>> replaced.select("x.b").collect()
Traceback (most recent call last):
...
pyspark.errors.exceptions.connect.AnalysisException: ...
>>> replaced.where("x.b = 20").collect()
[Row(a=2, b=200, c=1)]
>>> df.alias("x").withColumn("b", -F.col("b")).orderBy(F.col("x.b")).select("a").collect()
[Row(a=1), Row(a=2), Row(a=3)]
>>> df.alias("x").withColumn("d", F.col("a")).select("x.d").collect()
Traceback (most recent call last):
...
pyspark.errors.exceptions.connect.AnalysisException: ...
>>> df.alias("x").withMetadata("a", {"key": "value"}).select("x.a").collect()
Traceback (most recent call last):
...
pyspark.errors.exceptions.connect.AnalysisException: ...

A relation without a name has no qualifier, even for the columns that are kept.

>>> df.withColumn("d", F.lit(1)).select("`?table?`.a").collect()
Traceback (most recent call last):
...
pyspark.errors.exceptions.connect.AnalysisException: ...

A new column replaces an existing column whose name matches case-insensitively,
and the replacement takes the name of the new column.

>>> replaced_case = df.withColumn("B", F.col("b") * 10)
>>> replaced_case.columns
['a', 'B', 'c']
>>> sorted(replaced_case.select("a").where("b > 150").collect())
[Row(a=2), Row(a=3)]
>>> df.withColumn("A", -F.col("a")).select("b").where("a > 1").collect()
[]
>>> spark.conf.set("spark.sql.caseSensitive", "true")
>>> df.withColumn("A", F.lit(0)).columns
['a', 'b', 'c', 'A']
>>> spark.conf.unset("spark.sql.caseSensitive")

>>> df.alias("x").withColumn("A", F.col("a") * 100).select("x.a").collect()
Traceback (most recent call last):
...
pyspark.errors.exceptions.connect.AnalysisException: ...
>>> df.alias("x").withColumn("B", F.col("b") + 1).select("x.b").collect()
Traceback (most recent call last):
...
pyspark.errors.exceptions.connect.AnalysisException: ...
>>> df.alias("x").withMetadata("A", {"key": "value"}).groupBy("x.a").count().collect()
Traceback (most recent call last):
...
pyspark.errors.exceptions.connect.AnalysisException: ...
>>> sorted(df.alias("x").withColumn("A", F.col("a") * 100).select(F.col("x.*")).collect())
[Row(b=10, c=3), Row(b=20, c=1), Row(b=30, c=2)]
>>> sorted(df.alias("x").withColumn("A", F.col("a") * 100).where("x.a > 1").select("x.b").collect())
[Row(b=20), Row(b=30)]

>>> sorted(df.alias("x").withColumn("d", F.lit(1)).withColumn("e", F.lit(2)).select("x.a", "d", "e").collect())
[Row(a=1, d=1, e=2), Row(a=2, d=1, e=2), Row(a=3, d=1, e=2)]
>>> joined = df.alias("x").join(df.alias("y"), F.col("x.a") == F.col("y.a"))
>>> sorted(joined.withColumn("z", F.lit(1)).select("x.b", "y.c", "z").collect())
[Row(b=10, c=3, z=1), Row(b=20, c=1, z=1), Row(b=30, c=2, z=1)]

Bulk replacements retain input-column order and append new columns in the order
specified, using the configured name resolver.

>>> source = spark.createDataFrame([(1, 2, 3)], "a int, b int, c int")
>>> replacements = {"C": F.lit(30), "b": F.lit(20), "A": F.lit(10), "d": F.lit(40)}
>>> source.withColumns(replacements).collect()
[Row(A=10, b=20, C=30, d=40)]
>>> spark.conf.set("spark.sql.caseSensitive", "true")
>>> source.withColumns(replacements).collect()
[Row(a=1, b=20, c=3, C=30, A=10, d=40)]
>>> spark.conf.unset("spark.sql.caseSensitive")
