Skip to content
Open
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
35 changes: 25 additions & 10 deletions integration_tests/src/main/python/cast_test.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,9 @@
# See the License for the specific language governing permissions and
# limitations under the License.

from contextlib import redirect_stdout
from io import StringIO

import pytest

from asserts import *
Expand Down Expand Up @@ -913,19 +916,31 @@ def test_cast_date_integral_and_fp_ansi_off():
"cast(a as boolean)", "cast(a as byte)", "cast(a as short)", "cast(a as int)", "cast(a as long)", "cast(a as float)", "cast(a as double)"),
conf=ansi_disabled_conf)

# ToPrettyString is generated by df.show(), and there is no ToPrettyString function in pyspark.sql.functions package
# Here use df.show() to test ToPrettyString on date column. If fall back occurs, `with_gpu_session` will report error:
# Part of the plan is not columnar class
# ToPrettyString is generated by df.show(), and there is no ToPrettyString function in
# pyspark.sql.functions. Capture the rendered output while the prescribed collect assertion
# verifies that the query executes correctly on both CPU and GPU.
@allow_non_gpu('CollectLimitExec')
def test_to_pretty_string_for_date_and_test_cast_between_date_string():
def _query(spark):
spark.createDataFrame([('2025-01-02',), ('2025-01-03',)], 'str_col string').selectExpr(
"CAST(str_col as date)", "CAST(CAST(str_col as date) as string)").show()
show_outputs = {}

with_gpu_session(
lambda spark: _query(spark),
{'spark.rapids.sql.hasExtendedYearValues': False,
'spark.sql.session.timeZone': 'America/Los_Angeles'})
def _query(spark):
df = spark.createDataFrame(
[('2025-01-02',), ('2025-01-03',)], 'str_col string'
).orderBy('str_col').selectExpr(
"CAST(str_col as date)", "CAST(CAST(str_col as date) as string)"
)
output = StringIO()
with redirect_stdout(output):
df.show()
show_outputs[spark.conf.get('spark.rapids.sql.enabled')] = output.getvalue()
return df

conf = {
'spark.rapids.sql.hasExtendedYearValues': False,
'spark.sql.session.timeZone': 'America/Los_Angeles'
}
assert_gpu_and_cpu_are_equal_collect(_query, conf=conf)
assert show_outputs['false'] == show_outputs['true']


def test_cast_string_to_timestamp_valid():
Expand Down
Loading