Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions docs/source/contributor-guide/expression-audits/datetime_funcs.md
Original file line number Diff line number Diff line change
Expand Up @@ -95,4 +95,14 @@

- Rewrites to `Cast(..., EvalMode.LEGACY)` (no format, native) or `GetTimestamp(..., failOnError = false)` (with format, via the codegen dispatcher) before Comet sees the plan. In non-ANSI mode the rewritten tree is identical to `to_timestamp`; invalid inputs return NULL to match Spark.

## unix_timestamp

- Spark 3.4.3 (audited 2026-09-13): baseline. String inputs accept literal or column formats. Date, timestamp, and timestamp without time zone inputs ignore the format argument.
- Spark 3.5.8 (audited 2026-09-13): parsing failures use structured timestamp parsing errors.
- Spark 4.0.1 (audited 2026-09-13): `inputTypes` widened to `StringTypeWithCollation` for the input and format arguments.
- Spark 4.1.1 (audited 2026-09-13): same input types and parsing behavior as Spark 4.0.1.
- String inputs use Spark's generated parser through codegen dispatch, including collated strings and formats. Literal and column formats preserve null handling, ANSI errors, parser policy, and session time zone.
- Date, timestamp, and timestamp without time zone inputs retain native execution and ignore the format, including its collation. String input stays unsupported by the native serializer so `allowIncompatible=true` cannot send it to the native kernel.
- Native timestamp conversion truncates fractional seconds toward zero, matching Spark's `ToTimestamp`. This fixes the previous use of floor division for negative fractional timestamps: at UTC, `1969-12-31 23:59:58.5` produces `-1`, not `-2`. Casting a timestamp to `BIGINT` deliberately uses floor division in Spark and Comet, so that cast still produces `-2`.

[Spark Expression Support]: ../../user-guide/latest/expressions.md
3 changes: 2 additions & 1 deletion docs/source/contributor-guide/spark_configs_support.md
Original file line number Diff line number Diff line change
Expand Up @@ -124,7 +124,8 @@ and fallback paths:
`spark.comet.expression.FromUnixTime.allowIncompatible=true` is set; otherwise
it routes through the codegen dispatcher.
- `unix_timestamp(<timestamp_or_date>)` does not call the formatter at all; the
string-input overload falls back.
string-input overload routes through the codegen dispatcher and preserves Spark's
selected parser policy.
- `to_unix_timestamp` routes through the codegen dispatcher.

If a Comet contributor adds native string-format parsing or extends the date_format
Expand Down
2 changes: 1 addition & 1 deletion docs/source/user-guide/latest/expressions.md
Original file line number Diff line number Diff line change
Expand Up @@ -317,7 +317,7 @@ The type-name conversion functions (`bigint`, `binary`, `boolean`, `date`, `deci
| `unix_micros` | ✅ | Codegen dispatch | |
| `unix_millis` | ✅ | Codegen dispatch | |
| `unix_seconds` | ✅ | Codegen dispatch | |
| `unix_timestamp` | ✅ | Native | |
| `unix_timestamp` | ✅ | Hybrid | String parsing uses Spark's codegen and honors the time parser policy, ANSI mode, and session time zone. Date and timestamp inputs ignore the format and use native execution. |
| `weekday` | ✅ | Native | |
| `weekofyear` | ✅ | Native | |
| `window` | ✅ | — | Batch tumbling and sliding time-window grouping runs natively |
Expand Down
62 changes: 56 additions & 6 deletions native/spark-expr/src/datetime_funcs/unix_timestamp.rs
Original file line number Diff line number Diff line change
Expand Up @@ -23,7 +23,6 @@ use datafusion::common::{internal_datafusion_err, DataFusionError};
use datafusion::logical_expr::{
ColumnarValue, ScalarFunctionArgs, ScalarUDFImpl, Signature, Volatility,
};
use num::integer::div_floor;
use std::{fmt::Debug, sync::Arc};

const MICROS_PER_SECOND: i64 = 1_000_000;
Expand Down Expand Up @@ -84,12 +83,12 @@ impl ScalarUDFImpl for SparkUnixTimestamp {
timestamp_array
.values()
.iter()
.map(|&micros| div_floor(micros, MICROS_PER_SECOND))
.map(|&micros| micros / MICROS_PER_SECOND)
.collect()
} else {
timestamp_array
.iter()
.map(|v| v.map(|micros| div_floor(micros, MICROS_PER_SECOND)))
.map(|v| v.map(|micros| micros / MICROS_PER_SECOND))
.collect()
};

Expand All @@ -116,12 +115,12 @@ impl ScalarUDFImpl for SparkUnixTimestamp {
timestamp_array
.values()
.iter()
.map(|&micros| div_floor(micros, MICROS_PER_SECOND))
.map(|&micros| micros / MICROS_PER_SECOND)
.collect()
} else {
timestamp_array
.iter()
.map(|v| v.map(|micros| div_floor(micros, MICROS_PER_SECOND)))
.map(|v| v.map(|micros| micros / MICROS_PER_SECOND))
.collect()
};

Expand Down Expand Up @@ -153,7 +152,7 @@ impl ScalarUDFImpl for SparkUnixTimestamp {
} else {
timestamp_array
.iter()
.map(|v| v.map(|micros| div_floor(micros, MICROS_PER_SECOND)))
.map(|v| v.map(|micros| micros / MICROS_PER_SECOND))
.collect()
};

Expand Down Expand Up @@ -208,6 +207,57 @@ mod tests {
}
}

#[test]
fn test_unix_timestamp_truncates_fractional_seconds_toward_zero() {
for timezone in [None, Some("UTC")] {
for with_null in [false, true] {
let mut values = vec![
Some(-1_500_000),
Some(-1_000_000),
Some(-999_999),
Some(-1),
Some(0),
Some(1),
Some(999_999),
Some(1_000_000),
Some(1_500_000),
];
let mut expected = vec![
Some(-1),
Some(-1),
Some(0),
Some(0),
Some(0),
Some(0),
Some(0),
Some(1),
Some(1),
];
if with_null {
values.push(None);
expected.push(None);
}
let input = TimestampMicrosecondArray::from(values).with_timezone_opt(timezone);
let number_rows = input.len();
let udf = SparkUnixTimestamp::new("UTC".to_string());
let result = udf
.invoke_with_args(ScalarFunctionArgs {
args: vec![ColumnarValue::Array(Arc::new(input))],
number_rows,
return_field: Arc::new(Field::new("unix_timestamp", DataType::Int64, true)),
config_options: Arc::new(ConfigOptions::default()),
arg_fields: vec![],
})
.unwrap();
let ColumnarValue::Array(result) = result else {
panic!("Expected array result");
};
let actual = result.as_primitive::<Int64Type>();
assert_eq!(actual.iter().collect::<Vec<_>>(), expected);
}
}
}

#[test]
fn test_unix_timestamp_from_date() {
// Test with Date32
Expand Down
19 changes: 5 additions & 14 deletions spark/src/main/scala/org/apache/comet/serde/datetime.scala
Original file line number Diff line number Diff line change
Expand Up @@ -290,15 +290,12 @@ private[serde] object DatetimeCollation extends CometTypeShim {
expr.children.exists(c => hasNonDefaultStringCollation(c.dataType))
}

object CometUnixTimestamp extends CometExpressionSerde[UnixTimestamp] {

private val collationReason = DatetimeCollation.reason("unix_timestamp")
object CometUnixTimestamp
extends CometExpressionSerde[UnixTimestamp]
with CodegenDispatchFallback {

override def getUnsupportedReasons(): Seq[String] = Seq(
"Only `DateType`, `TimestampType`, and `TimestampNTZType` inputs are supported.")

override def getIncompatibleReasons(): Seq[String] =
DatetimeCollation.incompatibleReasons("unix_timestamp")
"String inputs, including collated strings, have no native implementation.")

private def isSupportedInputType(expr: UnixTimestamp): Boolean = {
expr.children.head.dataType match {
Expand All @@ -309,16 +306,10 @@ object CometUnixTimestamp extends CometExpressionSerde[UnixTimestamp] {
}

override def getSupportLevel(expr: UnixTimestamp): SupportLevel = {
// The input type is screened ahead of the collation check on purpose. A non-date/timestamp
// input has no native path at all, so it must report `Unsupported` rather than
// `Incompatible`: the latter is waved straight through to `convert` when
// `spark.comet.expression.UnixTimestamp.allowIncompatible=true`, and the native kernel then
// raises an execution error on the string child instead of falling back to Spark.
// Strings have no native path, even when incompatible expressions are allowed.
if (!isSupportedInputType(expr)) {
val inputType = expr.children.head.dataType
Unsupported(Some(s"unix_timestamp does not support input type: $inputType"))
} else if (DatetimeCollation.hasNonDefaultCollation(expr)) {
Incompatible(Some(collationReason))
} else {
Compatible()
}
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -15,15 +15,46 @@
-- specific language governing permissions and limitations
-- under the License.

-- Config: spark.sql.session.timeZone=UTC

statement
CREATE TABLE test_unix_ts(ts timestamp) USING parquet

statement
INSERT INTO test_unix_ts VALUES (timestamp('1970-01-01 00:00:00')), (timestamp('2024-06-15 10:30:45')), (NULL)

query
query expect_native(unix_timestamp)
SELECT unix_timestamp(ts) FROM test_unix_ts

-- literal arguments
query ignore(https://github.com/apache/datafusion-comet/issues/3336)
SELECT unix_timestamp(timestamp('1970-01-01 00:00:00')), unix_timestamp(timestamp('2024-06-15 10:30:45'))

-- Native timestamp conversion truncates fractional seconds toward zero, including before epoch.
statement
CREATE TABLE test_unix_ts_fractional(ts timestamp, ntz timestamp_ntz) USING parquet

statement
INSERT INTO test_unix_ts_fractional VALUES
(CAST('1969-12-31 23:59:58.500000' AS TIMESTAMP), CAST('1969-12-31 23:59:58.500000' AS TIMESTAMP_NTZ)),
(CAST('1969-12-31 23:59:59.000000' AS TIMESTAMP), CAST('1969-12-31 23:59:59.000000' AS TIMESTAMP_NTZ)),
(CAST('1969-12-31 23:59:59.500000' AS TIMESTAMP), CAST('1969-12-31 23:59:59.500000' AS TIMESTAMP_NTZ)),
(CAST('1969-12-31 23:59:59.999999' AS TIMESTAMP), CAST('1969-12-31 23:59:59.999999' AS TIMESTAMP_NTZ)),
(CAST('1970-01-01 00:00:00.000000' AS TIMESTAMP), CAST('1970-01-01 00:00:00.000000' AS TIMESTAMP_NTZ)),
(CAST('1970-01-01 00:00:00.000001' AS TIMESTAMP), CAST('1970-01-01 00:00:00.000001' AS TIMESTAMP_NTZ)),
(CAST('1970-01-01 00:00:01.500000' AS TIMESTAMP), CAST('1970-01-01 00:00:01.500000' AS TIMESTAMP_NTZ)),
(NULL, NULL)

query expect_native(unix_timestamp)
SELECT unix_timestamp(ts), unix_timestamp(ntz) FROM test_unix_ts_fractional

query expect_native(unix_timestamp)
SELECT unix_timestamp(ts), unix_timestamp(ntz) FROM test_unix_ts_fractional WHERE ts IS NOT NULL

-- unix_timestamp truncates toward zero, while casting a timestamp to BIGINT floors.
-- At -1.5 seconds the results are -1 and -2; at -0.5 seconds they are 0 and -1.
query expect_native(unix_timestamp, cast)
SELECT unix_timestamp(ts), CAST(ts AS BIGINT) FROM test_unix_ts_fractional

query expect_native(unix_timestamp, cast)
SELECT unix_timestamp(ts), CAST(ts AS BIGINT) FROM test_unix_ts_fractional WHERE ts IS NOT NULL
Original file line number Diff line number Diff line change
@@ -0,0 +1,42 @@
-- Licensed to the Apache Software Foundation (ASF) under one
-- or more contributor license agreements. See the NOTICE file
-- distributed with this work for additional information
-- regarding copyright ownership. The ASF licenses this file
-- to you under the Apache License, Version 2.0 (the
-- "License"); you may not use this file except in compliance
-- with the License. You may obtain a copy of the License at
--
-- http://www.apache.org/licenses/LICENSE-2.0
--
-- Unless required by applicable law or agreed to in writing,
-- software distributed under the License is distributed on an
-- "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
-- KIND, either express or implied. See the License for the
-- specific language governing permissions and limitations
-- under the License.

-- Config: spark.sql.ansi.enabled=true
-- Config: spark.sql.legacy.timeParserPolicy=CORRECTED
-- Config: spark.comet.exec.scalaUDF.codegen.enabled=true

statement
CREATE TABLE test_unix_ts_ansi(s string, fmt string) USING parquet

statement
INSERT INTO test_unix_ts_ansi VALUES ('not a date', 'yyyy-MM-dd')

query expect_error(could not be parsed)
SELECT unix_timestamp(s, 'yyyy-MM-dd') FROM test_unix_ts_ansi

query expect_error(could not be parsed)
SELECT unix_timestamp(s, fmt) FROM test_unix_ts_ansi

query expect_error(could not be parsed)
SELECT unix_timestamp('2024-13-99', 'yyyy-MM-dd')

-- Valid queries require Comet execution, so fallback cannot hide an error-path regression.
query expect_dispatch(unix_timestamp)
SELECT unix_timestamp('2024-06-15', 'yyyy-MM-dd'), unix_timestamp(CAST(NULL AS STRING))

query expect_dispatch(unix_timestamp)
SELECT unix_timestamp('2024-06-15', fmt) FROM test_unix_ts_ansi
Original file line number Diff line number Diff line change
@@ -0,0 +1,42 @@
-- Licensed to the Apache Software Foundation (ASF) under one
-- or more contributor license agreements. See the NOTICE file
-- distributed with this work for additional information
-- regarding copyright ownership. The ASF licenses this file
-- to you under the Apache License, Version 2.0 (the
-- "License"); you may not use this file except in compliance
-- with the License. You may obtain a copy of the License at
--
-- http://www.apache.org/licenses/LICENSE-2.0
--
-- Unless required by applicable law or agreed to in writing,
-- software distributed under the License is distributed on an
-- "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
-- KIND, either express or implied. See the License for the
-- specific language governing permissions and limitations
-- under the License.

-- Config: spark.comet.exec.scalaUDF.codegen.enabled=false

statement
CREATE TABLE test_unix_ts_fallback(s string, fmt string, d date, ts timestamp, ntz timestamp_ntz) USING parquet

statement
INSERT INTO test_unix_ts_fallback VALUES
('2024-06-15', 'yyyy-MM-dd', date('2024-06-15'), timestamp('2024-06-15 10:30:45'), CAST('2024-06-15 10:30:45' AS TIMESTAMP_NTZ)),
(NULL, NULL, NULL, NULL, NULL)

query expect_fallback(spark.comet.exec.scalaUDF.codegen.enabled)
SELECT unix_timestamp(s) FROM test_unix_ts_fallback

query expect_fallback(spark.comet.exec.scalaUDF.codegen.enabled)
SELECT unix_timestamp(s, fmt) FROM test_unix_ts_fallback

query expect_fallback(spark.comet.exec.scalaUDF.codegen.enabled)
SELECT unix_timestamp('2024-06-15', 'yyyy-MM-dd')

-- Date and timestamp inputs keep their native path and ignore the format.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The comment claims a native path but the assertion below cannot see it. Because this file disables the dispatcher, a flip to dispatch would surface as a fallback and the test would fail anyway, so it is covered today. query expect_native(unix_timestamp) would state it directly and keep holding if someone later changes the config header at the top of the file.

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Fair, updated

query expect_native(unix_timestamp)
SELECT unix_timestamp(d), unix_timestamp(ts), unix_timestamp(ntz) FROM test_unix_ts_fallback

query expect_native(unix_timestamp)
SELECT unix_timestamp(d, fmt), unix_timestamp(ts, fmt), unix_timestamp(ntz, fmt) FROM test_unix_ts_fallback
Original file line number Diff line number Diff line change
@@ -0,0 +1,65 @@
-- Licensed to the Apache Software Foundation (ASF) under one
-- or more contributor license agreements. See the NOTICE file
-- distributed with this work for additional information
-- regarding copyright ownership. The ASF licenses this file
-- to you under the Apache License, Version 2.0 (the
-- "License"); you may not use this file except in compliance
-- with the License. You may obtain a copy of the License at
--
-- http://www.apache.org/licenses/LICENSE-2.0
--
-- Unless required by applicable law or agreed to in writing,
-- software distributed under the License is distributed on an
-- "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
-- KIND, either express or implied. See the License for the
-- specific language governing permissions and limitations
-- under the License.

-- Config: spark.sql.legacy.timeParserPolicy=CORRECTED
-- ConfigMatrix: parquet.enable.dictionary=false,true
-- ConfigMatrix: spark.sql.session.timeZone=UTC,America/Los_Angeles

statement
CREATE TABLE test_unix_ts_strings(s string, fmt string) USING parquet

statement
INSERT INTO test_unix_ts_strings VALUES
('1970-01-01 00:00:00', 'yyyy-MM-dd HH:mm:ss'),
('1969-12-31 23:59:59', 'yyyy-MM-dd HH:mm:ss'),
('2024-02-29 12:30:45', 'yyyy-MM-dd HH:mm:ss'),
('2024-03-10 02:30:00', 'yyyy-MM-dd HH:mm:ss'),
('2024-11-03 01:30:00', 'yyyy-MM-dd HH:mm:ss'),
('1969-12-31 23:59:59.999999', 'yyyy-MM-dd HH:mm:ss.SSSSSS'),
('2024/06/15', 'yyyy/MM/dd'),
('2024-06-15T10:30:45+05:30', "yyyy-MM-dd'T'HH:mm:ssXXX"),
('1582-10-04', 'yyyy-MM-dd'),
('0001-01-01', 'yyyy-MM-dd'),
('9999-12-31', 'yyyy-MM-dd'),
('not a date', 'yyyy-MM-dd'),
('2024-02-30', 'yyyy-MM-dd'),
('', 'yyyy-MM-dd'),
(NULL, 'yyyy-MM-dd'),
('2024-06-15', NULL),
('2024-06-15', ''),
(NULL, NULL)

-- Exercise both the cached literal formatter and the per-row formatter.
query expect_dispatch(unix_timestamp)
SELECT unix_timestamp(s), unix_timestamp(s, 'yyyy-MM-dd HH:mm:ss') FROM test_unix_ts_strings

query expect_dispatch(unix_timestamp)
SELECT unix_timestamp(s, fmt) FROM test_unix_ts_strings

query expect_dispatch(unix_timestamp)
SELECT unix_timestamp('2024-06-15', fmt) FROM test_unix_ts_strings

-- Constant folding is disabled by the SQL test harness.
query expect_dispatch(unix_timestamp)
SELECT unix_timestamp('2024-06-15', 'yyyy-MM-dd'), unix_timestamp(''), unix_timestamp(CAST(NULL AS STRING)), unix_timestamp('2024-06-15', CAST(NULL AS STRING))

-- Parsing inside grouping must keep both the expression and aggregation in Comet.
query expect_dispatch(unix_timestamp)
SELECT unix_timestamp(s) AS u, count(*) FROM test_unix_ts_strings GROUP BY u

query expect_dispatch(unix_timestamp)
SELECT unix_timestamp(s, fmt) AS u, count(*) FROM test_unix_ts_strings GROUP BY u
Loading
Loading