Skip to main content

arrow_cast/
parse.rs

1// Licensed to the Apache Software Foundation (ASF) under one
2// or more contributor license agreements.  See the NOTICE file
3// distributed with this work for additional information
4// regarding copyright ownership.  The ASF licenses this file
5// to you under the Apache License, Version 2.0 (the
6// "License"); you may not use this file except in compliance
7// with the License.  You may obtain a copy of the License at
8//
9//   http://www.apache.org/licenses/LICENSE-2.0
10//
11// Unless required by applicable law or agreed to in writing,
12// software distributed under the License is distributed on an
13// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
14// KIND, either express or implied.  See the License for the
15// specific language governing permissions and limitations
16// under the License.
17
18//! [`Parser`] implementations for converting strings to Arrow types
19//!
20//! Used by the CSV and JSON readers to convert strings to Arrow types
21use arrow_array::ArrowNativeTypeOp;
22use arrow_array::timezone::Tz;
23use arrow_array::types::*;
24use arrow_buffer::ArrowNativeType;
25use arrow_schema::ArrowError;
26use chrono::prelude::*;
27use half::f16;
28use std::str::FromStr;
29
30/// Parse nanoseconds from the first `N` values in digits, subtracting the offset `O`
31#[inline]
32fn parse_nanos<const N: usize, const O: u8>(digits: &[u8]) -> u32 {
33    digits[..N]
34        .iter()
35        .fold(0_u32, |acc, v| acc * 10 + v.wrapping_sub(O) as u32)
36        * 10_u32.pow((9 - N) as _)
37}
38
39/// Helper for parsing RFC3339 timestamps
40struct TimestampParser {
41    /// The timestamp bytes to parse minus `b'0'`
42    ///
43    /// This makes interpretation as an integer inexpensive
44    digits: [u8; 32],
45    /// A mask containing a `1` bit where the corresponding byte is a valid ASCII digit
46    mask: u32,
47}
48
49impl TimestampParser {
50    fn new(bytes: &[u8]) -> Self {
51        let mut digits = [0; 32];
52        let mut mask = 0;
53
54        // Treating all bytes the same way, helps LLVM vectorise this correctly
55        for (idx, (o, i)) in digits.iter_mut().zip(bytes).enumerate() {
56            *o = i.wrapping_sub(b'0');
57            mask |= ((*o < 10) as u32) << idx
58        }
59
60        Self { digits, mask }
61    }
62
63    /// Returns true if the byte at `idx` in the original string equals `b`
64    fn test(&self, idx: usize, b: u8) -> bool {
65        self.digits[idx] == b.wrapping_sub(b'0')
66    }
67
68    /// Parses a date of the form `1997-01-31`
69    fn date(&self) -> Option<NaiveDate> {
70        if self.mask & 0b1111111111 != 0b1101101111 || !self.test(4, b'-') || !self.test(7, b'-') {
71            return None;
72        }
73
74        let year = self.digits[0] as u16 * 1000
75            + self.digits[1] as u16 * 100
76            + self.digits[2] as u16 * 10
77            + self.digits[3] as u16;
78
79        let month = self.digits[5] * 10 + self.digits[6];
80        let day = self.digits[8] * 10 + self.digits[9];
81
82        NaiveDate::from_ymd_opt(year as _, month as _, day as _)
83    }
84
85    /// Parses a time of any of forms
86    /// - `09:26:56`
87    /// - `09:26:56.123`
88    /// - `09:26:56.123456`
89    /// - `09:26:56.123456789`
90    /// - `092656`
91    ///
92    /// Returning the end byte offset
93    fn time(&self) -> Option<(NaiveTime, usize)> {
94        // Make a NaiveTime handling leap seconds
95        let time = |hour, min, sec, nano| match sec {
96            60 => {
97                let nano = 1_000_000_000 + nano;
98                NaiveTime::from_hms_nano_opt(hour as _, min as _, 59, nano)
99            }
100            _ => NaiveTime::from_hms_nano_opt(hour as _, min as _, sec as _, nano),
101        };
102
103        match (self.mask >> 11) & 0b11111111 {
104            // 09:26:56
105            0b11011011 if self.test(13, b':') && self.test(16, b':') => {
106                let hour = self.digits[11] * 10 + self.digits[12];
107                let minute = self.digits[14] * 10 + self.digits[15];
108                let second = self.digits[17] * 10 + self.digits[18];
109
110                match self.test(19, b'.') {
111                    true => {
112                        let digits = (self.mask >> 20).trailing_ones();
113                        let nanos = match digits {
114                            0 => return None,
115                            1 => parse_nanos::<1, 0>(&self.digits[20..21]),
116                            2 => parse_nanos::<2, 0>(&self.digits[20..22]),
117                            3 => parse_nanos::<3, 0>(&self.digits[20..23]),
118                            4 => parse_nanos::<4, 0>(&self.digits[20..24]),
119                            5 => parse_nanos::<5, 0>(&self.digits[20..25]),
120                            6 => parse_nanos::<6, 0>(&self.digits[20..26]),
121                            7 => parse_nanos::<7, 0>(&self.digits[20..27]),
122                            8 => parse_nanos::<8, 0>(&self.digits[20..28]),
123                            _ => parse_nanos::<9, 0>(&self.digits[20..29]),
124                        };
125                        Some((time(hour, minute, second, nanos)?, 20 + digits as usize))
126                    }
127                    false => Some((time(hour, minute, second, 0)?, 19)),
128                }
129            }
130            // 092656
131            0b111111 => {
132                let hour = self.digits[11] * 10 + self.digits[12];
133                let minute = self.digits[13] * 10 + self.digits[14];
134                let second = self.digits[15] * 10 + self.digits[16];
135                let time = time(hour, minute, second, 0)?;
136                Some((time, 17))
137            }
138            _ => None,
139        }
140    }
141}
142
143/// Accepts a string and parses it relative to the provided `timezone`
144///
145/// In addition to RFC3339 / ISO8601 standard timestamps, it also
146/// accepts strings that use a space ` ` to separate the date and time
147/// as well as strings that have no explicit timezone offset.
148///
149/// Examples of accepted inputs:
150/// * `1997-01-31T09:26:56.123Z`        # RCF3339
151/// * `1997-01-31T09:26:56.123-05:00`   # RCF3339
152/// * `1997-01-31 09:26:56.123-05:00`   # close to RCF3339 but with a space rather than T
153/// * `2023-01-01 04:05:06.789 -08`     # close to RCF3339, no fractional seconds or time separator
154/// * `1997-01-31T09:26:56.123`         # close to RCF3339 but no timezone offset specified
155/// * `1997-01-31 09:26:56.123`         # close to RCF3339 but uses a space and no timezone offset
156/// * `1997-01-31 09:26:56`             # close to RCF3339, no fractional seconds
157/// * `1997-01-31 092656`               # close to RCF3339, no fractional seconds
158/// * `1997-01-31 092656+04:00`         # close to RCF3339, no fractional seconds or time separator
159/// * `1997-01-31`                      # close to RCF3339, only date no time
160///
161/// [IANA timezones] are only supported if the `arrow-array/chrono-tz` feature is enabled
162///
163/// * `2023-01-01 040506 America/Los_Angeles`
164///
165/// If a timestamp is ambiguous, for example as a result of daylight-savings time, an error
166/// will be returned
167///
168/// Some formats supported by PostgresSql <https://www.postgresql.org/docs/current/datatype-datetime.html#DATATYPE-DATETIME-TIME-TABLE>
169/// are not supported, like
170///
171/// * "2023-01-01 04:05:06.789 +07:30:00",
172/// * "2023-01-01 040506 +07:30:00",
173/// * "2023-01-01 04:05:06.789 PST",
174///
175/// [IANA timezones]: https://www.iana.org/time-zones
176pub fn string_to_datetime<T: TimeZone>(timezone: &T, s: &str) -> Result<DateTime<T>, ArrowError> {
177    let err =
178        |ctx: &str| ArrowError::ParseError(format!("Error parsing timestamp from '{s}': {ctx}"));
179
180    let bytes = s.as_bytes();
181    if bytes.len() < 10 {
182        return Err(err("timestamp must contain at least 10 characters"));
183    }
184
185    let parser = TimestampParser::new(bytes);
186    let date = parser.date().ok_or_else(|| err("error parsing date"))?;
187    if bytes.len() == 10 {
188        let datetime = date.and_time(NaiveTime::MIN);
189        return timezone
190            .from_local_datetime(&datetime)
191            .single()
192            .ok_or_else(|| err("error computing timezone offset"));
193    }
194
195    if !parser.test(10, b'T') && !parser.test(10, b't') && !parser.test(10, b' ') {
196        return Err(err("invalid timestamp separator"));
197    }
198
199    let (time, mut tz_offset) = parser.time().ok_or_else(|| err("error parsing time"))?;
200    let datetime = date.and_time(time);
201
202    if tz_offset == 32 {
203        // Decimal overrun
204        while tz_offset < bytes.len() && bytes[tz_offset].is_ascii_digit() {
205            tz_offset += 1;
206        }
207    }
208
209    if bytes.len() <= tz_offset {
210        return timezone
211            .from_local_datetime(&datetime)
212            .single()
213            .ok_or_else(|| err("error computing timezone offset"));
214    }
215
216    if (bytes[tz_offset] == b'z' || bytes[tz_offset] == b'Z') && tz_offset == bytes.len() - 1 {
217        return Ok(timezone.from_utc_datetime(&datetime));
218    }
219
220    // Parse remainder of string as timezone
221    let parsed_tz: Tz = s[tz_offset..].trim_start().parse()?;
222    let parsed = parsed_tz
223        .from_local_datetime(&datetime)
224        .single()
225        .ok_or_else(|| err("error computing timezone offset"))?;
226
227    Ok(parsed.with_timezone(timezone))
228}
229
230/// Accepts a string in RFC3339 / ISO8601 standard format and some
231/// variants and converts it to a nanosecond precision timestamp.
232///
233/// See [`string_to_datetime`] for the full set of supported formats
234///
235/// Implements the `to_timestamp` function to convert a string to a
236/// timestamp, following the model of spark SQL’s to_`timestamp`.
237///
238/// Internally, this function uses the `chrono` library for the
239/// datetime parsing
240///
241/// We hope to extend this function in the future with a second
242/// parameter to specifying the format string.
243///
244/// ## Timestamp Precision
245///
246/// Function uses the maximum precision timestamps supported by
247/// Arrow (nanoseconds stored as a 64-bit integer) timestamps. This
248/// means the range of dates that timestamps can represent is ~1677 AD
249/// to 2262 AM
250///
251/// ## Timezone / Offset Handling
252///
253/// Numerical values of timestamps are stored compared to offset UTC.
254///
255/// This function interprets string without an explicit time zone as timestamps
256/// relative to UTC, see [`string_to_datetime`] for alternative semantics
257///
258/// In particular:
259///
260/// ```
261/// # use arrow_cast::parse::string_to_timestamp_nanos;
262/// // Note all three of these timestamps are parsed as the same value
263/// let a = string_to_timestamp_nanos("1997-01-31 09:26:56.123Z").unwrap();
264/// let b = string_to_timestamp_nanos("1997-01-31T09:26:56.123").unwrap();
265/// let c = string_to_timestamp_nanos("1997-01-31T14:26:56.123+05:00").unwrap();
266///
267/// assert_eq!(a, b);
268/// assert_eq!(b, c);
269/// ```
270///
271#[inline]
272pub fn string_to_timestamp_nanos(s: &str) -> Result<i64, ArrowError> {
273    to_timestamp_nanos(string_to_datetime(&Utc, s)?.naive_utc())
274}
275
276/// Fallible conversion of [`NaiveDateTime`] to `i64` nanoseconds
277#[inline]
278fn to_timestamp_nanos(dt: NaiveDateTime) -> Result<i64, ArrowError> {
279    dt.and_utc()
280        .timestamp_nanos_opt()
281        .ok_or_else(|| ArrowError::ParseError(ERR_NANOSECONDS_NOT_SUPPORTED.to_string()))
282}
283
284/// Accepts a string in ISO8601 standard format and some
285/// variants and converts it to nanoseconds since midnight.
286///
287/// Examples of accepted inputs:
288///
289/// * `09:26:56.123 AM`
290/// * `23:59:59`
291/// * `6:00 pm`
292///
293/// Internally, this function uses the `chrono` library for the time parsing
294///
295/// ## Timezone / Offset Handling
296///
297/// This function does not support parsing strings with a timezone
298/// or offset specified, as it considers only time since midnight.
299pub fn string_to_time_nanoseconds(s: &str) -> Result<i64, ArrowError> {
300    let nt = string_to_time(s)
301        .ok_or_else(|| ArrowError::ParseError(format!("Failed to parse \'{s}\' as time")))?;
302    Ok(nt.num_seconds_from_midnight() as i64 * 1_000_000_000 + nt.nanosecond() as i64)
303}
304
305fn string_to_time(s: &str) -> Option<NaiveTime> {
306    let bytes = s.as_bytes();
307    if bytes.len() < 4 {
308        return None;
309    }
310
311    let (am, bytes) = match bytes.get(bytes.len() - 3..) {
312        Some(b" AM" | b" am" | b" Am" | b" aM") => (Some(true), &bytes[..bytes.len() - 3]),
313        Some(b" PM" | b" pm" | b" pM" | b" Pm") => (Some(false), &bytes[..bytes.len() - 3]),
314        _ => (None, bytes),
315    };
316
317    if bytes.len() < 4 {
318        return None;
319    }
320
321    let mut digits = [b'0'; 6];
322
323    // Extract hour
324    let bytes = match (bytes[1], bytes[2]) {
325        (b':', _) => {
326            digits[1] = bytes[0];
327            &bytes[2..]
328        }
329        (_, b':') => {
330            digits[0] = bytes[0];
331            digits[1] = bytes[1];
332            &bytes[3..]
333        }
334        _ => return None,
335    };
336
337    if bytes.len() < 2 {
338        return None; // Minutes required
339    }
340
341    // Extract minutes
342    digits[2] = bytes[0];
343    digits[3] = bytes[1];
344
345    let nanoseconds = match bytes.get(2) {
346        Some(b':') => {
347            if bytes.len() < 5 {
348                return None;
349            }
350
351            // Extract seconds
352            digits[4] = bytes[3];
353            digits[5] = bytes[4];
354
355            // Extract sub-seconds if any
356            match bytes.get(5) {
357                Some(b'.') => {
358                    let decimal = &bytes[6..];
359                    if decimal.iter().any(|x| !x.is_ascii_digit()) {
360                        return None;
361                    }
362                    match decimal.len() {
363                        0 => return None,
364                        1 => parse_nanos::<1, b'0'>(decimal),
365                        2 => parse_nanos::<2, b'0'>(decimal),
366                        3 => parse_nanos::<3, b'0'>(decimal),
367                        4 => parse_nanos::<4, b'0'>(decimal),
368                        5 => parse_nanos::<5, b'0'>(decimal),
369                        6 => parse_nanos::<6, b'0'>(decimal),
370                        7 => parse_nanos::<7, b'0'>(decimal),
371                        8 => parse_nanos::<8, b'0'>(decimal),
372                        _ => parse_nanos::<9, b'0'>(decimal),
373                    }
374                }
375                Some(_) => return None,
376                None => 0,
377            }
378        }
379        Some(_) => return None,
380        None => 0,
381    };
382
383    digits.iter_mut().for_each(|x| *x = x.wrapping_sub(b'0'));
384    if digits.iter().any(|x| *x > 9) {
385        return None;
386    }
387
388    let hour = match (digits[0] * 10 + digits[1], am) {
389        (12, Some(true)) => 0,               // 12:00 AM -> 00:00
390        (h @ 1..=11, Some(true)) => h,       // 1:00 AM -> 01:00
391        (12, Some(false)) => 12,             // 12:00 PM -> 12:00
392        (h @ 1..=11, Some(false)) => h + 12, // 1:00 PM -> 13:00
393        (_, Some(_)) => return None,
394        (h, None) => h,
395    };
396
397    // Handle leap second
398    let (second, nanoseconds) = match digits[4] * 10 + digits[5] {
399        60 => (59, nanoseconds + 1_000_000_000),
400        s => (s, nanoseconds),
401    };
402
403    NaiveTime::from_hms_nano_opt(
404        hour as _,
405        (digits[2] * 10 + digits[3]) as _,
406        second as _,
407        nanoseconds,
408    )
409}
410
411/// Specialized parsing implementations to convert strings to Arrow types.
412///
413/// This is used by csv and json reader and can be used directly as well.
414///
415/// # Example
416///
417/// To parse a string to a [`Date32Type`]:
418///
419/// ```
420/// use arrow_cast::parse::Parser;
421/// use arrow_array::types::Date32Type;
422/// let date = Date32Type::parse("2021-01-01").unwrap();
423/// assert_eq!(date, 18628);
424/// ```
425///
426/// To parse a string to a [`TimestampNanosecondType`]:
427///
428/// ```
429/// use arrow_cast::parse::Parser;
430/// use arrow_array::types::TimestampNanosecondType;
431/// let ts = TimestampNanosecondType::parse("2021-01-01T00:00:00.123456789Z").unwrap();
432/// assert_eq!(ts, 1609459200123456789);
433/// ```
434pub trait Parser: ArrowPrimitiveType {
435    /// Parse a string to the native type
436    fn parse(string: &str) -> Option<Self::Native>;
437
438    /// Parse a string to the native type with a format string
439    ///
440    /// When not implemented, the format string is unused, and this method is equivalent to [parse](#tymethod.parse)
441    fn parse_formatted(string: &str, _format: &str) -> Option<Self::Native> {
442        Self::parse(string)
443    }
444}
445
446impl Parser for Float16Type {
447    fn parse(string: &str) -> Option<f16> {
448        if let Ok(raw_float) = lexical_core::parse(string.as_bytes()) {
449            return Some(f16::from_f32(raw_float));
450        }
451        let string = trim_pre_and_post_whitespace(string);
452        lexical_core::parse(string.as_bytes())
453            .ok()
454            .map(f16::from_f32)
455    }
456}
457
458impl Parser for Float32Type {
459    fn parse(string: &str) -> Option<f32> {
460        if let Ok(raw_float) = lexical_core::parse(string.as_bytes()) {
461            return Some(raw_float);
462        }
463        let string = trim_pre_and_post_whitespace(string);
464        lexical_core::parse(string.as_bytes()).ok()
465    }
466}
467
468impl Parser for Float64Type {
469    fn parse(string: &str) -> Option<f64> {
470        if let Ok(raw_float) = lexical_core::parse(string.as_bytes()) {
471            return Some(raw_float);
472        }
473        let string = trim_pre_and_post_whitespace(string);
474        lexical_core::parse(string.as_bytes()).ok()
475    }
476}
477
478/// this is a no-op if the string starts and ends with a digit, otherwise it will trim whitespace from the start and end of the string.
479#[inline]
480fn trim_pre_and_post_whitespace(string: &str) -> &str {
481    let bytes = string.as_bytes();
482    let prefix = bytes.first().is_some_and(|b| !b.is_ascii_digit());
483    let suffix = bytes.last().is_some_and(|b| !b.is_ascii_digit());
484    match (prefix, suffix) {
485        (false, false) => string,
486        (true, false) => string.trim_ascii_start(),
487        (false, true) => string.trim_ascii_end(),
488        (true, true) => string.trim_ascii(),
489    }
490}
491
492macro_rules! parser_primitive {
493    ($t:ty) => {
494        impl Parser for $t {
495            fn parse(string: &str) -> Option<Self::Native> {
496                let mut raw_bytes = string.as_bytes();
497                if !raw_bytes.last().is_some_and(|x| x.is_ascii_digit()) {
498                    raw_bytes = raw_bytes.trim_ascii_end();
499                    if !raw_bytes.last().is_some_and(|x| x.is_ascii_digit()) {
500                        return None;
501                    }
502                }
503                match atoi::FromRadix10SignedChecked::from_radix_10_signed_checked(raw_bytes) {
504                    (Some(n), x) if x == raw_bytes.len() => Some(n),
505                    _ => {
506                        let trimmed = raw_bytes.trim_ascii_start();
507                        match atoi::FromRadix10SignedChecked::from_radix_10_signed_checked(trimmed)
508                        {
509                            (Some(n), x) if x == trimmed.len() => Some(n),
510                            _ => None,
511                        }
512                    }
513                }
514            }
515        }
516    };
517}
518parser_primitive!(UInt64Type);
519parser_primitive!(UInt32Type);
520parser_primitive!(UInt16Type);
521parser_primitive!(UInt8Type);
522parser_primitive!(Int64Type);
523parser_primitive!(Int32Type);
524parser_primitive!(Int16Type);
525parser_primitive!(Int8Type);
526parser_primitive!(DurationNanosecondType);
527parser_primitive!(DurationMicrosecondType);
528parser_primitive!(DurationMillisecondType);
529parser_primitive!(DurationSecondType);
530
531impl Parser for TimestampNanosecondType {
532    fn parse(string: &str) -> Option<i64> {
533        if let Ok(nanos) = string_to_timestamp_nanos(string) {
534            return Some(nanos);
535        }
536
537        let trimmed = trim_pre_and_post_whitespace(string);
538        string_to_timestamp_nanos(trimmed).ok()
539    }
540}
541
542impl Parser for TimestampMicrosecondType {
543    fn parse(string: &str) -> Option<i64> {
544        if let Ok(nanos) = string_to_timestamp_nanos(string) {
545            return Some(nanos / 1_000);
546        }
547
548        let trimmed = trim_pre_and_post_whitespace(string);
549        string_to_timestamp_nanos(trimmed).ok().map(|x| x / 1_000)
550    }
551}
552
553impl Parser for TimestampMillisecondType {
554    fn parse(string: &str) -> Option<i64> {
555        if let Ok(nanos) = string_to_timestamp_nanos(string) {
556            return Some(nanos / 1_000_000);
557        }
558
559        let trimmed = trim_pre_and_post_whitespace(string);
560        string_to_timestamp_nanos(trimmed)
561            .ok()
562            .map(|x| x / 1_000_000)
563    }
564}
565
566impl Parser for TimestampSecondType {
567    fn parse(string: &str) -> Option<i64> {
568        if let Ok(nanos) = string_to_timestamp_nanos(string) {
569            return Some(nanos / 1_000_000_000);
570        }
571
572        let trimmed = trim_pre_and_post_whitespace(string);
573        string_to_timestamp_nanos(trimmed)
574            .ok()
575            .map(|x| x / 1_000_000_000)
576    }
577}
578
579impl Parser for Time64NanosecondType {
580    // Will truncate any fractions of a nanosecond
581    fn parse(string: &str) -> Option<Self::Native> {
582        let value = string_to_time_nanoseconds(string)
583            .ok()
584            .or_else(|| string.parse::<Self::Native>().ok());
585
586        if value.is_some() {
587            return value;
588        }
589
590        let trimmed = trim_pre_and_post_whitespace(string);
591        string_to_time_nanoseconds(trimmed)
592            .ok()
593            .or_else(|| string.parse::<Self::Native>().ok())
594    }
595
596    fn parse_formatted(string: &str, format: &str) -> Option<Self::Native> {
597        let nt = NaiveTime::parse_from_str(string, format).ok()?;
598        Some(nt.num_seconds_from_midnight() as i64 * 1_000_000_000 + nt.nanosecond() as i64)
599    }
600}
601
602impl Parser for Time64MicrosecondType {
603    // Will truncate any fractions of a microsecond
604    fn parse(string: &str) -> Option<Self::Native> {
605        let value = string_to_time_nanoseconds(string)
606            .ok()
607            .map(|nanos| nanos / 1_000)
608            .or_else(|| string.parse::<Self::Native>().ok());
609
610        if value.is_some() {
611            return value;
612        }
613
614        let trimmed = trim_pre_and_post_whitespace(string);
615        string_to_time_nanoseconds(trimmed)
616            .ok()
617            .map(|x| x / 1_000)
618            .or_else(|| string.parse::<Self::Native>().ok())
619    }
620
621    fn parse_formatted(string: &str, format: &str) -> Option<Self::Native> {
622        let nt = NaiveTime::parse_from_str(string, format).ok()?;
623        Some(nt.num_seconds_from_midnight() as i64 * 1_000_000 + nt.nanosecond() as i64 / 1_000)
624    }
625}
626
627impl Parser for Time32MillisecondType {
628    // Will truncate any fractions of a millisecond
629    fn parse(string: &str) -> Option<Self::Native> {
630        let value = string_to_time_nanoseconds(string)
631            .ok()
632            .map(|nanos| (nanos / 1_000_000) as i32)
633            .or_else(|| string.parse::<Self::Native>().ok());
634
635        if value.is_some() {
636            return value;
637        }
638
639        let trimmed = trim_pre_and_post_whitespace(string);
640        string_to_time_nanoseconds(trimmed)
641            .ok()
642            .map(|x| (x / 1_000_000) as i32)
643            .or_else(|| string.parse::<Self::Native>().ok())
644    }
645
646    fn parse_formatted(string: &str, format: &str) -> Option<Self::Native> {
647        let nt = NaiveTime::parse_from_str(string, format).ok()?;
648        Some(nt.num_seconds_from_midnight() as i32 * 1_000 + nt.nanosecond() as i32 / 1_000_000)
649    }
650}
651
652impl Parser for Time32SecondType {
653    // Will truncate any fractions of a second
654    fn parse(string: &str) -> Option<Self::Native> {
655        let value = string_to_time_nanoseconds(string)
656            .ok()
657            .map(|nanos| (nanos / 1_000_000_000) as i32)
658            .or_else(|| string.parse::<Self::Native>().ok());
659
660        if value.is_some() {
661            return value;
662        }
663
664        let trimmed = trim_pre_and_post_whitespace(string);
665        string_to_time_nanoseconds(trimmed)
666            .ok()
667            .map(|x| (x / 1_000_000_000) as i32)
668            .or_else(|| string.parse::<Self::Native>().ok())
669    }
670
671    fn parse_formatted(string: &str, format: &str) -> Option<Self::Native> {
672        let nt = NaiveTime::parse_from_str(string, format).ok()?;
673        Some(nt.num_seconds_from_midnight() as i32 + nt.nanosecond() as i32 / 1_000_000_000)
674    }
675}
676
677/// Number of days between 0001-01-01 and 1970-01-01
678const EPOCH_DAYS_FROM_CE: i32 = 719_163;
679
680/// Error message if nanosecond conversion request beyond supported interval
681const ERR_NANOSECONDS_NOT_SUPPORTED: &str = "The dates that can be represented as nanoseconds have to be between 1677-09-21T00:12:44.0 and 2262-04-11T23:47:16.854775804";
682
683/// Parse the ISO 8601 signed extended-year form (`±YYYY[Y...]-MM-DD`) into
684/// raw `(year, month, day)` components, without validating the calendar date.
685///
686/// The caller must have already verified that `string` begins with `+` or `-`;
687/// the year must have at least 4 digits. Returns `None` if the shape is
688/// malformed or any component fails to parse numerically.
689fn parse_extended_ymd(string: &str) -> Option<(i32, u32, u32)> {
690    debug_assert!(string.starts_with('+') || string.starts_with('-'));
691    // Skip the sign and look for the hyphen that terminates the year digits.
692    // Per ISO 8601 the unsigned year part must be at least 4 digits.
693    let rest = &string[1..];
694    let hyphen = rest.find('-')?;
695    if hyphen < 4 {
696        return None;
697    }
698    // The year substring is the sign and the digits (but not the separator),
699    // e.g. for "+10999-12-31", hyphen is 5 and s[..6] is "+10999".
700    let year: i32 = string[..hyphen + 1].parse().ok()?;
701    // The remainder should begin with a '-' which we strip off, leaving the month-day part.
702    let remainder = string[hyphen + 1..].strip_prefix('-')?;
703    let mut parts = remainder.splitn(2, '-');
704    let month: u32 = parts.next()?.parse().ok()?;
705    let day: u32 = parts.next()?.parse().ok()?;
706    Some((year, month, day))
707}
708
709fn parse_date(string: &str) -> Option<NaiveDate> {
710    // If the date has an extended (signed) year such as "+10999-12-31" or "-0012-05-06"
711    //
712    // According to [ISO 8601], years have:
713    //  Four digits or more for the year. Years in the range 0000 to 9999 will be pre-padded by
714    //  zero to ensure four digits. Years outside that range will have a prefixed positive or negative symbol.
715    //
716    // [ISO 8601]: https://docs.oracle.com/en/java/javase/17/docs/api/java.base/java/time/format/DateTimeFormatter.html#ISO_LOCAL_DATE
717    if string.starts_with('+') || string.starts_with('-') {
718        let (year, month, day) = parse_extended_ymd(string)?;
719        return NaiveDate::from_ymd_opt(year, month, day);
720    }
721
722    if string.len() > 10 {
723        // Try to parse as datetime and return just the date part
724        return string_to_datetime(&Utc, string)
725            .map(|dt| dt.date_naive())
726            .ok();
727    }
728    let mut digits = [0; 10];
729    let mut mask = 0;
730
731    // Treating all bytes the same way, helps LLVM vectorise this correctly
732    for (idx, (o, i)) in digits.iter_mut().zip(string.bytes()).enumerate() {
733        *o = i.wrapping_sub(b'0');
734        mask |= ((*o < 10) as u16) << idx
735    }
736
737    const HYPHEN: u8 = b'-'.wrapping_sub(b'0');
738
739    //  refer to https://www.rfc-editor.org/rfc/rfc3339#section-3
740    if digits[4] != HYPHEN {
741        let (year, month, day) = match (mask, string.len()) {
742            (0b11111111, 8) => (
743                digits[0] as u16 * 1000
744                    + digits[1] as u16 * 100
745                    + digits[2] as u16 * 10
746                    + digits[3] as u16,
747                digits[4] * 10 + digits[5],
748                digits[6] * 10 + digits[7],
749            ),
750            _ => return None,
751        };
752        return NaiveDate::from_ymd_opt(year as _, month as _, day as _);
753    }
754
755    let (month, day) = match mask {
756        0b1101101111 => {
757            if digits[7] != HYPHEN {
758                return None;
759            }
760            (digits[5] * 10 + digits[6], digits[8] * 10 + digits[9])
761        }
762        0b101101111 => {
763            if digits[7] != HYPHEN {
764                return None;
765            }
766            (digits[5] * 10 + digits[6], digits[8])
767        }
768        0b110101111 => {
769            if digits[6] != HYPHEN {
770                return None;
771            }
772            (digits[5], digits[7] * 10 + digits[8])
773        }
774        0b10101111 => {
775            if digits[6] != HYPHEN {
776                return None;
777            }
778            (digits[5], digits[7])
779        }
780        _ => return None,
781    };
782
783    let year =
784        digits[0] as u16 * 1000 + digits[1] as u16 * 100 + digits[2] as u16 * 10 + digits[3] as u16;
785
786    NaiveDate::from_ymd_opt(year as _, month as _, day as _)
787}
788
789/// Parse a date string into days since 1970-01-01, covering the full
790/// `Date32` range (years ≈ ±5,881,580) for the signed extended-year form.
791///
792/// The Gregorian calendar repeats exactly every 400 years (146,097 days), so
793/// we fold the year into `[0, 400)`, validate the folded date, and add
794/// `era * 146_097` to recover the absolute day count.
795///
796/// For all other inputs, behavior matches [`parse_date`].
797fn parse_date_to_days(string: &str) -> Option<i32> {
798    if string.starts_with('+') || string.starts_with('-') {
799        let (year, month, day) = parse_extended_ymd(string)?;
800        let y = year as i64;
801        let era = y.div_euclid(400);
802        let yoe = y.rem_euclid(400) as i32;
803        let naive_date = NaiveDate::from_ymd_opt(yoe, month, day)?;
804        let in_era = (naive_date.num_days_from_ce() - EPOCH_DAYS_FROM_CE) as i64;
805        return i32::try_from(era * 146_097 + in_era).ok();
806    }
807    parse_date(string).map(|naive_date| naive_date.num_days_from_ce() - EPOCH_DAYS_FROM_CE)
808}
809
810impl Parser for Date32Type {
811    fn parse(string: &str) -> Option<i32> {
812        if let Some(days) = parse_date_to_days(string) {
813            return Some(days);
814        }
815
816        let trimmed = trim_pre_and_post_whitespace(string);
817        parse_date_to_days(trimmed)
818    }
819
820    fn parse_formatted(string: &str, format: &str) -> Option<i32> {
821        let date = NaiveDate::parse_from_str(string, format).ok()?;
822        Some(date.num_days_from_ce() - EPOCH_DAYS_FROM_CE)
823    }
824}
825
826impl Parser for Date64Type {
827    fn parse(string: &str) -> Option<i64> {
828        if string.len() <= 10 {
829            let datetime = NaiveDateTime::new(parse_date(string)?, NaiveTime::default());
830            Some(datetime.and_utc().timestamp_millis())
831        } else {
832            let date_time = string_to_datetime(&Utc, string).ok()?;
833            Some(date_time.timestamp_millis())
834        }
835    }
836
837    fn parse_formatted(string: &str, format: &str) -> Option<i64> {
838        use chrono::format::Fixed;
839        use chrono::format::StrftimeItems;
840        let fmt = StrftimeItems::new(format);
841        let has_zone = fmt.into_iter().any(|item| match item {
842            chrono::format::Item::Fixed(fixed_item) => matches!(
843                fixed_item,
844                Fixed::RFC2822
845                    | Fixed::RFC3339
846                    | Fixed::TimezoneName
847                    | Fixed::TimezoneOffsetColon
848                    | Fixed::TimezoneOffsetColonZ
849                    | Fixed::TimezoneOffset
850                    | Fixed::TimezoneOffsetZ
851            ),
852            _ => false,
853        });
854        if has_zone {
855            let date_time = chrono::DateTime::parse_from_str(string, format).ok()?;
856            Some(date_time.timestamp_millis())
857        } else {
858            let date_time = NaiveDateTime::parse_from_str(string, format).ok()?;
859            Some(date_time.and_utc().timestamp_millis())
860        }
861    }
862}
863
864/// Parses the string representation of a decimal number into the unscaled
865/// native value of a decimal type with the given `precision` and `scale`.
866///
867/// The accepted syntax is:
868///
869/// ```text
870/// [whitespace] [+|-] digits [. [digits]] [(e|E) [+|-] digits] [whitespace]
871/// ```
872///
873/// or the same with the integer digits omitted (e.g. `.5`), as long as at
874/// least one digit is present in the mantissa. ASCII whitespace is trimmed
875/// from both ends. The exponent is applied before scaling, so `1.5e2` and
876/// `150` parse identically.
877///
878/// Fractional digits beyond `scale` are not stored but round the result half
879/// away from zero (e.g. `1.005` at scale 2 is `101`, `-1.005` is `-101`).
880/// Negative scales are supported and round the integer part in the same way
881/// (e.g. `150` at scale -2 is `2`).
882///
883/// Returns an error if the input is not a valid decimal string, or if the
884/// result does not fit the given precision.
885///
886/// # Example
887///
888/// ```
889/// # use arrow_array::types::Decimal128Type;
890/// # use arrow_cast::parse::parse_decimal;
891/// assert_eq!(parse_decimal::<Decimal128Type>("123.45", 10, 2).unwrap(), 12345);
892/// assert_eq!(parse_decimal::<Decimal128Type>("1.005", 10, 2).unwrap(), 101);
893/// assert_eq!(parse_decimal::<Decimal128Type>("1.5e2", 10, 0).unwrap(), 150);
894/// assert!(parse_decimal::<Decimal128Type>("1234.5", 5, 2).is_err()); // does not fit
895/// ```
896pub fn parse_decimal<T: DecimalType>(
897    s: &str,
898    precision: u8,
899    scale: i8,
900) -> Result<T::Native, ArrowError> {
901    parse_decimal_checked::<T>(s, precision, scale).map_err(|e| match e {
902        DecimalParseError::Overflow => ArrowError::ParseError(format!(
903            "{s:?} does not fit in {}({precision}, {scale})",
904            T::PREFIX
905        )),
906        DecimalParseError::InvalidFormat => {
907            ArrowError::ParseError(format!("Invalid decimal format: {s:?}"))
908        }
909    })
910}
911
912/// The reason a decimal string could not be parsed.
913#[derive(Debug, Clone, Copy, PartialEq, Eq)]
914pub(crate) enum DecimalParseError {
915    /// The input is not a valid decimal string
916    InvalidFormat,
917    /// The value does not fit in the precision or the native type of the decimal
918    Overflow,
919}
920
921/// Like [`parse_decimal`], but reports failures as a [`DecimalParseError`]
922/// instead of formatting an error message, for callers that discard or
923/// re-wrap the error.
924pub(crate) fn parse_decimal_checked<T: DecimalType>(
925    s: &str,
926    precision: u8,
927    scale: i8,
928) -> Result<T::Native, DecimalParseError> {
929    let (value, digits) = parse_decimal_native::<T>(s, scale)?;
930    // A value of at most `precision` digits is within the precision without
931    // inspecting it. A precision beyond the type's maximum is invalid.
932    let fits = precision <= T::MAX_PRECISION
933        && (digits <= precision as usize || T::is_valid_decimal_precision(value, precision));
934    if fits {
935        Ok(value)
936    } else {
937        Err(DecimalParseError::Overflow)
938    }
939}
940
941/// Parses `s` as a decimal with the given `scale` into the native type of `T`,
942/// checking only that the result fits the native type (not the precision),
943/// and returns it with an upper bound on its number of decimal digits.
944///
945/// See [`parse_decimal`] for the accepted syntax and rounding behaviour.
946#[inline]
947fn parse_decimal_native<T: DecimalType>(
948    s: &str,
949    scale: i8,
950) -> Result<(T::Native, usize), DecimalParseError> {
951    let bytes = s.as_bytes().trim_ascii();
952    let (negative, mut mantissa) = split_sign(bytes);
953
954    let mut scale = scale as i64;
955    loop {
956        let exponent_at = match parse_decimal_mantissa::<T>(mantissa, negative, scale) {
957            Ok(result) => return Ok(result),
958            Err(MantissaError::InvalidFormat) => return Err(DecimalParseError::InvalidFormat),
959            Err(MantissaError::Exponent(index)) => index,
960            // The digits before an exponent marker need not fit on their own
961            // (e.g. "4825037936439135476E-14"), so the overflow only stands
962            // if no marker follows
963            Err(MantissaError::Overflow) => mantissa
964                .iter()
965                .position(|b| matches!(b, b'e' | b'E'))
966                .ok_or(DecimalParseError::Overflow)?,
967        };
968
969        // If we saw an exponent, update the effective scale and rescan.
970        // Exponents are rare, so the risk of repeated work is preferable to
971        // the cost of scanning ahead for an exponent marker for every input.
972        let exponent = parse_decimal_exponent(&mantissa[exponent_at + 1..])?;
973        scale = scale.saturating_add(exponent);
974
975        // Trim the exponent so the next iteration succeeds without rescanning
976        mantissa = &mantissa[..exponent_at];
977    }
978}
979
980/// Why scanning the digits of a decimal string stopped.
981enum MantissaError {
982    /// The input is not a valid decimal string
983    InvalidFormat,
984    /// The value does not fit in the native type
985    Overflow,
986    /// An exponent marker (`e` or `E`) was found at the given byte offset
987    Exponent(usize),
988}
989
990impl From<DecimalParseError> for MantissaError {
991    fn from(e: DecimalParseError) -> Self {
992        match e {
993            DecimalParseError::InvalidFormat => Self::InvalidFormat,
994            DecimalParseError::Overflow => Self::Overflow,
995        }
996    }
997}
998
999/// The maximum number of decimal digits accumulated in a `u64` before the
1000/// chunk is folded into the native value: every 18-digit number fits in a
1001/// `u64`, and splits into two halves that each fit in a `u32` (see
1002/// [`decimal_chunk_to_native`]).
1003const MAX_CHUNK_DIGITS: usize = 18;
1004
1005/// Scans `mantissa` (digits with at most one decimal point; the sign has
1006/// already been removed) and folds the digits that are significant at
1007/// `scale` into a native value, rounding half away from zero on the first
1008/// digit that is not. Also returns an upper bound on the number of decimal
1009/// digits of the value: the digits kept, the zeros appended to reach the
1010/// scale, and the digit that rounding up can add.
1011#[inline]
1012fn parse_decimal_mantissa<T: DecimalType>(
1013    mantissa: &[u8],
1014    negative: bool,
1015    scale: i64,
1016) -> Result<(T::Native, usize), MantissaError> {
1017    // The number of integer and fractional digits that contribute to the
1018    // result. For a non-negative scale that is every integer digit and the
1019    // first `scale` fractional digits. For a negative scale the last `-scale`
1020    // integer digits (and every fractional digit) only matter for rounding.
1021    let (int_keep, frac_keep, mut round) = if scale >= 0 {
1022        (
1023            usize::MAX,
1024            usize::try_from(scale).unwrap_or(usize::MAX),
1025            true,
1026        )
1027    } else {
1028        let int_digits = mantissa.iter().take_while(|b| b.is_ascii_digit()).count();
1029        match usize::try_from(int_digits as i64 + scale) {
1030            Ok(keep) => (keep, 0, true),
1031            // Even the first digit is more than one position below the
1032            // least significant digit of the result: the value rounds to
1033            // zero regardless of what the digits are
1034            Err(_) => (0, 0, false),
1035        }
1036    };
1037
1038    let mut acc = DecimalAccumulator::<T> {
1039        value: T::Native::ZERO,
1040        chunk: 0,
1041        chunk_len: 0,
1042        negative,
1043    };
1044    let mut int_kept = 0_usize;
1045    let mut frac_kept = 0_usize;
1046    let mut first_discarded_digit = None;
1047
1048    // Digits before the decimal point
1049    let mut index = 0;
1050    while let Some(&b) = mantissa.get(index) {
1051        if !b.is_ascii_digit() {
1052            break;
1053        }
1054        if int_kept < int_keep {
1055            int_kept += 1;
1056            acc.push(b - b'0')?;
1057        } else {
1058            first_discarded_digit.get_or_insert(b - b'0');
1059        }
1060        index += 1;
1061    }
1062
1063    // Digits after the decimal point
1064    if mantissa.get(index) == Some(&b'.') {
1065        index += 1;
1066        while let Some(&b) = mantissa.get(index) {
1067            if !b.is_ascii_digit() {
1068                break;
1069            }
1070            if frac_kept < frac_keep {
1071                frac_kept += 1;
1072                acc.push(b - b'0')?;
1073            } else {
1074                first_discarded_digit.get_or_insert(b - b'0');
1075            }
1076            index += 1;
1077        }
1078    }
1079
1080    match mantissa.get(index) {
1081        None => {}
1082        Some(b'e' | b'E') => return Err(MantissaError::Exponent(index)),
1083        Some(_) => return Err(MantissaError::InvalidFormat),
1084    }
1085
1086    if int_kept == 0 && frac_kept == 0 && first_discarded_digit.is_none() {
1087        return Err(MantissaError::InvalidFormat);
1088    }
1089
1090    let mut value = acc.finish()?;
1091
1092    // Scale the value up to the target scale. Skipped for zero, where computing
1093    // 10^missing could overflow the native type even though the result (zero)
1094    // is always representable.
1095    let missing = scale - frac_kept as i64;
1096    if missing > 0 && !value.is_zero() {
1097        value = value
1098            .mul_checked(decimal_pow::<T>(missing)?)
1099            .map_err(|_| MantissaError::Overflow)?;
1100    }
1101
1102    round &= first_discarded_digit.is_some_and(|digit| digit >= 5);
1103    if round {
1104        value = if negative {
1105            value.sub_checked(T::Native::ONE)
1106        } else {
1107            value.add_checked(T::Native::ONE)
1108        }
1109        .map_err(|_| MantissaError::Overflow)?;
1110    }
1111
1112    let digits = usize::try_from(missing.max(0))
1113        .unwrap_or(usize::MAX)
1114        .saturating_add(int_kept + frac_kept + round as usize);
1115    Ok((value, digits))
1116}
1117
1118/// Parses the digits of an exponent (`[+|-] digits`), saturating at the bounds
1119/// of `i64`; any exponent that large scales every non-zero mantissa out of
1120/// range of every decimal type.
1121fn parse_decimal_exponent(exponent: &[u8]) -> Result<i64, DecimalParseError> {
1122    let (negative, digits) = split_sign(exponent);
1123    if digits.is_empty() {
1124        return Err(DecimalParseError::InvalidFormat);
1125    }
1126    let mut value = 0_i64;
1127    for &b in digits {
1128        if !b.is_ascii_digit() {
1129            return Err(DecimalParseError::InvalidFormat);
1130        }
1131        value = value.saturating_mul(10).saturating_add((b - b'0') as i64);
1132    }
1133    Ok(if negative { -value } else { value })
1134}
1135
1136/// Splits an optional leading sign from `bytes`, returning whether it is `-`
1137/// and the bytes that follow it.
1138#[inline]
1139fn split_sign(bytes: &[u8]) -> (bool, &[u8]) {
1140    match bytes.first() {
1141        Some(b'-') => (true, &bytes[1..]),
1142        Some(b'+') => (false, &bytes[1..]),
1143        _ => (false, bytes),
1144    }
1145}
1146
1147/// Accumulates decimal digits into `chunk`, folding it into `value` whenever
1148/// it reaches [`MAX_CHUNK_DIGITS`] digits
1149struct DecimalAccumulator<T: DecimalType> {
1150    value: T::Native,
1151    chunk: u64,
1152    chunk_len: usize,
1153    negative: bool,
1154}
1155
1156impl<T: DecimalType> DecimalAccumulator<T> {
1157    #[inline(always)]
1158    fn push(&mut self, digit: u8) -> Result<(), DecimalParseError> {
1159        // Cannot overflow: the chunk is folded into `value` before it exceeds
1160        // MAX_CHUNK_DIGITS digits, all of which fit in a u64
1161        self.chunk = self.chunk * 10 + digit as u64;
1162        self.chunk_len += 1;
1163        if self.chunk_len == MAX_CHUNK_DIGITS {
1164            self.value =
1165                fold_decimal_chunk::<T>(self.value, self.chunk, self.chunk_len, self.negative)?;
1166            self.chunk = 0;
1167            self.chunk_len = 0;
1168        }
1169        Ok(())
1170    }
1171
1172    /// Folds the digits still in `chunk` into the value
1173    #[inline]
1174    fn finish(self) -> Result<T::Native, DecimalParseError> {
1175        if self.chunk_len == 0 {
1176            return Ok(self.value);
1177        }
1178        fold_decimal_chunk::<T>(self.value, self.chunk, self.chunk_len, self.negative)
1179    }
1180}
1181
1182/// Folds a chunk of up to [`MAX_CHUNK_DIGITS`] digits into `value`, producing
1183/// `value * 10^chunk_len + chunk` (`chunk` is negated first when parsing a
1184/// negative number).
1185#[inline(always)]
1186fn fold_decimal_chunk<T: DecimalType>(
1187    value: T::Native,
1188    chunk: u64,
1189    chunk_len: usize,
1190    negative: bool,
1191) -> Result<T::Native, DecimalParseError> {
1192    let chunk = decimal_chunk_to_native::<T>(chunk, negative)?;
1193
1194    // When `value` is zero the multiply would be a no-op; skipping it avoids
1195    // computing 10^chunk_len, which can overflow a narrow native type even
1196    // though the result (the chunk itself) is representable.
1197    if value.is_zero() {
1198        return Ok(chunk);
1199    }
1200
1201    value
1202        .mul_checked(decimal_pow::<T>(chunk_len as i64)?)
1203        .map_err(|_| DecimalParseError::Overflow)?
1204        .add_checked(chunk)
1205        .map_err(|_| DecimalParseError::Overflow)
1206}
1207
1208/// Converts a chunk of at most [`MAX_CHUNK_DIGITS`] digits to the native type,
1209/// negated if `negative`.
1210#[inline]
1211fn decimal_chunk_to_native<T: DecimalType>(
1212    chunk: u64,
1213    negative: bool,
1214) -> Result<T::Native, DecimalParseError> {
1215    // Every native type can represent +/- 10^9, so a chunk below that converts
1216    // losslessly through usize on every target. So does any chunk when the
1217    // native type holds MAX_CHUNK_DIGITS digits and usize holds a u64.
1218    const HALF: u64 = 1_000_000_000;
1219    if chunk < HALF || (T::MAX_PRECISION as usize >= MAX_CHUNK_DIGITS && usize::BITS >= 64) {
1220        let chunk = T::Native::usize_as(chunk as usize);
1221        // `ZERO.sub_wrapping` rather than `neg_wrapping`: the latter compiles
1222        // to measurably slower code for i256 (~10% on casting strings to
1223        // Decimal256)
1224        return Ok(if negative {
1225            T::Native::ZERO.sub_wrapping(chunk)
1226        } else {
1227            chunk
1228        });
1229    }
1230    // Otherwise narrow the chunk in two halves that are each below 10^9
1231    let low = T::Native::usize_as((chunk % HALF) as usize);
1232    let high = T::Native::usize_as((chunk / HALF) as usize)
1233        .mul_checked(T::Native::usize_as(HALF as usize))
1234        .map_err(|_| DecimalParseError::Overflow)?;
1235    // Negate before combining so that a chunk with the magnitude of the
1236    // native type's minimum value (e.g. "2147483648" for Decimal32) remains
1237    // representable
1238    if negative {
1239        T::Native::ZERO
1240            .sub_checked(high)
1241            .map_err(|_| DecimalParseError::Overflow)?
1242            .sub_checked(low)
1243            .map_err(|_| DecimalParseError::Overflow)
1244    } else {
1245        high.add_checked(low)
1246            .map_err(|_| DecimalParseError::Overflow)
1247    }
1248}
1249
1250/// Returns `10^exp` as a `T::Native`, or an overflow error if the result does
1251/// not fit in the native type.
1252#[inline]
1253fn decimal_pow<T: DecimalType>(exp: i64) -> Result<T::Native, DecimalParseError> {
1254    // T::MAX_FOR_EACH_PRECISION[k] holds 10^k - 1, so adding one yields 10^k
1255    // without computing a power at runtime. Exponents beyond the table always
1256    // overflow: the native type cannot hold 10^(MAX_PRECISION + 1).
1257    usize::try_from(exp)
1258        .ok()
1259        .and_then(|exp| T::MAX_FOR_EACH_PRECISION.get(exp))
1260        .map(|max| max.add_wrapping(T::Native::ONE))
1261        .ok_or(DecimalParseError::Overflow)
1262}
1263
1264/// Parse human-readable interval string to Arrow [IntervalYearMonthType]
1265pub fn parse_interval_year_month(
1266    value: &str,
1267) -> Result<<IntervalYearMonthType as ArrowPrimitiveType>::Native, ArrowError> {
1268    let config = IntervalParseConfig::new(IntervalUnit::Year);
1269    let interval = Interval::parse(value, &config)?;
1270
1271    let months = interval.to_year_months().map_err(|_| {
1272        ArrowError::CastError(format!(
1273            "Cannot cast {value} to IntervalYearMonth. Only year and month fields are allowed."
1274        ))
1275    })?;
1276
1277    Ok(IntervalYearMonthType::make_value(0, months))
1278}
1279
1280/// Parse human-readable interval string to Arrow [IntervalDayTimeType]
1281pub fn parse_interval_day_time(
1282    value: &str,
1283) -> Result<<IntervalDayTimeType as ArrowPrimitiveType>::Native, ArrowError> {
1284    let config = IntervalParseConfig::new(IntervalUnit::Day);
1285    let interval = Interval::parse(value, &config)?;
1286
1287    let (days, millis) = interval.to_day_time().map_err(|_| ArrowError::CastError(format!(
1288        "Cannot cast {value} to IntervalDayTime because the nanos part isn't multiple of milliseconds"
1289    )))?;
1290
1291    Ok(IntervalDayTimeType::make_value(days, millis))
1292}
1293
1294/// Parse human-readable interval string to Arrow [IntervalMonthDayNanoType]
1295pub fn parse_interval_month_day_nano_config(
1296    value: &str,
1297    config: IntervalParseConfig,
1298) -> Result<<IntervalMonthDayNanoType as ArrowPrimitiveType>::Native, ArrowError> {
1299    let interval = Interval::parse(value, &config)?;
1300
1301    let (months, days, nanos) = interval.to_month_day_nanos();
1302
1303    Ok(IntervalMonthDayNanoType::make_value(months, days, nanos))
1304}
1305
1306/// Parse human-readable interval string to Arrow [IntervalMonthDayNanoType]
1307pub fn parse_interval_month_day_nano(
1308    value: &str,
1309) -> Result<<IntervalMonthDayNanoType as ArrowPrimitiveType>::Native, ArrowError> {
1310    parse_interval_month_day_nano_config(value, IntervalParseConfig::new(IntervalUnit::Month))
1311}
1312
1313const NANOS_PER_MILLIS: i64 = 1_000_000;
1314const NANOS_PER_SECOND: i64 = 1_000 * NANOS_PER_MILLIS;
1315const NANOS_PER_MINUTE: i64 = 60 * NANOS_PER_SECOND;
1316const NANOS_PER_HOUR: i64 = 60 * NANOS_PER_MINUTE;
1317#[cfg(test)]
1318const NANOS_PER_DAY: i64 = 24 * NANOS_PER_HOUR;
1319
1320/// Config to parse interval strings
1321///
1322/// Currently stores the `default_unit` to use if the string doesn't have one specified
1323#[derive(Debug, Clone)]
1324pub struct IntervalParseConfig {
1325    /// The default unit to use if none is specified
1326    /// e.g. `INTERVAL 1` represents `INTERVAL 1 SECOND` when default_unit = [IntervalUnit::Second]
1327    default_unit: IntervalUnit,
1328}
1329
1330impl IntervalParseConfig {
1331    /// Create a new [IntervalParseConfig] with the given default unit
1332    pub fn new(default_unit: IntervalUnit) -> Self {
1333        Self { default_unit }
1334    }
1335}
1336
1337#[rustfmt::skip]
1338#[derive(Debug, Clone, Copy)]
1339#[repr(u16)]
1340/// Represents the units of an interval, with each variant
1341/// corresponding to a bit in the interval's bitfield representation
1342pub enum IntervalUnit {
1343    /// A Century
1344    Century     = 0b_0000_0000_0001,
1345    /// A Decade
1346    Decade      = 0b_0000_0000_0010,
1347    /// A Year
1348    Year        = 0b_0000_0000_0100,
1349    /// A Month
1350    Month       = 0b_0000_0000_1000,
1351    /// A Week
1352    Week        = 0b_0000_0001_0000,
1353    /// A Day
1354    Day         = 0b_0000_0010_0000,
1355    /// An Hour
1356    Hour        = 0b_0000_0100_0000,
1357    /// A Minute
1358    Minute      = 0b_0000_1000_0000,
1359    /// A Second
1360    Second      = 0b_0001_0000_0000,
1361    /// A Millisecond
1362    Millisecond = 0b_0010_0000_0000,
1363    /// A Microsecond
1364    Microsecond = 0b_0100_0000_0000,
1365    /// A Nanosecond
1366    Nanosecond  = 0b_1000_0000_0000,
1367}
1368
1369/// Logic for parsing interval unit strings
1370///
1371/// See <https://github.com/postgres/postgres/blob/2caa85f4aae689e6f6721d7363b4c66a2a6417d6/src/backend/utils/adt/datetime.c#L189>
1372/// for a list of unit names supported by PostgreSQL which we try to match here.
1373impl FromStr for IntervalUnit {
1374    type Err = ArrowError;
1375
1376    fn from_str(s: &str) -> Result<Self, ArrowError> {
1377        match s.to_lowercase().as_str() {
1378            "c" | "cent" | "cents" | "century" | "centuries" => Ok(Self::Century),
1379            "dec" | "decs" | "decade" | "decades" => Ok(Self::Decade),
1380            "y" | "yr" | "yrs" | "year" | "years" => Ok(Self::Year),
1381            "mon" | "mons" | "month" | "months" => Ok(Self::Month),
1382            "w" | "week" | "weeks" => Ok(Self::Week),
1383            "d" | "day" | "days" => Ok(Self::Day),
1384            "h" | "hr" | "hrs" | "hour" | "hours" => Ok(Self::Hour),
1385            "m" | "min" | "mins" | "minute" | "minutes" => Ok(Self::Minute),
1386            "s" | "sec" | "secs" | "second" | "seconds" => Ok(Self::Second),
1387            "ms" | "msec" | "msecs" | "msecond" | "mseconds" | "millisecond" | "milliseconds" => {
1388                Ok(Self::Millisecond)
1389            }
1390            "us" | "usec" | "usecs" | "usecond" | "useconds" | "microsecond" | "microseconds" => {
1391                Ok(Self::Microsecond)
1392            }
1393            "nanosecond" | "nanoseconds" => Ok(Self::Nanosecond),
1394            _ => Err(ArrowError::InvalidArgumentError(format!(
1395                "Unknown interval type: {s}"
1396            ))),
1397        }
1398    }
1399}
1400
1401impl IntervalUnit {
1402    fn from_str_or_config(
1403        s: Option<&str>,
1404        config: &IntervalParseConfig,
1405    ) -> Result<Self, ArrowError> {
1406        match s {
1407            Some(s) => s.parse(),
1408            None => Ok(config.default_unit),
1409        }
1410    }
1411}
1412
1413/// A tuple representing (months, days, nanoseconds) in an interval
1414pub type MonthDayNano = (i32, i32, i64);
1415
1416/// Chosen based on the number of decimal digits in 1 week in nanoseconds
1417const INTERVAL_PRECISION: u32 = 15;
1418
1419#[derive(Clone, Copy, Debug, PartialEq)]
1420struct IntervalAmount {
1421    /// The integer component of the interval amount
1422    integer: i64,
1423    /// The fractional component multiplied by 10^INTERVAL_PRECISION
1424    frac: i64,
1425}
1426
1427#[cfg(test)]
1428impl IntervalAmount {
1429    fn new(integer: i64, frac: i64) -> Self {
1430        Self { integer, frac }
1431    }
1432}
1433
1434impl FromStr for IntervalAmount {
1435    type Err = ArrowError;
1436
1437    fn from_str(s: &str) -> Result<Self, Self::Err> {
1438        match s.split_once('.') {
1439            Some((integer, frac))
1440                if frac.len() <= INTERVAL_PRECISION as usize
1441                    && !frac.is_empty()
1442                    && !frac.starts_with('-') =>
1443            {
1444                // integer will be "" for values like ".5"
1445                // and "-" for values like "-.5"
1446                let explicit_neg = integer.starts_with('-');
1447                let integer = if integer.is_empty() || integer == "-" {
1448                    Ok(0)
1449                } else {
1450                    integer.parse::<i64>().map_err(|_| {
1451                        ArrowError::ParseError(format!("Failed to parse {s} as interval amount"))
1452                    })
1453                }?;
1454
1455                let frac_unscaled = frac.parse::<i64>().map_err(|_| {
1456                    ArrowError::ParseError(format!("Failed to parse {s} as interval amount"))
1457                })?;
1458
1459                // scale fractional part by interval precision
1460                let frac = frac_unscaled * 10_i64.pow(INTERVAL_PRECISION - frac.len() as u32);
1461
1462                // propagate the sign of the integer part to the fractional part
1463                let frac = if integer < 0 || explicit_neg {
1464                    -frac
1465                } else {
1466                    frac
1467                };
1468
1469                let result = Self { integer, frac };
1470
1471                Ok(result)
1472            }
1473            Some((_, frac)) if frac.starts_with('-') => Err(ArrowError::ParseError(format!(
1474                "Failed to parse {s} as interval amount"
1475            ))),
1476            Some((_, frac)) if frac.len() > INTERVAL_PRECISION as usize => {
1477                Err(ArrowError::ParseError(format!(
1478                    "{s} exceeds the precision available for interval amount"
1479                )))
1480            }
1481            Some(_) | None => {
1482                let integer = s.parse::<i64>().map_err(|_| {
1483                    ArrowError::ParseError(format!("Failed to parse {s} as interval amount"))
1484                })?;
1485
1486                let result = Self { integer, frac: 0 };
1487                Ok(result)
1488            }
1489        }
1490    }
1491}
1492
1493#[derive(Debug, Default, PartialEq)]
1494struct Interval {
1495    months: i32,
1496    days: i32,
1497    nanos: i64,
1498}
1499
1500impl Interval {
1501    fn new(months: i32, days: i32, nanos: i64) -> Self {
1502        Self {
1503            months,
1504            days,
1505            nanos,
1506        }
1507    }
1508
1509    fn to_year_months(&self) -> Result<i32, ArrowError> {
1510        match (self.months, self.days, self.nanos) {
1511            (months, days, nanos) if days == 0 && nanos == 0 => Ok(months),
1512            _ => Err(ArrowError::InvalidArgumentError(format!(
1513                "Unable to represent interval with days and nanos as year-months: {self:?}"
1514            ))),
1515        }
1516    }
1517
1518    fn to_day_time(&self) -> Result<(i32, i32), ArrowError> {
1519        let days = self.months.mul_checked(30)?.add_checked(self.days)?;
1520
1521        match self.nanos {
1522            nanos if nanos % NANOS_PER_MILLIS == 0 => {
1523                let millis = (self.nanos / 1_000_000).try_into().map_err(|_| {
1524                    ArrowError::InvalidArgumentError(format!(
1525                        "Unable to represent {} nanos as milliseconds in a signed 32-bit integer",
1526                        self.nanos
1527                    ))
1528                })?;
1529
1530                Ok((days, millis))
1531            }
1532            nanos => Err(ArrowError::InvalidArgumentError(format!(
1533                "Unable to represent {nanos} as milliseconds"
1534            ))),
1535        }
1536    }
1537
1538    fn to_month_day_nanos(&self) -> (i32, i32, i64) {
1539        (self.months, self.days, self.nanos)
1540    }
1541
1542    /// Parse string value in traditional Postgres format such as
1543    /// `1 year 2 months 3 days 4 hours 5 minutes 6 seconds`
1544    fn parse(value: &str, config: &IntervalParseConfig) -> Result<Self, ArrowError> {
1545        let components = parse_interval_components(value, config)?;
1546
1547        components
1548            .into_iter()
1549            .try_fold(Self::default(), |result, (amount, unit)| {
1550                result.add(amount, unit)
1551            })
1552    }
1553
1554    /// Interval addition following Postgres behavior. Fractional units will be spilled into smaller units.
1555    /// When the interval unit is larger than months, the result is rounded to total months and not spilled to days/nanos.
1556    /// Fractional parts of weeks and days are represented using days and nanoseconds.
1557    /// e.g. INTERVAL '0.5 MONTH' = 15 days, INTERVAL '1.5 MONTH' = 1 month 15 days
1558    /// e.g. INTERVAL '0.5 DAY' = 12 hours, INTERVAL '1.5 DAY' = 1 day 12 hours
1559    /// [Postgres reference](https://www.postgresql.org/docs/15/datatype-datetime.html#DATATYPE-INTERVAL-INPUT:~:text=Field%20values%20can,fractional%20on%20output.)
1560    fn add(&self, amount: IntervalAmount, unit: IntervalUnit) -> Result<Self, ArrowError> {
1561        let result = match unit {
1562            IntervalUnit::Century => {
1563                let months_int = amount.integer.mul_checked(100)?.mul_checked(12)?;
1564                let month_frac = amount.frac * 12 / 10_i64.pow(INTERVAL_PRECISION - 2);
1565                let months = months_int
1566                    .add_checked(month_frac)?
1567                    .try_into()
1568                    .map_err(|_| {
1569                        ArrowError::ParseError(format!(
1570                            "Unable to represent {} centuries as months in a signed 32-bit integer",
1571                            amount.integer
1572                        ))
1573                    })?;
1574
1575                Self::new(self.months.add_checked(months)?, self.days, self.nanos)
1576            }
1577            IntervalUnit::Decade => {
1578                let months_int = amount.integer.mul_checked(10)?.mul_checked(12)?;
1579
1580                let month_frac = amount.frac * 12 / 10_i64.pow(INTERVAL_PRECISION - 1);
1581                let months = months_int
1582                    .add_checked(month_frac)?
1583                    .try_into()
1584                    .map_err(|_| {
1585                        ArrowError::ParseError(format!(
1586                            "Unable to represent {} decades as months in a signed 32-bit integer",
1587                            amount.integer
1588                        ))
1589                    })?;
1590
1591                Self::new(self.months.add_checked(months)?, self.days, self.nanos)
1592            }
1593            IntervalUnit::Year => {
1594                let months_int = amount.integer.mul_checked(12)?;
1595                let month_frac = amount.frac * 12 / 10_i64.pow(INTERVAL_PRECISION);
1596                let months = months_int
1597                    .add_checked(month_frac)?
1598                    .try_into()
1599                    .map_err(|_| {
1600                        ArrowError::ParseError(format!(
1601                            "Unable to represent {} years as months in a signed 32-bit integer",
1602                            amount.integer
1603                        ))
1604                    })?;
1605
1606                Self::new(self.months.add_checked(months)?, self.days, self.nanos)
1607            }
1608            IntervalUnit::Month => {
1609                let months = amount.integer.try_into().map_err(|_| {
1610                    ArrowError::ParseError(format!(
1611                        "Unable to represent {} months in a signed 32-bit integer",
1612                        amount.integer
1613                    ))
1614                })?;
1615
1616                let days = amount.frac * 3 / 10_i64.pow(INTERVAL_PRECISION - 1);
1617                let days = days.try_into().map_err(|_| {
1618                    ArrowError::ParseError(format!(
1619                        "Unable to represent {} months as days in a signed 32-bit integer",
1620                        amount.frac / 10_i64.pow(INTERVAL_PRECISION)
1621                    ))
1622                })?;
1623
1624                Self::new(
1625                    self.months.add_checked(months)?,
1626                    self.days.add_checked(days)?,
1627                    self.nanos,
1628                )
1629            }
1630            IntervalUnit::Week => {
1631                let days = amount.integer.mul_checked(7)?.try_into().map_err(|_| {
1632                    ArrowError::ParseError(format!(
1633                        "Unable to represent {} weeks as days in a signed 32-bit integer",
1634                        amount.integer
1635                    ))
1636                })?;
1637
1638                let nanos = amount.frac * 7 * 24 * 6 * 6 / 10_i64.pow(INTERVAL_PRECISION - 11);
1639
1640                Self::new(
1641                    self.months,
1642                    self.days.add_checked(days)?,
1643                    self.nanos.add_checked(nanos)?,
1644                )
1645            }
1646            IntervalUnit::Day => {
1647                let days = amount.integer.try_into().map_err(|_| {
1648                    ArrowError::InvalidArgumentError(format!(
1649                        "Unable to represent {} days in a signed 32-bit integer",
1650                        amount.integer
1651                    ))
1652                })?;
1653
1654                let nanos = amount.frac * 24 * 6 * 6 / 10_i64.pow(INTERVAL_PRECISION - 11);
1655
1656                Self::new(
1657                    self.months,
1658                    self.days.add_checked(days)?,
1659                    self.nanos.add_checked(nanos)?,
1660                )
1661            }
1662            IntervalUnit::Hour => {
1663                let nanos_int = amount.integer.mul_checked(NANOS_PER_HOUR)?;
1664                let nanos_frac = amount.frac * 6 * 6 / 10_i64.pow(INTERVAL_PRECISION - 11);
1665                let nanos = nanos_int.add_checked(nanos_frac)?;
1666
1667                Interval::new(self.months, self.days, self.nanos.add_checked(nanos)?)
1668            }
1669            IntervalUnit::Minute => {
1670                let nanos_int = amount.integer.mul_checked(NANOS_PER_MINUTE)?;
1671                let nanos_frac = amount.frac * 6 / 10_i64.pow(INTERVAL_PRECISION - 10);
1672
1673                let nanos = nanos_int.add_checked(nanos_frac)?;
1674
1675                Interval::new(self.months, self.days, self.nanos.add_checked(nanos)?)
1676            }
1677            IntervalUnit::Second => {
1678                let nanos_int = amount.integer.mul_checked(NANOS_PER_SECOND)?;
1679                let nanos_frac = amount.frac / 10_i64.pow(INTERVAL_PRECISION - 9);
1680                let nanos = nanos_int.add_checked(nanos_frac)?;
1681
1682                Interval::new(self.months, self.days, self.nanos.add_checked(nanos)?)
1683            }
1684            IntervalUnit::Millisecond => {
1685                let nanos_int = amount.integer.mul_checked(NANOS_PER_MILLIS)?;
1686                let nanos_frac = amount.frac / 10_i64.pow(INTERVAL_PRECISION - 6);
1687                let nanos = nanos_int.add_checked(nanos_frac)?;
1688
1689                Interval::new(self.months, self.days, self.nanos.add_checked(nanos)?)
1690            }
1691            IntervalUnit::Microsecond => {
1692                let nanos_int = amount.integer.mul_checked(1_000)?;
1693                let nanos_frac = amount.frac / 10_i64.pow(INTERVAL_PRECISION - 3);
1694                let nanos = nanos_int.add_checked(nanos_frac)?;
1695
1696                Interval::new(self.months, self.days, self.nanos.add_checked(nanos)?)
1697            }
1698            IntervalUnit::Nanosecond => {
1699                let nanos_int = amount.integer;
1700                let nanos_frac = amount.frac / 10_i64.pow(INTERVAL_PRECISION);
1701                let nanos = nanos_int.add_checked(nanos_frac)?;
1702
1703                Interval::new(self.months, self.days, self.nanos.add_checked(nanos)?)
1704            }
1705        };
1706
1707        Ok(result)
1708    }
1709}
1710
1711/// parse the string into a vector of interval components i.e. (amount, unit) tuples
1712fn parse_interval_components(
1713    value: &str,
1714    config: &IntervalParseConfig,
1715) -> Result<Vec<(IntervalAmount, IntervalUnit)>, ArrowError> {
1716    let raw_pairs = split_interval_components(value);
1717
1718    // parse amounts and units
1719    let Ok(pairs): Result<Vec<(IntervalAmount, IntervalUnit)>, ArrowError> = raw_pairs
1720        .iter()
1721        .map(|(a, u)| Ok((a.parse()?, IntervalUnit::from_str_or_config(*u, config)?)))
1722        .collect()
1723    else {
1724        return Err(ArrowError::ParseError(format!(
1725            "Invalid input syntax for type interval: {value:?}"
1726        )));
1727    };
1728
1729    // collect parsed results
1730    let (amounts, units): (Vec<_>, Vec<_>) = pairs.into_iter().unzip();
1731
1732    // duplicate units?
1733    let mut observed_interval_types = 0;
1734    for (unit, (_, raw_unit)) in units.iter().zip(raw_pairs) {
1735        if observed_interval_types & (*unit as u16) != 0 {
1736            return Err(ArrowError::ParseError(format!(
1737                "Invalid input syntax for type interval: {:?}. Repeated type '{}'",
1738                value,
1739                raw_unit.unwrap_or_default(),
1740            )));
1741        }
1742
1743        observed_interval_types |= *unit as u16;
1744    }
1745
1746    let result = amounts.iter().copied().zip(units.iter().copied());
1747
1748    Ok(result.collect::<Vec<_>>())
1749}
1750
1751/// Split an interval into a vec of amounts and units.
1752///
1753/// Pairs are separated by spaces, but within a pair the amount and unit may or may not be separated by a space.
1754///
1755/// This should match the behavior of PostgreSQL's interval parser.
1756fn split_interval_components(value: &str) -> Vec<(&str, Option<&str>)> {
1757    let mut result = vec![];
1758    let mut words = value.split(char::is_whitespace);
1759    while let Some(word) = words.next() {
1760        if let Some(split_word_at) = word.find(not_interval_amount) {
1761            let (amount, unit) = word.split_at(split_word_at);
1762            result.push((amount, Some(unit)));
1763        } else if let Some(unit) = words.next() {
1764            result.push((word, Some(unit)));
1765        } else {
1766            result.push((word, None));
1767            break;
1768        }
1769    }
1770    result
1771}
1772
1773/// test if a character is NOT part of an interval numeric amount
1774fn not_interval_amount(c: char) -> bool {
1775    !c.is_ascii_digit() && c != '.' && c != '-'
1776}
1777
1778#[cfg(test)]
1779mod tests {
1780    use super::*;
1781    use arrow_array::temporal_conversions::date32_to_datetime;
1782    use arrow_buffer::i256;
1783
1784    /// Parses `s` without a precision check, for probing the native range
1785    fn parse_native<T: DecimalType>(s: &str, scale: i8) -> Result<T::Native, DecimalParseError> {
1786        parse_decimal_native::<T>(s, scale).map(|(value, _)| value)
1787    }
1788
1789    #[test]
1790    fn test_parse_nanos() {
1791        assert_eq!(parse_nanos::<3, 0>(&[1, 2, 3]), 123_000_000);
1792        assert_eq!(parse_nanos::<5, 0>(&[1, 2, 3, 4, 5]), 123_450_000);
1793        assert_eq!(parse_nanos::<6, b'0'>(b"123456"), 123_456_000);
1794    }
1795
1796    #[test]
1797    fn string_to_timestamp_timezone() {
1798        // Explicit timezone
1799        assert_eq!(
1800            1599572549190855000,
1801            parse_timestamp("2020-09-08T13:42:29.190855+00:00").unwrap()
1802        );
1803        assert_eq!(
1804            1599572549190855000,
1805            parse_timestamp("2020-09-08T13:42:29.190855Z").unwrap()
1806        );
1807        assert_eq!(
1808            1599572549000000000,
1809            parse_timestamp("2020-09-08T13:42:29Z").unwrap()
1810        ); // no fractional part
1811        assert_eq!(
1812            1599590549190855000,
1813            parse_timestamp("2020-09-08T13:42:29.190855-05:00").unwrap()
1814        );
1815    }
1816
1817    #[test]
1818    fn string_to_timestamp_timezone_space() {
1819        // Ensure space rather than T between time and date is accepted
1820        assert_eq!(
1821            1599572549190855000,
1822            parse_timestamp("2020-09-08 13:42:29.190855+00:00").unwrap()
1823        );
1824        assert_eq!(
1825            1599572549190855000,
1826            parse_timestamp("2020-09-08 13:42:29.190855Z").unwrap()
1827        );
1828        assert_eq!(
1829            1599572549000000000,
1830            parse_timestamp("2020-09-08 13:42:29Z").unwrap()
1831        ); // no fractional part
1832        assert_eq!(
1833            1599590549190855000,
1834            parse_timestamp("2020-09-08 13:42:29.190855-05:00").unwrap()
1835        );
1836    }
1837
1838    #[test]
1839    fn string_to_timestamp_no_timezone() {
1840        // This test is designed to succeed in regardless of the local
1841        // timezone the test machine is running. Thus it is still
1842        // somewhat susceptible to bugs in the use of chrono
1843        let naive_datetime = NaiveDateTime::new(
1844            NaiveDate::from_ymd_opt(2020, 9, 8).unwrap(),
1845            NaiveTime::from_hms_nano_opt(13, 42, 29, 190855000).unwrap(),
1846        );
1847
1848        // Ensure both T and ' ' variants work
1849        assert_eq!(
1850            naive_datetime.and_utc().timestamp_nanos_opt().unwrap(),
1851            parse_timestamp("2020-09-08T13:42:29.190855").unwrap()
1852        );
1853
1854        assert_eq!(
1855            naive_datetime.and_utc().timestamp_nanos_opt().unwrap(),
1856            parse_timestamp("2020-09-08 13:42:29.190855").unwrap()
1857        );
1858
1859        // Also ensure that parsing timestamps with no fractional
1860        // second part works as well
1861        let datetime_whole_secs = NaiveDateTime::new(
1862            NaiveDate::from_ymd_opt(2020, 9, 8).unwrap(),
1863            NaiveTime::from_hms_opt(13, 42, 29).unwrap(),
1864        )
1865        .and_utc();
1866
1867        // Ensure both T and ' ' variants work
1868        assert_eq!(
1869            datetime_whole_secs.timestamp_nanos_opt().unwrap(),
1870            parse_timestamp("2020-09-08T13:42:29").unwrap()
1871        );
1872
1873        assert_eq!(
1874            datetime_whole_secs.timestamp_nanos_opt().unwrap(),
1875            parse_timestamp("2020-09-08 13:42:29").unwrap()
1876        );
1877
1878        // ensure without time work
1879        // no time, should be the nano second at
1880        // 2020-09-08 0:0:0
1881        let datetime_no_time = NaiveDateTime::new(
1882            NaiveDate::from_ymd_opt(2020, 9, 8).unwrap(),
1883            NaiveTime::from_hms_opt(0, 0, 0).unwrap(),
1884        )
1885        .and_utc();
1886
1887        assert_eq!(
1888            datetime_no_time.timestamp_nanos_opt().unwrap(),
1889            parse_timestamp("2020-09-08").unwrap()
1890        )
1891    }
1892
1893    #[test]
1894    fn string_to_timestamp_chrono() {
1895        let cases = [
1896            "2020-09-08T13:42:29Z",
1897            "1969-01-01T00:00:00.1Z",
1898            "2020-09-08T12:00:12.12345678+00:00",
1899            "2020-09-08T12:00:12+00:00",
1900            "2020-09-08T12:00:12.1+00:00",
1901            "2020-09-08T12:00:12.12+00:00",
1902            "2020-09-08T12:00:12.123+00:00",
1903            "2020-09-08T12:00:12.1234+00:00",
1904            "2020-09-08T12:00:12.12345+00:00",
1905            "2020-09-08T12:00:12.123456+00:00",
1906            "2020-09-08T12:00:12.1234567+00:00",
1907            "2020-09-08T12:00:12.12345678+00:00",
1908            "2020-09-08T12:00:12.123456789+00:00",
1909            "2020-09-08T12:00:12.12345678912z",
1910            "2020-09-08T12:00:12.123456789123Z",
1911            "2020-09-08T12:00:12.123456789123+02:00",
1912            "2020-09-08T12:00:12.12345678912345Z",
1913            "2020-09-08T12:00:12.1234567891234567+02:00",
1914            "2020-09-08T12:00:60Z",
1915            "2020-09-08T12:00:60.123Z",
1916            "2020-09-08T12:00:60.123456+02:00",
1917            "2020-09-08T12:00:60.1234567891234567+02:00",
1918            "2020-09-08T12:00:60.999999999+02:00",
1919            "2020-09-08t12:00:12.12345678+00:00",
1920            "2020-09-08t12:00:12+00:00",
1921            "2020-09-08t12:00:12Z",
1922        ];
1923
1924        for case in cases {
1925            let chrono = DateTime::parse_from_rfc3339(case).unwrap();
1926            let chrono_utc = chrono.with_timezone(&Utc);
1927
1928            let custom = string_to_datetime(&Utc, case).unwrap();
1929            assert_eq!(chrono_utc, custom)
1930        }
1931    }
1932
1933    #[test]
1934    fn string_to_timestamp_naive() {
1935        let cases = [
1936            "2018-11-13T17:11:10.011375885995",
1937            "2030-12-04T17:11:10.123",
1938            "2030-12-04T17:11:10.1234",
1939            "2030-12-04T17:11:10.123456",
1940        ];
1941        for case in cases {
1942            let chrono = NaiveDateTime::parse_from_str(case, "%Y-%m-%dT%H:%M:%S%.f").unwrap();
1943            let custom = string_to_datetime(&Utc, case).unwrap();
1944            assert_eq!(chrono, custom.naive_utc())
1945        }
1946    }
1947
1948    #[test]
1949    fn string_to_timestamp_invalid() {
1950        // Test parsing invalid formats
1951        let cases = [
1952            ("", "timestamp must contain at least 10 characters"),
1953            ("SS", "timestamp must contain at least 10 characters"),
1954            ("Wed, 18 Feb 2015 23:16:09 GMT", "error parsing date"),
1955            ("1997-01-31H09:26:56.123Z", "invalid timestamp separator"),
1956            ("1997-01-31  09:26:56.123Z", "error parsing time"),
1957            ("1997:01:31T09:26:56.123Z", "error parsing date"),
1958            ("1997:1:31T09:26:56.123Z", "error parsing date"),
1959            ("1997-01-32T09:26:56.123Z", "error parsing date"),
1960            ("1997-13-32T09:26:56.123Z", "error parsing date"),
1961            ("1997-02-29T09:26:56.123Z", "error parsing date"),
1962            ("2015-02-30T17:35:20-08:00", "error parsing date"),
1963            ("1997-01-10T9:26:56.123Z", "error parsing time"),
1964            ("2015-01-20T25:35:20-08:00", "error parsing time"),
1965            ("1997-01-10T09:61:56.123Z", "error parsing time"),
1966            ("1997-01-10T09:61:90.123Z", "error parsing time"),
1967            ("1997-01-10T12:00:6.123Z", "error parsing time"),
1968            ("1997-01-31T092656.123Z", "error parsing time"),
1969            ("1997-01-10T12:00:06.", "error parsing time"),
1970            ("1997-01-10T12:00:06. ", "error parsing time"),
1971        ];
1972
1973        for (s, ctx) in cases {
1974            let expected = format!("Parser error: Error parsing timestamp from '{s}': {ctx}");
1975            let actual = string_to_datetime(&Utc, s).unwrap_err().to_string();
1976            assert_eq!(actual, expected)
1977        }
1978    }
1979
1980    // Parse a timestamp to timestamp int with a useful human readable error message
1981    fn parse_timestamp(s: &str) -> Result<i64, ArrowError> {
1982        let result = string_to_timestamp_nanos(s);
1983        if let Err(e) = &result {
1984            eprintln!("Error parsing timestamp '{s}': {e:?}");
1985        }
1986        result
1987    }
1988
1989    #[test]
1990    fn string_without_timezone_to_timestamp() {
1991        // string without timezone should always output the same regardless the local or session timezone
1992
1993        let naive_datetime = NaiveDateTime::new(
1994            NaiveDate::from_ymd_opt(2020, 9, 8).unwrap(),
1995            NaiveTime::from_hms_nano_opt(13, 42, 29, 190855000).unwrap(),
1996        );
1997
1998        // Ensure both T and ' ' variants work
1999        assert_eq!(
2000            naive_datetime.and_utc().timestamp_nanos_opt().unwrap(),
2001            parse_timestamp("2020-09-08T13:42:29.190855").unwrap()
2002        );
2003
2004        assert_eq!(
2005            naive_datetime.and_utc().timestamp_nanos_opt().unwrap(),
2006            parse_timestamp("2020-09-08 13:42:29.190855").unwrap()
2007        );
2008
2009        let naive_datetime = NaiveDateTime::new(
2010            NaiveDate::from_ymd_opt(2020, 9, 8).unwrap(),
2011            NaiveTime::from_hms_nano_opt(13, 42, 29, 0).unwrap(),
2012        );
2013
2014        // Ensure both T and ' ' variants work
2015        assert_eq!(
2016            naive_datetime.and_utc().timestamp_nanos_opt().unwrap(),
2017            parse_timestamp("2020-09-08T13:42:29").unwrap()
2018        );
2019
2020        assert_eq!(
2021            naive_datetime.and_utc().timestamp_nanos_opt().unwrap(),
2022            parse_timestamp("2020-09-08 13:42:29").unwrap()
2023        );
2024
2025        let tz: Tz = "+02:00".parse().unwrap();
2026        let date = string_to_datetime(&tz, "2020-09-08 13:42:29").unwrap();
2027        let utc = date.naive_utc().to_string();
2028        assert_eq!(utc, "2020-09-08 11:42:29");
2029        let local = date.naive_local().to_string();
2030        assert_eq!(local, "2020-09-08 13:42:29");
2031
2032        let date = string_to_datetime(&tz, "2020-09-08 13:42:29Z").unwrap();
2033        let utc = date.naive_utc().to_string();
2034        assert_eq!(utc, "2020-09-08 13:42:29");
2035        let local = date.naive_local().to_string();
2036        assert_eq!(local, "2020-09-08 15:42:29");
2037
2038        let dt =
2039            NaiveDateTime::parse_from_str("2020-09-08T13:42:29Z", "%Y-%m-%dT%H:%M:%SZ").unwrap();
2040        let local: Tz = "+08:00".parse().unwrap();
2041
2042        // Parsed as offset from UTC
2043        let date = string_to_datetime(&local, "2020-09-08T13:42:29Z").unwrap();
2044        assert_eq!(dt, date.naive_utc());
2045        assert_ne!(dt, date.naive_local());
2046
2047        // Parsed as offset from local
2048        let date = string_to_datetime(&local, "2020-09-08 13:42:29").unwrap();
2049        assert_eq!(dt, date.naive_local());
2050        assert_ne!(dt, date.naive_utc());
2051    }
2052
2053    #[test]
2054    fn parse_date32() {
2055        let cases = [
2056            "2020-09-08",
2057            "2020-9-8",
2058            "2020-09-8",
2059            "2020-9-08",
2060            "2020-12-1",
2061            "1690-2-5",
2062            "2020-09-08 01:02:03",
2063        ];
2064        for case in cases {
2065            let v = date32_to_datetime(Date32Type::parse(case).unwrap()).unwrap();
2066            let expected = NaiveDate::parse_from_str(case, "%Y-%m-%d")
2067                .or_else(|_| NaiveDate::parse_from_str(case, "%Y-%m-%d %H:%M:%S"))
2068                .unwrap();
2069            assert_eq!(v.date(), expected);
2070        }
2071
2072        let err_cases = [
2073            "",
2074            "80-01-01",
2075            "342",
2076            "Foo",
2077            "2020-09-08-03",
2078            "2020--04-03",
2079            "2020--",
2080            "2020-09-08 01",
2081            "2020-09-08 01:02",
2082            "2020-09-08 01-02-03",
2083            "2020-9-8 01:02:03",
2084            "2020-09-08 1:2:3",
2085        ];
2086        for case in err_cases {
2087            assert_eq!(Date32Type::parse(case), None);
2088        }
2089    }
2090
2091    #[test]
2092    fn parse_date32_extended_year() {
2093        // `Date32` covers any i32 days-from-epoch, verify we can parse it
2094        let cases: &[(&str, i32)] = &[
2095            ("+1970-01-01", 0),
2096            ("+2024-01-01", 19_723),
2097            ("-0001-01-01", -719_893),
2098            ("+29349-01-26", 10_000_000),
2099            ("+2739877-01-03", 1_000_000_000),
2100            // Extremes of the Date32 representable range.
2101            ("+5881580-07-11", i32::MAX),
2102            ("-5877641-06-23", i32::MIN),
2103        ];
2104        for (input, expected) in cases {
2105            assert_eq!(Date32Type::parse(input), Some(*expected), "input: {input}");
2106        }
2107
2108        // One past Date32::MAX / MIN overflows i32 days-from-epoch.
2109        assert_eq!(Date32Type::parse("+5881580-07-12"), None);
2110        assert_eq!(Date32Type::parse("-5877641-06-22"), None);
2111        // Invalid calendar dates still rejected regardless of year magnitude.
2112        assert_eq!(Date32Type::parse("+2739877-02-30"), None);
2113        assert_eq!(Date32Type::parse("+2739877-13-01"), None);
2114        assert_eq!(Date32Type::parse("-2739877-02-30"), None);
2115    }
2116
2117    #[test]
2118    fn parse_time64_nanos() {
2119        assert_eq!(
2120            Time64NanosecondType::parse("02:10:01.1234567899999999"),
2121            Some(7_801_123_456_789)
2122        );
2123        assert_eq!(
2124            Time64NanosecondType::parse("02:10:01.1234567"),
2125            Some(7_801_123_456_700)
2126        );
2127        assert_eq!(
2128            Time64NanosecondType::parse("2:10:01.1234567"),
2129            Some(7_801_123_456_700)
2130        );
2131        assert_eq!(
2132            Time64NanosecondType::parse("12:10:01.123456789 AM"),
2133            Some(601_123_456_789)
2134        );
2135        assert_eq!(
2136            Time64NanosecondType::parse("12:10:01.123456789 am"),
2137            Some(601_123_456_789)
2138        );
2139        assert_eq!(
2140            Time64NanosecondType::parse("2:10:01.12345678 PM"),
2141            Some(51_001_123_456_780)
2142        );
2143        assert_eq!(
2144            Time64NanosecondType::parse("2:10:01.12345678 pm"),
2145            Some(51_001_123_456_780)
2146        );
2147        assert_eq!(
2148            Time64NanosecondType::parse("02:10:01"),
2149            Some(7_801_000_000_000)
2150        );
2151        assert_eq!(
2152            Time64NanosecondType::parse("2:10:01"),
2153            Some(7_801_000_000_000)
2154        );
2155        assert_eq!(
2156            Time64NanosecondType::parse("12:10:01 AM"),
2157            Some(601_000_000_000)
2158        );
2159        assert_eq!(
2160            Time64NanosecondType::parse("12:10:01 am"),
2161            Some(601_000_000_000)
2162        );
2163        assert_eq!(
2164            Time64NanosecondType::parse("2:10:01 PM"),
2165            Some(51_001_000_000_000)
2166        );
2167        assert_eq!(
2168            Time64NanosecondType::parse("2:10:01 pm"),
2169            Some(51_001_000_000_000)
2170        );
2171        assert_eq!(
2172            Time64NanosecondType::parse("02:10"),
2173            Some(7_800_000_000_000)
2174        );
2175        assert_eq!(Time64NanosecondType::parse("2:10"), Some(7_800_000_000_000));
2176        assert_eq!(
2177            Time64NanosecondType::parse("12:10 AM"),
2178            Some(600_000_000_000)
2179        );
2180        assert_eq!(
2181            Time64NanosecondType::parse("12:10 am"),
2182            Some(600_000_000_000)
2183        );
2184        assert_eq!(
2185            Time64NanosecondType::parse("2:10 PM"),
2186            Some(51_000_000_000_000)
2187        );
2188        assert_eq!(
2189            Time64NanosecondType::parse("2:10 pm"),
2190            Some(51_000_000_000_000)
2191        );
2192
2193        // parse directly as nanoseconds
2194        assert_eq!(Time64NanosecondType::parse("1"), Some(1));
2195
2196        // leap second
2197        assert_eq!(
2198            Time64NanosecondType::parse("23:59:60"),
2199            Some(86_400_000_000_000)
2200        );
2201
2202        // custom format
2203        assert_eq!(
2204            Time64NanosecondType::parse_formatted("02 - 10 - 01 - .1234567", "%H - %M - %S - %.f"),
2205            Some(7_801_123_456_700)
2206        );
2207    }
2208
2209    #[test]
2210    fn parse_time64_micros() {
2211        // expected formats
2212        assert_eq!(
2213            Time64MicrosecondType::parse("02:10:01.1234"),
2214            Some(7_801_123_400)
2215        );
2216        assert_eq!(
2217            Time64MicrosecondType::parse("2:10:01.1234"),
2218            Some(7_801_123_400)
2219        );
2220        assert_eq!(
2221            Time64MicrosecondType::parse("12:10:01.123456 AM"),
2222            Some(601_123_456)
2223        );
2224        assert_eq!(
2225            Time64MicrosecondType::parse("12:10:01.123456 am"),
2226            Some(601_123_456)
2227        );
2228        assert_eq!(
2229            Time64MicrosecondType::parse("2:10:01.12345 PM"),
2230            Some(51_001_123_450)
2231        );
2232        assert_eq!(
2233            Time64MicrosecondType::parse("2:10:01.12345 pm"),
2234            Some(51_001_123_450)
2235        );
2236        assert_eq!(
2237            Time64MicrosecondType::parse("02:10:01"),
2238            Some(7_801_000_000)
2239        );
2240        assert_eq!(Time64MicrosecondType::parse("2:10:01"), Some(7_801_000_000));
2241        assert_eq!(
2242            Time64MicrosecondType::parse("12:10:01 AM"),
2243            Some(601_000_000)
2244        );
2245        assert_eq!(
2246            Time64MicrosecondType::parse("12:10:01 am"),
2247            Some(601_000_000)
2248        );
2249        assert_eq!(
2250            Time64MicrosecondType::parse("2:10:01 PM"),
2251            Some(51_001_000_000)
2252        );
2253        assert_eq!(
2254            Time64MicrosecondType::parse("2:10:01 pm"),
2255            Some(51_001_000_000)
2256        );
2257        assert_eq!(Time64MicrosecondType::parse("02:10"), Some(7_800_000_000));
2258        assert_eq!(Time64MicrosecondType::parse("2:10"), Some(7_800_000_000));
2259        assert_eq!(Time64MicrosecondType::parse("12:10 AM"), Some(600_000_000));
2260        assert_eq!(Time64MicrosecondType::parse("12:10 am"), Some(600_000_000));
2261        assert_eq!(
2262            Time64MicrosecondType::parse("2:10 PM"),
2263            Some(51_000_000_000)
2264        );
2265        assert_eq!(
2266            Time64MicrosecondType::parse("2:10 pm"),
2267            Some(51_000_000_000)
2268        );
2269
2270        // parse directly as microseconds
2271        assert_eq!(Time64MicrosecondType::parse("1"), Some(1));
2272
2273        // leap second
2274        assert_eq!(
2275            Time64MicrosecondType::parse("23:59:60"),
2276            Some(86_400_000_000)
2277        );
2278
2279        // custom format
2280        assert_eq!(
2281            Time64MicrosecondType::parse_formatted("02 - 10 - 01 - .1234", "%H - %M - %S - %.f"),
2282            Some(7_801_123_400)
2283        );
2284    }
2285
2286    #[test]
2287    fn parse_time32_millis() {
2288        // expected formats
2289        assert_eq!(Time32MillisecondType::parse("02:10:01.1"), Some(7_801_100));
2290        assert_eq!(Time32MillisecondType::parse("2:10:01.1"), Some(7_801_100));
2291        assert_eq!(
2292            Time32MillisecondType::parse("12:10:01.123 AM"),
2293            Some(601_123)
2294        );
2295        assert_eq!(
2296            Time32MillisecondType::parse("12:10:01.123 am"),
2297            Some(601_123)
2298        );
2299        assert_eq!(
2300            Time32MillisecondType::parse("2:10:01.12 PM"),
2301            Some(51_001_120)
2302        );
2303        assert_eq!(
2304            Time32MillisecondType::parse("2:10:01.12 pm"),
2305            Some(51_001_120)
2306        );
2307        assert_eq!(Time32MillisecondType::parse("02:10:01"), Some(7_801_000));
2308        assert_eq!(Time32MillisecondType::parse("2:10:01"), Some(7_801_000));
2309        assert_eq!(Time32MillisecondType::parse("12:10:01 AM"), Some(601_000));
2310        assert_eq!(Time32MillisecondType::parse("12:10:01 am"), Some(601_000));
2311        assert_eq!(Time32MillisecondType::parse("2:10:01 PM"), Some(51_001_000));
2312        assert_eq!(Time32MillisecondType::parse("2:10:01 pm"), Some(51_001_000));
2313        assert_eq!(Time32MillisecondType::parse("02:10"), Some(7_800_000));
2314        assert_eq!(Time32MillisecondType::parse("2:10"), Some(7_800_000));
2315        assert_eq!(Time32MillisecondType::parse("12:10 AM"), Some(600_000));
2316        assert_eq!(Time32MillisecondType::parse("12:10 am"), Some(600_000));
2317        assert_eq!(Time32MillisecondType::parse("2:10 PM"), Some(51_000_000));
2318        assert_eq!(Time32MillisecondType::parse("2:10 pm"), Some(51_000_000));
2319
2320        // parse directly as milliseconds
2321        assert_eq!(Time32MillisecondType::parse("1"), Some(1));
2322
2323        // leap second
2324        assert_eq!(Time32MillisecondType::parse("23:59:60"), Some(86_400_000));
2325
2326        // custom format
2327        assert_eq!(
2328            Time32MillisecondType::parse_formatted("02 - 10 - 01 - .1", "%H - %M - %S - %.f"),
2329            Some(7_801_100)
2330        );
2331    }
2332
2333    #[test]
2334    fn parse_time32_secs() {
2335        // expected formats
2336        assert_eq!(Time32SecondType::parse("02:10:01.1"), Some(7_801));
2337        assert_eq!(Time32SecondType::parse("02:10:01"), Some(7_801));
2338        assert_eq!(Time32SecondType::parse("2:10:01"), Some(7_801));
2339        assert_eq!(Time32SecondType::parse("12:10:01 AM"), Some(601));
2340        assert_eq!(Time32SecondType::parse("12:10:01 am"), Some(601));
2341        assert_eq!(Time32SecondType::parse("2:10:01 PM"), Some(51_001));
2342        assert_eq!(Time32SecondType::parse("2:10:01 pm"), Some(51_001));
2343        assert_eq!(Time32SecondType::parse("02:10"), Some(7_800));
2344        assert_eq!(Time32SecondType::parse("2:10"), Some(7_800));
2345        assert_eq!(Time32SecondType::parse("12:10 AM"), Some(600));
2346        assert_eq!(Time32SecondType::parse("12:10 am"), Some(600));
2347        assert_eq!(Time32SecondType::parse("2:10 PM"), Some(51_000));
2348        assert_eq!(Time32SecondType::parse("2:10 pm"), Some(51_000));
2349
2350        // parse directly as seconds
2351        assert_eq!(Time32SecondType::parse("1"), Some(1));
2352
2353        // leap second
2354        assert_eq!(Time32SecondType::parse("23:59:60"), Some(86400));
2355
2356        // custom format
2357        assert_eq!(
2358            Time32SecondType::parse_formatted("02 - 10 - 01", "%H - %M - %S"),
2359            Some(7_801)
2360        );
2361    }
2362
2363    #[test]
2364    fn test_string_to_time_invalid() {
2365        let cases = [
2366            "25:00",
2367            "9:00:",
2368            "009:00",
2369            "09:0:00",
2370            "25:00:00",
2371            "13:00 AM",
2372            "13:00 PM",
2373            "12:00. AM",
2374            "09:0:00",
2375            "09:01:0",
2376            "09:01:1",
2377            "9:1:0",
2378            "09:01:0",
2379            "1:00.123",
2380            "1:00:00.123f",
2381            " 9:00:00",
2382            ":09:00",
2383            "T9:00:00",
2384            "AM",
2385        ];
2386        for case in cases {
2387            assert!(string_to_time(case).is_none(), "{case}");
2388        }
2389    }
2390
2391    #[test]
2392    fn test_string_to_time_chrono() {
2393        let cases = [
2394            ("1:00", "%H:%M"),
2395            ("12:00", "%H:%M"),
2396            ("13:00", "%H:%M"),
2397            ("24:00", "%H:%M"),
2398            ("1:00:00", "%H:%M:%S"),
2399            ("12:00:30", "%H:%M:%S"),
2400            ("13:00:59", "%H:%M:%S"),
2401            ("24:00:60", "%H:%M:%S"),
2402            ("09:00:00", "%H:%M:%S%.f"),
2403            ("0:00:30.123456", "%H:%M:%S%.f"),
2404            ("0:00 AM", "%I:%M %P"),
2405            ("1:00 AM", "%I:%M %P"),
2406            ("12:00 AM", "%I:%M %P"),
2407            ("13:00 AM", "%I:%M %P"),
2408            ("0:00 PM", "%I:%M %P"),
2409            ("1:00 PM", "%I:%M %P"),
2410            ("12:00 PM", "%I:%M %P"),
2411            ("13:00 PM", "%I:%M %P"),
2412            ("1:00 pM", "%I:%M %P"),
2413            ("1:00 Pm", "%I:%M %P"),
2414            ("1:00 aM", "%I:%M %P"),
2415            ("1:00 Am", "%I:%M %P"),
2416            ("1:00:30.123456 PM", "%I:%M:%S%.f %P"),
2417            ("1:00:30.123456789 PM", "%I:%M:%S%.f %P"),
2418            ("1:00:30.123456789123 PM", "%I:%M:%S%.f %P"),
2419            ("1:00:30.1234 PM", "%I:%M:%S%.f %P"),
2420            ("1:00:30.123456 PM", "%I:%M:%S%.f %P"),
2421            ("1:00:30.123456789123456789 PM", "%I:%M:%S%.f %P"),
2422            ("1:00:30.12F456 PM", "%I:%M:%S%.f %P"),
2423        ];
2424        for (s, format) in cases {
2425            let chrono = NaiveTime::parse_from_str(s, format).ok();
2426            let custom = string_to_time(s);
2427            assert_eq!(chrono, custom, "{s}");
2428        }
2429    }
2430
2431    #[test]
2432    fn test_parse_interval() {
2433        let config = IntervalParseConfig::new(IntervalUnit::Month);
2434
2435        assert_eq!(
2436            Interval::new(1i32, 0i32, 0i64),
2437            Interval::parse("1 month", &config).unwrap(),
2438        );
2439
2440        assert_eq!(
2441            Interval::new(2i32, 0i32, 0i64),
2442            Interval::parse("2 month", &config).unwrap(),
2443        );
2444
2445        assert_eq!(
2446            Interval::new(-1i32, -18i32, -(NANOS_PER_DAY / 5)),
2447            Interval::parse("-1.5 months -3.2 days", &config).unwrap(),
2448        );
2449
2450        assert_eq!(
2451            Interval::new(0i32, 15i32, 0),
2452            Interval::parse("0.5 months", &config).unwrap(),
2453        );
2454
2455        assert_eq!(
2456            Interval::new(0i32, 15i32, 0),
2457            Interval::parse(".5 months", &config).unwrap(),
2458        );
2459
2460        assert_eq!(
2461            Interval::new(0i32, -15i32, 0),
2462            Interval::parse("-0.5 months", &config).unwrap(),
2463        );
2464
2465        assert_eq!(
2466            Interval::new(0i32, -15i32, 0),
2467            Interval::parse("-.5 months", &config).unwrap(),
2468        );
2469
2470        assert_eq!(
2471            Interval::new(2i32, 10i32, 9 * NANOS_PER_HOUR),
2472            Interval::parse("2.1 months 7.25 days 3 hours", &config).unwrap(),
2473        );
2474
2475        assert_eq!(
2476            Interval::parse("1 centurys 1 month", &config)
2477                .unwrap_err()
2478                .to_string(),
2479            r#"Parser error: Invalid input syntax for type interval: "1 centurys 1 month""#
2480        );
2481
2482        assert_eq!(
2483            Interval::new(37i32, 0i32, 0i64),
2484            Interval::parse("3 year 1 month", &config).unwrap(),
2485        );
2486
2487        assert_eq!(
2488            Interval::new(35i32, 0i32, 0i64),
2489            Interval::parse("3 year -1 month", &config).unwrap(),
2490        );
2491
2492        assert_eq!(
2493            Interval::new(-37i32, 0i32, 0i64),
2494            Interval::parse("-3 year -1 month", &config).unwrap(),
2495        );
2496
2497        assert_eq!(
2498            Interval::new(-35i32, 0i32, 0i64),
2499            Interval::parse("-3 year 1 month", &config).unwrap(),
2500        );
2501
2502        assert_eq!(
2503            Interval::new(0i32, 5i32, 0i64),
2504            Interval::parse("5 days", &config).unwrap(),
2505        );
2506
2507        assert_eq!(
2508            Interval::new(0i32, 7i32, 3 * NANOS_PER_HOUR),
2509            Interval::parse("7 days 3 hours", &config).unwrap(),
2510        );
2511
2512        assert_eq!(
2513            Interval::new(0i32, 7i32, 5 * NANOS_PER_MINUTE),
2514            Interval::parse("7 days 5 minutes", &config).unwrap(),
2515        );
2516
2517        assert_eq!(
2518            Interval::new(0i32, 7i32, -5 * NANOS_PER_MINUTE),
2519            Interval::parse("7 days -5 minutes", &config).unwrap(),
2520        );
2521
2522        assert_eq!(
2523            Interval::new(0i32, -7i32, 5 * NANOS_PER_HOUR),
2524            Interval::parse("-7 days 5 hours", &config).unwrap(),
2525        );
2526
2527        assert_eq!(
2528            Interval::new(
2529                0i32,
2530                -7i32,
2531                -5 * NANOS_PER_HOUR - 5 * NANOS_PER_MINUTE - 5 * NANOS_PER_SECOND
2532            ),
2533            Interval::parse("-7 days -5 hours -5 minutes -5 seconds", &config).unwrap(),
2534        );
2535
2536        assert_eq!(
2537            Interval::new(12i32, 0i32, 25 * NANOS_PER_MILLIS),
2538            Interval::parse("1 year 25 millisecond", &config).unwrap(),
2539        );
2540
2541        assert_eq!(
2542            Interval::new(
2543                12i32,
2544                1i32,
2545                (NANOS_PER_SECOND as f64 * 0.000000001_f64) as i64
2546            ),
2547            Interval::parse("1 year 1 day 0.000000001 seconds", &config).unwrap(),
2548        );
2549
2550        assert_eq!(
2551            Interval::new(12i32, 1i32, NANOS_PER_MILLIS / 10),
2552            Interval::parse("1 year 1 day 0.1 milliseconds", &config).unwrap(),
2553        );
2554
2555        assert_eq!(
2556            Interval::new(12i32, 1i32, 1000i64),
2557            Interval::parse("1 year 1 day 1 microsecond", &config).unwrap(),
2558        );
2559
2560        assert_eq!(
2561            Interval::new(12i32, 1i32, 1i64),
2562            Interval::parse("1 year 1 day 1 nanoseconds", &config).unwrap(),
2563        );
2564
2565        assert_eq!(
2566            Interval::new(1i32, 0i32, -NANOS_PER_SECOND),
2567            Interval::parse("1 month -1 second", &config).unwrap(),
2568        );
2569
2570        assert_eq!(
2571            Interval::new(
2572                -13i32,
2573                -8i32,
2574                -NANOS_PER_HOUR
2575                    - NANOS_PER_MINUTE
2576                    - NANOS_PER_SECOND
2577                    - (1.11_f64 * NANOS_PER_MILLIS as f64) as i64
2578            ),
2579            Interval::parse(
2580                "-1 year -1 month -1 week -1 day -1 hour -1 minute -1 second -1.11 millisecond",
2581                &config
2582            )
2583            .unwrap(),
2584        );
2585
2586        // no units
2587        assert_eq!(
2588            Interval::new(1, 0, 0),
2589            Interval::parse("1", &config).unwrap()
2590        );
2591        assert_eq!(
2592            Interval::new(42, 0, 0),
2593            Interval::parse("42", &config).unwrap()
2594        );
2595        assert_eq!(
2596            Interval::new(0, 0, 42_000_000_000),
2597            Interval::parse("42", &IntervalParseConfig::new(IntervalUnit::Second)).unwrap()
2598        );
2599
2600        // shorter units
2601        assert_eq!(
2602            Interval::new(1, 0, 0),
2603            Interval::parse("1 mon", &config).unwrap()
2604        );
2605        assert_eq!(
2606            Interval::new(1, 0, 0),
2607            Interval::parse("1 mons", &config).unwrap()
2608        );
2609        assert_eq!(
2610            Interval::new(0, 0, 1_000_000),
2611            Interval::parse("1 ms", &config).unwrap()
2612        );
2613        assert_eq!(
2614            Interval::new(0, 0, 1_000),
2615            Interval::parse("1 us", &config).unwrap()
2616        );
2617
2618        // no space
2619        assert_eq!(
2620            Interval::new(0, 0, 1_000),
2621            Interval::parse("1us", &config).unwrap()
2622        );
2623        assert_eq!(
2624            Interval::new(0, 0, NANOS_PER_SECOND),
2625            Interval::parse("1s", &config).unwrap()
2626        );
2627        assert_eq!(
2628            Interval::new(1, 2, 10_864_000_000_000),
2629            Interval::parse("1mon 2days 3hr 1min 4sec", &config).unwrap()
2630        );
2631
2632        assert_eq!(
2633            Interval::new(
2634                -13i32,
2635                -8i32,
2636                -NANOS_PER_HOUR
2637                    - NANOS_PER_MINUTE
2638                    - NANOS_PER_SECOND
2639                    - (1.11_f64 * NANOS_PER_MILLIS as f64) as i64
2640            ),
2641            Interval::parse(
2642                "-1year -1month -1week -1day -1 hour -1 minute -1 second -1.11millisecond",
2643                &config
2644            )
2645            .unwrap(),
2646        );
2647
2648        assert_eq!(
2649            Interval::parse("1h s", &config).unwrap_err().to_string(),
2650            r#"Parser error: Invalid input syntax for type interval: "1h s""#
2651        );
2652
2653        assert_eq!(
2654            Interval::parse("1XX", &config).unwrap_err().to_string(),
2655            r#"Parser error: Invalid input syntax for type interval: "1XX""#
2656        );
2657    }
2658
2659    #[test]
2660    fn test_duplicate_interval_type() {
2661        let config = IntervalParseConfig::new(IntervalUnit::Month);
2662
2663        let err = Interval::parse("1 month 1 second 1 second", &config)
2664            .expect_err("parsing interval should have failed");
2665        assert_eq!(
2666            r#"ParseError("Invalid input syntax for type interval: \"1 month 1 second 1 second\". Repeated type 'second'")"#,
2667            format!("{err:?}")
2668        );
2669
2670        // test with singular and plural forms
2671        let err = Interval::parse("1 century 2 centuries", &config)
2672            .expect_err("parsing interval should have failed");
2673        assert_eq!(
2674            r#"ParseError("Invalid input syntax for type interval: \"1 century 2 centuries\". Repeated type 'centuries'")"#,
2675            format!("{err:?}")
2676        );
2677    }
2678
2679    #[test]
2680    fn test_interval_amount_parsing() {
2681        // integer
2682        let result = IntervalAmount::from_str("123").unwrap();
2683        let expected = IntervalAmount::new(123, 0);
2684
2685        assert_eq!(result, expected);
2686
2687        // positive w/ fractional
2688        let result = IntervalAmount::from_str("0.3").unwrap();
2689        let expected = IntervalAmount::new(0, 3 * 10_i64.pow(INTERVAL_PRECISION - 1));
2690
2691        assert_eq!(result, expected);
2692
2693        // negative w/ fractional
2694        let result = IntervalAmount::from_str("-3.5").unwrap();
2695        let expected = IntervalAmount::new(-3, -5 * 10_i64.pow(INTERVAL_PRECISION - 1));
2696
2697        assert_eq!(result, expected);
2698
2699        // invalid: missing fractional
2700        let result = IntervalAmount::from_str("3.");
2701        assert!(result.is_err());
2702
2703        // invalid: sign in fractional
2704        let result = IntervalAmount::from_str("3.-5");
2705        assert!(result.is_err());
2706    }
2707
2708    #[test]
2709    fn test_interval_precision() {
2710        let config = IntervalParseConfig::new(IntervalUnit::Month);
2711
2712        let result = Interval::parse("100000.1 days", &config).unwrap();
2713        let expected = Interval::new(0_i32, 100_000_i32, NANOS_PER_DAY / 10);
2714
2715        assert_eq!(result, expected);
2716    }
2717
2718    #[test]
2719    fn test_interval_addition() {
2720        // add 4.1 centuries
2721        let start = Interval::new(1, 2, 3);
2722        let expected = Interval::new(4921, 2, 3);
2723
2724        let result = start
2725            .add(
2726                IntervalAmount::new(4, 10_i64.pow(INTERVAL_PRECISION - 1)),
2727                IntervalUnit::Century,
2728            )
2729            .unwrap();
2730
2731        assert_eq!(result, expected);
2732
2733        // add 10.25 decades
2734        let start = Interval::new(1, 2, 3);
2735        let expected = Interval::new(1231, 2, 3);
2736
2737        let result = start
2738            .add(
2739                IntervalAmount::new(10, 25 * 10_i64.pow(INTERVAL_PRECISION - 2)),
2740                IntervalUnit::Decade,
2741            )
2742            .unwrap();
2743
2744        assert_eq!(result, expected);
2745
2746        // add 30.3 years (reminder: Postgres logic does not spill to days/nanos when interval is larger than a month)
2747        let start = Interval::new(1, 2, 3);
2748        let expected = Interval::new(364, 2, 3);
2749
2750        let result = start
2751            .add(
2752                IntervalAmount::new(30, 3 * 10_i64.pow(INTERVAL_PRECISION - 1)),
2753                IntervalUnit::Year,
2754            )
2755            .unwrap();
2756
2757        assert_eq!(result, expected);
2758
2759        // add 1.5 months
2760        let start = Interval::new(1, 2, 3);
2761        let expected = Interval::new(2, 17, 3);
2762
2763        let result = start
2764            .add(
2765                IntervalAmount::new(1, 5 * 10_i64.pow(INTERVAL_PRECISION - 1)),
2766                IntervalUnit::Month,
2767            )
2768            .unwrap();
2769
2770        assert_eq!(result, expected);
2771
2772        // add -2 weeks
2773        let start = Interval::new(1, 25, 3);
2774        let expected = Interval::new(1, 11, 3);
2775
2776        let result = start
2777            .add(IntervalAmount::new(-2, 0), IntervalUnit::Week)
2778            .unwrap();
2779
2780        assert_eq!(result, expected);
2781
2782        // add 2.2 days
2783        let start = Interval::new(12, 15, 3);
2784        let expected = Interval::new(12, 17, 3 + 17_280 * NANOS_PER_SECOND);
2785
2786        let result = start
2787            .add(
2788                IntervalAmount::new(2, 2 * 10_i64.pow(INTERVAL_PRECISION - 1)),
2789                IntervalUnit::Day,
2790            )
2791            .unwrap();
2792
2793        assert_eq!(result, expected);
2794
2795        // add 12.5 hours
2796        let start = Interval::new(1, 2, 3);
2797        let expected = Interval::new(1, 2, 3 + 45_000 * NANOS_PER_SECOND);
2798
2799        let result = start
2800            .add(
2801                IntervalAmount::new(12, 5 * 10_i64.pow(INTERVAL_PRECISION - 1)),
2802                IntervalUnit::Hour,
2803            )
2804            .unwrap();
2805
2806        assert_eq!(result, expected);
2807
2808        // add -1.5 minutes
2809        let start = Interval::new(0, 0, -3);
2810        let expected = Interval::new(0, 0, -90_000_000_000 - 3);
2811
2812        let result = start
2813            .add(
2814                IntervalAmount::new(-1, -5 * 10_i64.pow(INTERVAL_PRECISION - 1)),
2815                IntervalUnit::Minute,
2816            )
2817            .unwrap();
2818
2819        assert_eq!(result, expected);
2820    }
2821
2822    #[test]
2823    fn string_to_timestamp_old() {
2824        parse_timestamp("1677-06-14T07:29:01.256")
2825            .map_err(|e| assert!(e.to_string().ends_with(ERR_NANOSECONDS_NOT_SUPPORTED)))
2826            .unwrap_err();
2827    }
2828
2829    #[test]
2830    fn test_parse_decimal_with_parameter() {
2831        let tests = [
2832            ("0", 0i128),
2833            ("123.123", 123123i128),
2834            ("123.1234", 123123i128),
2835            ("123.1", 123100i128),
2836            ("123", 123000i128),
2837            ("-123.123", -123123i128),
2838            ("-123.1234", -123123i128),
2839            ("-123.1", -123100i128),
2840            ("-123", -123000i128),
2841            ("0.0000123", 0i128),
2842            ("12.", 12000i128),
2843            ("-12.", -12000i128),
2844            ("00.1", 100i128),
2845            ("-00.1", -100i128),
2846            ("12345678912345678.1234", 12345678912345678123i128),
2847            ("-12345678912345678.1234", -12345678912345678123i128),
2848            ("99999999999999999.999", 99999999999999999999i128),
2849            ("-99999999999999999.999", -99999999999999999999i128),
2850            (".123", 123i128),
2851            ("-.123", -123i128),
2852            ("123.", 123000i128),
2853            ("-123.", -123000i128),
2854        ];
2855        for (s, i) in tests {
2856            let result_128 = parse_decimal::<Decimal128Type>(s, 20, 3);
2857            assert_eq!(i, result_128.unwrap());
2858            let result_256 = parse_decimal::<Decimal256Type>(s, 20, 3);
2859            assert_eq!(i256::from_i128(i), result_256.unwrap());
2860        }
2861
2862        let e_notation_tests = [
2863            ("1.23e3", "1230.0", 2),
2864            ("5.6714e+2", "567.14", 4),
2865            ("5.6714e-2", "0.056714", 4),
2866            ("5.6714e-2", "0.056714", 3),
2867            ("5.6741214125e2", "567.41214125", 4),
2868            ("8.91E4", "89100.0", 2),
2869            ("3.14E+5", "314000.0", 2),
2870            ("2.718e0", "2.718", 2),
2871            ("9.999999e-1", "0.9999999", 4),
2872            ("1.23e+3", "1230", 2),
2873            ("1.234559e+3", "1234.559", 2),
2874            ("1.00E-10", "0.0000000001", 11),
2875            ("1.23e-4", "0.000123", 2),
2876            ("9.876e7", "98760000.0", 2),
2877            ("5.432E+8", "543200000.0", 10),
2878            ("1.234567e9", "1234567000.0", 2),
2879            ("1.234567e2", "123.45670000", 2),
2880            ("4749.3e-5", "0.047493", 10),
2881            ("4749.3e+5", "474930000", 10),
2882            ("4749.3e-5", "0.047493", 1),
2883            ("4749.3e+5", "474930000", 1),
2884            ("0E-8", "0", 10),
2885            ("0E+6", "0", 10),
2886            ("0e0", "0", 10),
2887            ("-0e0", "0", 10),
2888            ("00e48", "0", 10),
2889            ("1E-8", "0.00000001", 10),
2890            ("12E+6", "12000000", 10),
2891            ("12E-6", "0.000012", 10),
2892            ("0.1e-6", "0.0000001", 10),
2893            ("0.1e+6", "100000", 10),
2894            ("0.12e-6", "0.00000012", 10),
2895            ("0.12e+6", "120000", 10),
2896            ("000000000001e0", "000000000001", 3),
2897            ("000001.1034567002e0", "000001.1034567002", 3),
2898            ("1.234e16", "12340000000000000", 0),
2899            ("123.4e16", "1234000000000000000", 0),
2900            ("15e-1", "1.5", 0),
2901            ("1.25e1", "12.5", 0),
2902            ("1.5e-1", "0.15", 1),
2903        ];
2904        for (e, d, scale) in e_notation_tests {
2905            let result_128_e = parse_decimal::<Decimal128Type>(e, 20, scale);
2906            let result_128_d = parse_decimal::<Decimal128Type>(d, 20, scale);
2907            assert_eq!(result_128_e.unwrap(), result_128_d.unwrap(), "{e} vs {d}");
2908            let result_256_e = parse_decimal::<Decimal256Type>(e, 20, scale);
2909            let result_256_d = parse_decimal::<Decimal256Type>(d, 20, scale);
2910            assert_eq!(result_256_e.unwrap(), result_256_d.unwrap(), "{e} vs {d}");
2911        }
2912        let can_not_parse_tests = [
2913            "123,123",
2914            ".",
2915            "123.123.123",
2916            "",
2917            "+",
2918            "-",
2919            "e",
2920            "e5",
2921            "-.",
2922            "+e-11",
2923            "-.E+3",
2924            ".e5",
2925            "1.3e+e3",
2926            "5.6714ee-2",
2927            "4.11ee-+4",
2928            "4.11e++4",
2929            "1.1e.12",
2930            "1.23e+3.",
2931            "1.23e+3.1",
2932            "1e",
2933            "1e+",
2934            "1e-",
2935            "1e5e5",
2936            "1 000",
2937            "1_000",
2938            "- 1",
2939            "1.5 x",
2940            "0x10",
2941            "NaN",
2942            "inf",
2943            "\u{661}\u{662}",
2944            "\u{ff11}",
2945        ];
2946        for s in can_not_parse_tests {
2947            let result_128 = parse_decimal::<Decimal128Type>(s, 20, 3);
2948            assert_eq!(
2949                format!("Parser error: Invalid decimal format: {s:?}"),
2950                result_128.unwrap_err().to_string()
2951            );
2952            let result_256 = parse_decimal::<Decimal256Type>(s, 20, 3);
2953            assert_eq!(
2954                format!("Parser error: Invalid decimal format: {s:?}"),
2955                result_256.unwrap_err().to_string()
2956            );
2957        }
2958        let overflow_parse_tests = [
2959            ("12345678", 3),
2960            ("1.2345678e7", 3),
2961            ("12345678.9", 3),
2962            ("1.23456789e+7", 3),
2963            ("99999999.99", 3),
2964            ("9.999999999e7", 3),
2965            ("12345678908765.123456", 3),
2966            ("123456789087651234.56e-4", 3),
2967            ("1234560000000", 0),
2968            ("12345678900.0", 0),
2969            ("1.23456e12", 0),
2970            ("9999999.9995", 3),
2971            ("1e99999", 0),
2972            ("1e40", 0),
2973        ];
2974        for (s, scale) in overflow_parse_tests {
2975            let result_128 = parse_decimal::<Decimal128Type>(s, 10, scale);
2976            let expected_128 =
2977                format!("Parser error: {s:?} does not fit in Decimal128(10, {scale})");
2978            assert_eq!(result_128.unwrap_err().to_string(), expected_128);
2979
2980            let result_256 = parse_decimal::<Decimal256Type>(s, 10, scale);
2981            let expected_256 =
2982                format!("Parser error: {s:?} does not fit in Decimal256(10, {scale})");
2983            assert_eq!(result_256.unwrap_err().to_string(), expected_256);
2984        }
2985
2986        let edge_tests_128 = [
2987            (
2988                "99999999999999999999999999999999999999",
2989                99999999999999999999999999999999999999i128,
2990                0,
2991            ),
2992            (
2993                "999999999999999999999999999999999999.99",
2994                99999999999999999999999999999999999999i128,
2995                2,
2996            ),
2997            (
2998                "9999999999999999999999999.9999999999999",
2999                99999999999999999999999999999999999999i128,
3000                13,
3001            ),
3002            (
3003                "9999999999999999999999999",
3004                99999999999999999999999990000000000000i128,
3005                13,
3006            ),
3007            (
3008                "0.99999999999999999999999999999999999999",
3009                99999999999999999999999999999999999999i128,
3010                38,
3011            ),
3012            (
3013                "0.00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000001016744",
3014                0i128,
3015                15,
3016            ),
3017            ("1.016744e-320", 0i128, 15),
3018            ("-1e3", -1000000000i128, 6),
3019            ("+1e3", 1000000000i128, 6),
3020            ("-1e31", -10000000000000000000000000000000000000i128, 6),
3021            // More digits than an i128 can hold, but a small value
3022            ("10000000000000000000000000000000000000000e-39", 10i128, 0),
3023            // Digits beyond the scale round; here the result still fits
3024            (
3025                "99999999999999999999999999999999999994e-1",
3026                9999999999999999999999999999999999999i128,
3027                0,
3028            ),
3029        ];
3030        for (s, i, scale) in edge_tests_128 {
3031            let result_128 = parse_decimal::<Decimal128Type>(s, 38, scale);
3032            assert_eq!(i, result_128.unwrap(), "{s}");
3033        }
3034        // Rounding carries into a 39th digit, which does not fit
3035        assert!(
3036            parse_decimal::<Decimal128Type>("999999999999999999999999999999999999999e-1", 38, 0)
3037                .is_err()
3038        );
3039
3040        let edge_tests_256 = [
3041            (
3042                "9999999999999999999999999999999999999999999999999999999999999999999999999999",
3043                i256::from_string(
3044                    "9999999999999999999999999999999999999999999999999999999999999999999999999999",
3045                )
3046                .unwrap(),
3047                0,
3048            ),
3049            (
3050                "999999999999999999999999999999999999999999999999999999999999999999999999.9999",
3051                i256::from_string(
3052                    "9999999999999999999999999999999999999999999999999999999999999999999999999999",
3053                )
3054                .unwrap(),
3055                4,
3056            ),
3057            (
3058                "99999999999999999999999999999999999999999999999999.99999999999999999999999999",
3059                i256::from_string(
3060                    "9999999999999999999999999999999999999999999999999999999999999999999999999999",
3061                )
3062                .unwrap(),
3063                26,
3064            ),
3065            (
3066                "9.999999999999999999999999999999999999999999999999999999999999999999999999999e49",
3067                i256::from_string(
3068                    "9999999999999999999999999999999999999999999999999999999999999999999999999999",
3069                )
3070                .unwrap(),
3071                26,
3072            ),
3073            (
3074                "99999999999999999999999999999999999999999999999999",
3075                i256::from_string(
3076                    "9999999999999999999999999999999999999999999999999900000000000000000000000000",
3077                )
3078                .unwrap(),
3079                26,
3080            ),
3081            (
3082                "9.9999999999999999999999999999999999999999999999999e+49",
3083                i256::from_string(
3084                    "9999999999999999999999999999999999999999999999999900000000000000000000000000",
3085                )
3086                .unwrap(),
3087                26,
3088            ),
3089        ];
3090        for (s, i, scale) in edge_tests_256 {
3091            let result = parse_decimal::<Decimal256Type>(s, 76, scale);
3092            assert_eq!(i, result.unwrap());
3093        }
3094
3095        let zero_scale_tests = [
3096            (".123", 0, 3),
3097            ("0.123", 0, 3),
3098            ("1.0", 1, 3),
3099            ("1.2", 1, 3),
3100            ("1.00", 1, 3),
3101            ("1.23", 1, 3),
3102            ("1.000", 1, 3),
3103            ("1.123", 1, 3),
3104            ("1.5", 2, 3),
3105            ("1.9", 2, 3),
3106            ("123.0", 123, 3),
3107            ("123.4", 123, 3),
3108            ("123.00", 123, 3),
3109            ("123.45", 123, 3),
3110            ("123.5", 124, 3),
3111            ("123.000000000000000000004", 123, 3),
3112            ("0.123e2", 12, 3),
3113            ("0.123e4", 1230, 10),
3114            ("1.23e4", 12300, 10),
3115            ("12.3e4", 123000, 10),
3116            ("123e4", 1230000, 10),
3117            (
3118                "20000000000000000000000000000000000002.0",
3119                20000000000000000000000000000000000002,
3120                38,
3121            ),
3122        ];
3123        for (s, i, precision) in zero_scale_tests {
3124            let result_128 = parse_decimal::<Decimal128Type>(s, precision, 0).unwrap();
3125            assert_eq!(i, result_128, "{s}");
3126        }
3127
3128        let can_not_parse_zero_scale = [".", "blag", "", "+", "-", "e"];
3129        for s in can_not_parse_zero_scale {
3130            let result_128 = parse_decimal::<Decimal128Type>(s, 5, 0);
3131            assert_eq!(
3132                format!("Parser error: Invalid decimal format: {s:?}"),
3133                result_128.unwrap_err().to_string(),
3134            );
3135        }
3136    }
3137
3138    #[test]
3139    fn test_parse_decimal_rounds_half_away_from_zero() {
3140        let tests = [
3141            ("1.234", 2, 123),
3142            ("1.235", 2, 124),
3143            ("1.2350000", 2, 124),
3144            ("1.2349999", 2, 123),
3145            ("-1.234", 2, -123),
3146            ("-1.235", 2, -124),
3147            ("-0.004", 2, 0),
3148            ("-0.005", 2, -1),
3149            (".5", 0, 1),
3150            ("-.5", 0, -1),
3151            ("0.5", 0, 1),
3152            ("1.5", 0, 2),
3153            ("2.5", 0, 3),
3154            ("-2.5", 0, -3),
3155            ("1.99", 1, 20),
3156            ("0.995", 2, 100),
3157            ("9.99", 1, 100),
3158            ("123.4567891", 5, 12345679),
3159            ("123.45", 0, 123),
3160            ("0.0000123", 3, 0),
3161            ("12.", 2, 1200),
3162            (".12", 2, 12),
3163            ("+.12", 2, 12),
3164            ("-.12", 2, -12),
3165        ];
3166        for (s, scale, expected) in tests {
3167            assert_eq!(
3168                parse_decimal::<Decimal128Type>(s, 38, scale).unwrap(),
3169                expected,
3170                "{s} at scale {scale}"
3171            );
3172            assert_eq!(
3173                parse_decimal::<Decimal256Type>(s, 76, scale).unwrap(),
3174                i256::from_i128(expected),
3175                "{s} at scale {scale}"
3176            );
3177        }
3178    }
3179
3180    #[test]
3181    fn test_parse_decimal_rounding_overflow() {
3182        // Rounding up can push the value past the precision ...
3183        assert!(parse_decimal::<Decimal128Type>("99999.5", 5, 0).is_err());
3184        assert!(parse_decimal::<Decimal128Type>("9.995", 3, 2).is_err());
3185        assert_eq!(parse_decimal::<Decimal128Type>("9.994", 3, 2).unwrap(), 999);
3186        assert_eq!(parse_decimal::<Decimal128Type>("0.995", 3, 2).unwrap(), 100);
3187
3188        // ... or past the native type itself
3189        assert_eq!(
3190            parse_native::<Decimal32Type>("2147483647.5", 0),
3191            Err(DecimalParseError::Overflow)
3192        );
3193        assert_eq!(
3194            parse_native::<Decimal32Type>("-2147483648.5", 0),
3195            Err(DecimalParseError::Overflow)
3196        );
3197        assert_eq!(
3198            parse_native::<Decimal128Type>(&format!("{}.5", i128::MAX), 0),
3199            Err(DecimalParseError::Overflow)
3200        );
3201        assert_eq!(
3202            parse_native::<Decimal128Type>(&format!("{}.5", i128::MIN), 0),
3203            Err(DecimalParseError::Overflow)
3204        );
3205        assert_eq!(
3206            parse_native::<Decimal256Type>(&format!("{}.5", i256::MAX), 0),
3207            Err(DecimalParseError::Overflow)
3208        );
3209        assert_eq!(
3210            parse_native::<Decimal256Type>(&format!("{}.5", i256::MIN), 0),
3211            Err(DecimalParseError::Overflow)
3212        );
3213    }
3214
3215    #[test]
3216    fn test_parse_decimal_precision_by_digit_count() {
3217        // Rounding up can add a digit
3218        assert_eq!(
3219            parse_decimal::<Decimal128Type>("99999.4", 5, 0).unwrap(),
3220            99999
3221        );
3222        assert!(parse_decimal::<Decimal128Type>("99999.5", 5, 0).is_err());
3223        assert!(parse_decimal::<Decimal128Type>("-99999.5", 5, 0).is_err());
3224        assert_eq!(
3225            parse_decimal::<Decimal128Type>("99999.5", 6, 0).unwrap(),
3226            100000
3227        );
3228        // Leading zeros count as digits only for the shortcut; the value is
3229        // then checked by its range
3230        assert_eq!(
3231            parse_decimal::<Decimal128Type>("000000000000000000000001", 1, 0).unwrap(),
3232            1
3233        );
3234        assert_eq!(
3235            parse_decimal::<Decimal128Type>("0.000000000000000000001", 1, 21).unwrap(),
3236            1
3237        );
3238        assert!(parse_decimal::<Decimal128Type>("0.0000000000000000000012", 1, 22).is_err());
3239        // The zeros appended to reach the scale count as digits
3240        assert_eq!(parse_decimal::<Decimal128Type>("1", 3, 2).unwrap(), 100);
3241        assert!(parse_decimal::<Decimal128Type>("1", 2, 2).is_err());
3242        assert!(parse_decimal::<Decimal128Type>("1e2", 2, 0).is_err());
3243        assert_eq!(parse_decimal::<Decimal32Type>("1e2", 3, 0).unwrap(), 100);
3244        // Scaling down leaves fewer digits
3245        assert_eq!(
3246            parse_decimal::<Decimal128Type>("123456", 2, -4).unwrap(),
3247            12
3248        );
3249        assert!(parse_decimal::<Decimal128Type>("123456", 1, -4).is_err());
3250        // A precision beyond the type's maximum is invalid
3251        assert!(parse_decimal::<Decimal32Type>("1", 10, 0).is_err());
3252        assert!(parse_decimal::<Decimal32Type>("00000000001", 10, 0).is_err());
3253        assert!(parse_decimal::<Decimal128Type>("1", 39, 0).is_err());
3254        assert!(parse_decimal::<Decimal256Type>("1", 77, 0).is_err());
3255    }
3256
3257    #[test]
3258    fn test_parse_decimal_native_full_range() {
3259        // The native range exceeds the largest precision; the precision check
3260        // is the caller's responsibility
3261        assert_eq!(
3262            parse_native::<Decimal32Type>("-2147483648", 0),
3263            Ok(i32::MIN)
3264        );
3265        assert_eq!(
3266            parse_native::<Decimal32Type>("2147483648", 0),
3267            Err(DecimalParseError::Overflow)
3268        );
3269        assert_eq!(
3270            parse_native::<Decimal64Type>("-9223372036854775808", 0),
3271            Ok(i64::MIN)
3272        );
3273        assert_eq!(
3274            parse_native::<Decimal64Type>("9223372036854775808", 0),
3275            Err(DecimalParseError::Overflow)
3276        );
3277        assert_eq!(
3278            parse_native::<Decimal128Type>(&i128::MAX.to_string(), 0),
3279            Ok(i128::MAX)
3280        );
3281        assert_eq!(
3282            parse_native::<Decimal128Type>(&i128::MIN.to_string(), 0),
3283            Ok(i128::MIN)
3284        );
3285        assert_eq!(
3286            parse_native::<Decimal256Type>(&i256::MAX.to_string(), 0),
3287            Ok(i256::MAX)
3288        );
3289        assert_eq!(
3290            parse_native::<Decimal256Type>(&i256::MIN.to_string(), 0),
3291            Ok(i256::MIN)
3292        );
3293        // The unscaled value (integer digits scaled by 10^21) far exceeds the
3294        // i256 range, so this must report overflow rather than wrapping to an
3295        // arbitrary (possibly in-range) value
3296        let input = format!("{}.12345678901234567890123", "7".repeat(71));
3297        assert_eq!(
3298            parse_native::<Decimal256Type>(&input, 21),
3299            Err(DecimalParseError::Overflow)
3300        );
3301
3302        assert!(parse_decimal::<Decimal128Type>(&i128::MAX.to_string(), 38, 0).is_err());
3303        assert!(parse_decimal::<Decimal32Type>("-2147483648", 9, 0).is_err());
3304    }
3305
3306    #[test]
3307    fn test_parse_decimal_integer_widths() {
3308        assert_eq!(
3309            parse_decimal::<Decimal32Type>("123.45", 9, 2).unwrap(),
3310            12_345_i32
3311        );
3312        assert_eq!(
3313            parse_decimal::<Decimal32Type>("-9999999.994", 9, 2).unwrap(),
3314            -999_999_999_i32
3315        );
3316        assert!(parse_decimal::<Decimal32Type>("9999999.995", 9, 2).is_err());
3317        assert!(parse_decimal::<Decimal32Type>("-9999999.995", 9, 2).is_err());
3318        assert_eq!(
3319            parse_decimal::<Decimal64Type>("123.45", 18, 2).unwrap(),
3320            12_345_i64
3321        );
3322        assert_eq!(
3323            parse_decimal::<Decimal64Type>("9999999999999999.99", 18, 2).unwrap(),
3324            999_999_999_999_999_999_i64
3325        );
3326        assert!(parse_decimal::<Decimal64Type>("10000000000000000.00", 18, 2).is_err());
3327        // Fractional parts longer than any native integer type parse fine;
3328        // digits beyond the scale only matter for rounding
3329        assert_eq!(
3330            parse_decimal::<Decimal64Type>(&format!(".{}", "5".repeat(100)), 18, 4).unwrap(),
3331            5_556_i64
3332        );
3333        assert_eq!(
3334            parse_decimal::<Decimal128Type>(&format!(".{}", "1".repeat(100)), 38, 4).unwrap(),
3335            1_111_i128
3336        );
3337    }
3338
3339    #[test]
3340    fn test_parse_decimal_exponent() {
3341        let tests = [
3342            ("1e2", 0, 100),
3343            ("1E2", 0, 100),
3344            ("1e+2", 0, 100),
3345            ("1e+02", 0, 100),
3346            ("1.5e2", 0, 150),
3347            ("1.5e2", 2, 15000),
3348            ("1.5e-1", 1, 2),
3349            ("15e-1", 0, 2),
3350            ("1e-2", 1, 0),
3351            ("1e-3", 2, 0),
3352            ("0e0", 2, 0),
3353            ("-0e0", 2, 0),
3354            ("0E5", 2, 0),
3355            ("0e99999", 2, 0),
3356            ("00e48", 8, 0),
3357            ("+00.0E+41", 12, 0),
3358            ("1.25e1", 0, 13),
3359            ("1e-99999", 2, 0),
3360            ("1.5e-400", 2, 0),
3361            ("123456789e-9", 9, 123456789),
3362            ("0.000000001e9", 0, 1),
3363            ("5e-1", 0, 1),
3364            ("4e-1", 0, 0),
3365            ("-5e-1", 0, -1),
3366        ];
3367        for (s, scale, expected) in tests {
3368            assert_eq!(
3369                parse_decimal::<Decimal128Type>(s, 38, scale).unwrap(),
3370                expected,
3371                "{s} at scale {scale}"
3372            );
3373            assert_eq!(
3374                parse_decimal::<Decimal32Type>(s, 9, scale).unwrap(),
3375                expected as i32,
3376                "{s} at scale {scale}"
3377            );
3378        }
3379
3380        // Exponents shift digits across the decimal point without losing any
3381        assert_eq!(
3382            parse_decimal::<Decimal32Type>("4825037936439135476.2609835314269495255615E-14", 9, 4)
3383                .unwrap(),
3384            482503794
3385        );
3386        assert_eq!(
3387            parse_decimal::<Decimal32Type>(
3388                "+18232335063972188138031550982650807591758238.0724251287782783777442440E-58",
3389                1,
3390                0
3391            )
3392            .unwrap(),
3393            0
3394        );
3395        assert_eq!(
3396            parse_decimal::<Decimal128Type>("4825037936439135476.2609835314269495255615E-14", 9, 1)
3397                .unwrap(),
3398            482504
3399        );
3400        assert!(
3401            parse_decimal::<Decimal128Type>("4825037936439135476.2609835314269495255615E-14", 5, 1)
3402                .is_err()
3403        );
3404        // Absurdly long exponents saturate rather than wrap
3405        assert!(parse_decimal::<Decimal128Type>(&format!("1e{}", "9".repeat(30)), 38, 0).is_err());
3406        assert_eq!(
3407            parse_decimal::<Decimal128Type>(&format!("1e-{}", "9".repeat(30)), 38, 0).unwrap(),
3408            0
3409        );
3410    }
3411
3412    #[test]
3413    fn test_parse_decimal_negative_scale() {
3414        let tests = [
3415            ("1234.5", -2, 12),
3416            ("150", -2, 2),
3417            ("149", -2, 1),
3418            ("-150", -2, -2),
3419            ("-149", -2, -1),
3420            ("50", -2, 1),
3421            ("49", -2, 0),
3422            ("5", -1, 1),
3423            ("4", -1, 0),
3424            ("0.5", -1, 0),
3425            ("5.9", -1, 1),
3426            ("1e5", -2, 1000),
3427            ("1.5e5", -2, 1500),
3428            ("0.9e2", -1, 9),
3429            (".5e3", -2, 5),
3430            ("12345", -5, 0),
3431            ("12345", -4, 1),
3432            ("000123456", -3, 123),
3433            ("0", -5, 0),
3434            ("-0.0", -5, 0),
3435        ];
3436        for (s, scale, expected) in tests {
3437            assert_eq!(
3438                parse_decimal::<Decimal128Type>(s, 38, scale).unwrap(),
3439                expected,
3440                "{s} at scale {scale}"
3441            );
3442            assert_eq!(
3443                parse_decimal::<Decimal32Type>(s, 9, scale).unwrap(),
3444                expected as i32,
3445                "{s} at scale {scale}"
3446            );
3447            assert_eq!(
3448                parse_decimal::<Decimal256Type>(s, 76, scale).unwrap(),
3449                i256::from_i128(expected),
3450                "{s} at scale {scale}"
3451            );
3452        }
3453        // The integer part can be wider than the native type as long as the
3454        // scaled value fits
3455        assert_eq!(
3456            parse_decimal::<Decimal128Type>(&format!("1{}", "0".repeat(50)), 38, -40).unwrap(),
3457            10_000_000_000
3458        );
3459        assert_eq!(
3460            parse_decimal::<Decimal32Type>("123456789012", 9, -5).unwrap(),
3461            1234568
3462        );
3463        assert!(parse_decimal::<Decimal32Type>("123456789012", 9, -2).is_err());
3464    }
3465
3466    #[test]
3467    fn test_parse_decimal_whitespace_and_long_input() {
3468        for s in [" 1.5", "1.5 ", " 1.5 ", "\t1.5\n", "\r\n1.5\x0c"] {
3469            assert_eq!(
3470                parse_decimal::<Decimal128Type>(s, 38, 1).unwrap(),
3471                15,
3472                "{s:?}"
3473            );
3474        }
3475        // Only ASCII whitespace is trimmed, as for the other CSV parsers
3476        assert!(parse_decimal::<Decimal128Type>("\u{a0}1.5", 38, 1).is_err());
3477        assert!(parse_decimal::<Decimal128Type>("1.5\u{2003}", 38, 1).is_err());
3478        assert!(parse_decimal::<Decimal128Type>(" ", 38, 1).is_err());
3479
3480        // Long inputs report overflow rather than wrapping or panicking
3481        for s in [
3482            "1".repeat(255),
3483            "1".repeat(256),
3484            "1".repeat(300),
3485            format!("{}.5", "1".repeat(300)),
3486            format!("1e{}", "9".repeat(300)),
3487        ] {
3488            let err = parse_decimal::<Decimal128Type>(&s, 38, 0).unwrap_err();
3489            assert!(err.to_string().contains("does not fit"), "{err}");
3490        }
3491        // Long fractions only matter for rounding
3492        assert_eq!(
3493            parse_decimal::<Decimal128Type>(&format!("0.{}", "0".repeat(200)), 38, 10).unwrap(),
3494            0
3495        );
3496        assert_eq!(
3497            parse_decimal::<Decimal128Type>(&format!("0.{}1", "0".repeat(200)), 38, 10).unwrap(),
3498            0
3499        );
3500        assert_eq!(
3501            parse_decimal::<Decimal128Type>(&format!("1.{}", "9".repeat(300)), 38, 2).unwrap(),
3502            200
3503        );
3504        // 10^scale overflows the native type, but zero is still representable
3505        assert_eq!(parse_decimal::<Decimal32Type>("0", 9, 10).unwrap(), 0);
3506        assert_eq!(parse_decimal::<Decimal32Type>("-0.0", 9, 10).unwrap(), 0);
3507        assert_eq!(parse_decimal::<Decimal64Type>("0", 18, 20).unwrap(), 0);
3508        assert_eq!(parse_decimal::<Decimal128Type>("0", 38, 40).unwrap(), 0);
3509        assert!(parse_decimal::<Decimal32Type>("1", 9, 10).is_err());
3510    }
3511
3512    #[test]
3513    #[cfg_attr(miri, ignore)] // Takes too long under Miri (adds ~1 hour to CI)
3514    fn test_parse_decimal_matches_bigint_reference() {
3515        use num_bigint::BigInt;
3516        use rand::rngs::StdRng;
3517        use rand::{RngExt, SeedableRng};
3518
3519        /// Generates random decimal strings with a known exact value and checks
3520        /// that `parse_decimal` rounds them correctly or reports overflow
3521        fn check<T: DecimalType>(rng: &mut StdRng, iterations: usize)
3522        where
3523            T::Native: std::fmt::Display,
3524        {
3525            let random_digits = |rng: &mut StdRng, len: usize| -> String {
3526                (0..len)
3527                    .map(|_| char::from(b'0' + rng.random_range(0..10u8)))
3528                    .collect()
3529            };
3530            for _ in 0..iterations {
3531                let sign = ["", "+", "-"][rng.random_range(0..3)];
3532                let int_len = rng.random_range(0..=40);
3533                let frac_len = if rng.random_bool(0.3) {
3534                    0
3535                } else {
3536                    rng.random_range(0..=40)
3537                };
3538                if int_len == 0 && frac_len == 0 {
3539                    continue;
3540                }
3541                let int = random_digits(rng, int_len);
3542                let frac = random_digits(rng, frac_len);
3543                let mut s = format!("{sign}{int}");
3544                if frac_len > 0 || rng.random_bool(0.2) {
3545                    s.push('.');
3546                    s.push_str(&frac);
3547                }
3548                let exponent: i64 = if rng.random_bool(0.3) {
3549                    rng.random_range(-60..=60)
3550                } else {
3551                    0
3552                };
3553                if exponent != 0 || rng.random_bool(0.1) {
3554                    s.push(if rng.random_bool(0.5) { 'e' } else { 'E' });
3555                    if exponent >= 0 && rng.random_bool(0.5) {
3556                        s.push('+');
3557                    }
3558                    s.push_str(&exponent.to_string());
3559                }
3560                let precision = rng.random_range(1..=T::MAX_PRECISION);
3561                let scale = rng.random_range(-10..=T::MAX_SCALE.min(precision as i8));
3562
3563                // value = mantissa * 10^(exponent - frac_len), scaled by 10^scale
3564                // and rounded half away from zero
3565                let mantissa: BigInt = format!("{int}{frac}").parse().unwrap();
3566                let shift = exponent - frac_len as i64 + scale as i64;
3567                let mut expected = if shift >= 0 {
3568                    mantissa * BigInt::from(10).pow(shift as u32)
3569                } else {
3570                    let divisor = BigInt::from(10).pow((-shift) as u32);
3571                    let quotient = &mantissa / &divisor;
3572                    if (&mantissa % &divisor) * 2 >= divisor {
3573                        quotient + 1
3574                    } else {
3575                        quotient
3576                    }
3577                };
3578                if sign == "-" {
3579                    expected = -expected;
3580                }
3581                let limit = BigInt::from(10).pow(precision as u32);
3582                let fits = expected < limit && expected > -limit;
3583
3584                match (fits, parse_decimal::<T>(&s, precision, scale)) {
3585                    (true, Ok(actual)) => {
3586                        let actual: BigInt = actual.to_string().parse().unwrap();
3587                        assert_eq!(
3588                            actual,
3589                            expected,
3590                            "{s:?} as {}({precision}, {scale})",
3591                            T::PREFIX
3592                        );
3593                    }
3594                    (false, Err(_)) => {}
3595                    (true, Err(e)) => {
3596                        panic!(
3597                            "{s:?} as {}({precision}, {scale}): expected {expected}, got {e}",
3598                            T::PREFIX
3599                        )
3600                    }
3601                    (false, Ok(actual)) => panic!(
3602                        "{s:?} as {}({precision}, {scale}): expected overflow, got {actual}",
3603                        T::PREFIX
3604                    ),
3605                }
3606            }
3607        }
3608
3609        let mut rng = StdRng::seed_from_u64(0xDEC1_3A15);
3610        check::<Decimal32Type>(&mut rng, 5_000);
3611        check::<Decimal64Type>(&mut rng, 5_000);
3612        check::<Decimal128Type>(&mut rng, 5_000);
3613        check::<Decimal256Type>(&mut rng, 5_000);
3614    }
3615
3616    #[test]
3617    fn test_parse_empty() {
3618        assert_eq!(Int32Type::parse(""), None);
3619        assert_eq!(Int64Type::parse(""), None);
3620        assert_eq!(UInt32Type::parse(""), None);
3621        assert_eq!(UInt64Type::parse(""), None);
3622        assert_eq!(Float32Type::parse(""), None);
3623        assert_eq!(Float64Type::parse(""), None);
3624        assert_eq!(Int32Type::parse("+"), None);
3625        assert_eq!(Int64Type::parse("+"), None);
3626        assert_eq!(UInt32Type::parse("+"), None);
3627        assert_eq!(UInt64Type::parse("+"), None);
3628        assert_eq!(Float32Type::parse("+"), None);
3629        assert_eq!(Float64Type::parse("+"), None);
3630        assert_eq!(TimestampNanosecondType::parse(""), None);
3631        assert_eq!(Date32Type::parse(""), None);
3632    }
3633
3634    #[test]
3635    fn test_parse_interval_month_day_nano_config() {
3636        let interval = parse_interval_month_day_nano_config(
3637            "1",
3638            IntervalParseConfig::new(IntervalUnit::Second),
3639        )
3640        .unwrap();
3641        assert_eq!(interval.months, 0);
3642        assert_eq!(interval.days, 0);
3643        assert_eq!(interval.nanoseconds, NANOS_PER_SECOND);
3644    }
3645    #[test]
3646    fn test_parse_prefix_white_space() {
3647        assert_eq!(Float64Type::parse(" 1.5"), Some(1.5));
3648        assert_eq!(Float64Type::parse("\t\n 20.54"), Some(20.54));
3649        assert_eq!(Float64Type::parse("\n2.5"), Some(2.5));
3650        assert_eq!(Float64Type::parse("\n-942.5423"), Some(-942.5423));
3651        assert_eq!(Float64Type::parse("\n\t\n\t\n40.5123"), Some(40.5123));
3652        assert_eq!(Float64Type::parse(" 1.5"), Some(1.5));
3653        assert_eq!(Float64Type::parse("\n\t\n\t\n-40.5123"), Some(-40.5123));
3654        assert_eq!(Float64Type::parse(" -1.5"), Some(-1.5));
3655        assert_eq!(Int32Type::parse(" 3"), Some(3));
3656        assert_eq!(Int32Type::parse("          30"), Some(30));
3657        assert_eq!(Int32Type::parse("\n \n 100"), Some(100));
3658        assert_eq!(Int32Type::parse(" \n25"), Some(25));
3659        assert_eq!(Int32Type::parse("\t800"), Some(800));
3660        assert_eq!(Int32Type::parse("\t  \n \t 851"), Some(851));
3661        assert_eq!(Int32Type::parse("\t\n\t\n\n\n\t1"), Some(1));
3662        assert_eq!(Int32Type::parse(" \n-25"), Some(-25));
3663        assert_eq!(Int32Type::parse("\t-800"), Some(-800));
3664
3665        // suffix whitespace
3666        assert_eq!(Float64Type::parse("1.5 "), Some(1.5));
3667        assert_eq!(Float64Type::parse("40.5123\n"), Some(40.5123));
3668        assert_eq!(Float64Type::parse("40.5123\n\t\n\t\n"), Some(40.5123));
3669        assert_eq!(Float64Type::parse("-942.5423\t"), Some(-942.5423));
3670        assert_eq!(Int32Type::parse("3 "), Some(3));
3671        assert_eq!(Int32Type::parse("30          "), Some(30));
3672        assert_eq!(Int32Type::parse("-25 \n"), Some(-25));
3673        assert_eq!(Int32Type::parse("800\t"), Some(800));
3674        // whitespace on both sides
3675        assert_eq!(Float64Type::parse(" 1.5 "), Some(1.5));
3676        assert_eq!(Float64Type::parse("\t\n 20.54 \t"), Some(20.54));
3677        assert_eq!(Float64Type::parse("\n-942.5423\n"), Some(-942.5423));
3678        assert_eq!(Int32Type::parse(" 3 "), Some(3));
3679        assert_eq!(Int32Type::parse("\n \n 100 \n"), Some(100));
3680        assert_eq!(Int32Type::parse("\t-800\t\n"), Some(-800));
3681
3682        // trailing non-whitespace chars should not parse
3683        assert_eq!(Float64Type::parse("1.5abc"), None);
3684        assert_eq!(Float64Type::parse("40.5123x"), None);
3685        assert_eq!(Int32Type::parse("30x"), None);
3686        assert_eq!(Int32Type::parse("100px"), None);
3687        assert_eq!(Int32Type::parse("-25!"), None);
3688        assert_eq!(Int32Type::parse("3j"), None);
3689        assert_eq!(Int32Type::parse("3"), Some(3));
3690    }
3691
3692    #[test]
3693    fn test_parse_temporal_with_surrounding_whitespace() {
3694        let date = Date32Type::parse("2024-01-05");
3695        assert_eq!(Date32Type::parse(" 2024-01-05 "), date);
3696        assert_eq!(Date32Type::parse("\t2024-01-05\n"), date);
3697
3698        let timestamp = "2024-01-05T10:00:00";
3699        let padded_timestamp = " 2024-01-05T10:00:00 ";
3700
3701        assert_eq!(
3702            TimestampNanosecondType::parse(padded_timestamp),
3703            TimestampNanosecondType::parse(timestamp)
3704        );
3705        assert_eq!(
3706            TimestampMicrosecondType::parse(padded_timestamp),
3707            TimestampMicrosecondType::parse(timestamp)
3708        );
3709        assert_eq!(
3710            TimestampMillisecondType::parse(padded_timestamp),
3711            TimestampMillisecondType::parse(timestamp)
3712        );
3713        assert_eq!(
3714            TimestampSecondType::parse(padded_timestamp),
3715            TimestampSecondType::parse(timestamp)
3716        );
3717        assert_eq!(
3718            TimestampNanosecondType::parse("\t2024-01-05T10:00:00\n"),
3719            TimestampNanosecondType::parse(timestamp)
3720        );
3721
3722        // leet :)
3723        let time = "13:37:00";
3724        let padded_time = " 13:37:00 ";
3725
3726        assert_eq!(
3727            Time64NanosecondType::parse(padded_time),
3728            Time64NanosecondType::parse(time)
3729        );
3730        assert_eq!(
3731            Time64MicrosecondType::parse(padded_time),
3732            Time64MicrosecondType::parse(time)
3733        );
3734        assert_eq!(
3735            Time64MicrosecondType::parse("\t13:37:00\n"),
3736            Time64MicrosecondType::parse(time)
3737        );
3738    }
3739
3740    #[test]
3741    fn test_parse_time_with_surrounding_ascii_whitespace() {
3742        let time = "10:00:00";
3743        let padded = " 10:00:00 ";
3744        let ascii_whitespace = "\t10:00:00\n";
3745
3746        assert_eq!(
3747            Time64NanosecondType::parse(padded),
3748            Time64NanosecondType::parse(time)
3749        );
3750        assert_eq!(
3751            Time64MicrosecondType::parse(padded),
3752            Time64MicrosecondType::parse(time)
3753        );
3754        assert_eq!(
3755            Time32MillisecondType::parse(padded),
3756            Time32MillisecondType::parse(time)
3757        );
3758        assert_eq!(
3759            Time32SecondType::parse(padded),
3760            Time32SecondType::parse(time)
3761        );
3762
3763        assert_eq!(
3764            Time64NanosecondType::parse(ascii_whitespace),
3765            Time64NanosecondType::parse(time)
3766        );
3767        assert_eq!(
3768            Time64MicrosecondType::parse(ascii_whitespace),
3769            Time64MicrosecondType::parse(time)
3770        );
3771        assert_eq!(
3772            Time32MillisecondType::parse(ascii_whitespace),
3773            Time32MillisecondType::parse(time)
3774        );
3775        assert_eq!(
3776            Time32SecondType::parse(ascii_whitespace),
3777            Time32SecondType::parse(time)
3778        );
3779    }
3780}