Skip to main content

iceberg/spec/values/
primitive.rs

1// Licensed to the Apache Software Foundation (ASF) under one
2// or more contributor license agreements.  See the NOTICE file
3// distributed with this work for additional information
4// regarding copyright ownership.  The ASF licenses this file
5// to you under the Apache License, Version 2.0 (the
6// "License"); you may not use this file except in compliance
7// with the License.  You may obtain a copy of the License at
8//
9//   http://www.apache.org/licenses/LICENSE-2.0
10//
11// Unless required by applicable law or agreed to in writing,
12// software distributed under the License is distributed on an
13// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
14// KIND, either express or implied.  See the License for the
15// specific language governing permissions and limitations
16// under the License.
17
18//! Primitive literal types
19
20use std::cmp::Ordering;
21
22use ordered_float::{FloatCore, OrderedFloat};
23
24/// Values present in iceberg type
25///
26/// `Float` and `Double` compare as Java's `Float.compare` and `Double.compare` do: `-0.0` is
27/// less than `0.0`, and every NaN is the same value, greater than all others. iceberg-java
28/// compares partition values this way, so `-0.0` and `0.0` are different partitions. The
29/// [spec](https://iceberg.apache.org/spec/#scan-planning) states the same rule: floating point
30/// partition values are equal if their IEEE 754 bit layouts are equal, with NaNs normalized.
31// The derived `Hash` gives `-0.0` and `0.0` the same hash. That is coarser than `eq`, which is
32// allowed: equal values still hash alike.
33#[allow(clippy::derived_hash_with_manual_eq)]
34#[derive(Clone, Debug, Hash, Eq)]
35pub enum PrimitiveLiteral {
36    /// 0x00 for false, non-zero byte for true
37    Boolean(bool),
38    /// Stored as 4-byte little-endian
39    Int(i32),
40    /// Stored as 8-byte little-endian
41    Long(i64),
42    /// Stored as 4-byte little-endian
43    Float(OrderedFloat<f32>),
44    /// Stored as 8-byte little-endian
45    Double(OrderedFloat<f64>),
46    /// UTF-8 bytes (without length)
47    String(String),
48    /// Binary value (without length)
49    Binary(Vec<u8>),
50    /// Stored as 16-byte big-endian
51    Int128(i128),
52    /// Stored as 16-byte big-endian
53    UInt128(u128),
54    /// When a number is larger than it can hold
55    AboveMax,
56    /// When a number is smaller than it can hold
57    BelowMin,
58}
59
60impl PartialEq for PrimitiveLiteral {
61    fn eq(&self, other: &Self) -> bool {
62        self.partial_cmp(other) == Some(Ordering::Equal)
63    }
64}
65
66impl PartialOrd for PrimitiveLiteral {
67    fn partial_cmp(&self, other: &Self) -> Option<Ordering> {
68        match (self, other) {
69            (Self::Boolean(a), Self::Boolean(b)) => a.partial_cmp(b),
70            (Self::Int(a), Self::Int(b)) => a.partial_cmp(b),
71            (Self::Long(a), Self::Long(b)) => a.partial_cmp(b),
72            (Self::Float(a), Self::Float(b)) => Some(float_cmp(a, b)),
73            (Self::Double(a), Self::Double(b)) => Some(float_cmp(a, b)),
74            (Self::String(a), Self::String(b)) => a.partial_cmp(b),
75            (Self::Binary(a), Self::Binary(b)) => a.partial_cmp(b),
76            (Self::Int128(a), Self::Int128(b)) => a.partial_cmp(b),
77            (Self::UInt128(a), Self::UInt128(b)) => a.partial_cmp(b),
78            (Self::AboveMax, Self::AboveMax) | (Self::BelowMin, Self::BelowMin) => {
79                Some(Ordering::Equal)
80            }
81            // Different variants order by declaration, as the derived impl did. A variant that
82            // lacks an arm above lands here against itself; fail closed instead of calling the
83            // two values equal.
84            _ => {
85                debug_assert_ne!(std::mem::discriminant(self), std::mem::discriminant(other));
86                match self.variant_index().cmp(&other.variant_index()) {
87                    Ordering::Equal => None,
88                    ordering => Some(ordering),
89                }
90            }
91        }
92    }
93}
94
95/// Compares floats as Java's `Float.compare` and `Double.compare` do. `OrderedFloat` already
96/// treats every NaN as one value above all others; it only lacks `-0.0` before `0.0`.
97fn float_cmp<T: FloatCore>(a: &OrderedFloat<T>, b: &OrderedFloat<T>) -> Ordering {
98    a.cmp(b).then_with(|| {
99        if a.is_nan() {
100            Ordering::Equal
101        } else {
102            b.is_sign_negative().cmp(&a.is_sign_negative())
103        }
104    })
105}
106
107impl PrimitiveLiteral {
108    /// Must follow the declaration order of the variants: `partial_cmp` uses it to order
109    /// different variants the way the derived `PartialOrd` did.
110    fn variant_index(&self) -> u8 {
111        match self {
112            Self::Boolean(_) => 0,
113            Self::Int(_) => 1,
114            Self::Long(_) => 2,
115            Self::Float(_) => 3,
116            Self::Double(_) => 4,
117            Self::String(_) => 5,
118            Self::Binary(_) => 6,
119            Self::Int128(_) => 7,
120            Self::UInt128(_) => 8,
121            Self::AboveMax => 9,
122            Self::BelowMin => 10,
123        }
124    }
125
126    /// Returns true if the Literal represents a primitive type
127    /// that can be a NaN, and that it's value is NaN
128    pub fn is_nan(&self) -> bool {
129        match self {
130            PrimitiveLiteral::Double(val) => val.is_nan(),
131            PrimitiveLiteral::Float(val) => val.is_nan(),
132            _ => false,
133        }
134    }
135}