Skip to main content

polars_core/utils/
mod.rs

1mod any_value;
2pub mod cut;
3use polars_arrow::compute::concatenate::concatenate_validities;
4use polars_arrow::compute::utils::combine_validities_and;
5pub mod flatten;
6pub(crate) mod series;
7mod supertype;
8use std::borrow::Cow;
9use std::ops::{Deref, DerefMut};
10mod schema;
11
12pub use any_value::*;
13use flatten::*;
14use num_traits::{One, Zero};
15pub use polars_arrow;
16use polars_arrow::bitmap::Bitmap;
17pub use polars_arrow::legacy::utils::*;
18pub use polars_arrow::trusted_len::TrustMyLength;
19pub use rayon;
20use rayon::prelude::*;
21pub use schema::*;
22pub use series::*;
23pub use supertype::*;
24
25use crate::prelude::*;
26use crate::runtime::RAYON;
27
28#[repr(transparent)]
29pub struct Wrap<T>(pub T);
30
31impl<T> Deref for Wrap<T> {
32    type Target = T;
33    fn deref(&self) -> &Self::Target {
34        &self.0
35    }
36}
37
38#[inline(always)]
39pub fn _set_partition_size() -> usize {
40    RAYON.current_num_threads()
41}
42
43/// Iterate `items` in parallel with the number of rayon tasks bounded by the thread count.
44///
45/// Use this when the length of `items` grows with the data. A worker adds a stack frame per
46/// stolen job, so an unbounded number of tasks can overflow a worker stack.
47pub fn par_iter_bounded<T: Sync>(items: &[T]) -> rayon::iter::MinLen<rayon::slice::Iter<'_, T>> {
48    const TASKS_PER_THREAD: usize = 8;
49    let min_len = items
50        .len()
51        .div_ceil(_set_partition_size() * TASKS_PER_THREAD);
52    items.par_iter().with_min_len(min_len.max(1))
53}
54
55/// Just a wrapper structure which is useful for certain impl specializations.
56///
57/// This is for instance use to implement
58/// `impl<T> FromIterator<T::Native> for NoNull<ChunkedArray<T>>`
59/// as `Option<T::Native>` was already implemented:
60/// `impl<T> FromIterator<Option<T::Native>> for ChunkedArray<T>`
61pub struct NoNull<T> {
62    inner: T,
63}
64
65impl<T> NoNull<T> {
66    pub fn new(inner: T) -> Self {
67        NoNull { inner }
68    }
69
70    pub fn into_inner(self) -> T {
71        self.inner
72    }
73}
74
75impl<T> Deref for NoNull<T> {
76    type Target = T;
77
78    fn deref(&self) -> &Self::Target {
79        &self.inner
80    }
81}
82
83impl<T> DerefMut for NoNull<T> {
84    fn deref_mut(&mut self) -> &mut Self::Target {
85        &mut self.inner
86    }
87}
88
89pub(crate) fn get_iter_capacity<T, I: Iterator<Item = T>>(iter: &I) -> usize {
90    match iter.size_hint() {
91        (_lower, Some(upper)) => upper,
92        (0, None) => 1024,
93        (lower, None) => lower,
94    }
95}
96
97// prefer this one over split_ca, as this can push the null_count into the thread pool
98// returns an `(offset, length)` tuple
99#[doc(hidden)]
100pub fn _split_offsets(len: usize, n: usize) -> Vec<(usize, usize)> {
101    if n == 1 {
102        vec![(0, len)]
103    } else {
104        let chunk_size = len / n;
105
106        (0..n)
107            .map(|partition| {
108                let offset = partition * chunk_size;
109                let len = if partition == (n - 1) {
110                    len - offset
111                } else {
112                    chunk_size
113                };
114                (partition * chunk_size, len)
115            })
116            .collect_trusted()
117    }
118}
119
120#[allow(clippy::len_without_is_empty)]
121pub trait Container: Clone {
122    fn slice(&self, offset: i64, len: usize) -> Self;
123
124    fn split_at(&self, offset: i64) -> (Self, Self);
125
126    fn len(&self) -> usize;
127
128    fn iter_chunks(&self) -> impl Iterator<Item = Self>;
129
130    fn should_rechunk(&self) -> bool;
131
132    fn n_chunks(&self) -> usize;
133
134    fn chunk_lengths(&self) -> impl Iterator<Item = usize>;
135}
136
137impl Container for DataFrame {
138    fn slice(&self, offset: i64, len: usize) -> Self {
139        DataFrame::slice(self, offset, len)
140    }
141
142    fn split_at(&self, offset: i64) -> (Self, Self) {
143        DataFrame::split_at(self, offset)
144    }
145
146    fn len(&self) -> usize {
147        self.height()
148    }
149
150    fn iter_chunks(&self) -> impl Iterator<Item = Self> {
151        flatten_df_iter(self)
152    }
153
154    fn should_rechunk(&self) -> bool {
155        self.should_rechunk()
156    }
157
158    fn n_chunks(&self) -> usize {
159        DataFrame::first_col_n_chunks(self)
160    }
161
162    fn chunk_lengths(&self) -> impl Iterator<Item = usize> {
163        self.first_col_chunk_lengths()
164    }
165}
166
167impl<T: PolarsDataType> Container for ChunkedArray<T> {
168    fn slice(&self, offset: i64, len: usize) -> Self {
169        ChunkedArray::slice(self, offset, len)
170    }
171
172    fn split_at(&self, offset: i64) -> (Self, Self) {
173        ChunkedArray::split_at(self, offset)
174    }
175
176    fn len(&self) -> usize {
177        ChunkedArray::len(self)
178    }
179
180    fn iter_chunks(&self) -> impl Iterator<Item = Self> {
181        self.downcast_iter()
182            .map(|arr| Self::with_chunk(self.name().clone(), arr.clone()))
183    }
184
185    fn should_rechunk(&self) -> bool {
186        false
187    }
188
189    fn n_chunks(&self) -> usize {
190        self.chunks().len()
191    }
192
193    fn chunk_lengths(&self) -> impl Iterator<Item = usize> {
194        ChunkedArray::chunk_lengths(self)
195    }
196}
197
198impl Container for Series {
199    fn slice(&self, offset: i64, len: usize) -> Self {
200        self.0.slice(offset, len)
201    }
202
203    fn split_at(&self, offset: i64) -> (Self, Self) {
204        self.0.split_at(offset)
205    }
206
207    fn len(&self) -> usize {
208        self.0.len()
209    }
210
211    fn iter_chunks(&self) -> impl Iterator<Item = Self> {
212        (0..self.0.n_chunks()).map(|i| self.select_chunk(i))
213    }
214
215    fn should_rechunk(&self) -> bool {
216        false
217    }
218
219    fn n_chunks(&self) -> usize {
220        self.chunks().len()
221    }
222
223    fn chunk_lengths(&self) -> impl Iterator<Item = usize> {
224        self.0.chunk_lengths()
225    }
226}
227
228fn split_impl<C: Container>(container: &C, target: usize, chunk_size: usize) -> Vec<C> {
229    if target == 1 {
230        return vec![container.clone()];
231    }
232    let mut out = Vec::with_capacity(target);
233    let chunk_size = chunk_size as i64;
234
235    // First split
236    let (chunk, mut remainder) = container.split_at(chunk_size);
237    out.push(chunk);
238
239    // Take the rest of the splits of exactly chunk size, but skip the last remainder as we won't split that.
240    for _ in 1..target - 1 {
241        let (a, b) = remainder.split_at(chunk_size);
242        out.push(a);
243        remainder = b
244    }
245    // This can be slightly larger than `chunk_size`, but is smaller than `2 * chunk_size`.
246    out.push(remainder);
247    out
248}
249
250/// Splits, but doesn't flatten chunks. E.g. a container can still have multiple chunks.
251pub fn split<C: Container>(container: &C, target: usize) -> Vec<C> {
252    let total_len = container.len();
253    if total_len == 0 {
254        return vec![container.clone()];
255    }
256
257    let chunk_size = std::cmp::max(total_len / target, 1);
258
259    if container.n_chunks() == target
260        && container
261            .chunk_lengths()
262            .all(|len| len.abs_diff(chunk_size) < 100)
263        // We cannot get chunks if they are misaligned
264        && !container.should_rechunk()
265    {
266        return container.iter_chunks().collect();
267    }
268    split_impl(container, target, chunk_size)
269}
270
271/// Split a [`Container`] in `target` elements. The target doesn't have to be respected if not
272/// Deviation of the target might be done to create more equal size chunks.
273pub fn split_and_flatten<C: Container>(container: &C, target: usize) -> Vec<C> {
274    let total_len = container.len();
275    if total_len == 0 {
276        return vec![container.clone()];
277    }
278
279    let chunk_size = std::cmp::max(total_len / target, 1);
280
281    if container.n_chunks() == target
282        && container
283            .chunk_lengths()
284            .all(|len| len.abs_diff(chunk_size) < 100)
285        // We cannot get chunks if they are misaligned
286        && !container.should_rechunk()
287    {
288        return container.iter_chunks().collect();
289    }
290
291    if container.n_chunks() == 1 {
292        split_impl(container, target, chunk_size)
293    } else {
294        let mut out = Vec::with_capacity(target);
295        let chunks = container.iter_chunks();
296
297        'new_chunk: for mut chunk in chunks {
298            loop {
299                let h = chunk.len();
300                if h < chunk_size {
301                    // TODO if the chunk is much smaller than chunk size, we should try to merge it with the next one.
302                    out.push(chunk);
303                    continue 'new_chunk;
304                }
305
306                // If a split leads to the next chunk being smaller than 30% take the whole chunk
307                if ((h - chunk_size) as f64 / chunk_size as f64) < 0.3 {
308                    out.push(chunk);
309                    continue 'new_chunk;
310                }
311
312                let (a, b) = chunk.split_at(chunk_size as i64);
313                out.push(a);
314                chunk = b;
315            }
316        }
317        out
318    }
319}
320
321/// Split a [`DataFrame`] in `target` elements. The target doesn't have to be respected if not
322/// strict. Deviation of the target might be done to create more equal size chunks.
323///
324/// # Panics
325/// if chunks are not aligned
326pub fn split_df_as_ref(df: &DataFrame, target: usize, strict: bool) -> Vec<DataFrame> {
327    if strict {
328        split(df, target)
329    } else {
330        split_and_flatten(df, target)
331    }
332}
333
334#[doc(hidden)]
335/// Split a [`DataFrame`] into `n` parts. We take a `&mut` to be able to repartition/align chunks.
336/// `strict` in that it respects `n` even if the chunks are suboptimal.
337pub fn split_df(df: &mut DataFrame, target: usize, strict: bool) -> Vec<DataFrame> {
338    if target == 0 || df.height() == 0 {
339        return vec![df.clone()];
340    }
341    // make sure that chunks are aligned.
342    df.align_chunks_par();
343    split_df_as_ref(df, target, strict)
344}
345
346pub fn slice_slice<T>(vals: &[T], offset: i64, len: usize) -> &[T] {
347    let (raw_offset, slice_len) = slice_offsets(offset, len, vals.len());
348    &vals[raw_offset..raw_offset + slice_len]
349}
350
351#[inline]
352pub fn slice_offsets(offset: i64, length: usize, array_len: usize) -> (usize, usize) {
353    let signed_start_offset = if offset < 0 {
354        offset.saturating_add_unsigned(array_len as u64)
355    } else {
356        offset
357    };
358    let signed_stop_offset = signed_start_offset.saturating_add_unsigned(length as u64);
359
360    let signed_array_len: i64 = array_len
361        .try_into()
362        .expect("array length larger than i64::MAX");
363    let clamped_start_offset = signed_start_offset.clamp(0, signed_array_len);
364    let clamped_stop_offset = signed_stop_offset.clamp(0, signed_array_len);
365
366    let slice_start_idx = clamped_start_offset as usize;
367    let slice_len = (clamped_stop_offset - clamped_start_offset) as usize;
368    (slice_start_idx, slice_len)
369}
370
371/// Apply a macro on the Series
372#[macro_export]
373macro_rules! match_dtype_to_physical_apply_macro {
374    ($obj:expr, $macro:ident, $macro_string:ident, $macro_bool:ident $(, $opt_args:expr)*) => {{
375        match $obj {
376            DataType::String => $macro_string!($($opt_args)*),
377            DataType::Boolean => $macro_bool!($($opt_args)*),
378            #[cfg(feature = "dtype-u8")]
379            DataType::UInt8 => $macro!(u8 $(, $opt_args)*),
380            #[cfg(feature = "dtype-u16")]
381            DataType::UInt16 => $macro!(u16 $(, $opt_args)*),
382            DataType::UInt32 => $macro!(u32 $(, $opt_args)*),
383            DataType::UInt64 => $macro!(u64 $(, $opt_args)*),
384            #[cfg(feature = "dtype-i8")]
385            DataType::Int8 => $macro!(i8 $(, $opt_args)*),
386            #[cfg(feature = "dtype-i16")]
387            DataType::Int16 => $macro!(i16 $(, $opt_args)*),
388            DataType::Int32 => $macro!(i32 $(, $opt_args)*),
389            DataType::Int64 => $macro!(i64 $(, $opt_args)*),
390            #[cfg(feature = "dtype-i128")]
391            DataType::Int128 => $macro!(i128 $(, $opt_args)*),
392            #[cfg(feature = "dtype-f16")]
393            DataType::Float16 => $macro!(pf16 $(, $opt_args)*),
394            DataType::Float32 => $macro!(f32 $(, $opt_args)*),
395            DataType::Float64 => $macro!(f64 $(, $opt_args)*),
396            dt => panic!("not implemented for dtype {:?}", dt),
397        }
398    }};
399}
400
401/// Apply a macro on the Series
402#[macro_export]
403macro_rules! match_dtype_to_logical_apply_macro {
404    ($obj:expr, $macro:ident, $macro_string:ident, $macro_binary:ident, $macro_bool:ident $(, $opt_args:expr)*) => {{
405        match $obj {
406            DataType::String => $macro_string!($($opt_args)*),
407            DataType::Binary => $macro_binary!($($opt_args)*),
408            DataType::Boolean => $macro_bool!($($opt_args)*),
409            #[cfg(feature = "dtype-u8")]
410            DataType::UInt8 => $macro!(UInt8Type $(, $opt_args)*),
411            #[cfg(feature = "dtype-u16")]
412            DataType::UInt16 => $macro!(UInt16Type $(, $opt_args)*),
413            DataType::UInt32 => $macro!(UInt32Type $(, $opt_args)*),
414            DataType::UInt64 => $macro!(UInt64Type $(, $opt_args)*),
415            #[cfg(feature = "dtype-u128")]
416            DataType::UInt128 => $macro!(UInt128Type $(, $opt_args)*),
417            #[cfg(feature = "dtype-i8")]
418            DataType::Int8 => $macro!(Int8Type $(, $opt_args)*),
419            #[cfg(feature = "dtype-i16")]
420            DataType::Int16 => $macro!(Int16Type $(, $opt_args)*),
421            DataType::Int32 => $macro!(Int32Type $(, $opt_args)*),
422            DataType::Int64 => $macro!(Int64Type $(, $opt_args)*),
423            #[cfg(feature = "dtype-i128")]
424            DataType::Int128 => $macro!(Int128Type $(, $opt_args)*),
425            #[cfg(feature = "dtype-f16")]
426            DataType::Float16 => $macro!(Float16Type $(, $opt_args)*),
427            DataType::Float32 => $macro!(Float32Type $(, $opt_args)*),
428            DataType::Float64 => $macro!(Float64Type $(, $opt_args)*),
429            dt => panic!("not implemented for dtype {:?}", dt),
430        }
431    }};
432}
433
434/// Apply a macro on the Downcasted ChunkedArrays
435#[macro_export]
436macro_rules! match_arrow_dtype_apply_macro_ca {
437    ($self:expr, $macro:ident, $macro_string:ident, $macro_bool:ident $(, $opt_args:expr)*) => {{
438        match $self.dtype() {
439            DataType::String => $macro_string!($self.str().unwrap() $(, $opt_args)*),
440            DataType::Boolean => $macro_bool!($self.bool().unwrap() $(, $opt_args)*),
441            #[cfg(feature = "dtype-u8")]
442            DataType::UInt8 => $macro!($self.u8().unwrap() $(, $opt_args)*),
443            #[cfg(feature = "dtype-u16")]
444            DataType::UInt16 => $macro!($self.u16().unwrap() $(, $opt_args)*),
445            DataType::UInt32 => $macro!($self.u32().unwrap() $(, $opt_args)*),
446            DataType::UInt64 => $macro!($self.u64().unwrap() $(, $opt_args)*),
447            #[cfg(feature = "dtype-u128")]
448            DataType::UInt128 => $macro!($self.u128().unwrap() $(, $opt_args)*),
449            #[cfg(feature = "dtype-i8")]
450            DataType::Int8 => $macro!($self.i8().unwrap() $(, $opt_args)*),
451            #[cfg(feature = "dtype-i16")]
452            DataType::Int16 => $macro!($self.i16().unwrap() $(, $opt_args)*),
453            DataType::Int32 => $macro!($self.i32().unwrap() $(, $opt_args)*),
454            DataType::Int64 => $macro!($self.i64().unwrap() $(, $opt_args)*),
455            #[cfg(feature = "dtype-i128")]
456            DataType::Int128 => $macro!($self.i128().unwrap() $(, $opt_args)*),
457            #[cfg(feature = "dtype-f16")]
458            DataType::Float16 => $macro!($self.f16().unwrap() $(, $opt_args)*),
459            DataType::Float32 => $macro!($self.f32().unwrap() $(, $opt_args)*),
460            DataType::Float64 => $macro!($self.f64().unwrap() $(, $opt_args)*),
461            dt => panic!("not implemented for dtype {:?}", dt),
462        }
463    }};
464}
465
466#[macro_export]
467macro_rules! with_match_physical_numeric_type {(
468    $dtype:expr, | $_:tt $T:ident | $($body:tt)*
469) => ({
470    macro_rules! __with_ty__ {( $_ $T:ident ) => ( $($body)* )}
471    #[cfg(feature = "dtype-f16")]
472    use polars_utils::float16::pf16;
473    use $crate::datatypes::DataType::*;
474    match $dtype {
475        #[cfg(feature = "dtype-i8")]
476        Int8 => __with_ty__! { i8 },
477        #[cfg(feature = "dtype-i16")]
478        Int16 => __with_ty__! { i16 },
479        Int32 => __with_ty__! { i32 },
480        Int64 => __with_ty__! { i64 },
481        #[cfg(feature = "dtype-i128")]
482        Int128 => __with_ty__! { i128 },
483        #[cfg(feature = "dtype-u8")]
484        UInt8 => __with_ty__! { u8 },
485        #[cfg(feature = "dtype-u16")]
486        UInt16 => __with_ty__! { u16 },
487        UInt32 => __with_ty__! { u32 },
488        UInt64 => __with_ty__! { u64 },
489        #[cfg(feature = "dtype-u128")]
490        UInt128 => __with_ty__! { u128 },
491        #[cfg(feature = "dtype-f16")]
492        Float16 => __with_ty__! { pf16 },
493        Float32 => __with_ty__! { f32 },
494        Float64 => __with_ty__! { f64 },
495        dt => panic!("not implemented for dtype {:?}", dt),
496    }
497})}
498
499#[macro_export]
500macro_rules! with_match_physical_integer_type {(
501    $dtype:expr, | $_:tt $T:ident | $($body:tt)*
502) => ({
503    macro_rules! __with_ty__ {( $_ $T:ident ) => ( $($body)* )}
504    #[cfg(feature = "dtype-f16")]
505    use polars_utils::float16::pf16;
506    use $crate::datatypes::DataType::*;
507    match $dtype {
508        #[cfg(feature = "dtype-i8")]
509        Int8 => __with_ty__! { i8 },
510        #[cfg(feature = "dtype-i16")]
511        Int16 => __with_ty__! { i16 },
512        Int32 => __with_ty__! { i32 },
513        Int64 => __with_ty__! { i64 },
514        #[cfg(feature = "dtype-i128")]
515        Int128 => __with_ty__! { i128 },
516        #[cfg(feature = "dtype-u8")]
517        UInt8 => __with_ty__! { u8 },
518        #[cfg(feature = "dtype-u16")]
519        UInt16 => __with_ty__! { u16 },
520        UInt32 => __with_ty__! { u32 },
521        UInt64 => __with_ty__! { u64 },
522        #[cfg(feature = "dtype-u128")]
523        UInt128 => __with_ty__! { u128 },
524        dt => panic!("not implemented for dtype {:?}", dt),
525    }
526})}
527
528#[macro_export]
529macro_rules! with_match_physical_float_type {(
530    $dtype:expr, | $_:tt $T:ident | $($body:tt)*
531) => ({
532    macro_rules! __with_ty__ {( $_ $T:ident ) => ( $($body)* )}
533    use polars_utils::float16::pf16;
534    use $crate::datatypes::DataType::*;
535    match $dtype {
536        #[cfg(feature = "dtype-f16")]
537        Float16 => __with_ty__! { pf16 },
538        Float32 => __with_ty__! { f32 },
539        Float64 => __with_ty__! { f64 },
540        dt => panic!("not implemented for dtype {:?}", dt),
541    }
542})}
543
544#[macro_export]
545macro_rules! with_match_physical_float_polars_type {(
546    $key_type:expr, | $_:tt $T:ident | $($body:tt)*
547) => ({
548    macro_rules! __with_ty__ {( $_ $T:ident ) => ( $($body)* )}
549    use $crate::datatypes::DataType::*;
550    match $key_type {
551        #[cfg(feature = "dtype-f16")]
552        Float16 => __with_ty__! { Float16Type },
553        Float32 => __with_ty__! { Float32Type },
554        Float64 => __with_ty__! { Float64Type },
555        dt => panic!("not implemented for dtype {:?}", dt),
556    }
557})}
558
559#[macro_export]
560macro_rules! with_match_physical_numeric_polars_type {(
561    $key_type:expr, | $_:tt $T:ident | $($body:tt)*
562) => ({
563    macro_rules! __with_ty__ {( $_ $T:ident ) => ( $($body)* )}
564    use $crate::datatypes::DataType::*;
565    match $key_type {
566            #[cfg(feature = "dtype-i8")]
567        Int8 => __with_ty__! { Int8Type },
568            #[cfg(feature = "dtype-i16")]
569        Int16 => __with_ty__! { Int16Type },
570        Int32 => __with_ty__! { Int32Type },
571        Int64 => __with_ty__! { Int64Type },
572            #[cfg(feature = "dtype-i128")]
573        Int128 => __with_ty__! { Int128Type },
574            #[cfg(feature = "dtype-u8")]
575        UInt8 => __with_ty__! { UInt8Type },
576            #[cfg(feature = "dtype-u16")]
577        UInt16 => __with_ty__! { UInt16Type },
578        UInt32 => __with_ty__! { UInt32Type },
579        UInt64 => __with_ty__! { UInt64Type },
580            #[cfg(feature = "dtype-u128")]
581        UInt128 => __with_ty__! { UInt128Type },
582            #[cfg(feature = "dtype-f16")]
583        Float16 => __with_ty__! { Float16Type },
584        Float32 => __with_ty__! { Float32Type },
585        Float64 => __with_ty__! { Float64Type },
586        dt => panic!("not implemented for dtype {:?}", dt),
587    }
588})}
589
590#[macro_export]
591macro_rules! with_match_physical_integer_polars_type {(
592    $key_type:expr, | $_:tt $T:ident | $($body:tt)*
593) => ({
594    macro_rules! __with_ty__ {( $_ $T:ident ) => ( $($body)* )}
595    use $crate::datatypes::DataType::*;
596    use $crate::datatypes::*;
597    match $key_type {
598        #[cfg(feature = "dtype-i8")]
599        Int8 => __with_ty__! { Int8Type },
600        #[cfg(feature = "dtype-i16")]
601        Int16 => __with_ty__! { Int16Type },
602        Int32 => __with_ty__! { Int32Type },
603        Int64 => __with_ty__! { Int64Type },
604        #[cfg(feature = "dtype-i128")]
605        Int128 => __with_ty__! { Int128Type },
606        #[cfg(feature = "dtype-u8")]
607        UInt8 => __with_ty__! { UInt8Type },
608        #[cfg(feature = "dtype-u16")]
609        UInt16 => __with_ty__! { UInt16Type },
610        UInt32 => __with_ty__! { UInt32Type },
611        UInt64 => __with_ty__! { UInt64Type },
612        #[cfg(feature = "dtype-u128")]
613        UInt128 => __with_ty__! { UInt128Type },
614        dt => panic!("not implemented for dtype {:?}", dt),
615    }
616})}
617
618#[macro_export]
619macro_rules! with_match_categorical_physical_type {(
620    $dtype:expr, | $_:tt $T:ident | $($body:tt)*
621) => ({
622    macro_rules! __with_ty__ {( $_ $T:ident ) => ( $($body)* )}
623    match $dtype {
624        CategoricalPhysical::U8 => __with_ty__! { Categorical8Type },
625        CategoricalPhysical::U16 => __with_ty__! { Categorical16Type },
626        CategoricalPhysical::U32 => __with_ty__! { Categorical32Type },
627    }
628})}
629
630/// Apply a macro on the Downcasted ChunkedArrays of DataTypes that are logical numerics.
631/// So no logical.
632#[macro_export]
633macro_rules! downcast_as_macro_arg_physical {
634    ($self:expr, $macro:ident $(, $opt_args:expr)*) => {{
635        match $self.dtype() {
636            #[cfg(feature = "dtype-u8")]
637            DataType::UInt8 => $macro!($self.u8().unwrap() $(, $opt_args)*),
638            #[cfg(feature = "dtype-u16")]
639            DataType::UInt16 => $macro!($self.u16().unwrap() $(, $opt_args)*),
640            DataType::UInt32 => $macro!($self.u32().unwrap() $(, $opt_args)*),
641            DataType::UInt64 => $macro!($self.u64().unwrap() $(, $opt_args)*),
642            #[cfg(feature = "dtype-u128")]
643            DataType::UInt128 => $macro!($self.u128().unwrap() $(, $opt_args)*),
644            #[cfg(feature = "dtype-i8")]
645            DataType::Int8 => $macro!($self.i8().unwrap() $(, $opt_args)*),
646            #[cfg(feature = "dtype-i16")]
647            DataType::Int16 => $macro!($self.i16().unwrap() $(, $opt_args)*),
648            DataType::Int32 => $macro!($self.i32().unwrap() $(, $opt_args)*),
649            DataType::Int64 => $macro!($self.i64().unwrap() $(, $opt_args)*),
650            #[cfg(feature = "dtype-i128")]
651            DataType::Int128 => $macro!($self.i128().unwrap() $(, $opt_args)*),
652            #[cfg(feature = "dtype-f16")]
653            DataType::Float16 => $macro!($self.f16().unwrap() $(, $opt_args)*),
654            DataType::Float32 => $macro!($self.f32().unwrap() $(, $opt_args)*),
655            DataType::Float64 => $macro!($self.f64().unwrap() $(, $opt_args)*),
656            dt => panic!("not implemented for {:?}", dt),
657        }
658    }};
659}
660
661/// Apply a macro on the Downcasted ChunkedArrays of DataTypes that are logical numerics.
662/// So no logical.
663#[macro_export]
664macro_rules! downcast_as_macro_arg_physical_mut {
665    ($self:expr, $macro:ident $(, $opt_args:expr)*) => {{
666        // clone so that we do not borrow
667        match $self.dtype().clone() {
668            #[cfg(feature = "dtype-u8")]
669            DataType::UInt8 => {
670                let ca: &mut UInt8Chunked = $self.as_mut();
671                $macro!(UInt8Type, ca $(, $opt_args)*)
672            },
673            #[cfg(feature = "dtype-u16")]
674            DataType::UInt16 => {
675                let ca: &mut UInt16Chunked = $self.as_mut();
676                $macro!(UInt16Type, ca $(, $opt_args)*)
677            },
678            DataType::UInt32 => {
679                let ca: &mut UInt32Chunked = $self.as_mut();
680                $macro!(UInt32Type, ca $(, $opt_args)*)
681            },
682            DataType::UInt64 => {
683                let ca: &mut UInt64Chunked = $self.as_mut();
684                $macro!(UInt64Type, ca $(, $opt_args)*)
685            },
686            #[cfg(feature = "dtype-u128")]
687            DataType::UInt128 => {
688                let ca: &mut UInt128Chunked = $self.as_mut();
689                $macro!(UInt128Type, ca $(, $opt_args)*)
690            },
691            #[cfg(feature = "dtype-i8")]
692            DataType::Int8 => {
693                let ca: &mut Int8Chunked = $self.as_mut();
694                $macro!(Int8Type, ca $(, $opt_args)*)
695            },
696            #[cfg(feature = "dtype-i16")]
697            DataType::Int16 => {
698                let ca: &mut Int16Chunked = $self.as_mut();
699                $macro!(Int16Type, ca $(, $opt_args)*)
700            },
701            DataType::Int32 => {
702                let ca: &mut Int32Chunked = $self.as_mut();
703                $macro!(Int32Type, ca $(, $opt_args)*)
704            },
705            DataType::Int64 => {
706                let ca: &mut Int64Chunked = $self.as_mut();
707                $macro!(Int64Type, ca $(, $opt_args)*)
708            },
709            #[cfg(feature = "dtype-i128")]
710            DataType::Int128 => {
711                let ca: &mut Int128Chunked = $self.as_mut();
712                $macro!(Int128Type, ca $(, $opt_args)*)
713            },
714            #[cfg(feature = "dtype-f16")]
715            DataType::Float16 => {
716                let ca: &mut Float16Chunked = $self.as_mut();
717                $macro!(Float16Type, ca $(, $opt_args)*)
718            },
719            DataType::Float32 => {
720                let ca: &mut Float32Chunked = $self.as_mut();
721                $macro!(Float32Type, ca $(, $opt_args)*)
722            },
723            DataType::Float64 => {
724                let ca: &mut Float64Chunked = $self.as_mut();
725                $macro!(Float64Type, ca $(, $opt_args)*)
726            },
727            dt => panic!("not implemented for {:?}", dt),
728        }
729    }};
730}
731
732#[macro_export]
733macro_rules! apply_method_all_arrow_series {
734    ($self:expr, $method:ident, $($args:expr),*) => {
735        match $self.dtype() {
736            DataType::Boolean => $self.bool().unwrap().$method($($args),*),
737            DataType::String => $self.str().unwrap().$method($($args),*),
738            #[cfg(feature = "dtype-u8")]
739            DataType::UInt8 => $self.u8().unwrap().$method($($args),*),
740            #[cfg(feature = "dtype-u16")]
741            DataType::UInt16 => $self.u16().unwrap().$method($($args),*),
742            DataType::UInt32 => $self.u32().unwrap().$method($($args),*),
743            DataType::UInt64 => $self.u64().unwrap().$method($($args),*),
744            #[cfg(feature = "dtype-u128")]
745            DataType::UInt128 => $self.u128().unwrap().$medthod($($args),*),
746            #[cfg(feature = "dtype-i8")]
747            DataType::Int8 => $self.i8().unwrap().$method($($args),*),
748            #[cfg(feature = "dtype-i16")]
749            DataType::Int16 => $self.i16().unwrap().$method($($args),*),
750            DataType::Int32 => $self.i32().unwrap().$method($($args),*),
751            DataType::Int64 => $self.i64().unwrap().$method($($args),*),
752            #[cfg(feature = "dtype-i128")]
753            DataType::Int128 => $self.i128().unwrap().$method($($args),*),
754            #[cfg(feature = "dtype-f16")]
755            DataType::Float16 => $self.f16().unwrap().$method($($args),*),
756            DataType::Float32 => $self.f32().unwrap().$method($($args),*),
757            DataType::Float64 => $self.f64().unwrap().$method($($args),*),
758            DataType::Time => $self.time().unwrap().$method($($args),*),
759            DataType::Date => $self.date().unwrap().$method($($args),*),
760            DataType::Datetime(_, _) => $self.datetime().unwrap().$method($($args),*),
761            DataType::List(_) => $self.list().unwrap().$method($($args),*),
762            DataType::Struct(_) => $self.struct_().unwrap().$method($($args),*),
763            dt => panic!("dtype {:?} not supported", dt)
764        }
765    }
766}
767
768#[macro_export]
769macro_rules! apply_method_physical_integer {
770    ($self:expr, $method:ident, $($args:expr),*) => {
771        match $self.dtype() {
772            #[cfg(feature = "dtype-u8")]
773            DataType::UInt8 => $self.u8().unwrap().$method($($args),*),
774            #[cfg(feature = "dtype-u16")]
775            DataType::UInt16 => $self.u16().unwrap().$method($($args),*),
776            DataType::UInt32 => $self.u32().unwrap().$method($($args),*),
777            DataType::UInt64 => $self.u64().unwrap().$method($($args),*),
778            #[cfg(feature = "dtype-u128")]
779            DataType::UInt128 => $self.u128().unwrap().$method($($args),*),
780            #[cfg(feature = "dtype-i8")]
781            DataType::Int8 => $self.i8().unwrap().$method($($args),*),
782            #[cfg(feature = "dtype-i16")]
783            DataType::Int16 => $self.i16().unwrap().$method($($args),*),
784            DataType::Int32 => $self.i32().unwrap().$method($($args),*),
785            DataType::Int64 => $self.i64().unwrap().$method($($args),*),
786            #[cfg(feature = "dtype-i128")]
787            DataType::Int128 => $self.i128().unwrap().$method($($args),*),
788            dt => panic!("not implemented for dtype {:?}", dt),
789        }
790    }
791}
792
793// doesn't include Bool and String
794#[macro_export]
795macro_rules! apply_method_physical_numeric {
796    ($self:expr, $method:ident, $($args:expr),*) => {
797        match $self.dtype() {
798            #[cfg(feature = "dtype-f16")]
799            DataType::Float16 => $self.f16().unwrap().$method($($args),*),
800            DataType::Float32 => $self.f32().unwrap().$method($($args),*),
801            DataType::Float64 => $self.f64().unwrap().$method($($args),*),
802            _ => apply_method_physical_integer!($self, $method, $($args),*),
803        }
804    }
805}
806
807#[macro_export]
808macro_rules! df {
809    ($($col_name:expr => $slice:expr), + $(,)?) => {
810        $crate::prelude::DataFrame::new_infer_height(vec![
811            $($crate::prelude::Column::from(<$crate::prelude::Series as $crate::prelude::NamedFrom::<_, _>>::new($col_name.into(), $slice)),)+
812        ])
813    }
814}
815
816pub fn get_time_units(tu_l: &TimeUnit, tu_r: &TimeUnit) -> TimeUnit {
817    use crate::datatypes::time_unit::TimeUnit::*;
818    match (tu_l, tu_r) {
819        (Nanoseconds, Microseconds) => Microseconds,
820        (_, Milliseconds) => Milliseconds,
821        _ => *tu_l,
822    }
823}
824
825#[cold]
826#[inline(never)]
827fn width_mismatch(df1: &DataFrame, df2: &DataFrame) -> PolarsError {
828    let mut df1_extra = Vec::new();
829    let mut df2_extra = Vec::new();
830
831    let s1 = df1.schema();
832    let s2 = df2.schema();
833
834    s1.field_compare(s2, &mut df1_extra, &mut df2_extra);
835
836    let df1_extra = df1_extra
837        .into_iter()
838        .map(|(_, (n, _))| n.as_str())
839        .collect::<Vec<_>>()
840        .join(", ");
841    let df2_extra = df2_extra
842        .into_iter()
843        .map(|(_, (n, _))| n.as_str())
844        .collect::<Vec<_>>()
845        .join(", ");
846
847    polars_err!(
848        SchemaMismatch: r#"unable to vstack, dataframes have different widths ({} != {}).
849One dataframe has additional columns: [{df1_extra}].
850Other dataframe has additional columns: [{df2_extra}]."#,
851        df1.width(),
852        df2.width(),
853    )
854}
855
856/// This takes ownership of the DataFrame so that drop is called earlier.
857/// Does not check if schema is correct
858pub fn accumulate_dataframes_vertical_unchecked<I>(dfs: I) -> DataFrame
859where
860    I: IntoIterator<Item = DataFrame>,
861{
862    let mut iter = dfs.into_iter();
863    let additional = iter.size_hint().0;
864    let mut acc_df = iter.next().unwrap();
865    acc_df.reserve_chunks(additional);
866
867    for df in iter {
868        if acc_df.width() != df.width() {
869            panic!("{}", width_mismatch(&acc_df, &df));
870        }
871
872        acc_df.vstack_mut_owned_unchecked(df);
873    }
874    acc_df
875}
876
877/// This takes ownership of the DataFrame so that drop is called earlier.
878/// # Panics
879/// Panics if `dfs` is empty.
880pub fn accumulate_dataframes_vertical<I>(dfs: I) -> PolarsResult<DataFrame>
881where
882    I: IntoIterator<Item = DataFrame>,
883{
884    let mut iter = dfs.into_iter();
885    let additional = iter.size_hint().0;
886    let mut acc_df = iter.next().unwrap();
887    acc_df.reserve_chunks(additional);
888    for df in iter {
889        if acc_df.width() != df.width() {
890            return Err(width_mismatch(&acc_df, &df));
891        }
892
893        acc_df.vstack_mut_owned(df)?;
894    }
895
896    Ok(acc_df)
897}
898
899/// Concat the DataFrames to a single DataFrame.
900pub fn concat_df<'a, I>(dfs: I) -> PolarsResult<DataFrame>
901where
902    I: IntoIterator<Item = &'a DataFrame>,
903{
904    let mut iter = dfs.into_iter();
905    let additional = iter.size_hint().0;
906    let mut acc_df = iter.next().unwrap().clone();
907    acc_df.reserve_chunks(additional);
908    for df in iter {
909        acc_df.vstack_mut(df)?;
910    }
911    Ok(acc_df)
912}
913
914/// Concat the DataFrames to a single DataFrame.
915pub fn concat_df_unchecked<'a, I>(dfs: I) -> DataFrame
916where
917    I: IntoIterator<Item = &'a DataFrame>,
918{
919    let mut iter = dfs.into_iter();
920    let additional = iter.size_hint().0;
921    let mut acc_df = iter.next().unwrap().clone();
922    acc_df.reserve_chunks(additional);
923    for df in iter {
924        acc_df.vstack_mut_unchecked(df);
925    }
926    acc_df
927}
928
929pub fn accumulate_dataframes_horizontal(dfs: Vec<DataFrame>) -> PolarsResult<DataFrame> {
930    let mut iter = dfs.into_iter();
931    let mut acc_df = iter.next().unwrap();
932    for df in iter {
933        acc_df.hstack_mut(df.columns())?;
934    }
935    Ok(acc_df)
936}
937
938/// Ensure the chunks in both ChunkedArrays have the same length.
939/// # Panics
940/// This will panic if `left.len() != right.len()` and array is chunked.
941pub fn align_chunks_binary<'a, T, B>(
942    left: &'a ChunkedArray<T>,
943    right: &'a ChunkedArray<B>,
944) -> (Cow<'a, ChunkedArray<T>>, Cow<'a, ChunkedArray<B>>)
945where
946    B: PolarsDataType,
947    T: PolarsDataType,
948{
949    let assert = || {
950        assert_eq!(
951            left.len(),
952            right.len(),
953            "expected arrays of the same length"
954        )
955    };
956    match (left.chunks.len(), right.chunks.len()) {
957        // All chunks are equal length
958        (1, 1) => (Cow::Borrowed(left), Cow::Borrowed(right)),
959        // All chunks are equal length
960        (a, b)
961            if a == b
962                && left
963                    .chunk_lengths()
964                    .zip(right.chunk_lengths())
965                    .all(|(l, r)| l == r) =>
966        {
967            (Cow::Borrowed(left), Cow::Borrowed(right))
968        },
969        (_, 1) => {
970            assert();
971            (
972                Cow::Borrowed(left),
973                Cow::Owned(right.match_chunks(left.chunk_lengths())),
974            )
975        },
976        (1, _) => {
977            assert();
978            (
979                Cow::Owned(left.match_chunks(right.chunk_lengths())),
980                Cow::Borrowed(right),
981            )
982        },
983        (_, _) => {
984            assert();
985            // could optimize to choose to rechunk a primitive and not a string or list type
986            let left = left.rechunk();
987            (
988                Cow::Owned(left.match_chunks(right.chunk_lengths())),
989                Cow::Borrowed(right),
990            )
991        },
992    }
993}
994
995/// Ensure the chunks in ChunkedArray and Series have the same length.
996/// # Panics
997/// This will panic if `left.len() != right.len()` and array is chunked.
998pub fn align_chunks_binary_ca_series<'a, T>(
999    left: &'a ChunkedArray<T>,
1000    right: &'a Series,
1001) -> (Cow<'a, ChunkedArray<T>>, Cow<'a, Series>)
1002where
1003    T: PolarsDataType,
1004{
1005    let assert = || {
1006        assert_eq!(
1007            left.len(),
1008            right.len(),
1009            "expected arrays of the same length"
1010        )
1011    };
1012    match (left.chunks.len(), right.chunks().len()) {
1013        // All chunks are equal length
1014        (1, 1) => (Cow::Borrowed(left), Cow::Borrowed(right)),
1015        // All chunks are equal length
1016        (a, b)
1017            if a == b
1018                && left
1019                    .chunk_lengths()
1020                    .zip(right.chunk_lengths())
1021                    .all(|(l, r)| l == r) =>
1022        {
1023            assert();
1024            (Cow::Borrowed(left), Cow::Borrowed(right))
1025        },
1026        (_, 1) => (left.rechunk(), Cow::Borrowed(right)),
1027        (1, _) => (Cow::Borrowed(left), Cow::Owned(right.rechunk())),
1028        (_, _) => {
1029            assert();
1030            (left.rechunk(), Cow::Owned(right.rechunk()))
1031        },
1032    }
1033}
1034
1035#[cfg(feature = "performant")]
1036pub(crate) fn align_chunks_binary_owned_series(left: Series, right: Series) -> (Series, Series) {
1037    match (left.chunks().len(), right.chunks().len()) {
1038        (1, 1) => (left, right),
1039        // All chunks are equal length
1040        (a, b)
1041            if a == b
1042                && left
1043                    .chunk_lengths()
1044                    .zip(right.chunk_lengths())
1045                    .all(|(l, r)| l == r) =>
1046        {
1047            (left, right)
1048        },
1049        (_, 1) => (left.rechunk(), right),
1050        (1, _) => (left, right.rechunk()),
1051        (_, _) => (left.rechunk(), right.rechunk()),
1052    }
1053}
1054
1055pub(crate) fn align_chunks_binary_owned<T, B>(
1056    left: ChunkedArray<T>,
1057    right: ChunkedArray<B>,
1058) -> (ChunkedArray<T>, ChunkedArray<B>)
1059where
1060    B: PolarsDataType,
1061    T: PolarsDataType,
1062{
1063    match (left.chunks.len(), right.chunks.len()) {
1064        (1, 1) => (left, right),
1065        // All chunks are equal length
1066        (a, b)
1067            if a == b
1068                && left
1069                    .chunk_lengths()
1070                    .zip(right.chunk_lengths())
1071                    .all(|(l, r)| l == r) =>
1072        {
1073            (left, right)
1074        },
1075        (_, 1) => (left.rechunk().into_owned(), right),
1076        (1, _) => (left, right.rechunk().into_owned()),
1077        (_, _) => (left.rechunk().into_owned(), right.rechunk().into_owned()),
1078    }
1079}
1080
1081/// # Panics
1082/// This will panic if `a.len() != b.len() || b.len() != c.len()` and array is chunked.
1083#[allow(clippy::type_complexity)]
1084pub fn align_chunks_ternary<'a, A, B, C>(
1085    a: &'a ChunkedArray<A>,
1086    b: &'a ChunkedArray<B>,
1087    c: &'a ChunkedArray<C>,
1088) -> (
1089    Cow<'a, ChunkedArray<A>>,
1090    Cow<'a, ChunkedArray<B>>,
1091    Cow<'a, ChunkedArray<C>>,
1092)
1093where
1094    A: PolarsDataType,
1095    B: PolarsDataType,
1096    C: PolarsDataType,
1097{
1098    if a.chunks.len() == 1 && b.chunks.len() == 1 && c.chunks.len() == 1 {
1099        return (Cow::Borrowed(a), Cow::Borrowed(b), Cow::Borrowed(c));
1100    }
1101
1102    assert!(
1103        a.len() == b.len() && b.len() == c.len(),
1104        "expected arrays of the same length"
1105    );
1106
1107    match (a.chunks.len(), b.chunks.len(), c.chunks.len()) {
1108        (_, 1, 1) => (
1109            Cow::Borrowed(a),
1110            Cow::Owned(b.match_chunks(a.chunk_lengths())),
1111            Cow::Owned(c.match_chunks(a.chunk_lengths())),
1112        ),
1113        (1, 1, _) => (
1114            Cow::Owned(a.match_chunks(c.chunk_lengths())),
1115            Cow::Owned(b.match_chunks(c.chunk_lengths())),
1116            Cow::Borrowed(c),
1117        ),
1118        (1, _, 1) => (
1119            Cow::Owned(a.match_chunks(b.chunk_lengths())),
1120            Cow::Borrowed(b),
1121            Cow::Owned(c.match_chunks(b.chunk_lengths())),
1122        ),
1123        (1, _, _) => {
1124            let b = b.rechunk();
1125            (
1126                Cow::Owned(a.match_chunks(c.chunk_lengths())),
1127                Cow::Owned(b.match_chunks(c.chunk_lengths())),
1128                Cow::Borrowed(c),
1129            )
1130        },
1131        (_, 1, _) => {
1132            let a = a.rechunk();
1133            (
1134                Cow::Owned(a.match_chunks(c.chunk_lengths())),
1135                Cow::Owned(b.match_chunks(c.chunk_lengths())),
1136                Cow::Borrowed(c),
1137            )
1138        },
1139        (_, _, 1) => {
1140            let b = b.rechunk();
1141            (
1142                Cow::Borrowed(a),
1143                Cow::Owned(b.match_chunks(a.chunk_lengths())),
1144                Cow::Owned(c.match_chunks(a.chunk_lengths())),
1145            )
1146        },
1147        (len_a, len_b, len_c)
1148            if len_a == len_b
1149                && len_b == len_c
1150                && a.chunk_lengths()
1151                    .zip(b.chunk_lengths())
1152                    .zip(c.chunk_lengths())
1153                    .all(|((a, b), c)| a == b && b == c) =>
1154        {
1155            (Cow::Borrowed(a), Cow::Borrowed(b), Cow::Borrowed(c))
1156        },
1157        _ => {
1158            // could optimize to choose to rechunk a primitive and not a string or list type
1159            let a = a.rechunk();
1160            let b = b.rechunk();
1161            (
1162                Cow::Owned(a.match_chunks(c.chunk_lengths())),
1163                Cow::Owned(b.match_chunks(c.chunk_lengths())),
1164                Cow::Borrowed(c),
1165            )
1166        },
1167    }
1168}
1169
1170pub fn binary_concatenate_validities<'a, T, B>(
1171    left: &'a ChunkedArray<T>,
1172    right: &'a ChunkedArray<B>,
1173) -> Option<Bitmap>
1174where
1175    B: PolarsDataType,
1176    T: PolarsDataType,
1177{
1178    let (left, right) = align_chunks_binary(left, right);
1179    let left_validity = concatenate_validities(left.chunks());
1180    let right_validity = concatenate_validities(right.chunks());
1181    combine_validities_and(left_validity.as_ref(), right_validity.as_ref())
1182}
1183
1184/// Convenience for `x.into_iter().map(Into::into).collect()` using an `into_vec()` function.
1185pub trait IntoVec<T> {
1186    fn into_vec(self) -> Vec<T>;
1187}
1188
1189impl<I, S> IntoVec<PlSmallStr> for I
1190where
1191    I: IntoIterator<Item = S>,
1192    S: Into<PlSmallStr>,
1193{
1194    fn into_vec(self) -> Vec<PlSmallStr> {
1195        self.into_iter().map(|s| s.into()).collect()
1196    }
1197}
1198
1199/// This logic is same as the impl on ChunkedArray
1200/// The difference is that there is less indirection because the caller should preallocate
1201/// `chunk_lens` once. On the `ChunkedArray` we indirect through an `ArrayRef` which is an indirection
1202/// and a vtable.
1203#[inline]
1204pub(crate) fn index_to_chunked_index<
1205    I: Iterator<Item = Idx>,
1206    Idx: PartialOrd + std::ops::AddAssign + std::ops::SubAssign + Zero + One,
1207>(
1208    chunk_lens: I,
1209    index: Idx,
1210) -> (Idx, Idx) {
1211    let mut index_remainder = index;
1212    let mut current_chunk_idx = Zero::zero();
1213
1214    for chunk_len in chunk_lens {
1215        if chunk_len > index_remainder {
1216            break;
1217        } else {
1218            index_remainder -= chunk_len;
1219            current_chunk_idx += One::one();
1220        }
1221    }
1222    (current_chunk_idx, index_remainder)
1223}
1224
1225pub(crate) fn index_to_chunked_index_rev<
1226    I: Iterator<Item = Idx>,
1227    Idx: PartialOrd
1228        + std::ops::AddAssign
1229        + std::ops::SubAssign
1230        + std::ops::Sub<Output = Idx>
1231        + Zero
1232        + One
1233        + Copy
1234        + std::fmt::Debug,
1235>(
1236    chunk_lens_rev: I,
1237    index_from_back: Idx,
1238    total_chunks: Idx,
1239) -> (Idx, Idx) {
1240    debug_assert!(index_from_back > Zero::zero(), "at least -1");
1241    let mut index_remainder = index_from_back;
1242    let mut current_chunk_idx = One::one();
1243    let mut current_chunk_len = Zero::zero();
1244
1245    for chunk_len in chunk_lens_rev {
1246        current_chunk_len = chunk_len;
1247        if chunk_len >= index_remainder {
1248            break;
1249        } else {
1250            index_remainder -= chunk_len;
1251            current_chunk_idx += One::one();
1252        }
1253    }
1254    (
1255        total_chunks - current_chunk_idx,
1256        current_chunk_len - index_remainder,
1257    )
1258}
1259
1260pub fn first_null<'a, I>(iter: I) -> Option<usize>
1261where
1262    I: Iterator<Item = &'a dyn Array>,
1263{
1264    let mut offset = 0;
1265    for arr in iter {
1266        if let Some(mask) = arr.validity() {
1267            let len_mask = mask.len();
1268            let n = mask.leading_ones();
1269            if n < len_mask {
1270                return Some(offset + n);
1271            }
1272            offset += len_mask
1273        } else {
1274            offset += arr.len();
1275        }
1276    }
1277    None
1278}
1279
1280pub fn first_non_null<'a, I>(iter: I) -> Option<usize>
1281where
1282    I: Iterator<Item = &'a dyn Array>,
1283{
1284    let mut offset = 0;
1285    for arr in iter {
1286        if let Some(mask) = arr.validity() {
1287            let len_mask = mask.len();
1288            let n = mask.leading_zeros();
1289            if n < len_mask {
1290                return Some(offset + n);
1291            }
1292            offset += len_mask
1293        } else if !arr.is_empty() {
1294            return Some(offset);
1295        }
1296    }
1297    None
1298}
1299
1300pub fn last_non_null<'a, I>(iter: I, len: usize) -> Option<usize>
1301where
1302    I: DoubleEndedIterator<Item = &'a dyn Array>,
1303{
1304    if len == 0 {
1305        return None;
1306    }
1307    let mut offset = 0;
1308    for arr in iter.rev() {
1309        if let Some(mask) = arr.validity() {
1310            let len_mask = mask.len();
1311            let n = mask.trailing_zeros();
1312            if n < len_mask {
1313                return Some(len - offset - n - 1);
1314            }
1315            offset += len_mask;
1316        } else if !arr.is_empty() {
1317            return Some(len - offset - 1);
1318        }
1319    }
1320    None
1321}
1322
1323pub fn coalesce_nulls_columns(a: &Column, b: &Column) -> (Column, Column) {
1324    if a.null_count() > 0 || b.null_count() > 0 {
1325        let mut a = a.as_materialized_series().rechunk();
1326        let mut b = b.as_materialized_series().rechunk();
1327        for (arr_a, arr_b) in unsafe { a.chunks_mut().iter_mut().zip(b.chunks_mut()) } {
1328            let validity = match (arr_a.validity(), arr_b.validity()) {
1329                (None, Some(b)) => Some(b.clone()),
1330                (Some(a), Some(b)) => Some(a & b),
1331                (Some(a), None) => Some(a.clone()),
1332                (None, None) => None,
1333            };
1334            *arr_a = arr_a.with_validity(validity.clone());
1335            *arr_b = arr_b.with_validity(validity);
1336        }
1337        a.compute_len();
1338        b.compute_len();
1339        (a.into(), b.into())
1340    } else {
1341        (a.clone(), b.clone())
1342    }
1343}
1344
1345#[cfg(test)]
1346mod test {
1347    use super::*;
1348
1349    #[test]
1350    fn test_split() {
1351        let ca: Int32Chunked = (0..10).collect_ca("a".into());
1352
1353        let out = split(&ca, 3);
1354        assert_eq!(out[0].len(), 3);
1355        assert_eq!(out[1].len(), 3);
1356        assert_eq!(out[2].len(), 4);
1357    }
1358
1359    #[test]
1360    fn test_align_chunks() -> PolarsResult<()> {
1361        let a = Int32Chunked::new(PlSmallStr::EMPTY, &[1, 2, 3, 4]);
1362        let mut b = Int32Chunked::new(PlSmallStr::EMPTY, &[1]);
1363        let b2 = Int32Chunked::new(PlSmallStr::EMPTY, &[2, 3, 4]);
1364
1365        b.append(&b2)?;
1366        let (a, b) = align_chunks_binary(&a, &b);
1367        assert_eq!(
1368            a.chunk_lengths().collect::<Vec<_>>(),
1369            b.chunk_lengths().collect::<Vec<_>>()
1370        );
1371
1372        let a = Int32Chunked::new(PlSmallStr::EMPTY, &[1, 2, 3, 4]);
1373        let mut b = Int32Chunked::new(PlSmallStr::EMPTY, &[1]);
1374        let b1 = b.clone();
1375        b.append(&b1)?;
1376        b.append(&b1)?;
1377        b.append(&b1)?;
1378        let (a, b) = align_chunks_binary(&a, &b);
1379        assert_eq!(
1380            a.chunk_lengths().collect::<Vec<_>>(),
1381            b.chunk_lengths().collect::<Vec<_>>()
1382        );
1383
1384        Ok(())
1385    }
1386}