Vendor dependencies

This commit is contained in:
2026-08-01 16:11:49 +03:00
parent 7f139a0241
commit 6b5e7f0f8b
29706 changed files with 9575646 additions and 0 deletions
File diff suppressed because one or more lines are too long
+1426
View File
File diff suppressed because it is too large Load Diff
+468
View File
@@ -0,0 +1,468 @@
# THIS FILE IS AUTOMATICALLY GENERATED BY CARGO
#
# When uploading crates to the registry Cargo will automatically
# "normalize" Cargo.toml files for maximal compatibility
# with all versions of Cargo and also rewrite `path` dependencies
# to registry (e.g., crates.io) dependencies.
#
# If you are reading this file be aware that the original Cargo.toml
# will likely look very different (and much more reasonable).
# See Cargo.toml.orig for the original contents.
[package]
edition = "2024"
rust-version = "1.85"
name = "arrow"
version = "57.3.1"
authors = ["Apache Arrow <dev@arrow.apache.org>"]
build = false
include = [
"benches/*.rs",
"src/**/*.rs",
"tests/*.rs",
"Cargo.toml",
"LICENSE.txt",
"NOTICE.txt",
]
autolib = false
autobins = false
autoexamples = false
autotests = false
autobenches = false
description = "Rust implementation of Apache Arrow"
homepage = "https://github.com/apache/arrow-rs"
readme = "README.md"
keywords = ["arrow"]
license = "Apache-2.0"
repository = "https://github.com/apache/arrow-rs"
resolver = "2"
[package.metadata.docs.rs]
all-features = true
[features]
canonical_extension_types = ["arrow-schema/canonical_extension_types"]
chrono-tz = ["arrow-array/chrono-tz"]
csv = ["arrow-csv"]
default = [
"csv",
"ipc",
"json",
]
ffi = [
"arrow-schema/ffi",
"arrow-data/ffi",
"arrow-array/ffi",
]
force_validate = [
"arrow-array/force_validate",
"arrow-data/force_validate",
]
ipc = ["arrow-ipc"]
ipc_compression = [
"ipc",
"arrow-ipc/lz4",
"arrow-ipc/zstd",
]
json = ["arrow-json"]
prettyprint = ["arrow-cast/prettyprint"]
pyarrow = [
"ffi",
"dep:arrow-pyarrow",
]
test_utils = [
"dep:rand",
"dep:half",
]
[lib]
name = "arrow"
path = "src/lib.rs"
bench = false
[[test]]
name = "arithmetic"
path = "tests/arithmetic.rs"
required-features = ["chrono-tz"]
[[test]]
name = "array_cast"
path = "tests/array_cast.rs"
required-features = [
"chrono-tz",
"prettyprint",
]
[[test]]
name = "array_equal"
path = "tests/array_equal.rs"
[[test]]
name = "array_transform"
path = "tests/array_transform.rs"
[[test]]
name = "array_validation"
path = "tests/array_validation.rs"
[[test]]
name = "csv"
path = "tests/csv.rs"
required-features = [
"csv",
"chrono-tz",
]
[[test]]
name = "schema"
path = "tests/schema.rs"
[[test]]
name = "shrink_to_fit"
path = "tests/shrink_to_fit.rs"
[[test]]
name = "timezone"
path = "tests/timezone.rs"
required-features = ["chrono-tz"]
[[bench]]
name = "aggregate_kernels"
path = "benches/aggregate_kernels.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "arithmetic_kernels"
path = "benches/arithmetic_kernels.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "array_data_validate"
path = "benches/array_data_validate.rs"
harness = false
[[bench]]
name = "array_from"
path = "benches/array_from.rs"
harness = false
[[bench]]
name = "array_iter"
path = "benches/array_iter.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "array_slice"
path = "benches/array_slice.rs"
harness = false
[[bench]]
name = "bit_length_kernel"
path = "benches/bit_length_kernel.rs"
harness = false
[[bench]]
name = "bitwise_kernel"
path = "benches/bitwise_kernel.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "boolean_append_packed"
path = "benches/boolean_append_packed.rs"
harness = false
[[bench]]
name = "boolean_kernels"
path = "benches/boolean_kernels.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "buffer_bit_ops"
path = "benches/buffer_bit_ops.rs"
harness = false
[[bench]]
name = "buffer_create"
path = "benches/buffer_create.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "builder"
path = "benches/builder.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "cast_kernels"
path = "benches/cast_kernels.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "coalesce_kernels"
path = "benches/coalesce_kernels.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "comparison_kernels"
path = "benches/comparison_kernels.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "concatenate_kernel"
path = "benches/concatenate_kernel.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "csv_reader"
path = "benches/csv_reader.rs"
harness = false
required-features = [
"test_utils",
"csv",
]
[[bench]]
name = "csv_writer"
path = "benches/csv_writer.rs"
harness = false
required-features = ["csv"]
[[bench]]
name = "decimal_validate"
path = "benches/decimal_validate.rs"
harness = false
[[bench]]
name = "equal"
path = "benches/equal.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "filter_kernels"
path = "benches/filter_kernels.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "interleave_kernels"
path = "benches/interleave_kernels.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "json_reader"
path = "benches/json_reader.rs"
harness = false
required-features = [
"test_utils",
"json",
]
[[bench]]
name = "json_writer"
path = "benches/json_writer.rs"
harness = false
required-features = [
"test_utils",
"json",
]
[[bench]]
name = "length_kernel"
path = "benches/length_kernel.rs"
harness = false
[[bench]]
name = "lexsort"
path = "benches/lexsort.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "merge_kernels"
path = "benches/merge_kernels.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "mutable_array"
path = "benches/mutable_array.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "partition_kernels"
path = "benches/partition_kernels.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "primitive_run_accessor"
path = "benches/primitive_run_accessor.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "primitive_run_take"
path = "benches/primitive_run_take.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "regexp_kernels"
path = "benches/regexp_kernels.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "row_format"
path = "benches/row_format.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "sort_kernel"
path = "benches/sort_kernel.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "string_dictionary_builder"
path = "benches/string_dictionary_builder.rs"
harness = false
[[bench]]
name = "string_run_builder"
path = "benches/string_run_builder.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "string_run_iterator"
path = "benches/string_run_iterator.rs"
harness = false
[[bench]]
name = "substring_kernels"
path = "benches/substring_kernels.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "take_kernels"
path = "benches/take_kernels.rs"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "zip_kernels"
path = "benches/zip_kernels.rs"
harness = false
required-features = ["test_utils"]
[dependencies.arrow-arith]
version = "57.3.1"
[dependencies.arrow-array]
version = "57.3.1"
[dependencies.arrow-buffer]
version = "57.3.1"
[dependencies.arrow-cast]
version = "57.3.1"
[dependencies.arrow-csv]
version = "57.3.1"
optional = true
[dependencies.arrow-data]
version = "57.3.1"
[dependencies.arrow-ipc]
version = "57.3.1"
optional = true
[dependencies.arrow-json]
version = "57.3.1"
optional = true
[dependencies.arrow-ord]
version = "57.3.1"
[dependencies.arrow-pyarrow]
version = "57.3.1"
optional = true
[dependencies.arrow-row]
version = "57.3.1"
[dependencies.arrow-schema]
version = "57.3.1"
[dependencies.arrow-select]
version = "57.3.1"
[dependencies.arrow-string]
version = "57.3.1"
[dependencies.half]
version = "2.1"
optional = true
default-features = false
[dependencies.rand]
version = "0.9"
features = [
"std",
"std_rng",
"thread_rng",
]
optional = true
default-features = false
[dev-dependencies.bytes]
version = "1.9"
[dev-dependencies.chrono]
version = "0.4.40"
features = ["clock"]
default-features = false
[dev-dependencies.criterion]
version = "0.8.0"
default-features = false
[dev-dependencies.half]
version = "2.1"
default-features = false
[dev-dependencies.memmap2]
version = "0.9.3"
[dev-dependencies.rand]
version = "0.9"
features = [
"std",
"std_rng",
"thread_rng",
]
default-features = false
[dev-dependencies.serde]
version = "1.0"
features = ["derive"]
default-features = false
[build-dependencies]
+328
View File
@@ -0,0 +1,328 @@
# Licensed to the Apache Software Foundation (ASF) under one
# or more contributor license agreements. See the NOTICE file
# distributed with this work for additional information
# regarding copyright ownership. The ASF licenses this file
# to you under the Apache License, Version 2.0 (the
# "License"); you may not use this file except in compliance
# with the License. You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing,
# software distributed under the License is distributed on an
# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
# KIND, either express or implied. See the License for the
# specific language governing permissions and limitations
# under the License.
[package]
name = "arrow"
version = { workspace = true }
description = "Rust implementation of Apache Arrow"
homepage = { workspace = true }
repository = { workspace = true }
authors = { workspace = true }
license = { workspace = true }
keywords = ["arrow"]
include = [
"benches/*.rs",
"src/**/*.rs",
"tests/*.rs",
"Cargo.toml",
"LICENSE.txt",
"NOTICE.txt",
]
edition = { workspace = true }
rust-version = { workspace = true }
[lib]
bench = false
[dependencies]
arrow-arith = { workspace = true }
arrow-array = { workspace = true }
arrow-buffer = { workspace = true }
arrow-cast = { workspace = true }
arrow-csv = { workspace = true, optional = true }
arrow-data = { workspace = true }
arrow-ipc = { workspace = true, optional = true }
arrow-json = { workspace = true, optional = true }
arrow-ord = { workspace = true }
arrow-pyarrow = { workspace = true, optional = true }
arrow-row = { workspace = true }
arrow-schema = { workspace = true }
arrow-select = { workspace = true }
arrow-string = { workspace = true }
rand = { version = "0.9", default-features = false, features = ["std", "std_rng", "thread_rng"], optional = true }
half = { version = "2.1", default-features = false, optional = true }
[package.metadata.docs.rs]
all-features = true
[features]
default = ["csv", "ipc", "json"]
ipc_compression = ["ipc", "arrow-ipc/lz4", "arrow-ipc/zstd"]
csv = ["arrow-csv"]
ipc = ["arrow-ipc"]
json = ["arrow-json"]
prettyprint = ["arrow-cast/prettyprint"]
# The test utils feature enables code used in benchmarks and tests but
# not the core arrow code itself. Be aware that `rand` must be kept as
# an optional dependency for supporting compile to wasm32-unknown-unknown
# target without assuming an environment containing JavaScript.
test_utils = ["dep:rand", "dep:half"]
pyarrow = ["ffi", "dep:arrow-pyarrow"]
# force_validate runs full data validation for all arrays that are created
# this is not enabled by default as it is too computationally expensive
# but is run as part of our CI checks
force_validate = ["arrow-array/force_validate", "arrow-data/force_validate"]
# Enable ffi support
ffi = ["arrow-schema/ffi", "arrow-data/ffi", "arrow-array/ffi"]
chrono-tz = ["arrow-array/chrono-tz"]
canonical_extension_types = ["arrow-schema/canonical_extension_types"]
[dev-dependencies]
chrono = { workspace = true }
criterion = { workspace = true, default-features = false }
half = { version = "2.1", default-features = false }
rand = { version = "0.9", default-features = false, features = ["std", "std_rng", "thread_rng"] }
serde = { version = "1.0", default-features = false, features = ["derive"] }
# used in examples
memmap2 = "0.9.3"
bytes = "1.9"
[build-dependencies]
[[example]]
name = "dynamic_types"
required-features = ["prettyprint"]
path = "./examples/dynamic_types.rs"
[[example]]
name = "read_csv"
required-features = ["prettyprint", "csv"]
path = "./examples/read_csv.rs"
[[example]]
name = "read_csv_infer_schema"
required-features = ["prettyprint", "csv"]
path = "./examples/read_csv_infer_schema.rs"
[[example]]
name = "zero_copy_ipc"
required-features = ["prettyprint"]
path = "examples/zero_copy_ipc.rs"
[[bench]]
name = "aggregate_kernels"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "array_from"
harness = false
[[bench]]
name = "array_iter"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "builder"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "buffer_bit_ops"
harness = false
[[bench]]
name = "boolean_kernels"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "boolean_append_packed"
harness = false
[[bench]]
name = "arithmetic_kernels"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "cast_kernels"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "comparison_kernels"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "filter_kernels"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "coalesce_kernels"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "take_kernels"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "interleave_kernels"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "merge_kernels"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "zip_kernels"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "length_kernel"
harness = false
[[bench]]
name = "bit_length_kernel"
harness = false
[[bench]]
name = "sort_kernel"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "partition_kernels"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "csv_writer"
harness = false
required-features = ["csv"]
[[bench]]
name = "csv_reader"
harness = false
required-features = ["test_utils", "csv"]
[[bench]]
name = "json_reader"
harness = false
required-features = ["test_utils", "json"]
[[bench]]
name = "json_writer"
harness = false
required-features = ["test_utils", "json"]
[[bench]]
name = "equal"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "array_slice"
harness = false
[[bench]]
name = "concatenate_kernel"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "mutable_array"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "buffer_create"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "string_dictionary_builder"
harness = false
[[bench]]
name = "string_run_builder"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "string_run_iterator"
harness = false
[[bench]]
name = "primitive_run_accessor"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "primitive_run_take"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "substring_kernels"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "regexp_kernels"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "array_data_validate"
harness = false
[[bench]]
name = "decimal_validate"
harness = false
[[bench]]
name = "row_format"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "bitwise_kernel"
harness = false
required-features = ["test_utils"]
[[bench]]
name = "lexsort"
harness = false
required-features = ["test_utils"]
[[test]]
name = "csv"
required-features = ["csv", "chrono-tz"]
[[test]]
name = "array_cast"
required-features = ["chrono-tz", "prettyprint"]
[[test]]
name = "timezone"
required-features = ["chrono-tz"]
[[test]]
name = "arithmetic"
required-features = ["chrono-tz"]
+202
View File
@@ -0,0 +1,202 @@
Apache License
Version 2.0, January 2004
http://www.apache.org/licenses/
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
1. Definitions.
"License" shall mean the terms and conditions for use, reproduction,
and distribution as defined by Sections 1 through 9 of this document.
"Licensor" shall mean the copyright owner or entity authorized by
the copyright owner that is granting the License.
"Legal Entity" shall mean the union of the acting entity and all
other entities that control, are controlled by, or are under common
control with that entity. For the purposes of this definition,
"control" means (i) the power, direct or indirect, to cause the
direction or management of such entity, whether by contract or
otherwise, or (ii) ownership of fifty percent (50%) or more of the
outstanding shares, or (iii) beneficial ownership of such entity.
"You" (or "Your") shall mean an individual or Legal Entity
exercising permissions granted by this License.
"Source" form shall mean the preferred form for making modifications,
including but not limited to software source code, documentation
source, and configuration files.
"Object" form shall mean any form resulting from mechanical
transformation or translation of a Source form, including but
not limited to compiled object code, generated documentation,
and conversions to other media types.
"Work" shall mean the work of authorship, whether in Source or
Object form, made available under the License, as indicated by a
copyright notice that is included in or attached to the work
(an example is provided in the Appendix below).
"Derivative Works" shall mean any work, whether in Source or Object
form, that is based on (or derived from) the Work and for which the
editorial revisions, annotations, elaborations, or other modifications
represent, as a whole, an original work of authorship. For the purposes
of this License, Derivative Works shall not include works that remain
separable from, or merely link (or bind by name) to the interfaces of,
the Work and Derivative Works thereof.
"Contribution" shall mean any work of authorship, including
the original version of the Work and any modifications or additions
to that Work or Derivative Works thereof, that is intentionally
submitted to Licensor for inclusion in the Work by the copyright owner
or by an individual or Legal Entity authorized to submit on behalf of
the copyright owner. For the purposes of this definition, "submitted"
means any form of electronic, verbal, or written communication sent
to the Licensor or its representatives, including but not limited to
communication on electronic mailing lists, source code control systems,
and issue tracking systems that are managed by, or on behalf of, the
Licensor for the purpose of discussing and improving the Work, but
excluding communication that is conspicuously marked or otherwise
designated in writing by the copyright owner as "Not a Contribution."
"Contributor" shall mean Licensor and any individual or Legal Entity
on behalf of whom a Contribution has been received by Licensor and
subsequently incorporated within the Work.
2. Grant of Copyright License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
copyright license to reproduce, prepare Derivative Works of,
publicly display, publicly perform, sublicense, and distribute the
Work and such Derivative Works in Source or Object form.
3. Grant of Patent License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
(except as stated in this section) patent license to make, have made,
use, offer to sell, sell, import, and otherwise transfer the Work,
where such license applies only to those patent claims licensable
by such Contributor that are necessarily infringed by their
Contribution(s) alone or by combination of their Contribution(s)
with the Work to which such Contribution(s) was submitted. If You
institute patent litigation against any entity (including a
cross-claim or counterclaim in a lawsuit) alleging that the Work
or a Contribution incorporated within the Work constitutes direct
or contributory patent infringement, then any patent licenses
granted to You under this License for that Work shall terminate
as of the date such litigation is filed.
4. Redistribution. You may reproduce and distribute copies of the
Work or Derivative Works thereof in any medium, with or without
modifications, and in Source or Object form, provided that You
meet the following conditions:
(a) You must give any other recipients of the Work or
Derivative Works a copy of this License; and
(b) You must cause any modified files to carry prominent notices
stating that You changed the files; and
(c) You must retain, in the Source form of any Derivative Works
that You distribute, all copyright, patent, trademark, and
attribution notices from the Source form of the Work,
excluding those notices that do not pertain to any part of
the Derivative Works; and
(d) If the Work includes a "NOTICE" text file as part of its
distribution, then any Derivative Works that You distribute must
include a readable copy of the attribution notices contained
within such NOTICE file, excluding those notices that do not
pertain to any part of the Derivative Works, in at least one
of the following places: within a NOTICE text file distributed
as part of the Derivative Works; within the Source form or
documentation, if provided along with the Derivative Works; or,
within a display generated by the Derivative Works, if and
wherever such third-party notices normally appear. The contents
of the NOTICE file are for informational purposes only and
do not modify the License. You may add Your own attribution
notices within Derivative Works that You distribute, alongside
or as an addendum to the NOTICE text from the Work, provided
that such additional attribution notices cannot be construed
as modifying the License.
You may add Your own copyright statement to Your modifications and
may provide additional or different license terms and conditions
for use, reproduction, or distribution of Your modifications, or
for any such Derivative Works as a whole, provided Your use,
reproduction, and distribution of the Work otherwise complies with
the conditions stated in this License.
5. Submission of Contributions. Unless You explicitly state otherwise,
any Contribution intentionally submitted for inclusion in the Work
by You to the Licensor shall be under the terms and conditions of
this License, without any additional terms or conditions.
Notwithstanding the above, nothing herein shall supersede or modify
the terms of any separate license agreement you may have executed
with Licensor regarding such Contributions.
6. Trademarks. This License does not grant permission to use the trade
names, trademarks, service marks, or product names of the Licensor,
except as required for reasonable and customary use in describing the
origin of the Work and reproducing the content of the NOTICE file.
7. Disclaimer of Warranty. Unless required by applicable law or
agreed to in writing, Licensor provides the Work (and each
Contributor provides its Contributions) on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
implied, including, without limitation, any warranties or conditions
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
PARTICULAR PURPOSE. You are solely responsible for determining the
appropriateness of using or redistributing the Work and assume any
risks associated with Your exercise of permissions under this License.
8. Limitation of Liability. In no event and under no legal theory,
whether in tort (including negligence), contract, or otherwise,
unless required by applicable law (such as deliberate and grossly
negligent acts) or agreed to in writing, shall any Contributor be
liable to You for damages, including any direct, indirect, special,
incidental, or consequential damages of any character arising as a
result of this License or out of the use or inability to use the
Work (including but not limited to damages for loss of goodwill,
work stoppage, computer failure or malfunction, or any and all
other commercial damages or losses), even if such Contributor
has been advised of the possibility of such damages.
9. Accepting Warranty or Additional Liability. While redistributing
the Work or Derivative Works thereof, You may choose to offer,
and charge a fee for, acceptance of support, warranty, indemnity,
or other liability obligations and/or rights consistent with this
License. However, in accepting such obligations, You may act only
on Your own behalf and on Your sole responsibility, not on behalf
of any other Contributor, and only if You agree to indemnify,
defend, and hold each Contributor harmless for any liability
incurred by, or claims asserted against, such Contributor by reason
of your accepting any such warranty or additional liability.
END OF TERMS AND CONDITIONS
APPENDIX: How to apply the Apache License to your work.
To apply the Apache License to your work, attach the following
boilerplate notice, with the fields enclosed by brackets "[]"
replaced with your own identifying information. (Don't include
the brackets!) The text should be enclosed in the appropriate
comment syntax for the file format. We also recommend that a
file or class name and description of purpose be included on the
same "printed page" as the copyright notice for easier
identification within third-party archives.
Copyright [yyyy] [name of copyright owner]
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
+84
View File
@@ -0,0 +1,84 @@
Apache Arrow
Copyright 2016-2019 The Apache Software Foundation
This product includes software developed at
The Apache Software Foundation (http://www.apache.org/).
This product includes software from the SFrame project (BSD, 3-clause).
* Copyright (C) 2015 Dato, Inc.
* Copyright (c) 2009 Carnegie Mellon University.
This product includes software from the Feather project (Apache 2.0)
https://github.com/wesm/feather
This product includes software from the DyND project (BSD 2-clause)
https://github.com/libdynd
This product includes software from the LLVM project
* distributed under the University of Illinois Open Source
This product includes software from the google-lint project
* Copyright (c) 2009 Google Inc. All rights reserved.
This product includes software from the mman-win32 project
* Copyright https://code.google.com/p/mman-win32/
* Licensed under the MIT License;
This product includes software from the LevelDB project
* Copyright (c) 2011 The LevelDB Authors. All rights reserved.
* Use of this source code is governed by a BSD-style license that can be
* Moved from Kudu http://github.com/cloudera/kudu
This product includes software from the CMake project
* Copyright 2001-2009 Kitware, Inc.
* Copyright 2012-2014 Continuum Analytics, Inc.
* All rights reserved.
This product includes software from https://github.com/matthew-brett/multibuild (BSD 2-clause)
* Copyright (c) 2013-2016, Matt Terry and Matthew Brett; all rights reserved.
This product includes software from the Ibis project (Apache 2.0)
* Copyright (c) 2015 Cloudera, Inc.
* https://github.com/cloudera/ibis
This product includes software from Dremio (Apache 2.0)
* Copyright (C) 2017-2018 Dremio Corporation
* https://github.com/dremio/dremio-oss
This product includes software from Google Guava (Apache 2.0)
* Copyright (C) 2007 The Guava Authors
* https://github.com/google/guava
This product include software from CMake (BSD 3-Clause)
* CMake - Cross Platform Makefile Generator
* Copyright 2000-2019 Kitware, Inc. and Contributors
The web site includes files generated by Jekyll.
--------------------------------------------------------------------------------
This product includes code from Apache Kudu, which includes the following in
its NOTICE file:
Apache Kudu
Copyright 2016 The Apache Software Foundation
This product includes software developed at
The Apache Software Foundation (http://www.apache.org/).
Portions of this software were developed at
Cloudera, Inc (http://www.cloudera.com/).
--------------------------------------------------------------------------------
This product includes code from Apache ORC, which includes the following in
its NOTICE file:
Apache ORC
Copyright 2013-2019 The Apache Software Foundation
This product includes software developed by The Apache Software
Foundation (http://www.apache.org/).
This product includes software developed by Hewlett-Packard:
(c) Copyright [2014-2015] Hewlett-Packard Development Company, L.P
+165
View File
@@ -0,0 +1,165 @@
<!---
Licensed to the Apache Software Foundation (ASF) under one
or more contributor license agreements. See the NOTICE file
distributed with this work for additional information
regarding copyright ownership. The ASF licenses this file
to you under the Apache License, Version 2.0 (the
"License"); you may not use this file except in compliance
with the License. You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing,
software distributed under the License is distributed on an
"AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
KIND, either express or implied. See the License for the
specific language governing permissions and limitations
under the License.
-->
# Apache Arrow Official Native Rust Implementation
[![crates.io](https://img.shields.io/crates/v/arrow.svg)](https://crates.io/crates/arrow)
[![docs.rs](https://img.shields.io/docsrs/arrow.svg)](https://docs.rs/arrow/latest/arrow/)
This crate contains the official Native Rust implementation of [Apache Arrow][arrow] in memory format, governed by the Apache Software Foundation.
The [API documentation](https://docs.rs/arrow/latest) contains examples and full API.
There are several [examples](https://github.com/apache/arrow-rs/tree/main/arrow/examples) to start from as well.
The API documentation for most recent, unreleased code is available [here](https://arrow.apache.org/rust/arrow/index.html).
## Arrow Implementation Status
Please see the [Implementation Status Page] on the Apache Arrow website for which
Arrow features are supported by this crate.
[Implementation Status Page]: https://arrow.apache.org/docs/status.html
## Rust Version Compatibility
This crate is tested with the latest stable version of Rust. We do not currently test against other, older versions.
## Versioning / Releases
The `arrow` crate follows the [SemVer standard] defined by Cargo and works well
within the Rust crate ecosystem. See the [repository README] for more details on
the release schedule, version and deprecation policy.
[SemVer standard]: https://doc.rust-lang.org/cargo/reference/semver.html
[repository README]: https://github.com/apache/arrow-rs
Note that for historical reasons, this crate uses versions with major numbers
greater than `0.x` (e.g. `19.0.0`), unlike many other crates in the Rust
ecosystem which spend extended time releasing versions `0.x` to signal planned
ongoing API changes. Minor arrow releases contain only compatible changes, while
major releases may contain breaking API changes.
## Feature Flags
The `arrow` crate provides the following features which may be enabled in your `Cargo.toml`:
- `csv` (default) - support for reading and writing Arrow arrays to/from csv files
- `json` (default) - support for reading and writing Arrow array to/from json files
- `ipc` (default) - support for reading [Arrow IPC Format](https://arrow.apache.org/docs/format/Columnar.html#serialization-and-interprocess-communication-ipc), also used as the wire protocol in [arrow-flight](https://crates.io/crates/arrow-flight)
- `ipc_compression` - Enables reading and writing compressed IPC streams (also enables `ipc`)
- `prettyprint` - support for formatting record batches as textual columns
implementations of some [compute](https://github.com/apache/arrow-rs/tree/main/arrow/src/compute/kernels)
- `chrono-tz` - support of parsing timezone using [chrono-tz](https://docs.rs/chrono-tz/0.6.0/chrono_tz/)
- `ffi` - bindings for the Arrow C [C Data Interface](https://arrow.apache.org/docs/format/CDataInterface.html)
- `pyarrow` - bindings for pyo3 to call arrow-rs from python
- `canonical_extension_types` - definitions for [canonical extension types](https://arrow.apache.org/docs/format/CanonicalExtensions.html#format-canonical-extensions)
## Arrow Feature Status
The [Apache Arrow Status](https://arrow.apache.org/docs/status.html) page lists which features of Arrow this crate supports.
## Safety
Arrow seeks to uphold the Rust Soundness Pledge as articulated eloquently [here](https://raphlinus.github.io/rust/2020/01/18/soundness-pledge.html). Specifically:
> The intent of this crate is to be free of soundness bugs. The developers will do their best to avoid them, and welcome help in analyzing and fixing them
Where soundness in turn is defined as:
> Code is unable to trigger undefined behavior using safe APIs
One way to ensure this would be to not use `unsafe`, however, as described in the opening chapter of the [Rustonomicon](https://doc.rust-lang.org/nomicon/meet-safe-and-unsafe.html) this is not a requirement, and flexibility in this regard is one of Rust's great strengths.
In particular there are a number of scenarios where `unsafe` is largely unavoidable:
- Invariants that cannot be statically verified by the compiler and unlock non-trivial performance wins, e.g. values in a StringArray are UTF-8, [TrustedLen](https://doc.rust-lang.org/std/iter/trait.TrustedLen.html) iterators, etc...
- FFI
Additionally, this crate exposes a number of `unsafe` APIs, allowing downstream crates to explicitly opt-out of potentially expensive invariant checking where appropriate.
We have a number of strategies to help reduce this risk:
- Provide strongly-typed `Array` and `ArrayBuilder` APIs to safely and efficiently interact with arrays
- Extensive validation logic to safely construct `ArrayData` from untrusted sources
- All commits are verified using [MIRI](https://github.com/rust-lang/miri) to detect undefined behaviour
- Use a `force_validate` feature that enables additional validation checks for use in test/debug builds
- There is ongoing work to reduce and better document the use of unsafe, and we welcome contributions in this space
## Building for WASM
Arrow can compile to WebAssembly using the `wasm32-unknown-unknown` and `wasm32-wasi` targets.
In order to compile Arrow for `wasm32-unknown-unknown` you will need to disable default features, then include the desired features, but exclude test dependencies (the `test_utils` feature). For example, use this snippet in your `Cargo.toml`:
```toml
[dependencies]
arrow = { version = "5.0", default-features = false, features = ["csv", "ipc"] }
```
## Examples
The examples folder shows how to construct some different types of Arrow
arrays, including dynamic arrays:
Examples can be run using the `cargo run --example` command. For example:
```bash
cargo run --example builders
cargo run --example dynamic_types
cargo run --example read_csv
```
[arrow]: https://arrow.apache.org/
## Performance Tips
Arrow aims to be as fast as possible out of the box, whilst not compromising on safety. However,
it relies heavily on LLVM auto-vectorisation to achieve this. Unfortunately the LLVM defaults,
particularly for x86_64, favour portability over performance, and LLVM will consequently avoid
using more recent instructions that would result in errors on older CPUs.
To address this it is recommended that you override the LLVM defaults either
by setting the `RUSTFLAGS` environment variable, or by setting `rustflags` in your
[Cargo configuration](https://doc.rust-lang.org/cargo/reference/config.html)
Enable all features supported by the current CPU
```ignore
RUSTFLAGS="-C target-cpu=native"
```
Enable all features supported by the current CPU, and enable full use of AVX512
```ignore
RUSTFLAGS="-C target-cpu=native -C target-feature=-prefer-256-bit"
```
Enable all features supported by CPUs more recent than haswell (2013)
```ignore
RUSTFLAGS="-C target-cpu=haswell"
```
For a full list of features and target CPUs use
```shell
$ rustc --print target-cpus
$ rustc --print target-features
```
+174
View File
@@ -0,0 +1,174 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use criterion::{Criterion, Throughput};
use rand::distr::{Distribution, StandardUniform};
extern crate arrow;
use arrow::compute::kernels::aggregate::*;
use arrow::util::bench_util::*;
use arrow::{array::*, datatypes::Float32Type};
use arrow_array::types::{Float64Type, Int8Type, Int16Type, Int32Type, Int64Type};
const BATCH_SIZE: usize = 64 * 1024;
fn primitive_benchmark<T: ArrowNumericType>(c: &mut Criterion, name: &str)
where
StandardUniform: Distribution<T::Native>,
{
let nonnull_array = create_primitive_array::<T>(BATCH_SIZE, 0.0);
let nullable_array = create_primitive_array::<T>(BATCH_SIZE, 0.5);
c.benchmark_group(name)
.throughput(Throughput::Bytes(
(std::mem::size_of::<T::Native>() * BATCH_SIZE) as u64,
))
.bench_function("sum nonnull", |b| b.iter(|| sum(&nonnull_array)))
.bench_function("min nonnull", |b| b.iter(|| min(&nonnull_array)))
.bench_function("max nonnull", |b| b.iter(|| max(&nonnull_array)))
.bench_function("sum nullable", |b| b.iter(|| sum(&nullable_array)))
.bench_function("min nullable", |b| b.iter(|| min(&nullable_array)))
.bench_function("max nullable", |b| b.iter(|| max(&nullable_array)));
}
fn add_benchmark(c: &mut Criterion) {
primitive_benchmark::<Float32Type>(c, "float32");
primitive_benchmark::<Float64Type>(c, "float64");
primitive_benchmark::<Int8Type>(c, "int8");
primitive_benchmark::<Int16Type>(c, "int16");
primitive_benchmark::<Int32Type>(c, "int32");
primitive_benchmark::<Int64Type>(c, "int64");
{
let nonnull_strings = create_string_array_with_len::<i32>(BATCH_SIZE, 0.0, 16);
let nullable_strings = create_string_array_with_len::<i32>(BATCH_SIZE, 0.5, 16);
c.benchmark_group("string")
.throughput(Throughput::Elements(BATCH_SIZE as u64))
.bench_function("min nonnull", |b| b.iter(|| min_string(&nonnull_strings)))
.bench_function("max nonnull", |b| b.iter(|| max_string(&nonnull_strings)))
.bench_function("min nullable", |b| b.iter(|| min_string(&nullable_strings)))
.bench_function("max nullable", |b| b.iter(|| max_string(&nullable_strings)));
}
{
let nonnull_strings = create_string_view_array_with_len(BATCH_SIZE, 0.0, 16, false);
let nullable_strings = create_string_view_array_with_len(BATCH_SIZE, 0.5, 16, false);
c.benchmark_group("string view")
.throughput(Throughput::Elements(BATCH_SIZE as u64))
.bench_function("min nonnull", |b| {
b.iter(|| min_string_view(&nonnull_strings))
})
.bench_function("max nonnull", |b| {
b.iter(|| max_string_view(&nonnull_strings))
})
.bench_function("min nullable", |b| {
b.iter(|| min_string_view(&nullable_strings))
})
.bench_function("max nullable", |b| {
b.iter(|| max_string_view(&nullable_strings))
});
}
{
let nonnull_bools_mixed = create_boolean_array(BATCH_SIZE, 0.0, 0.5);
let nonnull_bools_all_false = create_boolean_array(BATCH_SIZE, 0.0, 0.0);
let nonnull_bools_all_true = create_boolean_array(BATCH_SIZE, 0.0, 1.0);
let nullable_bool_mixed = create_boolean_array(BATCH_SIZE, 0.5, 0.5);
let nullable_bool_all_false = create_boolean_array(BATCH_SIZE, 0.5, 0.0);
let nullable_bool_all_true = create_boolean_array(BATCH_SIZE, 0.5, 1.0);
c.benchmark_group("bool")
.throughput(Throughput::Elements(BATCH_SIZE as u64))
.bench_function("min nonnull mixed", |b| {
b.iter(|| min_boolean(&nonnull_bools_mixed))
})
.bench_function("max nonnull mixed", |b| {
b.iter(|| max_boolean(&nonnull_bools_mixed))
})
.bench_function("or nonnull mixed", |b| {
b.iter(|| bool_or(&nonnull_bools_mixed))
})
.bench_function("and nonnull mixed", |b| {
b.iter(|| bool_and(&nonnull_bools_mixed))
})
.bench_function("min nonnull false", |b| {
b.iter(|| min_boolean(&nonnull_bools_all_false))
})
.bench_function("max nonnull false", |b| {
b.iter(|| max_boolean(&nonnull_bools_all_false))
})
.bench_function("or nonnull false", |b| {
b.iter(|| bool_or(&nonnull_bools_all_false))
})
.bench_function("and nonnull false", |b| {
b.iter(|| bool_and(&nonnull_bools_all_false))
})
.bench_function("min nonnull true", |b| {
b.iter(|| min_boolean(&nonnull_bools_all_true))
})
.bench_function("max nonnull true", |b| {
b.iter(|| max_boolean(&nonnull_bools_all_true))
})
.bench_function("or nonnull true", |b| {
b.iter(|| bool_or(&nonnull_bools_all_true))
})
.bench_function("and nonnull true", |b| {
b.iter(|| bool_and(&nonnull_bools_all_true))
})
.bench_function("min nullable mixed", |b| {
b.iter(|| min_boolean(&nullable_bool_mixed))
})
.bench_function("max nullable mixed", |b| {
b.iter(|| max_boolean(&nullable_bool_mixed))
})
.bench_function("or nullable mixed", |b| {
b.iter(|| bool_or(&nullable_bool_mixed))
})
.bench_function("and nullable mixed", |b| {
b.iter(|| bool_and(&nullable_bool_mixed))
})
.bench_function("min nullable false", |b| {
b.iter(|| min_boolean(&nullable_bool_all_false))
})
.bench_function("max nullable false", |b| {
b.iter(|| max_boolean(&nullable_bool_all_false))
})
.bench_function("or nullable false", |b| {
b.iter(|| bool_or(&nullable_bool_all_false))
})
.bench_function("and nullable false", |b| {
b.iter(|| bool_and(&nullable_bool_all_false))
})
.bench_function("min nullable true", |b| {
b.iter(|| min_boolean(&nullable_bool_all_true))
})
.bench_function("max nullable true", |b| {
b.iter(|| max_boolean(&nullable_bool_all_true))
})
.bench_function("or nullable true", |b| {
b.iter(|| bool_or(&nullable_bool_all_true))
})
.bench_function("and nullable true", |b| {
b.iter(|| bool_and(&nullable_bool_all_true))
});
}
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
@@ -0,0 +1,79 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use criterion::*;
extern crate arrow;
use arrow::compute::kernels::numeric::*;
use arrow::datatypes::Float32Type;
use arrow::util::bench_util::*;
use arrow_array::Scalar;
use std::hint;
fn add_benchmark(c: &mut Criterion) {
const BATCH_SIZE: usize = 64 * 1024;
for null_density in [0., 0.1, 0.5, 0.9, 1.0] {
let arr_a = create_primitive_array::<Float32Type>(BATCH_SIZE, null_density);
let arr_b = create_primitive_array::<Float32Type>(BATCH_SIZE, null_density);
let scalar_a = create_primitive_array::<Float32Type>(1, 0.);
let scalar = Scalar::new(&scalar_a);
c.bench_function(&format!("add({null_density})"), |b| {
b.iter(|| hint::black_box(add_wrapping(&arr_a, &arr_b).unwrap()))
});
c.bench_function(&format!("add_checked({null_density})"), |b| {
b.iter(|| hint::black_box(add(&arr_a, &arr_b).unwrap()))
});
c.bench_function(&format!("add_scalar({null_density})"), |b| {
b.iter(|| hint::black_box(add_wrapping(&arr_a, &scalar).unwrap()))
});
c.bench_function(&format!("subtract({null_density})"), |b| {
b.iter(|| hint::black_box(sub_wrapping(&arr_a, &arr_b).unwrap()))
});
c.bench_function(&format!("subtract_checked({null_density})"), |b| {
b.iter(|| hint::black_box(sub(&arr_a, &arr_b).unwrap()))
});
c.bench_function(&format!("subtract_scalar({null_density})"), |b| {
b.iter(|| hint::black_box(sub_wrapping(&arr_a, &scalar).unwrap()))
});
c.bench_function(&format!("multiply({null_density})"), |b| {
b.iter(|| hint::black_box(mul_wrapping(&arr_a, &arr_b).unwrap()))
});
c.bench_function(&format!("multiply_checked({null_density})"), |b| {
b.iter(|| hint::black_box(mul(&arr_a, &arr_b).unwrap()))
});
c.bench_function(&format!("multiply_scalar({null_density})"), |b| {
b.iter(|| hint::black_box(mul_wrapping(&arr_a, &scalar).unwrap()))
});
c.bench_function(&format!("divide({null_density})"), |b| {
b.iter(|| hint::black_box(div(&arr_a, &arr_b).unwrap()))
});
c.bench_function(&format!("divide_scalar({null_density})"), |b| {
b.iter(|| hint::black_box(div(&arr_a, &scalar).unwrap()))
});
c.bench_function(&format!("modulo({null_density})"), |b| {
b.iter(|| hint::black_box(rem(&arr_a, &arr_b).unwrap()))
});
c.bench_function(&format!("modulo_scalar({null_density})"), |b| {
b.iter(|| hint::black_box(rem(&arr_a, &scalar).unwrap()))
});
}
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
@@ -0,0 +1,63 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use criterion::Criterion;
extern crate arrow;
use arrow::{array::*, buffer::Buffer, datatypes::DataType};
fn create_binary_array_data(length: i32) -> ArrayData {
let value_buffer = Buffer::from_iter(0_i32..length);
let offsets_buffer = Buffer::from_iter(0_i32..length + 1);
ArrayData::try_new(
DataType::Binary,
length as usize,
None,
0,
vec![offsets_buffer, value_buffer],
vec![],
)
.unwrap()
}
fn validate_utf8_array(arr: &ArrayData) {
arr.validate_values().unwrap();
}
fn validate_benchmark(c: &mut Criterion) {
//Binary Array
c.bench_function("validate_binary_array_data 20000", |b| {
b.iter(|| create_binary_array_data(20000))
});
//Utf8 Array
let str_arr = StringArray::from(vec!["test"; 20000]).to_data();
c.bench_function("validate_utf8_array_data 20000", |b| {
b.iter(|| validate_utf8_array(&str_arr))
});
let byte_array = BinaryArray::from_iter_values(std::iter::repeat_n(b"test", 20000));
c.bench_function("byte_array_to_string_array 20000", |b| {
b.iter(|| StringArray::from(BinaryArray::from(byte_array.to_data())))
});
}
criterion_group!(benches, validate_benchmark);
criterion_main!(benches);
+253
View File
@@ -0,0 +1,253 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
extern crate arrow;
#[macro_use]
extern crate criterion;
use criterion::Criterion;
use arrow::array::*;
use arrow_buffer::i256;
use rand::Rng;
use std::iter::repeat_n;
use std::{hint, sync::Arc};
fn array_from_vec(n: usize) {
let v: Vec<i32> = (0..n as i32).collect();
hint::black_box(Int32Array::from(v));
}
fn array_string_from_vec(n: usize) {
let mut v: Vec<Option<&str>> = Vec::with_capacity(n);
for i in 0..n {
if i % 2 == 0 {
v.push(Some("hello world"));
} else {
v.push(None);
}
}
hint::black_box(StringArray::from(v));
}
fn struct_array_values(
n: usize,
) -> (
&'static str,
Vec<Option<&'static str>>,
&'static str,
Vec<Option<i32>>,
) {
let mut strings: Vec<Option<&str>> = Vec::with_capacity(n);
let mut ints: Vec<Option<i32>> = Vec::with_capacity(n);
for _ in 0..n / 4 {
strings.extend_from_slice(&[Some("joe"), None, None, Some("mark")]);
ints.extend_from_slice(&[Some(1), Some(2), None, Some(4)]);
}
("f1", strings, "f2", ints)
}
fn struct_array_from_vec(
field1: &str,
strings: &[Option<&str>],
field2: &str,
ints: &[Option<i32>],
) {
let strings: ArrayRef = Arc::new(StringArray::from(strings.to_owned()));
let ints: ArrayRef = Arc::new(Int32Array::from(ints.to_owned()));
hint::black_box(StructArray::try_from(vec![(field1, strings), (field2, ints)]).unwrap());
}
fn decimal32_array_from_vec(array: &[Option<i32>]) {
hint::black_box(
array
.iter()
.copied()
.collect::<Decimal32Array>()
.with_precision_and_scale(9, 2)
.unwrap(),
);
}
fn decimal64_array_from_vec(array: &[Option<i64>]) {
hint::black_box(
array
.iter()
.copied()
.collect::<Decimal64Array>()
.with_precision_and_scale(17, 2)
.unwrap(),
);
}
fn decimal128_array_from_vec(array: &[Option<i128>]) {
hint::black_box(
array
.iter()
.copied()
.collect::<Decimal128Array>()
.with_precision_and_scale(34, 2)
.unwrap(),
);
}
fn decimal256_array_from_vec(array: &[Option<i256>]) {
hint::black_box(
array
.iter()
.copied()
.collect::<Decimal256Array>()
.with_precision_and_scale(70, 2)
.unwrap(),
);
}
fn array_from_vec_decimal_benchmark(c: &mut Criterion) {
// bench decimal32 array
// create option<i32> array
let size: usize = 1 << 15;
let mut rng = rand::rng();
let mut array = vec![];
for _ in 0..size {
array.push(Some(rng.random_range::<i32, _>(0..99999999)));
}
c.bench_function("decimal32_array_from_vec 32768", |b| {
b.iter(|| decimal32_array_from_vec(array.as_slice()))
});
// bench decimal64 array
// create option<i64> array
let size: usize = 1 << 15;
let mut rng = rand::rng();
let mut array = vec![];
for _ in 0..size {
array.push(Some(rng.random_range::<i64, _>(0..9999999999)));
}
c.bench_function("decimal64_array_from_vec 32768", |b| {
b.iter(|| decimal64_array_from_vec(array.as_slice()))
});
// bench decimal128 array
// create option<i128> array
let size: usize = 1 << 15;
let mut rng = rand::rng();
let mut array = vec![];
for _ in 0..size {
array.push(Some(rng.random_range::<i128, _>(0..9999999999)));
}
c.bench_function("decimal128_array_from_vec 32768", |b| {
b.iter(|| decimal128_array_from_vec(array.as_slice()))
});
// bench decimal256array
// create option<into<decimal256>> array
let size = 1 << 10;
let mut array = vec![];
let mut rng = rand::rng();
for _ in 0..size {
let decimal = i256::from_i128(rng.random_range::<i128, _>(0..9999999999999));
array.push(Some(decimal));
}
// bench decimal256 array
c.bench_function("decimal256_array_from_vec 32768", |b| {
b.iter(|| decimal256_array_from_vec(array.as_slice()))
});
}
fn array_from_vec_benchmark(c: &mut Criterion) {
c.bench_function("array_from_vec 128", |b| b.iter(|| array_from_vec(128)));
c.bench_function("array_from_vec 256", |b| b.iter(|| array_from_vec(256)));
c.bench_function("array_from_vec 512", |b| b.iter(|| array_from_vec(512)));
c.bench_function("array_string_from_vec 128", |b| {
b.iter(|| array_string_from_vec(128))
});
c.bench_function("array_string_from_vec 256", |b| {
b.iter(|| array_string_from_vec(256))
});
c.bench_function("array_string_from_vec 512", |b| {
b.iter(|| array_string_from_vec(512))
});
let (field1, strings, field2, ints) = struct_array_values(128);
c.bench_function("struct_array_from_vec 128", |b| {
b.iter(|| struct_array_from_vec(field1, &strings, field2, &ints))
});
let (field1, strings, field2, ints) = struct_array_values(256);
c.bench_function("struct_array_from_vec 256", |b| {
b.iter(|| struct_array_from_vec(field1, &strings, field2, &ints))
});
let (field1, strings, field2, ints) = struct_array_values(512);
c.bench_function("struct_array_from_vec 512", |b| {
b.iter(|| struct_array_from_vec(field1, &strings, field2, &ints))
});
let (field1, strings, field2, ints) = struct_array_values(1024);
c.bench_function("struct_array_from_vec 1024", |b| {
b.iter(|| struct_array_from_vec(field1, &strings, field2, &ints))
});
}
fn gen_option_vector<TItem: Copy>(item: TItem, len: usize) -> Vec<Option<TItem>> {
hint::black_box(
repeat_n(item, len)
.enumerate()
.map(|(idx, item)| if idx % 3 == 0 { None } else { Some(item) })
.collect(),
)
}
fn from_iter_benchmark(c: &mut Criterion) {
const ITER_LEN: usize = 16_384;
// All ArrowPrimitiveType use the same implementation
c.bench_function("Int64Array::from_iter", |b| {
let values = gen_option_vector(1, ITER_LEN);
b.iter(|| hint::black_box(Int64Array::from_iter(values.iter())));
});
c.bench_function("Int64Array::from_trusted_len_iter", |b| {
let values = gen_option_vector(1, ITER_LEN);
b.iter(|| unsafe {
// SAFETY: values.iter() is a TrustedLenIterator
hint::black_box(Int64Array::from_trusted_len_iter(values.iter()))
});
});
c.bench_function("BooleanArray::from_iter", |b| {
let values = gen_option_vector(true, ITER_LEN);
b.iter(|| hint::black_box(BooleanArray::from_iter(values.iter())));
});
c.bench_function("BooleanArray::from_trusted_len_iter", |b| {
let values = gen_option_vector(true, ITER_LEN);
b.iter(|| unsafe {
// SAFETY: values.iter() is a TrustedLenIterator
hint::black_box(BooleanArray::from_trusted_len_iter(values.iter()))
});
});
}
criterion_group!(
benches,
array_from_vec_benchmark,
array_from_vec_decimal_benchmark,
from_iter_benchmark
);
criterion_main!(benches);
+305
View File
@@ -0,0 +1,305 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
extern crate arrow;
#[macro_use]
extern crate criterion;
use criterion::{Criterion, Throughput};
use std::hint;
use arrow::array::*;
use arrow::util::bench_util::*;
use arrow_array::types::{Int8Type, Int16Type, Int32Type, Int64Type};
const BATCH_SIZE: usize = 64 * 1024;
/// Run [`ArrayIter::fold`] while using black_box on each item and the result of the cb to prevent compiler optimizations.
fn fold_black_box_item_and_cb_res<ArrayAcc, F, B>(array: ArrayAcc, init: B, mut f: F)
where
ArrayAcc: ArrayAccessor,
F: FnMut(B, Option<ArrayAcc::Item>) -> B,
{
let result = ArrayIter::new(array).fold(hint::black_box(init), |acc, item| {
let res = f(acc, hint::black_box(item));
hint::black_box(res)
});
hint::black_box(result);
}
/// Run [`ArrayIter::fold`] while using black_box on each item to prevent compiler optimizations.
fn fold_black_box_item<ArrayAcc, F, B>(array: ArrayAcc, init: B, mut f: F)
where
ArrayAcc: ArrayAccessor,
F: FnMut(B, Option<ArrayAcc::Item>) -> B,
{
let result = ArrayIter::new(array).fold(hint::black_box(init), |acc, item| {
f(acc, hint::black_box(item))
});
hint::black_box(result);
}
/// Run [`ArrayIter::fold`] without using black_box on each item, but only on the result
/// to see if the compiler can do more optimizations.
fn fold_black_box_result<ArrayAcc, F, B>(array: ArrayAcc, init: B, f: F)
where
ArrayAcc: ArrayAccessor,
F: FnMut(B, Option<ArrayAcc::Item>) -> B,
{
let result = ArrayIter::new(array).fold(hint::black_box(init), f);
hint::black_box(result);
}
/// Run [`ArrayIter::any`] while using black_box on each item and the predicate return value to prevent compiler optimizations.
fn any_black_box_item_and_predicate<ArrayAcc>(
array: ArrayAcc,
mut any_predicate: impl FnMut(Option<ArrayAcc::Item>) -> bool,
) where
ArrayAcc: ArrayAccessor,
{
let any_res = ArrayIter::new(array).any(|item| {
let item = hint::black_box(item);
let res = any_predicate(item);
hint::black_box(res)
});
hint::black_box(any_res);
}
/// Run [`ArrayIter::any`] without using black_box in the loop, but only on the result
/// to see if the compiler can do more optimizations.
fn any_black_box_result<ArrayAcc>(
array: ArrayAcc,
any_predicate: impl FnMut(Option<ArrayAcc::Item>) -> bool,
) where
ArrayAcc: ArrayAccessor,
{
let any_res = ArrayIter::new(array).any(any_predicate);
hint::black_box(any_res);
}
/// Benchmark [`ArrayIter`] functions,
///
/// The passed `predicate_that_will_always_evaluate_to_false` function should be a predicate
/// that always returns `false` to ensure that the full array is always iterated over.
///
/// The predicate function should:
/// 1. always return false
/// 2. be impossible for the compiler to optimize away
/// 3. not use `hint::black_box` internally (unless impossible) to allow for more compiler optimizations
///
/// the way to achieve this is to make the predicate check for a value that is not presented in the array.
///
/// The reason for these requirements is that we want to iterate over the entire array while
/// letting the compiler have room for optimizations so it will be more representative of real world usage.
fn benchmark_array_iter<ArrayAcc, FoldFn, FoldInit>(
c: &mut Criterion,
name: &str,
nonnull_array: ArrayAcc,
nullable_array: ArrayAcc,
fold_init: FoldInit,
fold_fn: FoldFn,
predicate_that_will_always_evaluate_to_false: impl Fn(Option<ArrayAcc::Item>) -> bool,
) where
ArrayAcc: ArrayAccessor + Copy,
FoldInit: Copy,
FoldFn: Fn(FoldInit, Option<ArrayAcc::Item>) -> FoldInit,
{
let predicate_that_will_always_evaluate_to_false =
&predicate_that_will_always_evaluate_to_false;
let fold_fn = &fold_fn;
// Assert always false return false
{
let found = ArrayIter::new(nonnull_array).any(predicate_that_will_always_evaluate_to_false);
assert!(!found, "The predicate must always evaluate to false");
}
{
let found =
ArrayIter::new(nullable_array).any(predicate_that_will_always_evaluate_to_false);
assert!(!found, "The predicate must always evaluate to false");
}
c.benchmark_group(name)
.throughput(Throughput::Elements(BATCH_SIZE as u64))
// Most of the Rust default iterator functions are implemented on top of 2 functions:
// `fold` and `try_fold`
// so we are benchmarking `fold` first
.bench_function("nonnull fold black box item and fold result", |b| {
b.iter(|| fold_black_box_item_and_cb_res(nonnull_array, fold_init, fold_fn))
})
.bench_function("nonnull fold black box item", |b| {
b.iter(|| fold_black_box_item(nonnull_array, fold_init, fold_fn))
})
.bench_function("nonnull fold black box only result", |b| {
b.iter(|| fold_black_box_result(nonnull_array, fold_init, fold_fn))
})
.bench_function("null fold black box item and fold result", |b| {
b.iter(|| fold_black_box_item_and_cb_res(nullable_array, fold_init, fold_fn))
})
.bench_function("null fold black box item", |b| {
b.iter(|| fold_black_box_item(nullable_array, fold_init, fold_fn))
})
.bench_function("null fold black box only result", |b| {
b.iter(|| fold_black_box_result(nullable_array, fold_init, fold_fn))
})
// Due to `try_fold` not being available in stable Rust,
// we are benchmarking `any` instead which the default Rust implementation
// uses `try_fold` under the hood.
.bench_function("nonnull any black box item and predicate", |b| {
b.iter(|| {
any_black_box_item_and_predicate(
nonnull_array,
predicate_that_will_always_evaluate_to_false,
)
})
})
.bench_function("nonnull any black box only result", |b| {
b.iter(|| {
any_black_box_result(nonnull_array, predicate_that_will_always_evaluate_to_false)
})
})
.bench_function("null any black box item and predicate", |b| {
b.iter(|| {
any_black_box_item_and_predicate(
nullable_array,
predicate_that_will_always_evaluate_to_false,
)
})
})
.bench_function("null any black box only result", |b| {
b.iter(|| {
any_black_box_result(nullable_array, predicate_that_will_always_evaluate_to_false)
})
});
}
/// Replace all occurrences of `item_to_replace` with `replace_with` in the given `PrimitiveArray`.
/// will make it so we can filter by missing value
fn replace_primitive_value<T>(
array: PrimitiveArray<T>,
item_to_replace: T::Native,
replace_with: T::Native,
) -> PrimitiveArray<T>
where
T: ArrowPrimitiveType,
<T as ArrowPrimitiveType>::Native: Eq,
{
array.unary(|item| {
if item == item_to_replace {
replace_with
} else {
item
}
})
}
fn add_benchmark(c: &mut Criterion) {
benchmark_array_iter(
c,
"int8",
&replace_primitive_value(create_primitive_array::<Int8Type>(BATCH_SIZE, 0.0), 42, 1),
&replace_primitive_value(create_primitive_array::<Int8Type>(BATCH_SIZE, 0.5), 42, 1),
// fold init
0i8,
// fold function
|acc, item| acc.wrapping_add(item.unwrap_or_default()),
// predicate that will always evaluate to false while allowing us to avoid using hint::black_box and let the compiler optimize more
|item| item == Some(42),
);
benchmark_array_iter(
c,
"int16",
&replace_primitive_value(create_primitive_array::<Int16Type>(BATCH_SIZE, 0.0), 42, 1),
&replace_primitive_value(create_primitive_array::<Int16Type>(BATCH_SIZE, 0.5), 42, 1),
// fold init
0i16,
// fold function
|acc, item| acc.wrapping_add(item.unwrap_or_default()),
// predicate that will always evaluate to false while allowing us to avoid using hint::black_box and let the compiler optimize more
|item| item == Some(42),
);
benchmark_array_iter(
c,
"int32",
&replace_primitive_value(create_primitive_array::<Int32Type>(BATCH_SIZE, 0.0), 42, 1),
&replace_primitive_value(create_primitive_array::<Int32Type>(BATCH_SIZE, 0.5), 42, 1),
// fold init
0i32,
// fold function
|acc, item| acc.wrapping_add(item.unwrap_or_default()),
// predicate that will always evaluate to false while allowing us to avoid using hint::black_box and let the compiler optimize more
|item| item == Some(42),
);
benchmark_array_iter(
c,
"int64",
&replace_primitive_value(create_primitive_array::<Int64Type>(BATCH_SIZE, 0.0), 42, 1),
&replace_primitive_value(create_primitive_array::<Int64Type>(BATCH_SIZE, 0.5), 42, 1),
// fold init
0i64,
// fold function
|acc, item| acc.wrapping_add(item.unwrap_or_default()),
// predicate that will always evaluate to false while allowing us to avoid using hint::black_box and let the compiler optimize more
|item| item == Some(42),
);
benchmark_array_iter(
c,
"string with len 16",
&create_string_array_with_len::<i32>(BATCH_SIZE, 0.0, 16),
&create_string_array_with_len::<i32>(BATCH_SIZE, 0.5, 16),
// fold init
0_usize,
// fold function
|acc, item| acc.wrapping_add(item.map(|item| item.len()).unwrap_or_default()),
// predicate that will always evaluate to false while allowing us to avoid using hint::black_box and let the compiler optimize more
|item| item.is_some_and(|item| item.is_empty()),
);
benchmark_array_iter(
c,
"string view with len 16",
&create_string_view_array_with_len(BATCH_SIZE, 0.0, 16, false),
&create_string_view_array_with_len(BATCH_SIZE, 0.5, 16, false),
// fold init
0_usize,
// fold function
|acc, item| acc.wrapping_add(item.map(|item| item.len()).unwrap_or_default()),
// predicate that will always evaluate to false while allowing us to avoid using hint::black_box and let the compiler optimize more
|item| item.is_some_and(|item| item.is_empty()),
);
benchmark_array_iter(
c,
"boolean mixed true and false",
&create_boolean_array(BATCH_SIZE, 0.0, 0.5),
&create_boolean_array(BATCH_SIZE, 0.5, 0.5),
// fold init
0_usize,
// fold function
|acc, item| acc.wrapping_add(item.unwrap_or_default() as usize),
// Must use black_box here as this can be optimized away
|_item| hint::black_box(false),
);
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
+52
View File
@@ -0,0 +1,52 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use criterion::Criterion;
extern crate arrow;
use arrow::array::*;
use std::sync::Arc;
fn create_array_slice(array: &ArrayRef, length: usize) -> ArrayRef {
array.slice(0, length)
}
fn create_array_with_nulls(size: usize) -> ArrayRef {
let array: Float64Array = (0..size)
.map(|i| if i % 2 == 0 { Some(1.0) } else { None })
.collect();
Arc::new(array)
}
fn array_slice_benchmark(c: &mut Criterion) {
let array = create_array_with_nulls(4096);
c.bench_function("array_slice 128", |b| {
b.iter(|| create_array_slice(&array, 128))
});
c.bench_function("array_slice 512", |b| {
b.iter(|| create_array_slice(&array, 512))
});
c.bench_function("array_slice 2048", |b| {
b.iter(|| create_array_slice(&array, 2048))
});
}
criterion_group!(benches, array_slice_benchmark);
criterion_main!(benches);
+47
View File
@@ -0,0 +1,47 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use criterion::Criterion;
extern crate arrow;
use arrow::{array::*, compute::kernels::length::bit_length};
use std::hint;
fn bench_bit_length(array: &StringArray) {
hint::black_box(bit_length(array).unwrap());
}
fn add_benchmark(c: &mut Criterion) {
fn double_vec<T: Clone>(v: Vec<T>) -> Vec<T> {
[&v[..], &v[..]].concat()
}
// double ["hello", " ", "world", "!"] 10 times
let mut values = vec!["one", "on", "o", ""];
for _ in 0..10 {
values = double_vec(values);
}
let array = StringArray::from(values);
c.bench_function("bit_length", |b| b.iter(|| bench_bit_length(&array)));
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
+118
View File
@@ -0,0 +1,118 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use arrow::compute::kernels::bitwise::{
bitwise_and, bitwise_and_scalar, bitwise_not, bitwise_or, bitwise_or_scalar, bitwise_xor,
bitwise_xor_scalar,
};
use arrow::datatypes::Int64Type;
use criterion::Criterion;
use rand::RngCore;
use std::hint;
extern crate arrow;
use arrow::util::bench_util::create_primitive_array;
use arrow::util::test_util::seedable_rng;
fn bitwise_array_benchmark(c: &mut Criterion) {
let size = 64 * 1024_usize;
let left_without_null = create_primitive_array::<Int64Type>(size, 0 as f32);
let right_without_null = create_primitive_array::<Int64Type>(size, 0 as f32);
let left_with_null = create_primitive_array::<Int64Type>(size, 0.2_f32);
let right_with_null = create_primitive_array::<Int64Type>(size, 0.2_f32);
// array and
let mut group = c.benchmark_group("bench bitwise array: and");
group.bench_function("bitwise array and, no nulls", |b| {
b.iter(|| hint::black_box(bitwise_and(&left_without_null, &right_without_null).unwrap()))
});
group.bench_function("bitwise array and, 20% nulls", |b| {
b.iter(|| hint::black_box(bitwise_and(&left_with_null, &right_with_null).unwrap()))
});
group.finish();
// array or
let mut group = c.benchmark_group("bench bitwise: or");
group.bench_function("bitwise array or, no nulls", |b| {
b.iter(|| hint::black_box(bitwise_or(&left_without_null, &right_without_null).unwrap()))
});
group.bench_function("bitwise array or, 20% nulls", |b| {
b.iter(|| hint::black_box(bitwise_or(&left_with_null, &right_with_null).unwrap()))
});
group.finish();
// xor
let mut group = c.benchmark_group("bench bitwise: xor");
group.bench_function("bitwise array xor, no nulls", |b| {
b.iter(|| hint::black_box(bitwise_xor(&left_without_null, &right_without_null).unwrap()))
});
group.bench_function("bitwise array xor, 20% nulls", |b| {
b.iter(|| hint::black_box(bitwise_xor(&left_with_null, &right_with_null).unwrap()))
});
group.finish();
// not
let mut group = c.benchmark_group("bench bitwise: not");
group.bench_function("bitwise array not, no nulls", |b| {
b.iter(|| hint::black_box(bitwise_not(&left_without_null).unwrap()))
});
group.bench_function("bitwise array not, 20% nulls", |b| {
b.iter(|| hint::black_box(bitwise_not(&left_with_null).unwrap()))
});
group.finish();
}
fn bitwise_array_scalar_benchmark(c: &mut Criterion) {
let size = 64 * 1024_usize;
let array_without_null = create_primitive_array::<Int64Type>(size, 0 as f32);
let array_with_null = create_primitive_array::<Int64Type>(size, 0.2_f32);
let scalar = seedable_rng().next_u64() as i64;
// array scalar and
let mut group = c.benchmark_group("bench bitwise array scalar: and");
group.bench_function("bitwise array scalar and, no nulls", |b| {
b.iter(|| hint::black_box(bitwise_and_scalar(&array_without_null, scalar).unwrap()))
});
group.bench_function("bitwise array and, 20% nulls", |b| {
b.iter(|| hint::black_box(bitwise_and_scalar(&array_with_null, scalar).unwrap()))
});
group.finish();
// array scalar or
let mut group = c.benchmark_group("bench bitwise array scalar: or");
group.bench_function("bitwise array scalar or, no nulls", |b| {
b.iter(|| hint::black_box(bitwise_or_scalar(&array_without_null, scalar).unwrap()))
});
group.bench_function("bitwise array scalar or, 20% nulls", |b| {
b.iter(|| hint::black_box(bitwise_or_scalar(&array_with_null, scalar).unwrap()))
});
group.finish();
// array scalar xor
let mut group = c.benchmark_group("bench bitwise array scalar: xor");
group.bench_function("bitwise array scalar xor, no nulls", |b| {
b.iter(|| hint::black_box(bitwise_xor_scalar(&array_without_null, scalar).unwrap()))
});
group.bench_function("bitwise array scalar xor, 20% nulls", |b| {
b.iter(|| hint::black_box(bitwise_xor_scalar(&array_with_null, scalar).unwrap()))
});
group.finish();
}
criterion_group!(
benches,
bitwise_array_benchmark,
bitwise_array_scalar_benchmark
);
criterion_main!(benches);
@@ -0,0 +1,54 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use arrow::array::BooleanBufferBuilder;
use criterion::{Criterion, criterion_group, criterion_main};
use rand::{Rng, rng};
fn rand_bytes(len: usize) -> Vec<u8> {
let mut rng = rng();
let mut buf = vec![0_u8; len];
rng.fill(buf.as_mut_slice());
buf
}
fn boolean_append_packed(c: &mut Criterion) {
let mut rng = rng();
let source = rand_bytes(1024);
let ranges: Vec<_> = (0..100)
.map(|_| {
let start: usize = rng.random_range(0..1024 * 8);
let end: usize = rng.random_range(start..1024 * 8);
start..end
})
.collect();
let total_bits: usize = ranges.iter().map(|x| x.end - x.start).sum();
c.bench_function("boolean_append_packed", |b| {
b.iter(|| {
let mut buffer = BooleanBufferBuilder::new(total_bits);
for range in &ranges {
buffer.append_packed_range(range.clone(), &source);
}
assert_eq!(buffer.len(), total_bits);
})
});
}
criterion_group!(benches, boolean_append_packed);
criterion_main!(benches);
+83
View File
@@ -0,0 +1,83 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use criterion::Criterion;
use arrow::util::bench_util::create_boolean_array;
extern crate arrow;
use arrow::array::*;
use arrow::compute::kernels::boolean as boolean_kernels;
use std::hint;
fn bench_and(lhs: &BooleanArray, rhs: &BooleanArray) {
hint::black_box(boolean_kernels::and(lhs, rhs).unwrap());
}
fn bench_or(lhs: &BooleanArray, rhs: &BooleanArray) {
hint::black_box(boolean_kernels::or(lhs, rhs).unwrap());
}
fn bench_not(array: &BooleanArray) {
hint::black_box(boolean_kernels::not(array).unwrap());
}
fn add_benchmark(c: &mut Criterion) {
// allocate arrays of 32K elements
let size = 2usize.pow(15);
// Note we allocate all arrays before the benchmark to ensure the allocation of the arrays
// is not affected by allocations that happen during the benchmarked operation.
let array1 = create_boolean_array(size, 0.0, 0.5);
let array2 = create_boolean_array(size, 0.0, 0.5);
// Slice by 1 (not aligned to byte (8 bit) or word (64 bit) boundaries)
let offset = 1;
let array1_sliced_1 = array1.slice(offset, size - offset);
let array2_sliced_1 = array2.slice(offset, size - offset);
// Slice by 24 (aligned on byte (8 bit) but not word (64 bit) boundaries)
let offset = 24;
let array1_sliced_24 = array1.slice(offset, size - offset);
let array2_sliced_24 = array2.slice(offset, size - offset);
c.bench_function("and", |b| b.iter(|| bench_and(&array1, &array2)));
c.bench_function("or", |b| b.iter(|| bench_or(&array1, &array2)));
c.bench_function("not", |b| b.iter(|| bench_not(&array1)));
c.bench_function("and_sliced_1", |b| {
b.iter(|| bench_and(&array1_sliced_1, &array2_sliced_1))
});
c.bench_function("or_sliced_1", |b| {
b.iter(|| bench_or(&array1_sliced_1, &array2_sliced_1))
});
c.bench_function("not_sliced_1", |b| b.iter(|| bench_not(&array1_sliced_1)));
c.bench_function("and_sliced_24", |b| {
b.iter(|| bench_and(&array1_sliced_24, &array2_sliced_24))
});
c.bench_function("or_sliced_24", |b| {
b.iter(|| bench_or(&array1_sliced_24, &array2_sliced_24))
});
c.bench_function("not_slice_24", |b| b.iter(|| bench_not(&array1_sliced_24)));
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
+99
View File
@@ -0,0 +1,99 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use criterion::{Criterion, Throughput};
extern crate arrow;
use arrow::buffer::{Buffer, MutableBuffer, buffer_bin_and, buffer_bin_or, buffer_unary_not};
use std::hint;
/// Helper function to create arrays
fn create_buffer(size: usize) -> Buffer {
let mut result = MutableBuffer::new(size).with_bitset(size, false);
for i in 0..size {
result.as_slice_mut()[i] = 0b01010101 << i << (i % 4);
}
result.into()
}
fn bench_buffer_and(left: &Buffer, right: &Buffer) {
hint::black_box(buffer_bin_and(left, 0, right, 0, left.len() * 8));
}
fn bench_buffer_or(left: &Buffer, right: &Buffer) {
hint::black_box(buffer_bin_or(left, 0, right, 0, left.len() * 8));
}
fn bench_buffer_not(buffer: &Buffer) {
hint::black_box(buffer_unary_not(buffer, 0, buffer.len() * 8));
}
fn bench_buffer_and_with_offsets(
left: &Buffer,
left_offset: usize,
right: &Buffer,
right_offset: usize,
len: usize,
) {
hint::black_box(buffer_bin_and(left, left_offset, right, right_offset, len));
}
fn bench_buffer_or_with_offsets(
left: &Buffer,
left_offset: usize,
right: &Buffer,
right_offset: usize,
len: usize,
) {
hint::black_box(buffer_bin_or(left, left_offset, right, right_offset, len));
}
fn bench_buffer_not_with_offsets(buffer: &Buffer, offset: usize, len: usize) {
hint::black_box(buffer_unary_not(buffer, offset, len));
}
fn bit_ops_benchmark(c: &mut Criterion) {
let left = create_buffer(512 * 10);
let right = create_buffer(512 * 10);
c.benchmark_group("buffer_binary_ops")
.throughput(Throughput::Bytes(3 * left.len() as u64))
.bench_function("and", |b| b.iter(|| bench_buffer_and(&left, &right)))
.bench_function("or", |b| b.iter(|| bench_buffer_or(&left, &right)))
.bench_function("and_with_offset", |b| {
b.iter(|| bench_buffer_and_with_offsets(&left, 1, &right, 2, left.len() * 8 - 5))
})
.bench_function("or_with_offset", |b| {
b.iter(|| bench_buffer_or_with_offsets(&left, 1, &right, 2, left.len() * 8 - 5))
});
c.benchmark_group("buffer_unary_ops")
.throughput(Throughput::Bytes(2 * left.len() as u64))
.bench_function("not", |b| b.iter(|| bench_buffer_not(&left)))
.bench_function("not_with_offset", |b| {
b.iter(|| bench_buffer_not_with_offsets(&left, 1, left.len() * 8 - 5))
});
}
criterion_group!(benches, bit_ops_benchmark);
criterion_main!(benches);
+184
View File
@@ -0,0 +1,184 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use arrow::util::test_util::seedable_rng;
use criterion::Criterion;
use rand::Rng;
use rand::distr::Uniform;
extern crate arrow;
use arrow::{
buffer::{Buffer, MutableBuffer},
datatypes::ToByteSlice,
};
use std::hint;
fn mutable_buffer_from_iter(data: &[Vec<bool>]) -> Vec<Buffer> {
hint::black_box(
data.iter()
.map(|vec| vec.iter().copied().collect::<MutableBuffer>().into())
.collect::<Vec<_>>(),
)
}
fn buffer_from_iter(data: &[Vec<bool>]) -> Vec<Buffer> {
hint::black_box(
data.iter()
.map(|vec| vec.iter().copied().collect::<Buffer>())
.collect::<Vec<_>>(),
)
}
fn mutable_buffer_iter_bitset(data: &[Vec<bool>]) -> Vec<Buffer> {
hint::black_box({
data.iter()
.map(|datum| {
let mut result =
MutableBuffer::new(data.len().div_ceil(8)).with_bitset(datum.len(), false);
for (i, value) in datum.iter().enumerate() {
if *value {
unsafe {
arrow::util::bit_util::set_bit_raw(result.as_mut_ptr(), i);
}
}
}
result.into()
})
.collect::<Vec<_>>()
})
}
fn mutable_iter_extend_from_slice(data: &[Vec<u32>], capacity: usize) -> Buffer {
hint::black_box({
let mut result = MutableBuffer::new(capacity);
data.iter().for_each(|vec| {
vec.iter()
.for_each(|elem| result.extend_from_slice(elem.to_byte_slice()))
});
result.into()
})
}
fn mutable_buffer(data: &[Vec<u32>], capacity: usize) -> Buffer {
hint::black_box({
let mut result = MutableBuffer::new(capacity);
data.iter().for_each(|vec| result.extend_from_slice(vec));
result.into()
})
}
fn mutable_buffer_extend(data: &[Vec<u32>], capacity: usize) -> Buffer {
hint::black_box({
let mut result = MutableBuffer::new(capacity);
data.iter()
.for_each(|vec| result.extend(vec.iter().copied()));
result.into()
})
}
fn from_slice(data: &[Vec<u32>], capacity: usize) -> Buffer {
hint::black_box({
let mut a = Vec::<u32>::with_capacity(capacity);
data.iter().for_each(|vec| a.extend(vec));
Buffer::from(a.to_byte_slice())
})
}
fn create_data(size: usize) -> Vec<Vec<u32>> {
let rng = &mut seedable_rng();
let range = Uniform::new(0, 33).unwrap();
(0..size)
.map(|_| {
let size = rng.sample(range);
seedable_rng()
.sample_iter(&range)
.take(size as usize)
.collect()
})
.collect()
}
fn create_data_bool(size: usize) -> Vec<Vec<bool>> {
let rng = &mut seedable_rng();
let range = Uniform::new(0, 33).unwrap();
(0..size)
.map(|_| {
let size = rng.sample(range);
seedable_rng()
.sample_iter(&range)
.take(size as usize)
.map(|x| x > 15)
.collect()
})
.collect()
}
fn benchmark(c: &mut Criterion) {
let size = 2usize.pow(15);
let data = create_data(size);
let bool_data = create_data_bool(size);
let cap = data.iter().map(|i| i.len()).sum();
let byte_cap = cap * std::mem::size_of::<u32>();
c.bench_function("mutable iter extend_from_slice", |b| {
b.iter(|| mutable_iter_extend_from_slice(hint::black_box(&data), hint::black_box(0)))
});
c.bench_function("mutable", |b| {
b.iter(|| mutable_buffer(hint::black_box(&data), hint::black_box(0)))
});
c.bench_function("mutable extend", |b| {
b.iter(|| mutable_buffer_extend(&data, 0))
});
c.bench_function("mutable prepared", |b| {
b.iter(|| mutable_buffer(hint::black_box(&data), hint::black_box(byte_cap)))
});
c.bench_function("from_slice", |b| {
b.iter(|| from_slice(hint::black_box(&data), hint::black_box(0)))
});
c.bench_function("from_slice prepared", |b| {
b.iter(|| from_slice(hint::black_box(&data), hint::black_box(cap)))
});
c.bench_function("MutableBuffer iter bitset", |b| {
b.iter(|| mutable_buffer_iter_bitset(hint::black_box(&bool_data)))
});
c.bench_function("MutableBuffer::from_iter bool", |b| {
b.iter(|| mutable_buffer_from_iter(hint::black_box(&bool_data)))
});
c.bench_function("Buffer::from_iter bool", |b| {
b.iter(|| buffer_from_iter(hint::black_box(&bool_data)))
});
}
criterion_group!(benches, benchmark);
criterion_main!(benches);
+195
View File
@@ -0,0 +1,195 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
extern crate arrow;
extern crate criterion;
extern crate rand;
use std::mem::size_of;
use criterion::*;
use rand::distr::StandardUniform;
use arrow::array::*;
use arrow::util::test_util::seedable_rng;
use arrow_buffer::i256;
use rand::Rng;
use std::hint;
// Build arrays with 512k elements.
const BATCH_SIZE: usize = 8 << 10;
const NUM_BATCHES: usize = 64;
fn bench_primitive(c: &mut Criterion) {
let data: [i64; BATCH_SIZE] = [100; BATCH_SIZE];
let mut group = c.benchmark_group("bench_primitive");
group.throughput(Throughput::Bytes(
((data.len() * NUM_BATCHES * size_of::<i64>()) as u32).into(),
));
group.bench_function("bench_primitive", |b| {
b.iter(|| {
let mut builder = Int64Builder::with_capacity(64);
for _ in 0..NUM_BATCHES {
builder.append_slice(&data[..]);
}
hint::black_box(builder.finish());
})
});
group.finish();
}
fn bench_primitive_nulls(c: &mut Criterion) {
let mut group = c.benchmark_group("bench_primitive_nulls");
group.bench_function("bench_primitive_nulls", |b| {
b.iter(|| {
let mut builder = UInt8Builder::with_capacity(64);
for _ in 0..NUM_BATCHES * BATCH_SIZE {
builder.append_null();
}
hint::black_box(builder.finish());
})
});
group.finish();
}
fn bench_bool(c: &mut Criterion) {
let data: Vec<bool> = seedable_rng()
.sample_iter(&StandardUniform)
.take(BATCH_SIZE)
.collect();
let data_len = data.len();
let mut group = c.benchmark_group("bench_bool");
group.throughput(Throughput::Bytes(
((data_len * NUM_BATCHES * size_of::<bool>()) as u32).into(),
));
group.bench_function("bench_bool", |b| {
b.iter(|| {
let mut builder = BooleanBuilder::with_capacity(64);
for _ in 0..NUM_BATCHES {
builder.append_slice(&data[..]);
}
hint::black_box(builder.finish());
})
});
group.finish();
}
fn bench_string(c: &mut Criterion) {
const SAMPLE_STRING: &str = "sample string";
let mut group = c.benchmark_group("bench_primitive");
group.throughput(Throughput::Bytes(
((BATCH_SIZE * NUM_BATCHES * SAMPLE_STRING.len()) as u32).into(),
));
group.bench_function("bench_string", |b| {
b.iter(|| {
let mut builder = StringBuilder::new();
for _ in 0..NUM_BATCHES * BATCH_SIZE {
builder.append_value(SAMPLE_STRING);
}
hint::black_box(builder.finish());
})
});
group.finish();
}
fn bench_decimal32(c: &mut Criterion) {
c.bench_function("bench_decimal32_builder", |b| {
b.iter(|| {
let mut rng = rand::rng();
let mut decimal_builder = Decimal32Builder::with_capacity(BATCH_SIZE);
for _ in 0..BATCH_SIZE {
decimal_builder.append_value(rng.random_range::<i32, _>(0..999999999));
}
hint::black_box(
decimal_builder
.finish()
.with_precision_and_scale(9, 0)
.unwrap(),
);
})
});
}
fn bench_decimal64(c: &mut Criterion) {
c.bench_function("bench_decimal64_builder", |b| {
b.iter(|| {
let mut rng = rand::rng();
let mut decimal_builder = Decimal64Builder::with_capacity(BATCH_SIZE);
for _ in 0..BATCH_SIZE {
decimal_builder.append_value(rng.random_range::<i64, _>(0..9999999999));
}
hint::black_box(
decimal_builder
.finish()
.with_precision_and_scale(18, 0)
.unwrap(),
);
})
});
}
fn bench_decimal128(c: &mut Criterion) {
c.bench_function("bench_decimal128_builder", |b| {
b.iter(|| {
let mut rng = rand::rng();
let mut decimal_builder = Decimal128Builder::with_capacity(BATCH_SIZE);
for _ in 0..BATCH_SIZE {
decimal_builder.append_value(rng.random_range::<i128, _>(0..9999999999));
}
hint::black_box(
decimal_builder
.finish()
.with_precision_and_scale(38, 0)
.unwrap(),
);
})
});
}
fn bench_decimal256(c: &mut Criterion) {
c.bench_function("bench_decimal256_builder", |b| {
b.iter(|| {
let mut rng = rand::rng();
let mut decimal_builder = Decimal256Builder::with_capacity(BATCH_SIZE);
for _ in 0..BATCH_SIZE {
decimal_builder
.append_value(i256::from_i128(rng.random_range::<i128, _>(0..99999999999)));
}
hint::black_box(
decimal_builder
.finish()
.with_precision_and_scale(76, 10)
.unwrap(),
);
})
});
}
criterion_group!(
benches,
bench_primitive,
bench_primitive_nulls,
bench_bool,
bench_string,
bench_decimal32,
bench_decimal64,
bench_decimal128,
bench_decimal256,
);
criterion_main!(benches);
+405
View File
@@ -0,0 +1,405 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use criterion::Criterion;
use rand::Rng;
use rand::distr::{Distribution, StandardUniform, Uniform};
use std::hint;
use chrono::DateTime;
use std::sync::Arc;
extern crate arrow;
use arrow::array::*;
use arrow::compute::cast;
use arrow::datatypes::*;
use arrow::util::bench_util::*;
use arrow::util::test_util::seedable_rng;
fn build_array<T: ArrowPrimitiveType>(size: usize) -> ArrayRef
where
StandardUniform: Distribution<T::Native>,
{
let array = create_primitive_array::<T>(size, 0.1);
Arc::new(array)
}
fn build_utf8_date_array(size: usize, with_nulls: bool) -> ArrayRef {
use chrono::NaiveDate;
// use random numbers to avoid spurious compiler optimizations wrt to branching
let mut rng = seedable_rng();
let mut builder = StringBuilder::new();
let range = Uniform::new(0, 737776).unwrap();
for _ in 0..size {
if with_nulls && rng.random::<f32>() > 0.8 {
builder.append_null();
} else {
let string = NaiveDate::from_num_days_from_ce_opt(rng.sample(range))
.unwrap()
.format("%Y-%m-%d")
.to_string();
builder.append_value(&string);
}
}
Arc::new(builder.finish())
}
fn build_utf8_date_time_array(size: usize, with_nulls: bool) -> ArrayRef {
// use random numbers to avoid spurious compiler optimizations wrt to branching
let mut rng = seedable_rng();
let mut builder = StringBuilder::new();
let range = Uniform::new(0, 1608071414123).unwrap();
for _ in 0..size {
if with_nulls && rng.random::<f32>() > 0.8 {
builder.append_null();
} else {
let string = DateTime::from_timestamp(rng.sample(range), 0)
.unwrap()
.format("%Y-%m-%dT%H:%M:%S")
.to_string();
builder.append_value(&string);
}
}
Arc::new(builder.finish())
}
fn build_decimal32_array(size: usize, precision: u8, scale: i8) -> ArrayRef {
let mut rng = seedable_rng();
let mut builder = Decimal32Builder::with_capacity(size);
for _ in 0..size {
builder.append_value(rng.random_range::<i32, _>(0..1000000));
}
Arc::new(
builder
.finish()
.with_precision_and_scale(precision, scale)
.unwrap(),
)
}
fn build_decimal64_array(size: usize, precision: u8, scale: i8) -> ArrayRef {
let mut rng = seedable_rng();
let mut builder = Decimal64Builder::with_capacity(size);
for _ in 0..size {
builder.append_value(rng.random_range::<i64, _>(0..1000000000));
}
Arc::new(
builder
.finish()
.with_precision_and_scale(precision, scale)
.unwrap(),
)
}
fn build_decimal128_array(size: usize, precision: u8, scale: i8) -> ArrayRef {
let mut rng = seedable_rng();
let mut builder = Decimal128Builder::with_capacity(size);
for _ in 0..size {
builder.append_value(rng.random_range::<i128, _>(0..1000000000));
}
Arc::new(
builder
.finish()
.with_precision_and_scale(precision, scale)
.unwrap(),
)
}
fn build_decimal256_array(size: usize, precision: u8, scale: i8) -> ArrayRef {
let mut rng = seedable_rng();
let mut builder = Decimal256Builder::with_capacity(size);
let mut bytes = [0; 32];
for _ in 0..size {
let num = rng.random_range::<i128, _>(0..1000000000);
bytes[0..16].clone_from_slice(&num.to_le_bytes());
builder.append_value(i256::from_le_bytes(bytes));
}
Arc::new(
builder
.finish()
.with_precision_and_scale(precision, scale)
.unwrap(),
)
}
fn build_string_array(size: usize) -> ArrayRef {
let mut builder = StringBuilder::new();
for v in 0..size {
match v % 3 {
0 => builder.append_value("small"),
1 => builder.append_value("larger string more than 12 bytes"),
_ => builder.append_null(),
}
}
Arc::new(builder.finish())
}
fn build_dict_array(size: usize) -> ArrayRef {
let values = StringArray::from_iter([
Some("small"),
Some("larger string more than 12 bytes"),
None,
]);
let keys = UInt64Array::from_iter((0..size as u64).map(|v| v % 3));
Arc::new(DictionaryArray::new(keys, Arc::new(values)))
}
// cast array from specified primitive array type to desired data type
fn cast_array(array: &ArrayRef, to_type: DataType) {
hint::black_box(cast(array, &to_type).unwrap());
}
fn add_benchmark(c: &mut Criterion) {
let i32_array = build_array::<Int32Type>(512);
let i64_array = build_array::<Int64Type>(512);
let f32_array = build_array::<Float32Type>(512);
let f32_utf8_array = cast(&build_array::<Float32Type>(512), &DataType::Utf8).unwrap();
let f64_array = build_array::<Float64Type>(512);
let date64_array = build_array::<Date64Type>(512);
let date32_array = build_array::<Date32Type>(512);
let time32s_array = build_array::<Time32SecondType>(512);
let time64ns_array = build_array::<Time64NanosecondType>(512);
let time_ns_array = build_array::<TimestampNanosecondType>(512);
let time_ms_array = build_array::<TimestampMillisecondType>(512);
let utf8_date_array = build_utf8_date_array(512, true);
let utf8_date_time_array = build_utf8_date_time_array(512, true);
let decimal32_array = build_decimal32_array(512, 9, 3);
let decimal64_array = build_decimal64_array(512, 10, 3);
let decimal128_array = build_decimal128_array(512, 10, 3);
let decimal256_array = build_decimal256_array(512, 50, 3);
let string_array = build_string_array(512);
let wide_string_array = cast(&string_array, &DataType::LargeUtf8).unwrap();
let dict_array = build_dict_array(10_000);
let string_view_array = cast(&dict_array, &DataType::Utf8View).unwrap();
let binary_view_array = cast(&string_view_array, &DataType::BinaryView).unwrap();
c.bench_function("cast int32 to int32 512", |b| {
b.iter(|| cast_array(&i32_array, DataType::Int32))
});
c.bench_function("cast int32 to uint32 512", |b| {
b.iter(|| cast_array(&i32_array, DataType::UInt32))
});
c.bench_function("cast int32 to float32 512", |b| {
b.iter(|| cast_array(&i32_array, DataType::Float32))
});
c.bench_function("cast int32 to float64 512", |b| {
b.iter(|| cast_array(&i32_array, DataType::Float64))
});
c.bench_function("cast int32 to int64 512", |b| {
b.iter(|| cast_array(&i32_array, DataType::Int64))
});
c.bench_function("cast float32 to int32 512", |b| {
b.iter(|| cast_array(&f32_array, DataType::Int32))
});
c.bench_function("cast float64 to float32 512", |b| {
b.iter(|| cast_array(&f64_array, DataType::Float32))
});
c.bench_function("cast float64 to uint64 512", |b| {
b.iter(|| cast_array(&f64_array, DataType::UInt64))
});
c.bench_function("cast int64 to int32 512", |b| {
b.iter(|| cast_array(&i64_array, DataType::Int32))
});
c.bench_function("cast date64 to date32 512", |b| {
b.iter(|| cast_array(&date64_array, DataType::Date32))
});
c.bench_function("cast date32 to date64 512", |b| {
b.iter(|| cast_array(&date32_array, DataType::Date64))
});
c.bench_function("cast time32s to time32ms 512", |b| {
b.iter(|| cast_array(&time32s_array, DataType::Time32(TimeUnit::Millisecond)))
});
c.bench_function("cast time32s to time64us 512", |b| {
b.iter(|| cast_array(&time32s_array, DataType::Time64(TimeUnit::Microsecond)))
});
c.bench_function("cast time64ns to time32s 512", |b| {
b.iter(|| cast_array(&time64ns_array, DataType::Time32(TimeUnit::Second)))
});
c.bench_function("cast timestamp_ns to timestamp_s 512", |b| {
b.iter(|| {
cast_array(
&time_ns_array,
DataType::Timestamp(TimeUnit::Nanosecond, None),
)
})
});
c.bench_function("cast timestamp_ms to timestamp_ns 512", |b| {
b.iter(|| {
cast_array(
&time_ms_array,
DataType::Timestamp(TimeUnit::Nanosecond, None),
)
})
});
c.bench_function("cast utf8 to f32", |b| {
b.iter(|| cast_array(&f32_utf8_array, DataType::Float32))
});
c.bench_function("cast i64 to string 512", |b| {
b.iter(|| cast_array(&i64_array, DataType::Utf8))
});
c.bench_function("cast f32 to string 512", |b| {
b.iter(|| cast_array(&f32_array, DataType::Utf8))
});
c.bench_function("cast f64 to string 512", |b| {
b.iter(|| cast_array(&f64_array, DataType::Utf8))
});
c.bench_function("cast timestamp_ms to i64 512", |b| {
b.iter(|| cast_array(&time_ms_array, DataType::Int64))
});
c.bench_function("cast utf8 to date32 512", |b| {
b.iter(|| cast_array(&utf8_date_array, DataType::Date32))
});
c.bench_function("cast utf8 to date64 512", |b| {
b.iter(|| cast_array(&utf8_date_time_array, DataType::Date64))
});
c.bench_function("cast decimal32 to decimal32 512", |b| {
b.iter(|| cast_array(&decimal32_array, DataType::Decimal32(9, 4)))
});
c.bench_function("cast decimal32 to decimal32 512 lower precision", |b| {
b.iter(|| cast_array(&decimal32_array, DataType::Decimal32(6, 5)))
});
c.bench_function("cast decimal32 to decimal64 512", |b| {
b.iter(|| cast_array(&decimal32_array, DataType::Decimal64(11, 5)))
});
c.bench_function("cast decimal64 to decimal32 512", |b| {
b.iter(|| cast_array(&decimal64_array, DataType::Decimal32(9, 2)))
});
c.bench_function("cast decimal64 to decimal64 512", |b| {
b.iter(|| cast_array(&decimal64_array, DataType::Decimal64(12, 4)))
});
c.bench_function("cast decimal128 to decimal128 512", |b| {
b.iter(|| cast_array(&decimal128_array, DataType::Decimal128(30, 5)))
});
c.bench_function("cast decimal128 to decimal128 512 lower precision", |b| {
b.iter(|| cast_array(&decimal128_array, DataType::Decimal128(6, 5)))
});
c.bench_function("cast decimal128 to decimal256 512", |b| {
b.iter(|| cast_array(&decimal128_array, DataType::Decimal256(50, 5)))
});
c.bench_function("cast decimal256 to decimal128 512", |b| {
b.iter(|| cast_array(&decimal256_array, DataType::Decimal128(38, 2)))
});
c.bench_function("cast decimal256 to decimal256 512", |b| {
b.iter(|| cast_array(&decimal256_array, DataType::Decimal256(50, 5)))
});
c.bench_function("cast decimal128 to decimal128 512 with same scale", |b| {
b.iter(|| cast_array(&decimal128_array, DataType::Decimal128(30, 3)))
});
c.bench_function(
"cast decimal128 to decimal128 512 with lower scale (infallible)",
|b| b.iter(|| cast_array(&decimal128_array, DataType::Decimal128(7, -1))),
);
c.bench_function("cast decimal256 to decimal256 512 with same scale", |b| {
b.iter(|| cast_array(&decimal256_array, DataType::Decimal256(60, 3)))
});
c.bench_function("cast dict to string view", |b| {
b.iter(|| cast_array(&dict_array, DataType::Utf8View))
});
c.bench_function("cast string view to dict", |b| {
b.iter(|| {
cast_array(
&string_view_array,
DataType::Dictionary(Box::new(DataType::UInt64), Box::new(DataType::Utf8)),
)
})
});
c.bench_function("cast string view to string", |b| {
b.iter(|| cast_array(&string_view_array, DataType::Utf8))
});
c.bench_function("cast string view to wide string", |b| {
b.iter(|| cast_array(&string_view_array, DataType::LargeUtf8))
});
c.bench_function("cast binary view to string", |b| {
b.iter(|| cast_array(&binary_view_array, DataType::Utf8))
});
c.bench_function("cast binary view to wide string", |b| {
b.iter(|| cast_array(&binary_view_array, DataType::LargeUtf8))
});
c.bench_function("cast string to binary view 512", |b| {
b.iter(|| cast_array(&string_array, DataType::BinaryView))
});
c.bench_function("cast wide string to binary view 512", |b| {
b.iter(|| cast_array(&wide_string_array, DataType::BinaryView))
});
c.bench_function("cast string view to binary view", |b| {
b.iter(|| cast_array(&string_view_array, DataType::BinaryView))
});
c.bench_function("cast binary view to string view", |b| {
b.iter(|| cast_array(&binary_view_array, DataType::Utf8View))
});
c.bench_function("cast string single run to ree<int32>", |b| {
let source_array = StringArray::from(vec!["a"; 8192]);
let array_ref = Arc::new(source_array) as ArrayRef;
let target_type = DataType::RunEndEncoded(
Arc::new(Field::new("run_ends", DataType::Int32, false)),
Arc::new(Field::new("values", DataType::Utf8, true)),
);
b.iter(|| cast(&array_ref, &target_type).unwrap());
});
c.bench_function("cast runs of 10 string to ree<int32>", |b| {
let source_array: Int32Array = (0..8192).map(|i| i / 10).collect();
let array_ref = Arc::new(source_array) as ArrayRef;
let target_type = DataType::RunEndEncoded(
Arc::new(Field::new("run_ends", DataType::Int32, false)),
Arc::new(Field::new("values", DataType::Int32, true)),
);
b.iter(|| cast(&array_ref, &target_type).unwrap());
});
c.bench_function("cast runs of 1000 int32s to ree<int32>", |b| {
let source_array: Int32Array = (0..8192).map(|i| i / 1000).collect();
let array_ref = Arc::new(source_array) as ArrayRef;
let target_type = DataType::RunEndEncoded(
Arc::new(Field::new("run_ends", DataType::Int32, false)),
Arc::new(Field::new("values", DataType::Int32, true)),
);
b.iter(|| cast(&array_ref, &target_type).unwrap());
});
c.bench_function("cast no runs of int32s to ree<int32>", |b| {
let source_array: Int32Array = (0..8192).collect();
let array_ref = Arc::new(source_array) as ArrayRef;
let target_type = DataType::RunEndEncoded(
Arc::new(Field::new("run_ends", DataType::Int32, false)),
Arc::new(Field::new("values", DataType::Int32, true)),
);
b.iter(|| cast(&array_ref, &target_type).unwrap());
});
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
+471
View File
@@ -0,0 +1,471 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
//! Benchmarks for the `coalesce` kernels in Arrow.
use arrow::util::bench_util::*;
use std::sync::Arc;
use arrow::array::*;
use arrow_array::types::{Float64Type, Int32Type, TimestampNanosecondType};
use arrow_schema::{DataType, Field, Schema, SchemaRef, TimeUnit};
use arrow_select::coalesce::BatchCoalescer;
use criterion::{Criterion, criterion_group, criterion_main};
/// Benchmarks for generating evently sized output RecordBatches
/// from a sequence of filtered source batches
///
fn add_all_filter_benchmarks(c: &mut Criterion) {
let batch_size = 8192; // 8K rows is a commonly used size for batches
// Multiple primitive types
let primitive_schema = SchemaRef::new(Schema::new(vec![
Field::new("int32_val", DataType::Int32, true),
Field::new("float_val", DataType::Float64, true),
Field::new(
"timestamp_val",
DataType::Timestamp(TimeUnit::Nanosecond, Some("UTC".into())),
true,
),
]));
// Single StringViewArray
let single_schema = SchemaRef::new(Schema::new(vec![Field::new(
"value",
DataType::Utf8View,
true,
)]));
// Mixed primitive, StringViewArray
let mixed_utf8view_schema = SchemaRef::new(Schema::new(vec![
Field::new("int32_val", DataType::Int32, true),
Field::new("float_val", DataType::Float64, true),
Field::new("utf8view_val", DataType::Utf8View, true),
]));
// Mixed primitive, StringArray
let mixed_utf8_schema = SchemaRef::new(Schema::new(vec![
Field::new("int32_val", DataType::Int32, true),
Field::new("float_val", DataType::Float64, true),
Field::new("utf8", DataType::Utf8, true),
]));
// dictionary types
//
let mixed_dict_schema = SchemaRef::new(Schema::new(vec![
Field::new(
"string_dict",
DataType::Dictionary(Box::new(DataType::Int32), Box::new(DataType::Utf8)),
true,
),
Field::new("float_val1", DataType::Float64, true),
Field::new("float_val2", DataType::Float64, true),
// TODO model other dictionary types here (FixedSizeBinary for example)
]));
// Null density: 0, 10%
for null_density in [0.0, 0.1] {
// Selectivity: 0.1%, 1%, 10%, 80%
for selectivity in [0.001, 0.01, 0.1, 0.8] {
FilterBenchmarkBuilder {
c,
name: "primitive",
batch_size,
num_output_batches: 50,
null_density,
selectivity,
max_string_len: 30,
schema: &primitive_schema,
}
.build();
FilterBenchmarkBuilder {
c,
name: "single_utf8view",
batch_size,
num_output_batches: 50,
null_density,
selectivity,
max_string_len: 30,
schema: &single_schema,
}
.build();
// Model mostly short strings, but some longer ones
FilterBenchmarkBuilder {
c,
name: "mixed_utf8view (max_string_len=20)",
batch_size,
num_output_batches: 20,
null_density,
selectivity,
max_string_len: 20,
schema: &mixed_utf8view_schema,
}
.build();
// Model mostly longer strings
FilterBenchmarkBuilder {
c,
name: "mixed_utf8view (max_string_len=128)",
batch_size,
num_output_batches: 20,
null_density,
selectivity,
max_string_len: 128,
schema: &mixed_utf8view_schema,
}
.build();
FilterBenchmarkBuilder {
c,
name: "mixed_utf8",
batch_size,
num_output_batches: 20,
null_density,
selectivity,
max_string_len: 30,
schema: &mixed_utf8_schema,
}
.build();
FilterBenchmarkBuilder {
c,
name: "mixed_dict",
batch_size,
num_output_batches: 10,
null_density,
selectivity,
max_string_len: 30,
schema: &mixed_dict_schema,
}
.build();
}
}
}
criterion_group!(benches, add_all_filter_benchmarks);
criterion_main!(benches);
/// Run the filters with a batch_size, null_density, selectivity, and schema
struct FilterBenchmarkBuilder<'a> {
/// Benchmark criterion instance
c: &'a mut Criterion,
/// Name of the benchmark
name: &'a str,
/// Size of the input and output batches
batch_size: usize,
/// Number of output batches to collect (tuned to keep benchmark time reasonable)
num_output_batches: usize,
/// between 0.0 .. 1.0, percent of data rows (not filter rows) that should be null
null_density: f32,
/// between 0.0 .. 1.0, percent of rows that should be kept by the filter
selectivity: f32,
/// The maximum length of strings in the data stream
///
/// For StringViewArray, strings <= 12 bytes are stored inline, longer
/// strings are stored in a separate buffer so it is important to vary to
/// mix the relative paths
max_string_len: usize,
/// Schema of the data stream
schema: &'a SchemaRef,
}
impl FilterBenchmarkBuilder<'_> {
fn build(self) {
let Self {
c,
name,
batch_size,
num_output_batches,
null_density,
selectivity,
max_string_len,
schema,
} = self;
let filters = FilterStreamBuilder::new()
.with_batch_size(batch_size)
.with_true_density(selectivity)
.with_null_density(0.0) // no nulls in the filter
.build();
let data = DataStreamBuilder::new(Arc::clone(schema))
.with_batch_size(batch_size)
.with_null_density(null_density)
.with_max_string_len(max_string_len)
.build();
// Keep feeding the filter stream into the coalescer until we hit a total number of output batches
let id = format!(
"filter: {name}, {batch_size}, nulls: {null_density}, selectivity: {selectivity}"
);
c.bench_function(&id, |b| {
b.iter(|| {
filter_streams(num_output_batches, filters.clone(), data.clone());
})
});
}
}
/// Pull RecordBatches from a data stream and apply a sequence of
/// filters from a filter stream until we have a specified number of output
/// batches.
fn filter_streams(
mut num_output_batches: usize,
mut filter_stream: FilterStream,
mut data_stream: DataStream,
) {
let schema = data_stream.schema();
let batch_size = data_stream.batch_size();
let mut coalescer = BatchCoalescer::new(Arc::clone(schema), batch_size);
while num_output_batches > 0 {
let filter = filter_stream.next_filter();
let batch = data_stream.next_batch();
coalescer
.push_batch_with_filter(batch.clone(), filter)
.unwrap();
// consume (but discard) the output batch
if coalescer.next_completed_batch().is_some() {
num_output_batches -= 1;
}
}
}
/// Stream of filters to apply to a sequence of input RecordBatches
///
/// This pre-computes a sequence of filters and then repeats it forever.
#[derive(Debug, Clone)]
struct FilterStream {
index: usize,
// arc'd so it is cheaply cloned
batches: Arc<[BooleanArray]>,
}
impl FilterStream {
pub fn next_filter(&mut self) -> &BooleanArray {
let current_index = self.index;
self.index += 1;
if self.index >= self.batches.len() {
self.index = 0; // loop back to the start
}
self.batches
.get(current_index)
.expect("No more filters available")
}
}
#[derive(Debug)]
struct FilterStreamBuilder {
batch_size: usize,
num_batches: usize, // number of unique batches to create
null_density: f32,
true_density: f32,
}
impl FilterStreamBuilder {
fn new() -> Self {
FilterStreamBuilder {
batch_size: 8192, // default batch size
num_batches: 11, // default number of unique batches (different than data stream)
null_density: 0.0, // default null density
true_density: 0.5, // default true density
}
}
/// set the batch size for the filter stream
fn with_batch_size(mut self, batch_size: usize) -> Self {
self.batch_size = batch_size;
self
}
/// set the null density for the filter stream
fn with_null_density(mut self, null_density: f32) -> Self {
assert!((0.0..=1.0).contains(&null_density));
self.null_density = null_density;
self
}
/// set the true density for the filter stream
fn with_true_density(mut self, true_density: f32) -> Self {
assert!((0.0..=1.0).contains(&true_density));
self.true_density = true_density;
self
}
fn build(self) -> FilterStream {
let Self {
batch_size,
num_batches,
null_density,
true_density,
} = self;
let batches = (0..num_batches)
.map(|_| create_boolean_array(batch_size, null_density, true_density))
.collect::<Vec<_>>();
FilterStream {
index: 0,
batches: Arc::from(batches),
}
}
}
#[derive(Debug, Clone)]
struct DataStream {
schema: SchemaRef,
index: usize,
batch_size: usize,
// arc'd so it is cheaply cloned
batches: Arc<[RecordBatch]>,
}
impl DataStream {
/// Returns the schema for this data stream
pub fn schema(&self) -> &SchemaRef {
&self.schema
}
/// Returns the batch size
pub fn batch_size(&self) -> usize {
self.batch_size
}
fn next_batch(&mut self) -> &RecordBatch {
let current_index = self.index;
self.index += 1;
if self.index >= self.batches.len() {
self.index = 0; // loop back to the start
}
self.batches
.get(current_index)
.expect("No more batches available")
}
}
#[derive(Debug, Clone)]
struct DataStreamBuilder {
schema: SchemaRef,
batch_size: usize,
null_density: f32,
num_batches: usize, // number of unique batches to create
max_string_len: usize, // maximum length of strings in the data stream
}
impl DataStreamBuilder {
fn new(schema: SchemaRef) -> Self {
DataStreamBuilder {
schema,
batch_size: 8192,
null_density: 0.0,
num_batches: 10,
max_string_len: 30,
}
}
/// set the batch size for the data stream
fn with_batch_size(mut self, batch_size: usize) -> Self {
self.batch_size = batch_size;
self
}
/// set the null density for the data stream
fn with_null_density(mut self, null_density: f32) -> Self {
assert!((0.0..=1.0).contains(&null_density));
self.null_density = null_density;
self
}
fn with_max_string_len(mut self, max_string_len: usize) -> Self {
self.max_string_len = max_string_len;
self
}
/// build the data stream (not implemented yet)
fn build(self) -> DataStream {
let batches = (0..self.num_batches)
.map(|seed| {
let columns = self
.schema
.fields()
.iter()
.map(|field| self.create_input_array(field, seed as u64))
.collect::<Vec<_>>();
RecordBatch::try_new(self.schema.clone(), columns).unwrap()
})
.collect::<Vec<_>>();
let Self {
schema,
batch_size,
null_density: _,
num_batches: _,
max_string_len: _,
} = self;
DataStream {
schema,
index: 0,
batch_size,
batches: Arc::from(batches),
}
}
fn create_input_array(&self, field: &Field, seed: u64) -> ArrayRef {
match field.data_type() {
DataType::Int32 => Arc::new(create_primitive_array_with_seed::<Int32Type>(
self.batch_size,
self.null_density,
seed,
)),
DataType::Float64 => Arc::new(create_primitive_array_with_seed::<Float64Type>(
self.batch_size,
self.null_density,
seed,
)),
DataType::Timestamp(TimeUnit::Nanosecond, Some(tz)) => Arc::new(
create_primitive_array_with_seed::<TimestampNanosecondType>(
self.batch_size,
self.null_density,
seed,
)
.with_timezone(Arc::clone(tz)),
),
DataType::Utf8 => Arc::new(create_string_array::<i32>(
self.batch_size,
self.null_density,
)), // TODO seed
DataType::Utf8View => {
Arc::new(create_string_view_array_with_max_len(
self.batch_size,
self.null_density,
self.max_string_len,
)) // TODO seed
}
DataType::Dictionary(key_type, value_type)
if key_type.as_ref() == &DataType::Int32
&& value_type.as_ref() == &DataType::Utf8 =>
{
Arc::new(create_string_dict_array::<Int32Type>(
self.batch_size,
self.null_density,
self.max_string_len,
)) // TODO seed
}
_ => panic!("Unsupported data type: {field:?}"),
}
}
}
+536
View File
@@ -0,0 +1,536 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
extern crate arrow;
#[macro_use]
extern crate criterion;
use arrow::compute::kernels::cmp::*;
use arrow::util::bench_util::*;
use arrow::util::test_util::seedable_rng;
use arrow::{array::*, datatypes::Float32Type, datatypes::Int32Type};
use arrow_buffer::IntervalMonthDayNano;
use arrow_string::like::*;
use arrow_string::regexp::regexp_is_match_scalar;
use criterion::Criterion;
use rand::Rng;
use rand::rngs::StdRng;
use std::hint;
const SIZE: usize = 65536;
fn bench_like_utf8_scalar(arr_a: &StringArray, value_b: &str) {
like(arr_a, &StringArray::new_scalar(value_b)).unwrap();
}
fn bench_like_utf8view_scalar(arr_a: &StringViewArray, value_b: &str) {
like(arr_a, &StringViewArray::new_scalar(value_b)).unwrap();
}
fn bench_nlike_utf8_scalar(arr_a: &StringArray, value_b: &str) {
nlike(arr_a, &StringArray::new_scalar(value_b)).unwrap();
}
fn bench_ilike_utf8_scalar(arr_a: &StringArray, value_b: &str) {
ilike(arr_a, &StringArray::new_scalar(value_b)).unwrap();
}
fn bench_nilike_utf8_scalar(arr_a: &StringArray, value_b: &str) {
nilike(arr_a, &StringArray::new_scalar(value_b)).unwrap();
}
fn bench_stringview_regexp_is_match_scalar(arr_a: &StringViewArray, value_b: &str) {
regexp_is_match_scalar(hint::black_box(arr_a), hint::black_box(value_b), None).unwrap();
}
fn bench_string_regexp_is_match_scalar(arr_a: &StringArray, value_b: &str) {
regexp_is_match_scalar(hint::black_box(arr_a), hint::black_box(value_b), None).unwrap();
}
fn make_string_array(size: usize, rng: &mut StdRng) -> impl Iterator<Item = Option<String>> + '_ {
(0..size).map(|_| {
let len = rng.random_range(0..64);
let bytes = (0..len).map(|_| rng.random_range(0..128)).collect();
Some(String::from_utf8(bytes).unwrap())
})
}
fn make_inlined_string_array(
size: usize,
rng: &mut StdRng,
) -> impl Iterator<Item = Option<String>> + '_ {
(0..size).map(|_| {
let len = rng.random_range(0..12);
let bytes = (0..len).map(|_| rng.random_range(0..128)).collect();
Some(String::from_utf8(bytes).unwrap())
})
}
fn add_benchmark(c: &mut Criterion) {
let arr_a = create_primitive_array_with_seed::<Float32Type>(SIZE, 0.0, 42);
let arr_b = create_primitive_array_with_seed::<Float32Type>(SIZE, 0.0, 43);
let arr_month_day_nano_a = create_month_day_nano_array_with_seed(SIZE, 0.0, 43);
let arr_month_day_nano_b = create_month_day_nano_array_with_seed(SIZE, 0.0, 43);
let arr_string = create_string_array::<i32>(SIZE, 0.0);
let arr_string_view = create_string_view_array(SIZE, 0.0);
// create long string arrays with the same prefix
let arr_long_string = create_longer_string_array_with_same_prefix::<i32>(SIZE, 0.0);
let arr_long_string_view = create_longer_string_view_array_with_same_prefix(SIZE, 0.0);
let left_arr_long_string = create_longer_string_array_with_same_prefix::<i32>(SIZE, 0.0);
let right_arr_long_string = create_longer_string_array_with_same_prefix::<i32>(SIZE, 0.0);
let left_arr_long_string_view = create_longer_string_view_array_with_same_prefix(SIZE, 0.0);
let right_arr_long_string_view = create_longer_string_view_array_with_same_prefix(SIZE, 0.0);
let scalar = Float32Array::from(vec![1.0]);
// eq benchmarks
c.bench_function("eq Float32", |b| b.iter(|| eq(&arr_a, &arr_b)));
c.bench_function("eq scalar Float32", |b| {
b.iter(|| eq(&arr_a, &Scalar::new(&scalar)).unwrap())
});
c.bench_function("neq Float32", |b| b.iter(|| neq(&arr_a, &arr_b)));
c.bench_function("neq scalar Float32", |b| {
b.iter(|| neq(&arr_a, &Scalar::new(&scalar)).unwrap())
});
c.bench_function("lt Float32", |b| b.iter(|| lt(&arr_a, &arr_b)));
c.bench_function("lt scalar Float32", |b| {
b.iter(|| lt(&arr_a, &Scalar::new(&scalar)).unwrap())
});
c.bench_function("lt_eq Float32", |b| b.iter(|| lt_eq(&arr_a, &arr_b)));
c.bench_function("lt_eq scalar Float32", |b| {
b.iter(|| lt_eq(&arr_a, &Scalar::new(&scalar)).unwrap())
});
c.bench_function("gt Float32", |b| b.iter(|| gt(&arr_a, &arr_b)));
c.bench_function("gt scalar Float32", |b| {
b.iter(|| gt(&arr_a, &Scalar::new(&scalar)).unwrap())
});
c.bench_function("gt_eq Float32", |b| b.iter(|| gt_eq(&arr_a, &arr_b)));
c.bench_function("gt_eq scalar Float32", |b| {
b.iter(|| gt_eq(&arr_a, &Scalar::new(&scalar)).unwrap())
});
let arr_a = create_primitive_array_with_seed::<Int32Type>(SIZE, 0.0, 42);
let arr_b = create_primitive_array_with_seed::<Int32Type>(SIZE, 0.0, 43);
let scalar = Int32Array::new_scalar(1);
c.bench_function("eq Int32", |b| b.iter(|| eq(&arr_a, &arr_b)));
c.bench_function("eq scalar Int32", |b| {
b.iter(|| eq(&arr_a, &scalar).unwrap())
});
c.bench_function("neq Int32", |b| b.iter(|| neq(&arr_a, &arr_b)));
c.bench_function("neq scalar Int32", |b| {
b.iter(|| neq(&arr_a, &scalar).unwrap())
});
c.bench_function("lt Int32", |b| b.iter(|| lt(&arr_a, &arr_b)));
c.bench_function("lt scalar Int32", |b| {
b.iter(|| lt(&arr_a, &scalar).unwrap())
});
c.bench_function("lt_eq Int32", |b| b.iter(|| lt_eq(&arr_a, &arr_b)));
c.bench_function("lt_eq scalar Int32", |b| {
b.iter(|| lt_eq(&arr_a, &scalar).unwrap())
});
c.bench_function("gt Int32", |b| b.iter(|| gt(&arr_a, &arr_b)));
c.bench_function("gt scalar Int32", |b| {
b.iter(|| gt(&arr_a, &scalar).unwrap())
});
c.bench_function("gt_eq Int32", |b| b.iter(|| gt_eq(&arr_a, &arr_b)));
c.bench_function("gt_eq scalar Int32", |b| {
b.iter(|| gt_eq(&arr_a, &scalar).unwrap())
});
c.bench_function("eq MonthDayNano", |b| {
b.iter(|| eq(&arr_month_day_nano_a, &arr_month_day_nano_b))
});
let scalar = IntervalMonthDayNanoArray::new_scalar(IntervalMonthDayNano::new(123, 0, 0));
c.bench_function("eq scalar MonthDayNano", |b| {
b.iter(|| eq(&arr_month_day_nano_b, &scalar).unwrap())
});
let mut rng = seedable_rng();
let mut array_gen = make_string_array(1024 * 1024 * 8, &mut rng);
let string_left = StringArray::from_iter(array_gen);
let string_view_left = StringViewArray::from_iter(string_left.iter());
// reference to the same rng to make sure we generate **different** array data,
// ow. the left and right will be identical
array_gen = make_string_array(1024 * 1024 * 8, &mut rng);
let string_right = StringArray::from_iter(array_gen);
let string_view_right = StringViewArray::from_iter(string_right.iter());
let string_scalar = StringArray::new_scalar("xxxx");
c.bench_function("eq scalar StringArray", |b| {
b.iter(|| eq(&string_scalar, &string_left).unwrap())
});
c.bench_function("lt scalar StringViewArray", |b| {
b.iter(|| {
lt(
&Scalar::new(StringViewArray::from_iter_values(["xxxx"])),
&string_view_left,
)
.unwrap()
})
});
c.bench_function("lt scalar StringArray", |b| {
b.iter(|| {
lt(
&Scalar::new(StringArray::from_iter_values(["xxxx"])),
&string_left,
)
.unwrap()
})
});
// StringViewArray has special handling for strings with length <= 12 and length <= 4
let string_view_scalar = StringViewArray::new_scalar("xxxx");
c.bench_function("eq scalar StringViewArray 4 bytes", |b| {
b.iter(|| eq(&string_view_scalar, &string_view_left).unwrap())
});
let string_view_scalar = StringViewArray::new_scalar("xxxxxx");
c.bench_function("eq scalar StringViewArray 6 bytes", |b| {
b.iter(|| eq(&string_view_scalar, &string_view_left).unwrap())
});
let string_view_scalar = StringViewArray::new_scalar("xxxxxxxxxxxxx");
c.bench_function("eq scalar StringViewArray 13 bytes", |b| {
b.iter(|| eq(&string_view_scalar, &string_view_left).unwrap())
});
c.bench_function("eq StringArray StringArray", |b| {
b.iter(|| eq(&string_left, &string_right).unwrap())
});
c.bench_function("eq StringViewArray StringViewArray", |b| {
b.iter(|| eq(&string_view_left, &string_view_right).unwrap())
});
let array_gen = make_inlined_string_array(1024 * 1024 * 8, &mut rng);
let string_left = StringArray::from_iter(array_gen);
let string_view_inlined_left = StringViewArray::from_iter(string_left.iter());
let array_gen = make_inlined_string_array(1024 * 1024 * 8, &mut rng);
let string_right = StringArray::from_iter(array_gen);
let string_view_inlined_right = StringViewArray::from_iter(string_right.iter());
// Add fast path benchmarks for StringViewArray, both side are inlined views < 12 bytes
c.bench_function("eq StringViewArray StringViewArray inlined bytes", |b| {
b.iter(|| eq(&string_view_inlined_left, &string_view_inlined_right).unwrap())
});
c.bench_function("lt StringViewArray StringViewArray inlined bytes", |b| {
b.iter(|| lt(&string_view_inlined_left, &string_view_inlined_right).unwrap())
});
// eq benchmarks for long strings with the same prefix
c.bench_function("eq long same prefix strings StringArray", |b| {
b.iter(|| eq(&left_arr_long_string, &right_arr_long_string).unwrap())
});
c.bench_function("neq long same prefix strings StringArray", |b| {
b.iter(|| neq(&left_arr_long_string, &right_arr_long_string).unwrap())
});
c.bench_function("lt long same prefix strings StringArray", |b| {
b.iter(|| lt(&left_arr_long_string, &right_arr_long_string).unwrap())
});
c.bench_function("eq long same prefix strings StringViewArray", |b| {
b.iter(|| eq(&left_arr_long_string_view, &right_arr_long_string_view).unwrap())
});
c.bench_function("neq long same prefix strings StringViewArray", |b| {
b.iter(|| neq(&left_arr_long_string_view, &right_arr_long_string_view).unwrap())
});
c.bench_function("lt long same prefix strings StringViewArray", |b| {
b.iter(|| lt(&left_arr_long_string_view, &right_arr_long_string_view).unwrap())
});
// StringArray: LIKE benchmarks
c.bench_function("like_utf8 scalar equals", |b| {
b.iter(|| bench_like_utf8_scalar(&arr_string, "xxxx"))
});
c.bench_function("like_utf8 scalar contains", |b| {
b.iter(|| bench_like_utf8_scalar(&arr_string, "%xxxx%"))
});
c.bench_function("like_utf8 scalar ends with", |b| {
b.iter(|| bench_like_utf8_scalar(&arr_string, "%xxxx"))
});
c.bench_function("like_utf8 scalar starts with", |b| {
b.iter(|| bench_like_utf8_scalar(&arr_string, "xxxx%"))
});
c.bench_function("like_utf8 scalar complex", |b| {
b.iter(|| bench_like_utf8_scalar(&arr_string, "%xx_xx%xxx"))
});
// StringArray: LIKE benchmarks with long strings 4 bytes prefix
// Note:
// long strings mean strings start with same 4 bytes prefix such as "test",
// followed by a tail, ensuring the total length is greater than 12 bytes.
c.bench_function("long same prefix strings like_utf8 scalar equals", |b| {
b.iter(|| bench_like_utf8_scalar(&arr_long_string, "prefix_1234"))
});
c.bench_function("long same prefix strings like_utf8 scalar contains", |b| {
b.iter(|| bench_like_utf8_scalar(&arr_long_string, "%prefix_1234%"))
});
c.bench_function("long same prefix strings like_utf8 scalar ends with", |b| {
b.iter(|| bench_like_utf8_scalar(&arr_long_string, "%prefix_1234"))
});
c.bench_function(
"long same prefix strings like_utf8 scalar starts with",
|b| b.iter(|| bench_like_utf8_scalar(&arr_long_string, "prefix_1234%")),
);
c.bench_function("long same prefix strings like_utf8 scalar complex", |b| {
b.iter(|| bench_like_utf8_scalar(&arr_long_string, "%prefix_1234%xxx"))
});
// StringViewArray: LIKE benchmarks with long strings 4 bytes prefix
// Note:
// long strings mean strings start with same 4 bytes prefix such as "test",
// followed by a tail, ensuring the total length is greater than 12 bytes.
c.bench_function(
"long same prefix strings like_utf8view scalar equals",
|b| b.iter(|| bench_like_utf8view_scalar(&arr_long_string_view, "prefix_1234")),
);
c.bench_function(
"long same prefix strings like_utf8view scalar contains",
|b| b.iter(|| bench_like_utf8view_scalar(&arr_long_string_view, "%prefix_1234%")),
);
c.bench_function(
"long same prefix strings like_utf8view scalar ends with",
|b| b.iter(|| bench_like_utf8view_scalar(&arr_long_string_view, "%prefix_1234")),
);
c.bench_function(
"long same prefix strings like_utf8view scalar starts with",
|b| b.iter(|| bench_like_utf8view_scalar(&arr_long_string_view, "prefix_1234%")),
);
c.bench_function(
"long same prefix strings like_utf8view scalar complex",
|b| b.iter(|| bench_like_utf8view_scalar(&arr_long_string_view, "%prefix_1234%xxx")),
);
// StringViewArray: LIKE benchmarks
// Note: since like/nlike share the same implementation, we only benchmark one
c.bench_function("like_utf8view scalar equals", |b| {
b.iter(|| bench_like_utf8view_scalar(&string_view_left, "xxxx"))
});
c.bench_function("like_utf8view scalar contains", |b| {
b.iter(|| bench_like_utf8view_scalar(&string_view_left, "%xxxx%"))
});
// StringView has special handling for strings with length <= 12 and length <= 4
c.bench_function("like_utf8view scalar ends with 4 bytes", |b| {
b.iter(|| bench_like_utf8view_scalar(&string_view_left, "%xxxx"))
});
c.bench_function("like_utf8view scalar ends with 6 bytes", |b| {
b.iter(|| bench_like_utf8view_scalar(&string_view_left, "%xxxxxx"))
});
c.bench_function("like_utf8view scalar ends with 13 bytes", |b| {
b.iter(|| bench_like_utf8view_scalar(&string_view_left, "%xxxxxxxxxxxxx"))
});
c.bench_function("like_utf8view scalar starts with 4 bytes", |b| {
b.iter(|| bench_like_utf8view_scalar(&string_view_left, "xxxx%"))
});
c.bench_function("like_utf8view scalar starts with 6 bytes", |b| {
b.iter(|| bench_like_utf8view_scalar(&string_view_left, "xxxxxx%"))
});
c.bench_function("like_utf8view scalar starts with 13 bytes", |b| {
b.iter(|| bench_like_utf8view_scalar(&string_view_left, "xxxxxxxxxxxxx%"))
});
c.bench_function("like_utf8view scalar complex", |b| {
b.iter(|| bench_like_utf8view_scalar(&string_view_left, "%xx_xx%xxx"))
});
// StringArray: NOT LIKE benchmarks
c.bench_function("nlike_utf8 scalar equals", |b| {
b.iter(|| bench_nlike_utf8_scalar(&arr_string, "xxxx"))
});
c.bench_function("nlike_utf8 scalar contains", |b| {
b.iter(|| bench_nlike_utf8_scalar(&arr_string, "%xxxx%"))
});
c.bench_function("nlike_utf8 scalar ends with", |b| {
b.iter(|| bench_nlike_utf8_scalar(&arr_string, "%xxxx"))
});
c.bench_function("nlike_utf8 scalar starts with", |b| {
b.iter(|| bench_nlike_utf8_scalar(&arr_string, "xxxx%"))
});
c.bench_function("nlike_utf8 scalar complex", |b| {
b.iter(|| bench_nlike_utf8_scalar(&arr_string, "%xx_xx%xxx"))
});
// StringArray: ILIKE benchmarks
c.bench_function("ilike_utf8 scalar equals", |b| {
b.iter(|| bench_ilike_utf8_scalar(&arr_string, "xxXX"))
});
c.bench_function("ilike_utf8 scalar contains", |b| {
b.iter(|| bench_ilike_utf8_scalar(&arr_string, "%xxXX%"))
});
c.bench_function("ilike_utf8 scalar ends with", |b| {
b.iter(|| bench_ilike_utf8_scalar(&arr_string, "%xXXx"))
});
c.bench_function("ilike_utf8 scalar starts with", |b| {
b.iter(|| bench_ilike_utf8_scalar(&arr_string, "XXXx%"))
});
c.bench_function("ilike_utf8 scalar complex", |b| {
b.iter(|| bench_ilike_utf8_scalar(&arr_string, "%xx_xX%xXX"))
});
// StringArray: NOT ILIKE benchmarks
c.bench_function("nilike_utf8 scalar equals", |b| {
b.iter(|| bench_nilike_utf8_scalar(&arr_string, "xxXX"))
});
c.bench_function("nilike_utf8 scalar contains", |b| {
b.iter(|| bench_nilike_utf8_scalar(&arr_string, "%xxXX%"))
});
c.bench_function("nilike_utf8 scalar ends with", |b| {
b.iter(|| bench_nilike_utf8_scalar(&arr_string, "%xXXx"))
});
c.bench_function("nilike_utf8 scalar starts with", |b| {
b.iter(|| bench_nilike_utf8_scalar(&arr_string, "XXXx%"))
});
c.bench_function("nilike_utf8 scalar complex", |b| {
b.iter(|| bench_nilike_utf8_scalar(&arr_string, "%xx_xX%xXX"))
});
// StringArray: regexp_matches_utf8 scalar benchmarks
let mut group =
c.benchmark_group("StringArray: regexp_matches_utf8 scalar benchmarks".to_string());
group
.bench_function("regexp_matches_utf8 scalar starts with", |b| {
b.iter(|| bench_string_regexp_is_match_scalar(&arr_string, "^xx"))
})
.bench_function("regexp_matches_utf8 scalar contains", |b| {
b.iter(|| bench_string_regexp_is_match_scalar(&arr_string, ".*xxXX.*"))
})
.bench_function("regexp_matches_utf8 scalar ends with", |b| {
b.iter(|| bench_string_regexp_is_match_scalar(&arr_string, "xx$"))
})
.bench_function("regexp_matches_utf8 scalar complex", |b| {
b.iter(|| bench_string_regexp_is_match_scalar(&arr_string, ".*x{2}.xX.*xXX"))
});
group.finish();
// StringViewArray: regexp_matches_utf8view scalar benchmarks
group =
c.benchmark_group("StringViewArray: regexp_matches_utf8view scalar benchmarks".to_string());
group
.bench_function("regexp_matches_utf8view scalar starts with", |b| {
b.iter(|| bench_stringview_regexp_is_match_scalar(&arr_string_view, "^xx"))
})
.bench_function("regexp_matches_utf8view scalar contains", |b| {
b.iter(|| bench_stringview_regexp_is_match_scalar(&arr_string_view, ".*xxXX.*"))
})
.bench_function("regexp_matches_utf8view scalar ends with", |b| {
b.iter(|| bench_stringview_regexp_is_match_scalar(&arr_string_view, "xx$"))
})
.bench_function("regexp_matches_utf8view scalar complex", |b| {
b.iter(|| bench_stringview_regexp_is_match_scalar(&arr_string_view, ".*x{2}.xX.*xXX"))
});
group.finish();
// DictionaryArray benchmarks
let strings = create_string_array::<i32>(20, 0.);
let dict_arr_a = create_dict_from_values::<Int32Type>(SIZE, 0., &strings);
let scalar = StringArray::from(vec!["test"]);
c.bench_function("eq_dyn_utf8_scalar dictionary[10] string[4])", |b| {
b.iter(|| eq(&dict_arr_a, &Scalar::new(&scalar)))
});
c.bench_function(
"gt_eq_dyn_utf8_scalar scalar dictionary[10] string[4])",
|b| b.iter(|| gt_eq(&dict_arr_a, &Scalar::new(&scalar))),
);
c.bench_function("like_utf8_scalar_dyn dictionary[10] string[4])", |b| {
b.iter(|| like(&dict_arr_a, &StringArray::new_scalar("test")))
});
c.bench_function("ilike_utf8_scalar_dyn dictionary[10] string[4])", |b| {
b.iter(|| ilike(&dict_arr_a, &StringArray::new_scalar("test")))
});
let strings = create_string_array::<i32>(20, 0.);
let dict_arr_a = create_dict_from_values::<Int32Type>(SIZE, 0., &strings);
let dict_arr_b = create_dict_from_values::<Int32Type>(SIZE, 0., &strings);
c.bench_function("eq dictionary[10] string[4])", |b| {
b.iter(|| eq(&dict_arr_a, &dict_arr_b).unwrap())
});
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
+242
View File
@@ -0,0 +1,242 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
extern crate arrow;
#[macro_use]
extern crate criterion;
use std::sync::Arc;
use criterion::Criterion;
use arrow::array::*;
use arrow::compute::concat;
use arrow::datatypes::*;
use arrow::util::bench_util::*;
use std::hint;
fn bench_concat(v1: &dyn Array, v2: &dyn Array) {
hint::black_box(concat(&[v1, v2]).unwrap());
}
fn bench_concat_arrays(arrays: &[&dyn Array]) {
hint::black_box(concat(arrays).unwrap());
}
fn add_benchmark(c: &mut Criterion) {
let v1 = create_primitive_array::<Int32Type>(1024, 0.0);
let v2 = create_primitive_array::<Int32Type>(1024, 0.0);
c.bench_function("concat i32 1024", |b| b.iter(|| bench_concat(&v1, &v2)));
let v1 = create_primitive_array::<Int32Type>(1024, 0.5);
let v2 = create_primitive_array::<Int32Type>(1024, 0.5);
c.bench_function("concat i32 nulls 1024", |b| {
b.iter(|| bench_concat(&v1, &v2))
});
let small_array = create_primitive_array::<Int32Type>(4, 0.0);
let arrays: Vec<_> = (0..1024).map(|_| &small_array as &dyn Array).collect();
c.bench_function("concat 1024 arrays i32 4", |b| {
b.iter(|| bench_concat_arrays(&arrays))
});
{
let input = (0..100)
.map(|_| create_primitive_array::<Int32Type>(8192, 0.0))
.collect::<Vec<_>>();
let arrays: Vec<_> = input.iter().map(|arr| arr as &dyn Array).collect();
c.bench_function("concat i32 8192 over 100 arrays", |b| {
b.iter(|| bench_concat_arrays(&arrays))
});
}
{
let input = (0..100)
.map(|_| create_primitive_array::<Int32Type>(8192, 0.5))
.collect::<Vec<_>>();
let arrays: Vec<_> = input.iter().map(|arr| arr as &dyn Array).collect();
c.bench_function("concat i32 nulls 8192 over 100 arrays", |b| {
b.iter(|| bench_concat_arrays(&arrays))
});
}
let v1 = create_boolean_array(1024, 0.0, 0.5);
let v2 = create_boolean_array(1024, 0.0, 0.5);
c.bench_function("concat boolean 1024", |b| b.iter(|| bench_concat(&v1, &v2)));
let v1 = create_boolean_array(1024, 0.5, 0.5);
let v2 = create_boolean_array(1024, 0.5, 0.5);
c.bench_function("concat boolean nulls 1024", |b| {
b.iter(|| bench_concat(&v1, &v2))
});
let small_array = create_boolean_array(4, 0.0, 0.5);
let arrays: Vec<_> = (0..1024).map(|_| &small_array as &dyn Array).collect();
c.bench_function("concat 1024 arrays boolean 4", |b| {
b.iter(|| bench_concat_arrays(&arrays))
});
{
let input = (0..100)
.map(|_| create_boolean_array(8192, 0.0, 0.5))
.collect::<Vec<_>>();
let arrays: Vec<_> = input.iter().map(|arr| arr as &dyn Array).collect();
c.bench_function("concat boolean 8192 over 100 arrays", |b| {
b.iter(|| bench_concat_arrays(&arrays))
});
}
{
let input = (0..100)
.map(|_| create_boolean_array(8192, 0.5, 0.5))
.collect::<Vec<_>>();
let arrays: Vec<_> = input.iter().map(|arr| arr as &dyn Array).collect();
c.bench_function("concat boolean nulls 8192 over 100 arrays", |b| {
b.iter(|| bench_concat_arrays(&arrays))
});
}
let v1 = create_string_array::<i32>(1024, 0.0);
let v2 = create_string_array::<i32>(1024, 0.0);
c.bench_function("concat str 1024", |b| b.iter(|| bench_concat(&v1, &v2)));
let v1 = create_string_array::<i32>(1024, 0.5);
let v2 = create_string_array::<i32>(1024, 0.5);
c.bench_function("concat str nulls 1024", |b| {
b.iter(|| bench_concat(&v1, &v2))
});
let small_array = create_string_array::<i32>(4, 0.0);
let arrays: Vec<_> = (0..1024).map(|_| &small_array as &dyn Array).collect();
c.bench_function("concat 1024 arrays str 4", |b| {
b.iter(|| bench_concat_arrays(&arrays))
});
{
let input = (0..100)
.map(|_| create_string_array::<i32>(8192, 0.0))
.collect::<Vec<_>>();
let arrays: Vec<_> = input.iter().map(|arr| arr as &dyn Array).collect();
c.bench_function("concat str 8192 over 100 arrays", |b| {
b.iter(|| bench_concat_arrays(&arrays))
});
}
{
let input = (0..100)
.map(|_| create_string_array::<i32>(8192, 0.5))
.collect::<Vec<_>>();
let arrays: Vec<_> = input.iter().map(|arr| arr as &dyn Array).collect();
c.bench_function("concat str nulls 8192 over 100 arrays", |b| {
b.iter(|| bench_concat_arrays(&arrays))
});
}
// String view arrays
for null_density in [0.0, 0.2] {
// Any strings less than 12 characters are stored as prefix only, so specially
// benchmark cases that have different mixes of lengths.
for (name, str_len) in [("all_inline", 12), ("", 20), ("", 128)] {
let array = create_string_view_array_with_len(8192, null_density, str_len, false);
let arrays = (0..10).map(|_| &array as &dyn Array).collect::<Vec<_>>();
let id = format!(
"concat utf8_view {name} max_str_len={str_len} null_density={null_density}"
);
c.bench_function(&id, |b| b.iter(|| bench_concat_arrays(&arrays)));
}
}
let v1 = create_string_array_with_len::<i32>(10, 0.0, 20);
let v1 = create_dict_from_values::<Int32Type>(1024, 0.0, &v1);
let v2 = create_string_array_with_len::<i32>(10, 0.0, 20);
let v2 = create_dict_from_values::<Int32Type>(1024, 0.0, &v2);
c.bench_function("concat str_dict 1024", |b| {
b.iter(|| bench_concat(&v1, &v2))
});
let v1 = create_string_array_with_len::<i32>(1024, 0.0, 20);
let v1 = create_sparse_dict_from_values::<Int32Type>(1024, 0.0, &v1, 10..20);
let v2 = create_string_array_with_len::<i32>(1024, 0.0, 20);
let v2 = create_sparse_dict_from_values::<Int32Type>(1024, 0.0, &v2, 30..40);
c.bench_function("concat str_dict_sparse 1024", |b| {
b.iter(|| bench_concat(&v1, &v2))
});
let v1 = FixedSizeListArray::try_new(
Arc::new(Field::new_list_field(DataType::Int32, true)),
1024,
Arc::new(create_primitive_array::<Int32Type>(1024 * 1024, 0.0)),
None,
)
.unwrap();
let v2 = FixedSizeListArray::try_new(
Arc::new(Field::new_list_field(DataType::Int32, true)),
1024,
Arc::new(create_primitive_array::<Int32Type>(1024 * 1024, 0.0)),
None,
)
.unwrap();
c.bench_function("concat fixed size lists", |b| {
b.iter(|| bench_concat(&v1, &v2))
});
{
let batch_size = 1024;
let batch_count = 2;
let struct_arrays = (0..batch_count)
.map(|_| {
let ints = create_primitive_array::<Int32Type>(batch_size, 0.0);
let string_dict = create_sparse_dict_from_values::<Int32Type>(
batch_size,
0.0,
&create_string_array_with_len::<i32>(20, 0.0, 10),
0..10,
);
let int_dict = create_sparse_dict_from_values::<UInt16Type>(
batch_size,
0.0,
&create_primitive_array::<Int64Type>(20, 0.0),
0..10,
);
let fields = vec![
Field::new("int_field", ints.data_type().clone(), false),
Field::new("strings_dict_field", string_dict.data_type().clone(), false),
Field::new("int_dict_field", int_dict.data_type().clone(), false),
];
StructArray::try_new(
fields.clone().into(),
vec![Arc::new(ints), Arc::new(string_dict), Arc::new(int_dict)],
None,
)
.unwrap()
})
.collect::<Vec<_>>();
let array_refs = struct_arrays
.iter()
.map(|a| a as &dyn Array)
.collect::<Vec<_>>();
c.bench_function(
&format!("concat struct with int32 and dicts size={batch_size} count={batch_count}"),
|b| b.iter(|| bench_concat_arrays(&array_refs)),
);
}
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
+184
View File
@@ -0,0 +1,184 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
extern crate arrow;
extern crate criterion;
use std::io::Cursor;
use std::sync::Arc;
use arrow::util::bench_util::create_string_view_array_with_len;
use criterion::*;
use rand::Rng;
use arrow::array::*;
use arrow::csv;
use arrow::datatypes::*;
use arrow::util::bench_util::{create_primitive_array, create_string_array_with_len};
use arrow::util::test_util::seedable_rng;
fn do_bench(c: &mut Criterion, name: &str, cols: Vec<ArrayRef>) {
let batch = RecordBatch::try_from_iter(cols.into_iter().map(|a| ("col", a))).unwrap();
let mut buf = Vec::with_capacity(1024);
let mut csv = csv::Writer::new(&mut buf);
csv.write(&batch).unwrap();
drop(csv);
for batch_size in [128, 1024, 4096] {
c.bench_function(&format!("{name} - {batch_size}"), |b| {
b.iter(|| {
let cursor = Cursor::new(buf.as_slice());
let reader = csv::ReaderBuilder::new(batch.schema())
.with_batch_size(batch_size)
.with_header(true)
.build_buffered(cursor)
.unwrap();
for next in reader {
next.unwrap();
}
});
});
}
}
fn criterion_benchmark(c: &mut Criterion) {
let mut rng = seedable_rng();
// Single Primitive Column tests
let values = Int32Array::from_iter_values((0..4096).map(|_| rng.random_range(0..1024)));
let cols = vec![Arc::new(values) as ArrayRef];
do_bench(c, "4096 i32_small(0)", cols);
let values = Int32Array::from_iter_values((0..4096).map(|_| rng.random()));
let cols = vec![Arc::new(values) as ArrayRef];
do_bench(c, "4096 i32(0)", cols);
let values = UInt64Array::from_iter_values((0..4096).map(|_| rng.random_range(0..1024)));
let cols = vec![Arc::new(values) as ArrayRef];
do_bench(c, "4096 u64_small(0)", cols);
let values = UInt64Array::from_iter_values((0..4096).map(|_| rng.random()));
let cols = vec![Arc::new(values) as ArrayRef];
do_bench(c, "4096 u64(0)", cols);
let values = Int64Array::from_iter_values((0..4096).map(|_| rng.random_range(0..1024) - 512));
let cols = vec![Arc::new(values) as ArrayRef];
do_bench(c, "4096 i64_small(0)", cols);
let values = Int64Array::from_iter_values((0..4096).map(|_| rng.random()));
let cols = vec![Arc::new(values) as ArrayRef];
do_bench(c, "4096 i64(0)", cols);
let cols = vec![Arc::new(Float32Array::from_iter_values(
(0..4096).map(|_| rng.random_range(0..1024000) as f32 / 1000.),
)) as _];
do_bench(c, "4096 f32_small(0)", cols);
let values = Float32Array::from_iter_values((0..4096).map(|_| rng.random()));
let cols = vec![Arc::new(values) as ArrayRef];
do_bench(c, "4096 f32(0)", cols);
let cols = vec![Arc::new(Float64Array::from_iter_values(
(0..4096).map(|_| rng.random_range(0..1024000) as f64 / 1000.),
)) as _];
do_bench(c, "4096 f64_small(0)", cols);
let values = Float64Array::from_iter_values((0..4096).map(|_| rng.random()));
let cols = vec![Arc::new(values) as ArrayRef];
do_bench(c, "4096 f64(0)", cols);
// Single String Column tests
let cols = vec![Arc::new(create_string_array_with_len::<i32>(4096, 0., 10)) as ArrayRef];
do_bench(c, "4096 string(10, 0)", cols);
let cols = vec![Arc::new(create_string_array_with_len::<i32>(4096, 0., 30)) as ArrayRef];
do_bench(c, "4096 string(30, 0)", cols);
let cols = vec![Arc::new(create_string_array_with_len::<i32>(4096, 0., 100)) as ArrayRef];
do_bench(c, "4096 string(100, 0)", cols);
let cols = vec![Arc::new(create_string_array_with_len::<i32>(4096, 0.5, 100)) as ArrayRef];
do_bench(c, "4096 string(100, 0.5)", cols);
// Single StringView Column tests
let cols = vec![Arc::new(create_string_view_array_with_len(4096, 0., 10, false)) as ArrayRef];
do_bench(c, "4096 StringView(10, 0)", cols);
let cols = vec![Arc::new(create_string_view_array_with_len(4096, 0., 30, false)) as ArrayRef];
do_bench(c, "4096 StringView(30, 0)", cols);
let cols = vec![Arc::new(create_string_view_array_with_len(4096, 0., 100, false)) as ArrayRef];
do_bench(c, "4096 StringView(100, 0)", cols);
let cols = vec![Arc::new(create_string_view_array_with_len(4096, 0.5, 100, false)) as ArrayRef];
do_bench(c, "4096 StringView(100, 0.5)", cols);
// Multi-Column(with String) tests
let cols = vec![
Arc::new(create_string_array_with_len::<i32>(4096, 0.5, 20)) as ArrayRef,
Arc::new(create_string_array_with_len::<i32>(4096, 0., 30)) as ArrayRef,
Arc::new(create_string_array_with_len::<i32>(4096, 0., 100)) as ArrayRef,
Arc::new(create_primitive_array::<Int64Type>(4096, 0.)) as ArrayRef,
];
do_bench(
c,
"4096 string(20, 0.5), string(30, 0), string(100, 0), i64(0)",
cols,
);
let cols = vec![
Arc::new(create_string_array_with_len::<i32>(4096, 0.5, 20)) as ArrayRef,
Arc::new(create_string_array_with_len::<i32>(4096, 0., 30)) as ArrayRef,
Arc::new(create_primitive_array::<Float64Type>(4096, 0.)) as ArrayRef,
Arc::new(create_primitive_array::<Int64Type>(4096, 0.)) as ArrayRef,
];
do_bench(
c,
"4096 string(20, 0.5), string(30, 0), f64(0), i64(0)",
cols,
);
// Multi-Column(with StringView) tests
let cols = vec![
Arc::new(create_string_view_array_with_len(4096, 0.5, 20, false)) as ArrayRef,
Arc::new(create_string_view_array_with_len(4096, 0., 30, false)) as ArrayRef,
Arc::new(create_string_view_array_with_len(4096, 0., 100, false)) as ArrayRef,
Arc::new(create_primitive_array::<Int64Type>(4096, 0.)) as ArrayRef,
];
do_bench(
c,
"4096 StringView(20, 0.5), StringView(30, 0), StringView(100, 0), i64(0)",
cols,
);
let cols = vec![
Arc::new(create_string_view_array_with_len(4096, 0.5, 20, false)) as ArrayRef,
Arc::new(create_string_view_array_with_len(4096, 0., 30, false)) as ArrayRef,
Arc::new(create_primitive_array::<Float64Type>(4096, 0.)) as ArrayRef,
Arc::new(create_primitive_array::<Int64Type>(4096, 0.)) as ArrayRef,
];
do_bench(
c,
"4096 StringView(20, 0.5), StringView(30, 0), f64(0), i64(0)",
cols,
);
}
criterion_group!(benches, criterion_benchmark);
criterion_main!(benches);
+69
View File
@@ -0,0 +1,69 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
extern crate arrow;
extern crate criterion;
use criterion::*;
use arrow::array::*;
use arrow::csv;
use arrow::datatypes::*;
use std::env;
use std::fs::File;
use std::hint;
use std::sync::Arc;
fn criterion_benchmark(c: &mut Criterion) {
let schema = Schema::new(vec![
Field::new("c1", DataType::Utf8, false),
Field::new("c2", DataType::Float64, true),
Field::new("c3", DataType::UInt32, false),
Field::new("c4", DataType::Boolean, true),
]);
let c1 = StringArray::from(vec![
"Lorem ipsum dolor sit amet",
"consectetur adipiscing elit",
"sed do eiusmod tempor",
]);
let c2 = PrimitiveArray::<Float64Type>::from(vec![Some(123.564532), None, Some(-556132.25)]);
let c3 = PrimitiveArray::<UInt32Type>::from(vec![3, 2, 1]);
let c4 = BooleanArray::from(vec![Some(true), Some(false), None]);
let b = RecordBatch::try_new(
Arc::new(schema),
vec![Arc::new(c1), Arc::new(c2), Arc::new(c3), Arc::new(c4)],
)
.unwrap();
let path = env::temp_dir().join("bench_write_csv.csv");
let file = File::create(path).unwrap();
let mut writer = csv::Writer::new(file);
let batches = vec![&b, &b, &b, &b, &b, &b, &b, &b, &b, &b, &b];
c.bench_function("record_batches_to_csv", |b| {
b.iter(|| {
#[allow(clippy::unit_arg)]
hint::black_box(for batch in &batches {
writer.write(batch).unwrap()
});
});
});
}
criterion_group!(benches, criterion_benchmark);
criterion_main!(benches);
+137
View File
@@ -0,0 +1,137 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use arrow::array::{
Array, Decimal32Array, Decimal32Builder, Decimal64Array, Decimal64Builder, Decimal128Array,
Decimal128Builder, Decimal256Array, Decimal256Builder,
};
use criterion::Criterion;
use rand::Rng;
extern crate arrow;
use arrow_buffer::i256;
fn validate_decimal32_array(array: Decimal32Array) {
array.with_precision_and_scale(8, 0).unwrap();
}
fn validate_decimal64_array(array: Decimal64Array) {
array.with_precision_and_scale(16, 0).unwrap();
}
fn validate_decimal128_array(array: Decimal128Array) {
array.with_precision_and_scale(35, 0).unwrap();
}
fn validate_decimal256_array(array: Decimal256Array) {
array.with_precision_and_scale(35, 0).unwrap();
}
fn validate_decimal32_benchmark(c: &mut Criterion) {
let mut rng = rand::rng();
let size: i32 = 20000;
let mut decimal_builder = Decimal32Builder::with_capacity(size as usize);
for _ in 0..size {
decimal_builder.append_value(rng.random_range::<i32, _>(0..99999999));
}
let decimal_array = decimal_builder
.finish()
.with_precision_and_scale(9, 0)
.unwrap();
let data = decimal_array.into_data();
c.bench_function("validate_decimal32_array 20000", |b| {
b.iter(|| {
let array = Decimal32Array::from(data.clone());
validate_decimal32_array(array);
})
});
}
fn validate_decimal64_benchmark(c: &mut Criterion) {
let mut rng = rand::rng();
let size: i64 = 20000;
let mut decimal_builder = Decimal64Builder::with_capacity(size as usize);
for _ in 0..size {
decimal_builder.append_value(rng.random_range::<i64, _>(0..999999999999));
}
let decimal_array = decimal_builder
.finish()
.with_precision_and_scale(18, 0)
.unwrap();
let data = decimal_array.into_data();
c.bench_function("validate_decimal64_array 20000", |b| {
b.iter(|| {
let array = Decimal64Array::from(data.clone());
validate_decimal64_array(array);
})
});
}
fn validate_decimal128_benchmark(c: &mut Criterion) {
let mut rng = rand::rng();
let size: i128 = 20000;
let mut decimal_builder = Decimal128Builder::with_capacity(size as usize);
for _ in 0..size {
decimal_builder.append_value(rng.random_range::<i128, _>(0..999999999999));
}
let decimal_array = decimal_builder
.finish()
.with_precision_and_scale(38, 0)
.unwrap();
let data = decimal_array.into_data();
c.bench_function("validate_decimal128_array 20000", |b| {
b.iter(|| {
let array = Decimal128Array::from(data.clone());
validate_decimal128_array(array);
})
});
}
fn validate_decimal256_benchmark(c: &mut Criterion) {
let mut rng = rand::rng();
let size: i128 = 20000;
let mut decimal_builder = Decimal256Builder::with_capacity(size as usize);
for _ in 0..size {
let v = rng.random_range::<i128, _>(0..999999999999999);
let decimal = i256::from_i128(v);
decimal_builder.append_value(decimal);
}
let decimal_array256_data = decimal_builder
.finish()
.with_precision_and_scale(76, 0)
.unwrap();
let data = decimal_array256_data.into_data();
c.bench_function("validate_decimal256_array 20000", |b| {
b.iter(|| {
let array = Decimal256Array::from(data.clone());
validate_decimal256_array(array);
})
});
}
criterion_group!(
benches,
validate_decimal32_benchmark,
validate_decimal64_benchmark,
validate_decimal128_benchmark,
validate_decimal256_benchmark,
);
criterion_main!(benches);
+61
View File
@@ -0,0 +1,61 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
// Allowed because we use `arr == arr` in benchmarks
#![allow(clippy::eq_op)]
#[macro_use]
extern crate criterion;
use criterion::Criterion;
extern crate arrow;
use arrow::util::bench_util::*;
use arrow::{array::*, datatypes::Float32Type};
use std::hint;
fn bench_equal<A: Array + PartialEq<A>>(arr_a: &A) {
hint::black_box(arr_a == arr_a);
}
fn add_benchmark(c: &mut Criterion) {
let arr_a = create_primitive_array::<Float32Type>(512, 0.0);
c.bench_function("equal_512", |b| b.iter(|| bench_equal(&arr_a)));
let arr_a_nulls = create_primitive_array::<Float32Type>(512, 0.5);
c.bench_function("equal_nulls_512", |b| b.iter(|| bench_equal(&arr_a_nulls)));
let arr_a = create_primitive_array::<Float32Type>(51200, 0.1);
c.bench_function("equal_51200", |b| b.iter(|| bench_equal(&arr_a)));
let arr_a = create_string_array::<i32>(512, 0.0);
c.bench_function("equal_string_512", |b| b.iter(|| bench_equal(&arr_a)));
let arr_a_nulls = create_string_array::<i32>(512, 0.5);
c.bench_function("equal_string_nulls_512", |b| {
b.iter(|| bench_equal(&arr_a_nulls))
});
let arr_a = create_boolean_array(512, 0.0, 0.5);
c.bench_function("equal_bool_512", |b| b.iter(|| bench_equal(&arr_a)));
let arr_a = create_boolean_array(513, 0.0, 0.5);
c.bench_function("equal_bool_513", |b| b.iter(|| bench_equal(&arr_a)));
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
+301
View File
@@ -0,0 +1,301 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
extern crate arrow;
use std::sync::Arc;
use arrow::compute::{FilterBuilder, FilterPredicate, filter_record_batch};
use arrow::util::bench_util::*;
use arrow::array::*;
use arrow::compute::filter;
use arrow::datatypes::{Field, Float32Type, Int32Type, Int64Type, Schema, UInt8Type};
use arrow_array::types::Decimal128Type;
use criterion::{Criterion, criterion_group, criterion_main};
use std::hint;
fn bench_filter(data_array: &dyn Array, filter_array: &BooleanArray) {
hint::black_box(filter(data_array, filter_array).unwrap());
}
fn bench_built_filter(filter: &FilterPredicate, array: &dyn Array) {
hint::black_box(filter.filter(array).unwrap());
}
fn add_benchmark(c: &mut Criterion) {
let size = 65536;
let filter_array = create_boolean_array(size, 0.0, 0.5);
let dense_filter_array = create_boolean_array(size, 0.0, 1.0 - 1.0 / 1024.0);
let sparse_filter_array = create_boolean_array(size, 0.0, 1.0 / 1024.0);
let filter = FilterBuilder::new(&filter_array).optimize().build();
let dense_filter = FilterBuilder::new(&dense_filter_array).optimize().build();
let sparse_filter = FilterBuilder::new(&sparse_filter_array).optimize().build();
let data_array = create_primitive_array::<UInt8Type>(size, 0.0);
c.bench_function("filter optimize (kept 1/2)", |b| {
b.iter(|| FilterBuilder::new(&filter_array).optimize().build())
});
c.bench_function("filter optimize high selectivity (kept 1023/1024)", |b| {
b.iter(|| FilterBuilder::new(&dense_filter_array).optimize().build())
});
c.bench_function("filter optimize low selectivity (kept 1/1024)", |b| {
b.iter(|| FilterBuilder::new(&sparse_filter_array).optimize().build())
});
c.bench_function("filter u8 (kept 1/2)", |b| {
b.iter(|| bench_filter(&data_array, &filter_array))
});
c.bench_function("filter u8 high selectivity (kept 1023/1024)", |b| {
b.iter(|| bench_filter(&data_array, &dense_filter_array))
});
c.bench_function("filter u8 low selectivity (kept 1/1024)", |b| {
b.iter(|| bench_filter(&data_array, &sparse_filter_array))
});
c.bench_function("filter context u8 (kept 1/2)", |b| {
b.iter(|| bench_built_filter(&filter, &data_array))
});
c.bench_function("filter context u8 high selectivity (kept 1023/1024)", |b| {
b.iter(|| bench_built_filter(&dense_filter, &data_array))
});
c.bench_function("filter context u8 low selectivity (kept 1/1024)", |b| {
b.iter(|| bench_built_filter(&sparse_filter, &data_array))
});
let data_array = create_primitive_array::<Int32Type>(size, 0.0);
c.bench_function("filter i32 (kept 1/2)", |b| {
b.iter(|| bench_filter(&data_array, &filter_array))
});
c.bench_function("filter i32 high selectivity (kept 1023/1024)", |b| {
b.iter(|| bench_filter(&data_array, &dense_filter_array))
});
c.bench_function("filter i32 low selectivity (kept 1/1024)", |b| {
b.iter(|| bench_filter(&data_array, &sparse_filter_array))
});
c.bench_function("filter context i32 (kept 1/2)", |b| {
b.iter(|| bench_built_filter(&filter, &data_array))
});
c.bench_function(
"filter context i32 high selectivity (kept 1023/1024)",
|b| b.iter(|| bench_built_filter(&dense_filter, &data_array)),
);
c.bench_function("filter context i32 low selectivity (kept 1/1024)", |b| {
b.iter(|| bench_built_filter(&sparse_filter, &data_array))
});
let data_array = create_primitive_array::<Int32Type>(size, 0.5);
c.bench_function("filter context i32 w NULLs (kept 1/2)", |b| {
b.iter(|| bench_built_filter(&filter, &data_array))
});
c.bench_function(
"filter context i32 w NULLs high selectivity (kept 1023/1024)",
|b| b.iter(|| bench_built_filter(&dense_filter, &data_array)),
);
c.bench_function(
"filter context i32 w NULLs low selectivity (kept 1/1024)",
|b| b.iter(|| bench_built_filter(&sparse_filter, &data_array)),
);
let data_array = create_primitive_array::<UInt8Type>(size, 0.5);
c.bench_function("filter context u8 w NULLs (kept 1/2)", |b| {
b.iter(|| bench_built_filter(&filter, &data_array))
});
c.bench_function(
"filter context u8 w NULLs high selectivity (kept 1023/1024)",
|b| b.iter(|| bench_built_filter(&dense_filter, &data_array)),
);
c.bench_function(
"filter context u8 w NULLs low selectivity (kept 1/1024)",
|b| b.iter(|| bench_built_filter(&sparse_filter, &data_array)),
);
let data_array = create_primitive_array::<Float32Type>(size, 0.5);
c.bench_function("filter f32 (kept 1/2)", |b| {
b.iter(|| bench_filter(&data_array, &filter_array))
});
c.bench_function("filter context f32 (kept 1/2)", |b| {
b.iter(|| bench_built_filter(&filter, &data_array))
});
c.bench_function(
"filter context f32 high selectivity (kept 1023/1024)",
|b| b.iter(|| bench_built_filter(&dense_filter, &data_array)),
);
c.bench_function("filter context f32 low selectivity (kept 1/1024)", |b| {
b.iter(|| bench_built_filter(&sparse_filter, &data_array))
});
let data_array = create_primitive_array::<Decimal128Type>(size, 0.0);
c.bench_function("filter decimal128 (kept 1/2)", |b| {
b.iter(|| bench_filter(&data_array, &filter_array))
});
c.bench_function("filter decimal128 high selectivity (kept 1023/1024)", |b| {
b.iter(|| bench_filter(&data_array, &dense_filter_array))
});
c.bench_function("filter decimal128 low selectivity (kept 1/1024)", |b| {
b.iter(|| bench_filter(&data_array, &sparse_filter_array))
});
c.bench_function("filter context decimal128 (kept 1/2)", |b| {
b.iter(|| bench_built_filter(&filter, &data_array))
});
c.bench_function(
"filter context decimal128 high selectivity (kept 1023/1024)",
|b| b.iter(|| bench_built_filter(&dense_filter, &data_array)),
);
c.bench_function(
"filter context decimal128 low selectivity (kept 1/1024)",
|b| b.iter(|| bench_built_filter(&sparse_filter, &data_array)),
);
let data_array = create_string_array::<i32>(size, 0.5);
c.bench_function("filter context string (kept 1/2)", |b| {
b.iter(|| bench_built_filter(&filter, &data_array))
});
c.bench_function(
"filter context string high selectivity (kept 1023/1024)",
|b| b.iter(|| bench_built_filter(&dense_filter, &data_array)),
);
c.bench_function("filter context string low selectivity (kept 1/1024)", |b| {
b.iter(|| bench_built_filter(&sparse_filter, &data_array))
});
let data_array = create_string_dict_array::<Int32Type>(size, 0.0, 4);
c.bench_function("filter context string dictionary (kept 1/2)", |b| {
b.iter(|| bench_built_filter(&filter, &data_array))
});
c.bench_function(
"filter context string dictionary high selectivity (kept 1023/1024)",
|b| b.iter(|| bench_built_filter(&dense_filter, &data_array)),
);
c.bench_function(
"filter context string dictionary low selectivity (kept 1/1024)",
|b| b.iter(|| bench_built_filter(&sparse_filter, &data_array)),
);
let data_array = create_string_dict_array::<Int32Type>(size, 0.5, 4);
c.bench_function("filter context string dictionary w NULLs (kept 1/2)", |b| {
b.iter(|| bench_built_filter(&filter, &data_array))
});
c.bench_function(
"filter context string dictionary w NULLs high selectivity (kept 1023/1024)",
|b| b.iter(|| bench_built_filter(&dense_filter, &data_array)),
);
c.bench_function(
"filter context string dictionary w NULLs low selectivity (kept 1/1024)",
|b| b.iter(|| bench_built_filter(&sparse_filter, &data_array)),
);
let mut add_benchmark_for_fsb_with_length = |value_length: usize| {
let data_array = create_fsb_array(size, 0.0, value_length);
c.bench_function(
format!("filter fsb with value length {value_length} (kept 1/2)").as_str(),
|b| b.iter(|| bench_filter(&data_array, &filter_array)),
);
c.bench_function(
format!(
"filter fsb with value length {value_length} high selectivity (kept 1023/1024)"
)
.as_str(),
|b| b.iter(|| bench_filter(&data_array, &dense_filter_array)),
);
c.bench_function(
format!("filter fsb with value length {value_length} low selectivity (kept 1/1024)")
.as_str(),
|b| b.iter(|| bench_filter(&data_array, &sparse_filter_array)),
);
c.bench_function(
format!("filter context fsb with value length {value_length} (kept 1/2)").as_str(),
|b| b.iter(|| bench_built_filter(&filter, &filter_array)),
);
c.bench_function(
format!(
"filter context fsb with value length {value_length} high selectivity (kept 1023/1024)"
)
.as_str(),
|b| b.iter(|| bench_built_filter(&filter, &dense_filter_array)),
);
c.bench_function(
format!(
"filter context fsb with value length {value_length} low selectivity (kept 1/1024)"
)
.as_str(),
|b| b.iter(|| bench_built_filter(&filter, &sparse_filter_array)),
);
};
add_benchmark_for_fsb_with_length(5);
add_benchmark_for_fsb_with_length(20);
add_benchmark_for_fsb_with_length(50);
let data_array = create_primitive_array::<Float32Type>(size, 0.0);
let field = Field::new("c1", data_array.data_type().clone(), true);
let schema = Schema::new(vec![field]);
let batch = RecordBatch::try_new(Arc::new(schema), vec![Arc::new(data_array)]).unwrap();
c.bench_function("filter single record batch", |b| {
b.iter(|| filter_record_batch(&batch, &filter_array))
});
let data_array = create_string_view_array_with_len(size, 0.5, 4, false);
c.bench_function("filter context short string view (kept 1/2)", |b| {
b.iter(|| bench_built_filter(&filter, &data_array))
});
c.bench_function(
"filter context short string view high selectivity (kept 1023/1024)",
|b| b.iter(|| bench_built_filter(&dense_filter, &data_array)),
);
c.bench_function(
"filter context short string view low selectivity (kept 1/1024)",
|b| b.iter(|| bench_built_filter(&sparse_filter, &data_array)),
);
let data_array = create_string_view_array_with_len(size, 0.5, 4, true);
c.bench_function("filter context mixed string view (kept 1/2)", |b| {
b.iter(|| bench_built_filter(&filter, &data_array))
});
c.bench_function(
"filter context mixed string view high selectivity (kept 1023/1024)",
|b| b.iter(|| bench_built_filter(&dense_filter, &data_array)),
);
c.bench_function(
"filter context mixed string view low selectivity (kept 1/1024)",
|b| b.iter(|| bench_built_filter(&sparse_filter, &data_array)),
);
let data_array = create_primitive_run_array::<Int32Type, Int64Type>(size, size);
c.bench_function("filter run array (kept 1/2)", |b| {
b.iter(|| bench_built_filter(&filter, &data_array))
});
c.bench_function("filter run array high selectivity (kept 1023/1024)", |b| {
b.iter(|| bench_built_filter(&dense_filter, &data_array))
});
c.bench_function("filter run array low selectivity (kept 1/1024)", |b| {
b.iter(|| bench_built_filter(&sparse_filter, &data_array))
});
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
+172
View File
@@ -0,0 +1,172 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use criterion::Criterion;
use std::ops::Range;
use rand::Rng;
extern crate arrow;
use arrow::datatypes::*;
use arrow::util::test_util::seedable_rng;
use arrow::{array::*, util::bench_util::*};
use arrow_select::interleave::interleave;
use std::hint;
use std::sync::Arc;
fn do_bench(
c: &mut Criterion,
prefix: &str,
len: usize,
base: &dyn Array,
slices: &[Range<usize>],
) {
let arrays: Vec<_> = slices
.iter()
.map(|r| base.slice(r.start, r.end - r.start))
.collect();
let values: Vec<_> = arrays.iter().map(|x| x.as_ref()).collect();
bench_values(
c,
&format!("interleave {prefix} {len} {slices:?}"),
len,
&values,
);
}
fn bench_values(c: &mut Criterion, name: &str, len: usize, values: &[&dyn Array]) {
let mut rng = seedable_rng();
let indices: Vec<_> = (0..len)
.map(|_| {
let array_idx = rng.random_range(0..values.len());
let value_idx = rng.random_range(0..values[array_idx].len());
(array_idx, value_idx)
})
.collect();
c.bench_function(name, |b| {
b.iter(|| hint::black_box(interleave(values, &indices).unwrap()))
});
}
fn add_benchmark(c: &mut Criterion) {
let i32 = create_primitive_array::<Int32Type>(1024, 0.);
let i32_opt = create_primitive_array::<Int32Type>(1024, 0.5);
let string = create_string_array_with_len::<i32>(1024, 0., 20);
let string_opt = create_string_array_with_len::<i32>(1024, 0.5, 20);
let values = create_string_array_with_len::<i32>(10, 0.0, 20);
let dict = create_dict_from_values::<Int32Type>(1024, 0.0, &values);
let struct_i32_no_nulls_i32_no_nulls = StructArray::new(
Fields::from(vec![
Field::new("a", Int32Type::DATA_TYPE, false),
Field::new("b", Int32Type::DATA_TYPE, false),
]),
vec![
Arc::new(create_primitive_array::<Int32Type>(1024, 0.)),
Arc::new(create_primitive_array::<Int32Type>(1024, 0.)),
],
None,
);
let struct_string_no_nulls_string_no_nulls = StructArray::new(
Fields::from(vec![
Field::new("a", DataType::Utf8, false),
Field::new("b", DataType::Utf8, false),
]),
vec![
Arc::new(create_string_array_with_len::<i32>(1024, 0., 20)),
Arc::new(create_string_array_with_len::<i32>(1024, 0., 20)),
],
None,
);
let struct_i32_no_nulls_string_no_nulls = StructArray::new(
Fields::from(vec![
Field::new("a", DataType::Int32, false),
Field::new("b", DataType::Utf8, false),
]),
vec![
Arc::new(create_primitive_array::<Int32Type>(1024, 0.)),
Arc::new(create_string_array_with_len::<i32>(1024, 0., 20)),
],
None,
);
let values = create_string_array_with_len::<i32>(1024, 0.0, 20);
let sparse_dict = create_sparse_dict_from_values::<Int32Type>(1024, 0.0, &values, 10..20);
let string_view = create_string_view_array(1024, 0.0);
// use 8192 as a standard list size for better coverage
let list_i64 = create_primitive_list_array_with_seed::<i32, Int64Type>(8192, 0.1, 0.1, 20, 42);
let list_i64_no_nulls =
create_primitive_list_array_with_seed::<i32, Int64Type>(8192, 0.0, 0.0, 20, 42);
let cases: &[(&str, &dyn Array)] = &[
("i32(0.0)", &i32),
("i32(0.5)", &i32_opt),
("str(20, 0.0)", &string),
("str(20, 0.5)", &string_opt),
("dict(20, 0.0)", &dict),
("dict_sparse(20, 0.0)", &sparse_dict),
("str_view(0.0)", &string_view),
(
"struct(i32(0.0), i32(0.0)",
&struct_i32_no_nulls_i32_no_nulls,
),
(
"struct(str(20, 0.0), str(20, 0.0))",
&struct_string_no_nulls_string_no_nulls,
),
(
"struct(i32(0.0), str(20, 0.0)",
&struct_i32_no_nulls_string_no_nulls,
),
("list<i64>(0.1,0.1,20)", &list_i64),
("list<i64>(0.0,0.0,20)", &list_i64_no_nulls),
];
for (prefix, base) in cases {
let slices: &[(usize, &[_])] = &[
(100, &[0..100, 100..230, 450..1000]),
(400, &[0..100, 100..230, 450..1000]),
(1024, &[0..100, 100..230, 450..1000]),
(1024, &[0..100, 100..230, 450..1000, 0..1000]),
];
for (len, slice) in slices {
do_bench(c, prefix, *len, *base, slice);
}
}
for len in [100, 1024, 2048] {
bench_values(
c,
&format!("interleave dict_distinct {len}"),
100,
&[&dict, &sparse_dict],
);
}
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
+179
View File
@@ -0,0 +1,179 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use criterion::*;
use arrow::datatypes::*;
use arrow::util::bench_util::{
create_primitive_array, create_string_array, create_string_array_with_len,
};
use arrow_array::RecordBatch;
use arrow_json::{LineDelimitedWriter, ReaderBuilder};
use std::hint;
use std::io::Cursor;
use std::sync::Arc;
#[allow(deprecated)]
fn do_bench(c: &mut Criterion, name: &str, json: &str, schema: SchemaRef) {
c.bench_function(name, |b| {
b.iter(|| {
let cursor = Cursor::new(hint::black_box(json));
let builder = ReaderBuilder::new(schema.clone()).with_batch_size(64);
let reader = builder.build(cursor).unwrap();
for next in reader {
next.unwrap();
}
})
});
}
fn small_bench_primitive(c: &mut Criterion) {
let schema = Arc::new(Schema::new(vec![
Field::new("c1", DataType::Utf8, true),
Field::new("c2", DataType::Float64, true),
Field::new("c3", DataType::UInt32, true),
Field::new("c4", DataType::Boolean, true),
]));
let json_content = r#"
{"c1": "eleven", "c2": 6.2222222225, "c3": 5.0, "c4": false}
{"c1": "twelve", "c2": -55555555555555.2, "c3": 3}
{"c1": null, "c2": 3, "c3": 125, "c4": null}
{"c2": -35, "c3": 100.0, "c4": true}
{"c1": "fifteen", "c2": null, "c4": true}
{"c1": "eleven", "c2": 6.2222222225, "c3": 5.0, "c4": false}
{"c1": "twelve", "c2": -55555555555555.2, "c3": 3}
{"c1": null, "c2": 3, "c3": 125, "c4": null}
{"c2": -35, "c3": 100.0, "c4": true}
{"c1": "fifteen", "c2": null, "c4": true}
"#;
do_bench(c, "small_bench_primitive", json_content, schema)
}
fn small_bench_primitive_with_utf8view(c: &mut Criterion) {
let schema = Arc::new(Schema::new(vec![
Field::new("c1", DataType::Utf8View, true),
Field::new("c2", DataType::Float64, true),
Field::new("c3", DataType::UInt32, true),
Field::new("c4", DataType::Boolean, true),
]));
let json_content = r#"
{"c1": "eleven", "c2": 6.2222222225, "c3": 5.0, "c4": false}
{"c1": "twelve", "c2": -55555555555555.2, "c3": 3}
{"c1": null, "c2": 3, "c3": 125, "c4": null}
{"c2": -35, "c3": 100.0, "c4": true}
{"c1": "fifteen", "c2": null, "c4": true}
{"c1": "eleven", "c2": 6.2222222225, "c3": 5.0, "c4": false}
{"c1": "twelve", "c2": -55555555555555.2, "c3": 3}
{"c1": null, "c2": 3, "c3": 125, "c4": null}
{"c2": -35, "c3": 100.0, "c4": true}
{"c1": "fifteen", "c2": null, "c4": true}
"#;
do_bench(
c,
"small_bench_primitive_with_utf8view",
json_content,
schema,
)
}
fn large_bench_primitive(c: &mut Criterion) {
let schema = Arc::new(Schema::new(vec![
Field::new("c1", DataType::Utf8, true),
Field::new("c2", DataType::Int32, true),
Field::new("c3", DataType::UInt32, true),
Field::new("c4", DataType::Utf8, true),
Field::new("c5", DataType::Utf8, true),
Field::new("c6", DataType::Float32, true),
]));
let c1 = Arc::new(create_string_array::<i32>(4096, 0.));
let c2 = Arc::new(create_primitive_array::<Int32Type>(4096, 0.));
let c3 = Arc::new(create_primitive_array::<UInt32Type>(4096, 0.));
let c4 = Arc::new(create_string_array_with_len::<i32>(4096, 0.2, 10));
let c5 = Arc::new(create_string_array_with_len::<i32>(4096, 0.2, 20));
let c6 = Arc::new(create_primitive_array::<Float32Type>(4096, 0.2));
let batch = RecordBatch::try_from_iter([
("c1", c1 as _),
("c2", c2 as _),
("c3", c3 as _),
("c4", c4 as _),
("c5", c5 as _),
("c6", c6 as _),
])
.unwrap();
let mut out = Vec::with_capacity(1024);
LineDelimitedWriter::new(&mut out).write(&batch).unwrap();
let json = std::str::from_utf8(&out).unwrap();
do_bench(c, "large_bench_primitive", json, schema)
}
fn small_bench_list(c: &mut Criterion) {
let schema = Arc::new(Schema::new(vec![
Field::new(
"c1",
DataType::List(Arc::new(Field::new_list_field(DataType::Utf8, true))),
true,
),
Field::new(
"c2",
DataType::List(Arc::new(Field::new_list_field(DataType::Float64, true))),
true,
),
Field::new(
"c3",
DataType::List(Arc::new(Field::new_list_field(DataType::UInt32, true))),
true,
),
Field::new(
"c4",
DataType::List(Arc::new(Field::new_list_field(DataType::Boolean, true))),
true,
),
]));
let json = r#"
{"c1": ["eleven"], "c2": [6.2222222225, -3.2, null], "c3": [5.0, 6], "c4": [false, true]}
{"c1": ["twelve"], "c2": [-55555555555555.2, 12500000.0], "c3": [3, 4, 5]}
{"c1": null, "c2": [3], "c3": [125, 127, 129], "c4": [null, false, true]}
{"c2": [-35], "c3": [100.0, 200.0], "c4": null}
{"c1": ["fifteen"], "c2": [null, 2.1, 1.5, -3], "c4": [true, false, null]}
{"c1": ["fifteen"], "c2": [], "c4": [true, false, null]}
{"c1": ["eleven"], "c2": [6.2222222225, -3.2, null], "c3": [5.0, 6], "c4": [false, true]}
{"c1": ["twelve"], "c2": [-55555555555555.2, 12500000.0], "c3": [3, 4, 5]}
{"c1": null, "c2": [3], "c3": [125, 127, 129], "c4": [null, false, true]}
{"c2": [-35], "c3": [100.0, 200.0], "c4": null}
{"c1": ["fifteen"], "c2": [null, 2.1, 1.5, -3], "c4": [true, false, null]}
{"c1": ["fifteen"], "c2": [], "c4": [true, false, null]}
"#;
do_bench(c, "small_bench_list", json, schema)
}
fn criterion_benchmark(c: &mut Criterion) {
small_bench_primitive(c);
large_bench_primitive(c);
small_bench_list(c);
small_bench_primitive_with_utf8view(c);
}
criterion_group!(benches, criterion_benchmark);
criterion_main!(benches);
+336
View File
@@ -0,0 +1,336 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use criterion::*;
use arrow::datatypes::*;
use arrow::util::bench_util::{
create_primitive_array, create_string_array, create_string_array_with_len,
create_string_dict_array,
};
use arrow::util::test_util::seedable_rng;
use arrow_array::{Array, ListArray, RecordBatch, StructArray};
use arrow_buffer::{BooleanBuffer, NullBuffer, OffsetBuffer};
use arrow_json::{LineDelimitedWriter, ReaderBuilder};
use rand::Rng;
use serde::Serialize;
use std::sync::Arc;
const NUM_ROWS: usize = 65536;
fn do_bench(c: &mut Criterion, name: &str, batch: &RecordBatch) {
c.bench_function(name, |b| {
b.iter(|| {
let mut out = Vec::with_capacity(1024);
LineDelimitedWriter::new(&mut out).write(batch).unwrap();
out
})
});
}
fn create_mixed(len: usize) -> RecordBatch {
let c1 = Arc::new(create_string_array::<i32>(len, 0.));
let c2 = Arc::new(create_primitive_array::<Int32Type>(len, 0.));
let c3 = Arc::new(create_primitive_array::<UInt32Type>(len, 0.));
let c4 = Arc::new(create_string_array_with_len::<i32>(len, 0.2, 10));
let c5 = Arc::new(create_string_array_with_len::<i32>(len, 0.2, 20));
let c6 = Arc::new(create_primitive_array::<Float32Type>(len, 0.2));
RecordBatch::try_from_iter([
("c1", c1 as _),
("c2", c2 as _),
("c3", c3 as _),
("c4", c4 as _),
("c5", c5 as _),
("c6", c6 as _),
])
.unwrap()
}
fn create_nulls(len: usize) -> NullBuffer {
let mut rng = seedable_rng();
BooleanBuffer::from_iter((0..len).map(|_| rng.random_bool(0.2))).into()
}
fn create_offsets(len: usize) -> (usize, OffsetBuffer<i32>) {
let mut rng = seedable_rng();
let mut last_offset = 0;
let mut offsets = Vec::with_capacity(len + 1);
offsets.push(0);
for _ in 0..len {
let len = rng.random_range(0..10);
offsets.push(last_offset + len);
last_offset += len;
}
(
*offsets.last().unwrap() as _,
OffsetBuffer::new(offsets.into()),
)
}
fn create_nullable_struct(len: usize) -> StructArray {
let c2 = StructArray::from(create_mixed(len));
StructArray::new(
c2.fields().clone(),
c2.columns().to_vec(),
Some(create_nulls(c2.len())),
)
}
fn bench_float(c: &mut Criterion) {
let c1 = Arc::new(create_primitive_array::<Float32Type>(NUM_ROWS, 0.));
let c2 = Arc::new(create_primitive_array::<Float64Type>(NUM_ROWS, 0.));
let batch = RecordBatch::try_from_iter([("c1", c1 as _), ("c2", c2 as _)]).unwrap();
do_bench(c, "bench_float", &batch)
}
fn bench_integer(c: &mut Criterion) {
let c1 = Arc::new(create_primitive_array::<UInt64Type>(NUM_ROWS, 0.));
let c2 = Arc::new(create_primitive_array::<Int32Type>(NUM_ROWS, 0.));
let c3 = Arc::new(create_primitive_array::<UInt32Type>(NUM_ROWS, 0.));
let batch =
RecordBatch::try_from_iter([("c1", c1 as _), ("c2", c2 as _), ("c3", c3 as _)]).unwrap();
do_bench(c, "bench_integer", &batch)
}
fn bench_mixed(c: &mut Criterion) {
let batch = create_mixed(NUM_ROWS);
do_bench(c, "bench_mixed", &batch)
}
fn bench_dict_array(c: &mut Criterion) {
let c1 = Arc::new(create_string_dict_array::<Int32Type>(NUM_ROWS, 0., 30));
let c2 = Arc::new(create_string_dict_array::<Int32Type>(NUM_ROWS, 0., 20));
let c3 = Arc::new(create_string_dict_array::<Int32Type>(NUM_ROWS, 0.1, 20));
let batch =
RecordBatch::try_from_iter([("c1", c1 as _), ("c2", c2 as _), ("c3", c3 as _)]).unwrap();
do_bench(c, "bench_dict_array", &batch)
}
fn bench_string(c: &mut Criterion) {
let c1 = Arc::new(create_string_array::<i32>(NUM_ROWS, 0.));
let c2 = Arc::new(create_string_array_with_len::<i32>(NUM_ROWS, 0., 10));
let c3 = Arc::new(create_string_array_with_len::<i32>(NUM_ROWS, 0.1, 20));
let batch =
RecordBatch::try_from_iter([("c1", c1 as _), ("c2", c2 as _), ("c3", c3 as _)]).unwrap();
do_bench(c, "bench_string", &batch)
}
fn bench_struct(c: &mut Criterion) {
let c1 = Arc::new(create_string_array::<i32>(NUM_ROWS, 0.));
let c2 = Arc::new(StructArray::from(create_mixed(NUM_ROWS)));
let batch = RecordBatch::try_from_iter([("c1", c1 as _), ("c2", c2 as _)]).unwrap();
do_bench(c, "bench_struct", &batch)
}
fn bench_nullable_struct(c: &mut Criterion) {
let c1 = Arc::new(create_string_array::<i32>(NUM_ROWS, 0.));
let c2 = Arc::new(create_nullable_struct(NUM_ROWS));
let batch = RecordBatch::try_from_iter([("c1", c1 as _), ("c2", c2 as _)]).unwrap();
do_bench(c, "bench_nullable_struct", &batch)
}
fn bench_list(c: &mut Criterion) {
let (values_len, offsets) = create_offsets(NUM_ROWS);
let c1_values = Arc::new(create_string_array::<i32>(values_len, 0.));
let c1_field = Arc::new(Field::new_list_field(c1_values.data_type().clone(), false));
let c1 = Arc::new(ListArray::new(c1_field, offsets, c1_values, None));
let batch = RecordBatch::try_from_iter([("c1", c1 as _)]).unwrap();
do_bench(c, "bench_list", &batch)
}
fn bench_nullable_list(c: &mut Criterion) {
let (values_len, offsets) = create_offsets(NUM_ROWS);
let c1_values = Arc::new(create_string_array::<i32>(values_len, 0.1));
let c1_field = Arc::new(Field::new_list_field(c1_values.data_type().clone(), true));
let c1_nulls = create_nulls(NUM_ROWS);
let c1 = Arc::new(ListArray::new(c1_field, offsets, c1_values, Some(c1_nulls)));
let batch = RecordBatch::try_from_iter([("c1", c1 as _)]).unwrap();
do_bench(c, "bench_nullable_list", &batch)
}
fn bench_struct_list(c: &mut Criterion) {
let (values_len, offsets) = create_offsets(NUM_ROWS);
let c1_values = Arc::new(create_nullable_struct(values_len));
let c1_field = Arc::new(Field::new_list_field(c1_values.data_type().clone(), true));
let c1_nulls = create_nulls(NUM_ROWS);
let c1 = Arc::new(ListArray::new(c1_field, offsets, c1_values, Some(c1_nulls)));
let batch = RecordBatch::try_from_iter([("c1", c1 as _)]).unwrap();
do_bench(c, "bench_struct_list", &batch)
}
fn do_number_to_string_bench<S: Serialize>(
name: &str,
c: &mut Criterion,
schema: Arc<Schema>,
rows: Vec<S>,
) {
c.bench_function(name, |b| {
b.iter(|| {
let mut decoder = ReaderBuilder::new(schema.clone())
.with_coerce_primitive(true) // important for coercion
.build_decoder()
.expect("Failed to build decoder");
decoder.serialize(&rows).expect("Failed to serialize rows");
decoder
.flush()
.expect("Failed to flush")
.expect("No RecordBatch produced");
})
});
}
fn bench_i64_to_string(c: &mut Criterion) {
#[derive(Serialize)]
struct TestRow {
val: i64,
}
let schema = Arc::new(Schema::new(vec![Field::new("val", DataType::Utf8, false)]));
let a_bunch_of_numbers = create_primitive_array::<Int64Type>(NUM_ROWS, 0.0);
let rows: Vec<TestRow> = (0..NUM_ROWS)
.map(|i| TestRow {
val: a_bunch_of_numbers.value(i),
})
.collect();
do_number_to_string_bench("i64_to_string", c, schema, rows)
}
fn bench_i32_to_string(c: &mut Criterion) {
#[derive(Serialize)]
struct TestRow {
val: i32,
}
let schema = Arc::new(Schema::new(vec![Field::new("val", DataType::Utf8, false)]));
let a_bunch_of_numbers = create_primitive_array::<Int32Type>(NUM_ROWS, 0.0);
let rows: Vec<TestRow> = (0..NUM_ROWS)
.map(|i| TestRow {
val: a_bunch_of_numbers.value(i),
})
.collect();
do_number_to_string_bench("i32_to_string", c, schema, rows)
}
fn bench_f32_to_string(c: &mut Criterion) {
#[derive(Serialize)]
struct TestRow {
val: f32,
}
let schema = Arc::new(Schema::new(vec![Field::new("val", DataType::Utf8, false)]));
let a_bunch_of_numbers = create_primitive_array::<Float32Type>(NUM_ROWS, 0.0);
let rows: Vec<TestRow> = (0..NUM_ROWS)
.map(|i| TestRow {
val: a_bunch_of_numbers.value(i),
})
.collect();
do_number_to_string_bench("f32_to_string", c, schema, rows)
}
fn bench_f64_to_string(c: &mut Criterion) {
#[derive(Serialize)]
struct TestRow {
val: f64,
}
let schema = Arc::new(Schema::new(vec![Field::new("val", DataType::Utf8, false)]));
let a_bunch_of_numbers = create_primitive_array::<Float64Type>(NUM_ROWS, 0.0);
let rows: Vec<TestRow> = (0..NUM_ROWS)
.map(|i| TestRow {
val: a_bunch_of_numbers.value(i),
})
.collect();
do_number_to_string_bench("f64_to_string", c, schema, rows)
}
fn bench_mixed_numbers_to_string(c: &mut Criterion) {
#[derive(Serialize)]
struct TestRow {
val1: f64,
val2: f32,
val3: i64,
val4: i32,
}
let schema = Arc::new(Schema::new(vec![
Field::new("val1", DataType::Utf8, false),
Field::new("val2", DataType::Utf8, false),
Field::new("val3", DataType::Utf8, false),
Field::new("val4", DataType::Utf8, false),
]));
let f64_array = create_primitive_array::<Float64Type>(NUM_ROWS, 0.0);
let f32_array = create_primitive_array::<Float32Type>(NUM_ROWS, 0.0);
let i64_array = create_primitive_array::<Int64Type>(NUM_ROWS, 0.0);
let i32_array = create_primitive_array::<Int32Type>(NUM_ROWS, 0.0);
let rows: Vec<TestRow> = (0..NUM_ROWS)
.map(|i| TestRow {
val1: f64_array.value(i),
val2: f32_array.value(i),
val3: i64_array.value(i),
val4: i32_array.value(i),
})
.collect();
do_number_to_string_bench("mixed_numbers_to_string", c, schema, rows)
}
fn criterion_benchmark(c: &mut Criterion) {
bench_integer(c);
bench_float(c);
bench_string(c);
bench_mixed(c);
bench_dict_array(c);
bench_struct(c);
bench_nullable_struct(c);
bench_list(c);
bench_nullable_list(c);
bench_struct_list(c);
bench_f64_to_string(c);
bench_f32_to_string(c);
bench_i64_to_string(c);
bench_i32_to_string(c);
bench_mixed_numbers_to_string(c);
}
criterion_group!(benches, criterion_benchmark);
criterion_main!(benches);
+48
View File
@@ -0,0 +1,48 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use criterion::Criterion;
extern crate arrow;
use arrow::array::*;
use arrow::compute::kernels::length::length;
use std::hint;
fn bench_length(array: &StringArray) {
hint::black_box(length(array).unwrap());
}
fn add_benchmark(c: &mut Criterion) {
fn double_vec<T: Clone>(v: Vec<T>) -> Vec<T> {
[&v[..], &v[..]].concat()
}
// double ["hello", " ", "world", "!"] 10 times
let mut values = vec!["one", "on", "o", ""];
for _ in 0..10 {
values = double_vec(values);
}
let array = StringArray::from(values);
c.bench_function("length", |b| b.iter(|| bench_length(&array)));
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
+221
View File
@@ -0,0 +1,221 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use arrow::compute::{SortColumn, lexsort_to_indices};
use arrow::row::{RowConverter, SortField};
use arrow::util::bench_util::{
create_dict_from_values, create_primitive_array, create_string_array_with_len,
};
use arrow::util::data_gen::create_random_array;
use arrow_array::types::Int32Type;
use arrow_array::{Array, ArrayRef, UInt32Array};
use arrow_schema::{DataType, Field};
use criterion::{Criterion, criterion_group, criterion_main};
use std::{hint, sync::Arc};
#[derive(Copy, Clone)]
enum Column {
RequiredI32,
OptionalI32,
Required16CharString,
Optional16CharString,
Optional50CharString,
Optional100Value50CharStringDict,
RequiredI32List,
OptionalI32List,
Required4CharStringList,
Optional4CharStringList,
}
impl std::fmt::Debug for Column {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
let s = match self {
Column::RequiredI32 => "i32",
Column::OptionalI32 => "i32_opt",
Column::Required16CharString => "str(16)",
Column::Optional16CharString => "str_opt(16)",
Column::Optional50CharString => "str_opt(50)",
Column::Optional100Value50CharStringDict => "dict(100,str_opt(50))",
Column::RequiredI32List => "i32_list",
Column::OptionalI32List => "i32_list_opt",
Column::Required4CharStringList => "str_list(4)",
Column::Optional4CharStringList => "str_list_opt(4)",
};
f.write_str(s)
}
}
impl Column {
fn generate(self, size: usize) -> ArrayRef {
match self {
Column::RequiredI32 => Arc::new(create_primitive_array::<Int32Type>(size, 0.)),
Column::OptionalI32 => Arc::new(create_primitive_array::<Int32Type>(size, 0.2)),
Column::Required16CharString => {
Arc::new(create_string_array_with_len::<i32>(size, 0., 16))
}
Column::Optional16CharString => {
Arc::new(create_string_array_with_len::<i32>(size, 0.2, 16))
}
Column::Optional50CharString => {
Arc::new(create_string_array_with_len::<i32>(size, 0., 50))
}
Column::Optional100Value50CharStringDict => {
Arc::new(create_dict_from_values::<Int32Type>(
size,
0.1,
&create_string_array_with_len::<i32>(100, 0., 50),
))
}
Column::RequiredI32List => {
let field = Field::new(
"_1",
DataType::List(Arc::new(Field::new_list_field(DataType::Int32, false))),
true,
);
create_random_array(&field, size, 0., 1.).unwrap()
}
Column::OptionalI32List => {
let field = Field::new(
"_1",
DataType::List(Arc::new(Field::new_list_field(DataType::Int32, true))),
true,
);
create_random_array(&field, size, 0.2, 1.).unwrap()
}
Column::Required4CharStringList => {
let field = Field::new(
"_1",
DataType::List(Arc::new(Field::new_list_field(DataType::Utf8, false))),
true,
);
create_random_array(&field, size, 0., 1.).unwrap()
}
Column::Optional4CharStringList => {
let field = Field::new(
"_1",
DataType::List(Arc::new(Field::new_list_field(DataType::Utf8, true))),
true,
);
create_random_array(&field, size, 0.2, 1.).unwrap()
}
}
}
}
fn do_bench(c: &mut Criterion, columns: &[Column], len: usize) {
let arrays: Vec<_> = columns.iter().map(|x| x.generate(len)).collect();
let sort_columns: Vec<_> = arrays
.iter()
.cloned()
.map(|values| SortColumn {
values,
options: None,
})
.collect();
c.bench_function(&format!("lexsort_to_indices({columns:?}): {len}"), |b| {
b.iter(|| hint::black_box(lexsort_to_indices(&sort_columns, None).unwrap()))
});
c.bench_function(&format!("lexsort_rows({columns:?}): {len}"), |b| {
b.iter(|| {
hint::black_box({
let fields = arrays
.iter()
.map(|a| SortField::new(a.data_type().clone()))
.collect();
let converter = RowConverter::new(fields).unwrap();
let rows = converter.convert_columns(&arrays).unwrap();
let mut sort: Vec<_> = rows.iter().enumerate().collect();
sort.sort_unstable_by(|(_, a), (_, b)| a.cmp(b));
UInt32Array::from_iter_values(sort.iter().map(|(i, _)| *i as u32))
})
})
});
}
fn add_benchmark(c: &mut Criterion) {
let cases: &[&[Column]] = &[
&[Column::RequiredI32, Column::OptionalI32],
&[Column::RequiredI32, Column::Optional16CharString],
&[Column::RequiredI32, Column::Required16CharString],
&[Column::Optional16CharString, Column::Required16CharString],
&[
Column::Optional16CharString,
Column::Optional50CharString,
Column::Required16CharString,
],
&[
Column::Optional16CharString,
Column::Required16CharString,
Column::Optional16CharString,
Column::Optional16CharString,
Column::Optional16CharString,
],
&[
Column::OptionalI32,
Column::Optional100Value50CharStringDict,
],
&[
Column::Optional100Value50CharStringDict,
Column::Optional100Value50CharStringDict,
],
&[
Column::Optional100Value50CharStringDict,
Column::Optional100Value50CharStringDict,
Column::Optional100Value50CharStringDict,
Column::Required16CharString,
],
&[
Column::Optional100Value50CharStringDict,
Column::Optional100Value50CharStringDict,
Column::Optional100Value50CharStringDict,
Column::Optional50CharString,
],
&[
Column::Optional100Value50CharStringDict,
Column::Optional100Value50CharStringDict,
Column::Optional100Value50CharStringDict,
Column::Optional50CharString,
],
&[Column::OptionalI32, Column::RequiredI32List],
&[Column::OptionalI32, Column::OptionalI32List],
&[Column::OptionalI32List, Column::OptionalI32],
&[Column::RequiredI32, Column::Required4CharStringList],
&[Column::Required4CharStringList, Column::RequiredI32],
&[Column::RequiredI32, Column::Optional4CharStringList],
&[Column::Optional4CharStringList, Column::RequiredI32],
&[
Column::RequiredI32,
Column::RequiredI32List,
Column::Required16CharString,
],
&[
Column::OptionalI32,
Column::OptionalI32List,
Column::Optional50CharString,
],
];
for case in cases {
do_bench(c, case, 4096);
do_bench(c, case, 4096 * 8);
}
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
+280
View File
@@ -0,0 +1,280 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use criterion::measurement::WallTime;
use criterion::{BenchmarkGroup, BenchmarkId, Criterion, criterion_group, criterion_main};
use rand::distr::{Distribution, StandardUniform};
use rand::prelude::StdRng;
use rand::{Rng, SeedableRng};
use std::hint;
use std::sync::Arc;
use arrow::array::*;
use arrow::datatypes::*;
use arrow::util::bench_util::*;
use arrow_select::merge::merge;
trait InputGenerator {
fn name(&self) -> &str;
/// Return an ArrayRef containing a single null value
fn generate_scalar_with_null_value(&self) -> ArrayRef;
/// Generate a `number_of_scalars` unique scalars
fn generate_non_null_scalars(&self, seed: u64, number_of_scalars: usize) -> Vec<ArrayRef>;
/// Generate an array with the specified length and null percentage
fn generate_array(&self, seed: u64, array_length: usize, null_percentage: f32) -> ArrayRef;
}
struct GeneratePrimitive<T: ArrowPrimitiveType> {
description: String,
_marker: std::marker::PhantomData<T>,
}
impl<T> InputGenerator for GeneratePrimitive<T>
where
T: ArrowPrimitiveType,
StandardUniform: Distribution<T::Native>,
{
fn name(&self) -> &str {
self.description.as_str()
}
fn generate_scalar_with_null_value(&self) -> ArrayRef {
new_null_array(&T::DATA_TYPE, 1)
}
fn generate_non_null_scalars(&self, seed: u64, number_of_scalars: usize) -> Vec<ArrayRef> {
let rng = StdRng::seed_from_u64(seed);
rng.sample_iter::<T::Native, _>(StandardUniform)
.take(number_of_scalars)
.map(|v: T::Native| {
Arc::new(PrimitiveArray::<T>::new_scalar(v).into_inner()) as ArrayRef
})
.collect()
}
fn generate_array(&self, seed: u64, array_length: usize, null_percentage: f32) -> ArrayRef {
Arc::new(create_primitive_array_with_seed::<T>(
array_length,
null_percentage,
seed,
))
}
}
struct GenerateBytes<Byte: ByteArrayType> {
range_length: std::ops::Range<usize>,
description: String,
_marker: std::marker::PhantomData<Byte>,
}
impl<Byte> InputGenerator for GenerateBytes<Byte>
where
Byte: ByteArrayType,
{
fn name(&self) -> &str {
self.description.as_str()
}
fn generate_scalar_with_null_value(&self) -> ArrayRef {
new_null_array(&Byte::DATA_TYPE, 1)
}
fn generate_non_null_scalars(&self, seed: u64, number_of_scalars: usize) -> Vec<ArrayRef> {
let array = self.generate_array(seed, number_of_scalars, 0.0);
(0..number_of_scalars).map(|i| array.slice(i, 1)).collect()
}
fn generate_array(&self, seed: u64, array_length: usize, null_percentage: f32) -> ArrayRef {
let is_binary =
Byte::DATA_TYPE == DataType::Binary || Byte::DATA_TYPE == DataType::LargeBinary;
if is_binary {
Arc::new(create_binary_array_with_len_range_and_prefix_and_seed::<
Byte::Offset,
>(
array_length,
null_percentage,
self.range_length.start,
self.range_length.end - 1,
&[],
seed,
))
} else {
Arc::new(create_string_array_with_len_range_and_prefix_and_seed::<
Byte::Offset,
>(
array_length,
null_percentage,
self.range_length.start,
self.range_length.end - 1,
"",
seed,
))
}
}
}
fn mask_cases(len: usize) -> Vec<(&'static str, BooleanArray)> {
vec![
("all_true", create_boolean_array(len, 0.0, 1.0)),
("99pct_true", create_boolean_array(len, 0.0, 0.99)),
("90pct_true", create_boolean_array(len, 0.0, 0.9)),
("50pct_true", create_boolean_array(len, 0.0, 0.5)),
("10pct_true", create_boolean_array(len, 0.0, 0.1)),
("1pct_true", create_boolean_array(len, 0.0, 0.01)),
("all_false", create_boolean_array(len, 0.0, 0.0)),
("50pct_nulls", create_boolean_array(len, 0.5, 0.5)),
]
}
fn bench_merge_on_input_generator(c: &mut Criterion, input_generator: &impl InputGenerator) {
const ARRAY_LEN: usize = 8192;
let mut group =
c.benchmark_group(format!("merge_{ARRAY_LEN}_from_{}", input_generator.name()).as_str());
let null_scalar = input_generator.generate_scalar_with_null_value();
let [non_null_scalar_1, non_null_scalar_2]: [_; 2] = input_generator
.generate_non_null_scalars(42, 2)
.try_into()
.unwrap();
// For simplicity, we generate arrays with length ARRAY_LEN. Not all input values will be used.
let array_1_10pct_nulls = input_generator.generate_array(42, ARRAY_LEN, 0.1);
let array_2_10pct_nulls = input_generator.generate_array(18, ARRAY_LEN, 0.1);
let masks = mask_cases(ARRAY_LEN);
// Benchmarks for different scalar combinations
for (description, truthy, falsy) in &[
("null_vs_non_null_scalar", &null_scalar, &non_null_scalar_1),
(
"non_null_scalar_vs_null_scalar",
&non_null_scalar_1,
&null_scalar,
),
("non_nulls_scalars", &non_null_scalar_1, &non_null_scalar_2),
] {
bench_merge_input_on_all_masks(
description,
&mut group,
&masks,
&Scalar::new(truthy),
&Scalar::new(falsy),
);
}
bench_merge_input_on_all_masks(
"array_vs_non_null_scalar",
&mut group,
&masks,
&array_1_10pct_nulls,
&non_null_scalar_1,
);
bench_merge_input_on_all_masks(
"non_null_scalar_vs_array",
&mut group,
&masks,
&non_null_scalar_1,
&array_1_10pct_nulls,
);
bench_merge_input_on_all_masks(
"array_vs_array",
&mut group,
&masks,
&array_1_10pct_nulls,
&array_2_10pct_nulls,
);
group.finish();
}
fn bench_merge_input_on_all_masks(
description: &str,
group: &mut BenchmarkGroup<WallTime>,
masks: &[(&str, BooleanArray)],
truthy: &impl Datum,
falsy: &impl Datum,
) {
for (mask_description, mask) in masks {
let id = BenchmarkId::new(description, mask_description);
group.bench_with_input(id, mask, |b, mask| {
b.iter(|| hint::black_box(merge(mask, truthy, falsy)))
});
}
}
fn add_benchmark(c: &mut Criterion) {
// Primitive
bench_merge_on_input_generator(
c,
&GeneratePrimitive::<Int32Type> {
description: "i32".to_string(),
_marker: std::marker::PhantomData,
},
);
// Short strings
bench_merge_on_input_generator(
c,
&GenerateBytes::<GenericStringType<i32>> {
description: "short strings (3..10)".to_string(),
range_length: 3..10,
_marker: std::marker::PhantomData,
},
);
// Long strings
bench_merge_on_input_generator(
c,
&GenerateBytes::<GenericStringType<i32>> {
description: "long strings (100..400)".to_string(),
range_length: 100..400,
_marker: std::marker::PhantomData,
},
);
// Short Bytes
bench_merge_on_input_generator(
c,
&GenerateBytes::<GenericBinaryType<i32>> {
description: "short bytes (3..10)".to_string(),
range_length: 3..10,
_marker: std::marker::PhantomData,
},
);
// Long Bytes
bench_merge_on_input_generator(
c,
&GenerateBytes::<GenericBinaryType<i32>> {
description: "long bytes (100..400)".to_string(),
range_length: 100..400,
_marker: std::marker::PhantomData,
},
);
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
+61
View File
@@ -0,0 +1,61 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use criterion::Criterion;
use rand::Rng;
extern crate arrow;
use arrow::util::test_util::seedable_rng;
use arrow::{array::*, util::bench_util::create_string_array};
fn create_slices(size: usize) -> Vec<(usize, usize)> {
let rng = &mut seedable_rng();
(0..size)
.map(|_| {
let start = rng.random_range(0..size / 2);
let end = rng.random_range(start + 1..size);
(start, end)
})
.collect()
}
fn bench<T: Array>(v1: &T, slices: &[(usize, usize)]) {
let data = v1.to_data();
let mut mutable = MutableArrayData::new(vec![&data], false, 5);
for (start, end) in slices {
mutable.extend(0, *start, *end)
}
mutable.freeze();
}
fn add_benchmark(c: &mut Criterion) {
let v1 = create_string_array::<i32>(1024, 0.0);
let v2 = create_slices(1024);
c.bench_function("mutable str 1024", |b| b.iter(|| bench(&v1, &v2)));
let v1 = create_string_array::<i32>(1024, 0.5);
let v2 = create_slices(1024);
c.bench_function("mutable str nulls 1024", |b| b.iter(|| bench(&v1, &v2)));
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
+130
View File
@@ -0,0 +1,130 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use criterion::Criterion;
use std::sync::Arc;
extern crate arrow;
use arrow::compute::kernels::sort::{SortColumn, lexsort};
use arrow::util::bench_util::*;
use arrow::{
array::*,
datatypes::{Float64Type, UInt8Type},
};
use arrow_ord::partition::partition;
use rand::distr::{Distribution, StandardUniform};
use std::hint;
fn create_array<T: ArrowPrimitiveType>(size: usize, with_nulls: bool) -> ArrayRef
where
StandardUniform: Distribution<T::Native>,
{
let null_density = if with_nulls { 0.5 } else { 0.0 };
let array = create_primitive_array::<T>(size, null_density);
Arc::new(array)
}
fn bench_partition(sorted_columns: &[ArrayRef]) {
hint::black_box(partition(sorted_columns).unwrap().ranges());
}
fn create_sorted_low_cardinality_data(length: usize) -> Vec<ArrayRef> {
let arr = Int64Array::from_iter_values(
std::iter::repeat_n(1, length / 4)
.chain(std::iter::repeat_n(2, length / 4))
.chain(std::iter::repeat_n(3, length / 4))
.chain(std::iter::repeat_n(4, length / 4)),
);
lexsort(
&[SortColumn {
values: Arc::new(arr),
options: None,
}],
None,
)
.unwrap()
}
fn create_sorted_float_data(pow: u32, with_nulls: bool) -> Vec<ArrayRef> {
lexsort(
&[
SortColumn {
values: create_array::<Float64Type>(2u64.pow(pow) as usize, with_nulls),
options: None,
},
SortColumn {
values: create_array::<Float64Type>(2u64.pow(pow) as usize, with_nulls),
options: None,
},
],
None,
)
.unwrap()
}
fn create_sorted_data(pow: u32, with_nulls: bool) -> Vec<ArrayRef> {
lexsort(
&[
SortColumn {
values: create_array::<UInt8Type>(2u64.pow(pow) as usize, with_nulls),
options: None,
},
SortColumn {
values: create_array::<UInt8Type>(2u64.pow(pow) as usize, with_nulls),
options: None,
},
],
None,
)
.unwrap()
}
fn add_benchmark(c: &mut Criterion) {
let sorted_columns = create_sorted_data(10, false);
c.bench_function("partition(u8) 2^10", |b| {
b.iter(|| bench_partition(&sorted_columns))
});
let sorted_columns = create_sorted_data(12, false);
c.bench_function("partition(u8) 2^12", |b| {
b.iter(|| bench_partition(&sorted_columns))
});
let sorted_columns = create_sorted_data(10, true);
c.bench_function("partition(u8) 2^10 with nulls", |b| {
b.iter(|| bench_partition(&sorted_columns))
});
let sorted_columns = create_sorted_data(12, true);
c.bench_function("partition(u8) 2^12 with nulls", |b| {
b.iter(|| bench_partition(&sorted_columns))
});
let sorted_columns = create_sorted_float_data(10, false);
c.bench_function("partition(f64) 2^10", |b| {
b.iter(|| bench_partition(&sorted_columns))
});
let sorted_columns = create_sorted_low_cardinality_data(1024);
c.bench_function("partition(low cardinality) 1024", |b| {
b.iter(|| bench_partition(&sorted_columns))
});
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
@@ -0,0 +1,54 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use arrow::datatypes::Int32Type;
use arrow::{array::PrimitiveArray, util::bench_util::create_primitive_run_array};
use arrow_array::ArrayAccessor;
use criterion::{Criterion, criterion_group, criterion_main};
fn criterion_benchmark(c: &mut Criterion) {
let mut group = c.benchmark_group("primitive_run_accessor");
let mut do_bench = |physical_array_len: usize, logical_array_len: usize| {
group.bench_function(
format!("(run_array_len:{logical_array_len}, physical_array_len:{physical_array_len})"),
|b| {
let run_array = create_primitive_run_array::<Int32Type, Int32Type>(
logical_array_len,
physical_array_len,
);
let typed = run_array.downcast::<PrimitiveArray<Int32Type>>().unwrap();
b.iter(|| {
for i in 0..logical_array_len {
let _ = unsafe { typed.value_unchecked(i) };
}
})
},
);
};
do_bench(128, 512);
do_bench(256, 1024);
do_bench(512, 2048);
do_bench(1024, 4096);
do_bench(2048, 8192);
group.finish();
}
criterion_group!(benches, criterion_benchmark);
criterion_main!(benches);
@@ -0,0 +1,78 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use arrow::array::UInt32Builder;
use arrow::compute::take;
use arrow::datatypes::{Int32Type, Int64Type};
use arrow::util::bench_util::*;
use arrow::util::test_util::seedable_rng;
use arrow_array::UInt32Array;
use criterion::{Criterion, criterion_group, criterion_main};
use rand::Rng;
use std::hint;
fn create_random_index(size: usize, null_density: f32, max_value: usize) -> UInt32Array {
let mut rng = seedable_rng();
let mut builder = UInt32Builder::with_capacity(size);
for _ in 0..size {
if rng.random::<f32>() < null_density {
builder.append_null();
} else {
let value = rng.random_range::<u32, _>(0u32..max_value as u32);
builder.append_value(value);
}
}
builder.finish()
}
fn criterion_benchmark(c: &mut Criterion) {
let mut group = c.benchmark_group("primitive_run_take");
let mut do_bench = |physical_array_len: usize, logical_array_len: usize, take_len: usize| {
let run_array = create_primitive_run_array::<Int32Type, Int64Type>(
logical_array_len,
physical_array_len,
);
let indices = create_random_index(take_len, 0.0, logical_array_len);
group.bench_function(
format!(
"(run_array_len:{logical_array_len}, physical_array_len:{physical_array_len}, take_len:{take_len})"),
|b| {
b.iter(|| {
hint::black_box(take(&run_array, &indices, None).unwrap());
})
},
);
};
do_bench(64, 512, 512);
do_bench(128, 512, 512);
do_bench(256, 1024, 512);
do_bench(256, 1024, 1024);
do_bench(512, 2048, 512);
do_bench(512, 2048, 1024);
do_bench(1024, 4096, 512);
do_bench(1024, 4096, 1024);
group.finish();
}
criterion_group!(benches, criterion_benchmark);
criterion_main!(benches);
+52
View File
@@ -0,0 +1,52 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use criterion::Criterion;
extern crate arrow;
use arrow::array::*;
use arrow::compute::kernels::regexp::*;
use arrow::util::bench_util::*;
use std::hint;
fn bench_regexp(arr: &GenericStringArray<i32>, regex_array: &dyn Datum) {
regexp_match(hint::black_box(arr), regex_array, None).unwrap();
}
fn add_benchmark(c: &mut Criterion) {
let size = 65536;
let val_len = 1000;
let arr_string = create_string_array_with_len::<i32>(size, 0.0, val_len);
let pattern_values = vec![r".*-(\d*)-.*"; size];
let pattern = GenericStringArray::<i32>::from(pattern_values);
c.bench_function("regexp", |b| b.iter(|| bench_regexp(&arr_string, &pattern)));
let pattern_values = vec![r".*-(\d*)-.*"];
let pattern = Scalar::new(GenericStringArray::<i32>::from(pattern_values));
c.bench_function("regexp scalar", |b| {
b.iter(|| bench_regexp(&arr_string, &pattern))
});
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
+286
View File
@@ -0,0 +1,286 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
extern crate core;
use arrow::array::ArrayRef;
use arrow::datatypes::{Int64Type, UInt64Type};
use arrow::row::{RowConverter, SortField};
use arrow::util::bench_util::{
create_boolean_array, create_dict_from_values, create_primitive_array,
create_string_array_with_len, create_string_dict_array, create_string_view_array_with_len,
create_string_view_array_with_max_len,
};
use arrow::util::data_gen::create_random_array;
use arrow_array::Array;
use arrow_array::types::Int32Type;
use arrow_schema::{DataType, Field};
use criterion::Criterion;
use std::{hint, sync::Arc};
fn do_bench(c: &mut Criterion, name: &str, cols: Vec<ArrayRef>) {
let fields: Vec<_> = cols
.iter()
.map(|x| SortField::new(x.data_type().clone()))
.collect();
c.bench_function(&format!("convert_columns {name}"), |b| {
b.iter(|| {
let converter = RowConverter::new(fields.clone()).unwrap();
hint::black_box(converter.convert_columns(&cols).unwrap())
});
});
let converter = RowConverter::new(fields).unwrap();
let rows = converter.convert_columns(&cols).unwrap();
// using a pre-prepared row converter should be faster than the first time
c.bench_function(&format!("convert_columns_prepared {name}"), |b| {
b.iter(|| hint::black_box(converter.convert_columns(&cols).unwrap()));
});
c.bench_function(&format!("convert_rows {name}"), |b| {
b.iter(|| hint::black_box(converter.convert_rows(&rows).unwrap()));
});
let mut rows = converter.empty_rows(0, 0);
c.bench_function(&format!("append_rows {name}"), |b| {
let cols = cols.clone();
b.iter(|| {
rows.clear();
converter.append(&mut rows, &cols).unwrap();
hint::black_box(&mut rows);
});
});
}
fn bench_iter(c: &mut Criterion) {
let col = create_string_view_array_with_len(4096, 0., 100, false);
let converter = RowConverter::new(vec![SortField::new(col.data_type().clone())]).unwrap();
let rows = converter
.convert_columns(&[Arc::new(col) as ArrayRef])
.unwrap();
c.bench_function("iterate rows", |b| {
b.iter(|| {
for r in rows.iter() {
hint::black_box(r.as_ref());
}
})
});
}
fn row_bench(c: &mut Criterion) {
let cols = vec![Arc::new(create_primitive_array::<UInt64Type>(4096, 0.)) as ArrayRef];
do_bench(c, "4096 u64(0)", cols);
let cols = vec![Arc::new(create_primitive_array::<UInt64Type>(4096, 0.3)) as ArrayRef];
do_bench(c, "4096 u64(0.3)", cols);
let cols = vec![Arc::new(create_primitive_array::<Int64Type>(4096, 0.)) as ArrayRef];
do_bench(c, "4096 i64(0)", cols);
let cols = vec![Arc::new(create_primitive_array::<Int64Type>(4096, 0.3)) as ArrayRef];
do_bench(c, "4096 i64(0.3)", cols);
let cols = vec![Arc::new(create_boolean_array(4096, 0., 0.5)) as ArrayRef];
do_bench(c, "4096 bool(0, 0.5)", cols);
let cols = vec![Arc::new(create_boolean_array(4096, 0.3, 0.5)) as ArrayRef];
do_bench(c, "4096 bool(0.3, 0.5)", cols);
let cols = vec![Arc::new(create_string_array_with_len::<i32>(4096, 0., 10)) as ArrayRef];
do_bench(c, "4096 string(10, 0)", cols);
let cols = vec![Arc::new(create_string_array_with_len::<i32>(4096, 0., 30)) as ArrayRef];
do_bench(c, "4096 string(30, 0)", cols);
let cols = vec![Arc::new(create_string_array_with_len::<i32>(4096, 0., 100)) as ArrayRef];
do_bench(c, "4096 string(100, 0)", cols);
let cols = vec![Arc::new(create_string_array_with_len::<i32>(4096, 0.5, 100)) as ArrayRef];
do_bench(c, "4096 string(100, 0.5)", cols);
let cols = vec![Arc::new(create_string_view_array_with_len(4096, 0., 10, false)) as ArrayRef];
do_bench(c, "4096 string view(10, 0)", cols);
let cols = vec![Arc::new(create_string_view_array_with_len(4096, 0., 30, false)) as ArrayRef];
do_bench(c, "4096 string view(30, 0)", cols);
let cols = vec![Arc::new(create_string_view_array_with_len(4096, 0., 100, false)) as ArrayRef];
do_bench(c, "4096 string view(100, 0)", cols);
let cols = vec![Arc::new(create_string_view_array_with_len(4096, 0.5, 100, false)) as ArrayRef];
do_bench(c, "4096 string view(100, 0.5)", cols);
let cols = vec![Arc::new(create_string_view_array_with_max_len(4096, 0., 100)) as ArrayRef];
do_bench(c, "4096 string view(1..100, 0)", cols);
let cols = vec![Arc::new(create_string_view_array_with_max_len(4096, 0.5, 100)) as ArrayRef];
do_bench(c, "4096 string view(1..100, 0.5)", cols);
let cols = vec![Arc::new(create_string_dict_array::<Int32Type>(4096, 0., 10)) as ArrayRef];
do_bench(c, "4096 string_dictionary(10, 0)", cols);
let cols = vec![Arc::new(create_string_dict_array::<Int32Type>(4096, 0., 30)) as ArrayRef];
do_bench(c, "4096 string_dictionary(30, 0)", cols);
let cols = vec![Arc::new(create_string_dict_array::<Int32Type>(4096, 0., 100)) as ArrayRef];
do_bench(c, "4096 string_dictionary(100, 0)", cols.clone());
let cols = vec![Arc::new(create_string_dict_array::<Int32Type>(4096, 0.5, 100)) as ArrayRef];
do_bench(c, "4096 string_dictionary(100, 0.5)", cols.clone());
let values = create_string_array_with_len::<i32>(10, 0., 10);
let dict = create_dict_from_values::<Int32Type>(4096, 0., &values);
let cols = vec![Arc::new(dict) as ArrayRef];
do_bench(c, "4096 string_dictionary_low_cardinality(10, 0)", cols);
let values = create_string_array_with_len::<i32>(10, 0., 30);
let dict = create_dict_from_values::<Int32Type>(4096, 0., &values);
let cols = vec![Arc::new(dict) as ArrayRef];
do_bench(c, "4096 string_dictionary_low_cardinality(30, 0)", cols);
let values = create_string_array_with_len::<i32>(10, 0., 100);
let dict = create_dict_from_values::<Int32Type>(4096, 0., &values);
let cols = vec![Arc::new(dict) as ArrayRef];
do_bench(c, "4096 string_dictionary_low_cardinality(100, 0)", cols);
let cols = vec![
Arc::new(create_string_array_with_len::<i32>(4096, 0.5, 20)) as ArrayRef,
Arc::new(create_string_array_with_len::<i32>(4096, 0., 30)) as ArrayRef,
Arc::new(create_string_array_with_len::<i32>(4096, 0., 100)) as ArrayRef,
Arc::new(create_primitive_array::<Int64Type>(4096, 0.)) as ArrayRef,
];
do_bench(
c,
"4096 string(20, 0.5), string(30, 0), string(100, 0), i64(0)",
cols,
);
let cols = vec![
Arc::new(create_string_dict_array::<Int32Type>(4096, 0.5, 20)) as ArrayRef,
Arc::new(create_string_dict_array::<Int32Type>(4096, 0., 30)) as ArrayRef,
Arc::new(create_string_dict_array::<Int32Type>(4096, 0., 100)) as ArrayRef,
Arc::new(create_primitive_array::<Int64Type>(4096, 0.)) as ArrayRef,
];
do_bench(
c,
"4096 4096 string_dictionary(20, 0.5), string_dictionary(30, 0), string_dictionary(100, 0), i64(0)",
cols,
);
// List
let cols = vec![
create_random_array(
&Field::new(
"list",
DataType::List(Arc::new(Field::new_list_field(DataType::UInt64, false))),
false,
),
4096,
0.,
1.0,
)
.unwrap(),
];
do_bench(c, "4096 list(0) of u64(0)", cols);
let cols = vec![
create_random_array(
&Field::new(
"list",
DataType::LargeList(Arc::new(Field::new_list_field(DataType::UInt64, false))),
false,
),
4096,
0.,
1.0,
)
.unwrap(),
];
do_bench(c, "4096 large_list(0) of u64(0)", cols);
let cols = vec![
create_random_array(
&Field::new(
"list",
DataType::List(Arc::new(Field::new_list_field(DataType::UInt64, false))),
false,
),
10,
0.,
1.0,
)
.unwrap(),
];
do_bench(c, "10 list(0) of u64(0)", cols);
let cols = vec![
create_random_array(
&Field::new(
"list",
DataType::LargeList(Arc::new(Field::new_list_field(DataType::UInt64, false))),
false,
),
10,
0.,
1.0,
)
.unwrap(),
];
do_bench(c, "10 large_list(0) of u64(0)", cols);
let cols = vec![
create_random_array(
&Field::new(
"list",
DataType::List(Arc::new(Field::new_list_field(DataType::UInt64, false))),
false,
),
4096,
0.,
1.0,
)
.unwrap()
.slice(10, 20),
];
do_bench(c, "4096 list(0) sliced to 10 of u64(0)", cols);
let cols = vec![
create_random_array(
&Field::new(
"list",
DataType::LargeList(Arc::new(Field::new_list_field(DataType::UInt64, false))),
false,
),
4096,
0.,
1.0,
)
.unwrap()
.slice(10, 20),
];
do_bench(c, "4096 large_list(0) sliced to 10 of u64(0)", cols);
bench_iter(c);
}
criterion_group!(benches, row_bench);
criterion_main!(benches);
+326
View File
@@ -0,0 +1,326 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use criterion::Criterion;
use std::sync::Arc;
extern crate arrow;
use arrow::compute::{SortColumn, lexsort, sort, sort_to_indices};
use arrow::datatypes::{Int16Type, Int32Type};
use arrow::util::bench_util::*;
use arrow::{array::*, datatypes::Float32Type};
use arrow_ord::rank::rank;
use std::hint;
fn create_f32_array(size: usize, with_nulls: bool) -> ArrayRef {
let null_density = if with_nulls { 0.5 } else { 0.0 };
let array = create_primitive_array::<Float32Type>(size, null_density);
Arc::new(array)
}
fn create_bool_array(size: usize, with_nulls: bool) -> ArrayRef {
let null_density = if with_nulls { 0.5 } else { 0.0 };
let true_density = 0.5;
let array = create_boolean_array(size, null_density, true_density);
Arc::new(array)
}
fn bench_sort(array: &dyn Array) {
hint::black_box(sort(array, None).unwrap());
}
fn bench_lexsort(array_a: &ArrayRef, array_b: &ArrayRef, limit: Option<usize>) {
let columns = vec![
SortColumn {
values: array_a.clone(),
options: None,
},
SortColumn {
values: array_b.clone(),
options: None,
},
];
hint::black_box(lexsort(&columns, limit).unwrap());
}
fn bench_sort_to_indices(array: &dyn Array, limit: Option<usize>) {
hint::black_box(sort_to_indices(array, None, limit).unwrap());
}
fn add_benchmark(c: &mut Criterion) {
let arr = create_primitive_array::<Int32Type>(2usize.pow(10), 0.0);
c.bench_function("sort i32 2^10", |b| b.iter(|| bench_sort(&arr)));
c.bench_function("sort i32 to indices 2^10", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_primitive_array::<Int32Type>(2usize.pow(12), 0.0);
c.bench_function("sort i32 2^12", |b| b.iter(|| bench_sort(&arr)));
c.bench_function("sort i32 to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_primitive_array::<Int32Type>(2usize.pow(10), 0.5);
c.bench_function("sort i32 nulls 2^10", |b| b.iter(|| bench_sort(&arr)));
c.bench_function("sort i32 nulls to indices 2^10", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_primitive_array::<Int32Type>(2usize.pow(12), 0.5);
c.bench_function("sort i32 nulls 2^12", |b| b.iter(|| bench_sort(&arr)));
c.bench_function("sort i32 nulls to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_f32_array(2_usize.pow(12), false);
c.bench_function("sort f32 2^12", |b| b.iter(|| bench_sort(&arr)));
c.bench_function("sort f32 to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_f32_array(2usize.pow(12), true);
c.bench_function("sort f32 nulls 2^12", |b| b.iter(|| bench_sort(&arr)));
c.bench_function("sort f32 nulls to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_string_array_with_max_len::<i32>(2usize.pow(12), 0.0, 10);
c.bench_function("sort string[0-10] to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_string_array_with_max_len::<i32>(2usize.pow(12), 0.5, 10);
c.bench_function("sort string[0-10] nulls to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_string_array_with_max_len::<i32>(2usize.pow(12), 0.0, 100);
c.bench_function("sort string[0-100] to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_string_array_with_max_len::<i32>(2usize.pow(12), 0.5, 100);
c.bench_function("sort string[0-100] nulls to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_string_array::<i32>(2usize.pow(12), 0.0);
c.bench_function("sort string[0-400] to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_string_array::<i32>(2usize.pow(12), 0.5);
c.bench_function("sort string[0-400] nulls to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_string_array_with_len::<i32>(2usize.pow(12), 0.0, 10);
c.bench_function("sort string[10] to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_string_array_with_len::<i32>(2usize.pow(12), 0.5, 10);
c.bench_function("sort string[10] nulls to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_string_array_with_len::<i32>(2usize.pow(12), 0.0, 100);
c.bench_function("sort string[100] to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_string_array_with_len::<i32>(2usize.pow(12), 0.5, 100);
c.bench_function("sort string[100] nulls to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_string_array_with_len::<i32>(2usize.pow(12), 0.0, 1000);
c.bench_function("sort string[1000] to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_string_array_with_len::<i32>(2usize.pow(12), 0.5, 1000);
c.bench_function("sort string[1000] nulls to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
// This will generate string view arrays with 2^12 elements, each with a length fixed 10, and without nulls.
let arr = create_string_view_array_with_fixed_len(2usize.pow(12), 0.0, 10);
c.bench_function("sort string_view[10] to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
// This will generate string view arrays with 2^12 elements, each with a length fixed 10, and with 50% nulls.
let arr = create_string_view_array_with_fixed_len(2usize.pow(12), 0.5, 10);
c.bench_function("sort string_view[10] nulls to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
// This will generate string view arrays with 2^12 elements, each with a length randomly chosen from 0 to max 400, and without nulls.
let arr = create_string_view_array(2usize.pow(12), 0.0);
c.bench_function("sort string_view[0-400] to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
// This will generate string view arrays with 2^12 elements, each with a length randomly chosen from 0 to max 400, and with 50% nulls.
let arr = create_string_view_array(2usize.pow(12), 0.5);
c.bench_function("sort string_view[0-400] nulls to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
// This will generate string view arrays with 2^12 elements, each with a length < 12 bytes which is inlined data, and without nulls.
let arr = create_string_view_array_with_max_len(2usize.pow(12), 0.0, 12);
c.bench_function("sort string_view_inlined[0-12] to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
// This will generate string view arrays with 2^12 elements, each with a length < 12 bytes which is inlined data, and with 50% nulls.
let arr = create_string_view_array_with_max_len(2usize.pow(12), 0.5, 12);
c.bench_function(
"sort string_view_inlined[0-12] nulls to indices 2^12",
|b| b.iter(|| bench_sort_to_indices(&arr, None)),
);
let arr = create_string_dict_array::<Int32Type>(2usize.pow(12), 0.0, 10);
c.bench_function("sort string[10] dict to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let arr = create_string_dict_array::<Int32Type>(2usize.pow(12), 0.5, 10);
c.bench_function("sort string[10] dict nulls to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&arr, None))
});
let run_encoded_array =
create_primitive_run_array::<Int16Type, Int32Type>(2usize.pow(12), 2usize.pow(10));
c.bench_function("sort primitive run 2^12", |b| {
b.iter(|| bench_sort(&run_encoded_array))
});
c.bench_function("sort primitive run to indices 2^12", |b| {
b.iter(|| bench_sort_to_indices(&run_encoded_array, None))
});
let arr_a = create_f32_array(2usize.pow(10), false);
let arr_b = create_f32_array(2usize.pow(10), false);
c.bench_function("lexsort (f32, f32) 2^10", |b| {
b.iter(|| bench_lexsort(&arr_a, &arr_b, None))
});
let arr_a = create_f32_array(2usize.pow(12), false);
let arr_b = create_f32_array(2usize.pow(12), false);
c.bench_function("lexsort (f32, f32) 2^12", |b| {
b.iter(|| bench_lexsort(&arr_a, &arr_b, None))
});
let arr_a = create_f32_array(2usize.pow(10), true);
let arr_b = create_f32_array(2usize.pow(10), true);
c.bench_function("lexsort (f32, f32) nulls 2^10", |b| {
b.iter(|| bench_lexsort(&arr_a, &arr_b, None))
});
let arr_a = create_f32_array(2usize.pow(12), true);
let arr_b = create_f32_array(2usize.pow(12), true);
c.bench_function("lexsort (f32, f32) nulls 2^12", |b| {
b.iter(|| bench_lexsort(&arr_a, &arr_b, None))
});
let arr_a = create_bool_array(2usize.pow(12), false);
let arr_b = create_bool_array(2usize.pow(12), false);
c.bench_function("lexsort (bool, bool) 2^12", |b| {
b.iter(|| bench_lexsort(&arr_a, &arr_b, None))
});
let arr_a = create_bool_array(2usize.pow(12), true);
let arr_b = create_bool_array(2usize.pow(12), true);
c.bench_function("lexsort (bool, bool) nulls 2^12", |b| {
b.iter(|| bench_lexsort(&arr_a, &arr_b, None))
});
let arr_a = create_f32_array(2usize.pow(12), false);
let arr_b = create_f32_array(2usize.pow(12), false);
c.bench_function("lexsort (f32, f32) 2^12 limit 10", |b| {
b.iter(|| bench_lexsort(&arr_a, &arr_b, Some(10)))
});
let arr_a = create_f32_array(2usize.pow(12), false);
let arr_b = create_f32_array(2usize.pow(12), false);
c.bench_function("lexsort (f32, f32) 2^12 limit 100", |b| {
b.iter(|| bench_lexsort(&arr_a, &arr_b, Some(100)))
});
let arr_a = create_f32_array(2usize.pow(12), false);
let arr_b = create_f32_array(2usize.pow(12), false);
c.bench_function("lexsort (f32, f32) 2^12 limit 1000", |b| {
b.iter(|| bench_lexsort(&arr_a, &arr_b, Some(1000)))
});
let arr_a = create_f32_array(2usize.pow(12), false);
let arr_b = create_f32_array(2usize.pow(12), false);
c.bench_function("lexsort (f32, f32) 2^12 limit 2^12", |b| {
b.iter(|| bench_lexsort(&arr_a, &arr_b, Some(2usize.pow(12))))
});
let arr_a = create_f32_array(2usize.pow(12), true);
let arr_b = create_f32_array(2usize.pow(12), true);
c.bench_function("lexsort (f32, f32) nulls 2^12 limit 10", |b| {
b.iter(|| bench_lexsort(&arr_a, &arr_b, Some(10)))
});
c.bench_function("lexsort (f32, f32) nulls 2^12 limit 100", |b| {
b.iter(|| bench_lexsort(&arr_a, &arr_b, Some(100)))
});
c.bench_function("lexsort (f32, f32) nulls 2^12 limit 1000", |b| {
b.iter(|| bench_lexsort(&arr_a, &arr_b, Some(1000)))
});
c.bench_function("lexsort (f32, f32) nulls 2^12 limit 2^12", |b| {
b.iter(|| bench_lexsort(&arr_a, &arr_b, Some(2usize.pow(12))))
});
let arr = create_f32_array(2usize.pow(12), false);
c.bench_function("rank f32 2^12", |b| {
b.iter(|| hint::black_box(rank(&arr, None).unwrap()))
});
let arr = create_f32_array(2usize.pow(12), true);
c.bench_function("rank f32 nulls 2^12", |b| {
b.iter(|| hint::black_box(rank(&arr, None).unwrap()))
});
let arr = create_string_array_with_len::<i32>(2usize.pow(12), 0.0, 10);
c.bench_function("rank string[10] 2^12", |b| {
b.iter(|| hint::black_box(rank(&arr, None).unwrap()))
});
let arr = create_string_array_with_len::<i32>(2usize.pow(12), 0.5, 10);
c.bench_function("rank string[10] nulls 2^12", |b| {
b.iter(|| hint::black_box(rank(&arr, None).unwrap()))
});
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
@@ -0,0 +1,70 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use arrow::array::StringDictionaryBuilder;
use arrow::datatypes::Int32Type;
use criterion::{Criterion, criterion_group, criterion_main};
use rand::{Rng, rng};
/// Note: this is best effort, not all keys are necessarily present or unique
fn build_strings(dict_size: usize, total_size: usize, key_len: usize) -> Vec<String> {
let mut rng = rng();
let values: Vec<String> = (0..dict_size)
.map(|_| (0..key_len).map(|_| rng.random::<char>()).collect())
.collect();
(0..total_size)
.map(|_| values[rng.random_range(0..dict_size)].clone())
.collect()
}
fn criterion_benchmark(c: &mut Criterion) {
let mut group = c.benchmark_group("string_dictionary_builder");
let mut do_bench = |dict_size: usize, total_size: usize, key_len: usize| {
group.bench_function(
format!("(dict_size:{dict_size}, len:{total_size}, key_len: {key_len})"),
|b| {
let strings = build_strings(dict_size, total_size, key_len);
b.iter(|| {
let mut builder = StringDictionaryBuilder::<Int32Type>::with_capacity(
strings.len(),
key_len + 1,
(key_len + 1) * dict_size,
);
for val in &strings {
builder.append(val).unwrap();
}
builder.finish();
})
},
);
};
do_bench(20, 1000, 5);
do_bench(100, 1000, 5);
do_bench(100, 1000, 10);
do_bench(100, 10000, 10);
do_bench(100, 10000, 100);
group.finish();
}
criterion_group!(benches, criterion_benchmark);
criterion_main!(benches);
@@ -0,0 +1,60 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use arrow::array::StringRunBuilder;
use arrow::datatypes::Int32Type;
use arrow::util::bench_util::create_string_array_for_runs;
use criterion::{Criterion, criterion_group, criterion_main};
fn criterion_benchmark(c: &mut Criterion) {
let mut group = c.benchmark_group("string_run_builder");
let mut do_bench = |physical_array_len: usize, logical_array_len: usize, string_len: usize| {
group.bench_function(
format!(
"(run_array_len:{logical_array_len}, physical_array_len:{physical_array_len}, string_len: {string_len})",
),
|b| {
let strings =
create_string_array_for_runs(physical_array_len, logical_array_len, string_len);
b.iter(|| {
let mut builder = StringRunBuilder::<Int32Type>::with_capacity(
physical_array_len,
(string_len + 1) * physical_array_len,
);
for val in &strings {
builder.append_value(val);
}
builder.finish();
})
},
);
};
do_bench(20, 1000, 5);
do_bench(100, 1000, 5);
do_bench(100, 1000, 10);
do_bench(100, 10000, 10);
do_bench(100, 10000, 100);
group.finish();
}
criterion_group!(benches, criterion_benchmark);
criterion_main!(benches);
@@ -0,0 +1,82 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use arrow::array::{Int32RunArray, StringArray, StringRunBuilder};
use arrow::datatypes::Int32Type;
use criterion::{Criterion, criterion_group, criterion_main};
use rand::{Rng, rng};
fn build_strings_runs(
physical_array_len: usize,
logical_array_len: usize,
string_len: usize,
) -> Int32RunArray {
let mut rng = rng();
let run_len = logical_array_len / physical_array_len;
let mut values: Vec<String> = (0..physical_array_len)
.map(|_| (0..string_len).map(|_| rng.random::<char>()).collect())
.flat_map(|s| std::iter::repeat_n(s, run_len))
.collect();
while values.len() < logical_array_len {
let last_val = values[values.len() - 1].clone();
values.push(last_val);
}
let mut builder = StringRunBuilder::<Int32Type>::with_capacity(
physical_array_len,
(string_len + 1) * physical_array_len,
);
builder.extend(values.into_iter().map(Some));
builder.finish()
}
fn criterion_benchmark(c: &mut Criterion) {
let mut group = c.benchmark_group("string_run_iterator");
let mut do_bench = |physical_array_len: usize, logical_array_len: usize, string_len: usize| {
group.bench_function(
format!(
"(run_array_len:{logical_array_len}, physical_array_len:{physical_array_len}, string_len: {string_len})"),
|b| {
let run_array =
build_strings_runs(physical_array_len, logical_array_len, string_len);
let typed = run_array.downcast::<StringArray>().unwrap();
b.iter(|| {
let iter = typed.into_iter();
for _ in iter {}
})
},
);
};
do_bench(256, 1024, 5);
do_bench(256, 1024, 25);
do_bench(256, 1024, 100);
do_bench(512, 2048, 5);
do_bench(512, 2048, 25);
do_bench(512, 2048, 100);
do_bench(1024, 4096, 5);
do_bench(1024, 4096, 25);
do_bench(1024, 4096, 100);
group.finish();
}
criterion_group!(benches, criterion_benchmark);
criterion_main!(benches);
+66
View File
@@ -0,0 +1,66 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use criterion::Criterion;
extern crate arrow;
use arrow::array::*;
use arrow::compute::kernels::substring::*;
use arrow::util::bench_util::*;
use std::hint;
fn bench_substring(arr: &dyn Array, start: i64, length: Option<u64>) {
substring(hint::black_box(arr), start, length).unwrap();
}
fn bench_substring_by_char<O: OffsetSizeTrait>(
arr: &GenericStringArray<O>,
start: i64,
length: Option<u64>,
) {
substring_by_char(hint::black_box(arr), start, length).unwrap();
}
fn add_benchmark(c: &mut Criterion) {
let size = 65536;
let val_len = 1000;
let arr_string = create_string_array_with_len::<i32>(size, 0.0, val_len);
let arr_fsb = create_fsb_array(size, 0.0, val_len);
c.bench_function("substring utf8 (start = 0, length = None)", |b| {
b.iter(|| bench_substring(&arr_string, 0, None))
});
c.bench_function("substring utf8 (start = 1, length = str_len - 1)", |b| {
b.iter(|| bench_substring(&arr_string, 1, Some((val_len - 1) as u64)))
});
c.bench_function("substring utf8 by char", |b| {
b.iter(|| bench_substring_by_char(&arr_string, 1, Some((val_len - 1) as u64)))
});
c.bench_function("substring fixed size binary array", |b| {
b.iter(|| bench_substring(&arr_fsb, 1, Some((val_len - 1) as u64)))
});
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
+211
View File
@@ -0,0 +1,211 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
#[macro_use]
extern crate criterion;
use criterion::Criterion;
use rand::Rng;
extern crate arrow;
use arrow::compute::{TakeOptions, take};
use arrow::datatypes::*;
use arrow::util::test_util::seedable_rng;
use arrow::{array::*, util::bench_util::*};
use std::hint;
fn create_random_index(size: usize, null_density: f32) -> UInt32Array {
let mut rng = seedable_rng();
let mut builder = UInt32Builder::with_capacity(size);
for _ in 0..size {
if rng.random::<f32>() < null_density {
builder.append_null();
} else {
let value = rng.random_range::<u32, _>(0u32..size as u32);
builder.append_value(value);
}
}
builder.finish()
}
fn bench_take(values: &dyn Array, indices: &UInt32Array) {
hint::black_box(take(values, indices, None).unwrap());
}
fn bench_take_bounds_check(values: &dyn Array, indices: &UInt32Array) {
hint::black_box(take(values, indices, Some(TakeOptions { check_bounds: true })).unwrap());
}
fn add_benchmark(c: &mut Criterion) {
let values = create_primitive_array::<Int32Type>(512, 0.0);
let indices = create_random_index(512, 0.0);
c.bench_function("take i32 512", |b| b.iter(|| bench_take(&values, &indices)));
let values = create_primitive_array::<Int32Type>(1024, 0.0);
let indices = create_random_index(1024, 0.0);
c.bench_function("take i32 1024", |b| {
b.iter(|| bench_take(&values, &indices))
});
let indices = create_random_index(1024, 0.5);
c.bench_function("take i32 null indices 1024", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_primitive_array::<Int32Type>(1024, 0.5);
let indices = create_random_index(1024, 0.0);
c.bench_function("take i32 null values 1024", |b| {
b.iter(|| bench_take(&values, &indices))
});
let indices = create_random_index(1024, 0.5);
c.bench_function("take i32 null values null indices 1024", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_primitive_array::<Int32Type>(512, 0.0);
let indices = create_random_index(512, 0.0);
c.bench_function("take check bounds i32 512", |b| {
b.iter(|| bench_take_bounds_check(&values, &indices))
});
let values = create_primitive_array::<Int32Type>(1024, 0.0);
let indices = create_random_index(1024, 0.0);
c.bench_function("take check bounds i32 1024", |b| {
b.iter(|| bench_take_bounds_check(&values, &indices))
});
let values = create_boolean_array(512, 0.0, 0.5);
let indices = create_random_index(512, 0.0);
c.bench_function("take bool 512", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_boolean_array(1024, 0.0, 0.5);
let indices = create_random_index(1024, 0.0);
c.bench_function("take bool 1024", |b| {
b.iter(|| bench_take(&values, &indices))
});
let indices = create_random_index(1024, 0.5);
c.bench_function("take bool null indices 1024", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_boolean_array(1024, 0.5, 0.5);
let indices = create_random_index(1024, 0.0);
c.bench_function("take bool null values 1024", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_boolean_array(1024, 0.5, 0.5);
let indices = create_random_index(1024, 0.5);
c.bench_function("take bool null values null indices 1024", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_string_array::<i32>(512, 0.0);
let indices = create_random_index(512, 0.0);
c.bench_function("take str 512", |b| b.iter(|| bench_take(&values, &indices)));
let values = create_string_array::<i32>(1024, 0.0);
let indices = create_random_index(1024, 0.0);
c.bench_function("take str 1024", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_string_array::<i32>(512, 0.0);
let indices = create_random_index(512, 0.5);
c.bench_function("take str null indices 512", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_string_array::<i32>(1024, 0.0);
let indices = create_random_index(1024, 0.5);
c.bench_function("take str null indices 1024", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_string_array::<i32>(1024, 0.5);
let indices = create_random_index(1024, 0.0);
c.bench_function("take str null values 1024", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_string_array::<i32>(1024, 0.5);
let indices = create_random_index(1024, 0.5);
c.bench_function("take str null values null indices 1024", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_string_view_array(512, 0.0);
let indices = create_random_index(512, 0.0);
c.bench_function("take stringview 512", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_string_view_array(1024, 0.0);
let indices = create_random_index(1024, 0.0);
c.bench_function("take stringview 1024", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_string_view_array(512, 0.0);
let indices = create_random_index(512, 0.5);
c.bench_function("take stringview null indices 512", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_string_view_array(1024, 0.0);
let indices = create_random_index(1024, 0.5);
c.bench_function("take stringview null indices 1024", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_string_view_array(1024, 0.5);
let indices = create_random_index(1024, 0.0);
c.bench_function("take stringview null values 1024", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_string_view_array(1024, 0.5);
let indices = create_random_index(1024, 0.5);
c.bench_function("take stringview null values null indices 1024", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_primitive_run_array::<Int32Type, Int32Type>(1024, 512);
let indices = create_random_index(1024, 0.0);
c.bench_function(
"take primitive run logical len: 1024, physical len: 512, indices: 1024",
|b| b.iter(|| bench_take(&values, &indices)),
);
let values = create_fsb_array(1024, 0.0, 12);
let indices = create_random_index(1024, 0.0);
c.bench_function("take primitive fsb value len: 12, indices: 1024", |b| {
b.iter(|| bench_take(&values, &indices))
});
let values = create_fsb_array(1024, 0.5, 12);
let indices = create_random_index(1024, 0.0);
c.bench_function(
"take primitive fsb value len: 12, null values, indices: 1024",
|b| b.iter(|| bench_take(&values, &indices)),
);
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
+327
View File
@@ -0,0 +1,327 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use criterion::measurement::WallTime;
use criterion::{BenchmarkGroup, BenchmarkId, Criterion, criterion_group, criterion_main};
use rand::distr::{Distribution, StandardUniform};
use rand::prelude::StdRng;
use rand::{Rng, SeedableRng};
use std::hint;
use std::ops::Range;
use std::sync::Arc;
use arrow::array::*;
use arrow::datatypes::*;
use arrow::util::bench_util::*;
use arrow_select::zip::zip;
trait InputGenerator {
fn name(&self) -> &str;
/// Return an ArrayRef containing a single null value
fn generate_scalar_with_null_value(&self) -> ArrayRef;
/// Generate a `number_of_scalars` unique scalars
fn generate_non_null_scalars(&self, seed: u64, number_of_scalars: usize) -> Vec<ArrayRef>;
/// Generate array with specified length and null percentage
fn generate_array(&self, seed: u64, array_length: usize, null_percentage: f32) -> ArrayRef;
}
struct GeneratePrimitive<T: ArrowPrimitiveType> {
description: String,
_marker: std::marker::PhantomData<T>,
}
impl<T> InputGenerator for GeneratePrimitive<T>
where
T: ArrowPrimitiveType,
StandardUniform: Distribution<T::Native>,
{
fn name(&self) -> &str {
self.description.as_str()
}
fn generate_scalar_with_null_value(&self) -> ArrayRef {
new_null_array(&T::DATA_TYPE, 1)
}
fn generate_non_null_scalars(&self, seed: u64, number_of_scalars: usize) -> Vec<ArrayRef> {
let rng = StdRng::seed_from_u64(seed);
rng.sample_iter::<T::Native, _>(StandardUniform)
.take(number_of_scalars)
.map(|v: T::Native| {
Arc::new(PrimitiveArray::<T>::new_scalar(v).into_inner()) as ArrayRef
})
.collect()
}
fn generate_array(&self, seed: u64, array_length: usize, null_percentage: f32) -> ArrayRef {
Arc::new(create_primitive_array_with_seed::<T>(
array_length,
null_percentage,
seed,
))
}
}
struct GenerateBytes<Byte: ByteArrayType> {
range_length: std::ops::Range<usize>,
description: String,
_marker: std::marker::PhantomData<Byte>,
}
impl<Byte> InputGenerator for GenerateBytes<Byte>
where
Byte: ByteArrayType,
{
fn name(&self) -> &str {
self.description.as_str()
}
fn generate_scalar_with_null_value(&self) -> ArrayRef {
new_null_array(&Byte::DATA_TYPE, 1)
}
fn generate_non_null_scalars(&self, seed: u64, number_of_scalars: usize) -> Vec<ArrayRef> {
let array = self.generate_array(seed, number_of_scalars, 0.0);
(0..number_of_scalars).map(|i| array.slice(i, 1)).collect()
}
fn generate_array(&self, seed: u64, array_length: usize, null_percentage: f32) -> ArrayRef {
let is_binary =
Byte::DATA_TYPE == DataType::Binary || Byte::DATA_TYPE == DataType::LargeBinary;
if is_binary {
Arc::new(create_binary_array_with_len_range_and_prefix_and_seed::<
Byte::Offset,
>(
array_length,
null_percentage,
self.range_length.start,
self.range_length.end - 1,
&[],
seed,
))
} else {
Arc::new(create_string_array_with_len_range_and_prefix_and_seed::<
Byte::Offset,
>(
array_length,
null_percentage,
self.range_length.start,
self.range_length.end - 1,
"",
seed,
))
}
}
}
struct GenerateStringView {
range: Range<usize>,
description: String,
_marker: std::marker::PhantomData<StringViewType>,
}
impl InputGenerator for GenerateStringView {
fn name(&self) -> &str {
self.description.as_str()
}
fn generate_scalar_with_null_value(&self) -> ArrayRef {
new_null_array(&DataType::Utf8View, 1)
}
fn generate_non_null_scalars(&self, seed: u64, number_of_scalars: usize) -> Vec<ArrayRef> {
let array = self.generate_array(seed, number_of_scalars, 0.0);
(0..number_of_scalars).map(|i| array.slice(i, 1)).collect()
}
fn generate_array(&self, seed: u64, array_length: usize, null_percentage: f32) -> ArrayRef {
Arc::new(create_string_view_array_with_len_range_and_seed(
array_length,
null_percentage,
self.range.clone(),
seed,
))
}
}
fn mask_cases(len: usize) -> Vec<(&'static str, BooleanArray)> {
vec![
("all_true", create_boolean_array(len, 0.0, 1.0)),
("99pct_true", create_boolean_array(len, 0.0, 0.99)),
("90pct_true", create_boolean_array(len, 0.0, 0.9)),
("50pct_true", create_boolean_array(len, 0.0, 0.5)),
("10pct_true", create_boolean_array(len, 0.0, 0.1)),
("1pct_true", create_boolean_array(len, 0.0, 0.01)),
("all_false", create_boolean_array(len, 0.0, 0.0)),
("50pct_nulls", create_boolean_array(len, 0.5, 0.5)),
]
}
fn bench_zip_on_input_generator(c: &mut Criterion, input_generator: &impl InputGenerator) {
const ARRAY_LEN: usize = 8192;
let mut group =
c.benchmark_group(format!("zip_{ARRAY_LEN}_from_{}", input_generator.name()).as_str());
let null_scalar = input_generator.generate_scalar_with_null_value();
let [non_null_scalar_1, non_null_scalar_2]: [_; 2] = input_generator
.generate_non_null_scalars(42, 2)
.try_into()
.unwrap();
let array_1_10pct_nulls = input_generator.generate_array(42, ARRAY_LEN, 0.1);
let array_2_10pct_nulls = input_generator.generate_array(18, ARRAY_LEN, 0.1);
let masks = mask_cases(ARRAY_LEN);
// Benchmarks for different scalar combinations
for (description, truthy, falsy) in &[
("null_vs_non_null_scalar", &null_scalar, &non_null_scalar_1),
(
"non_null_scalar_vs_null_scalar",
&non_null_scalar_1,
&null_scalar,
),
("non_nulls_scalars", &non_null_scalar_1, &non_null_scalar_2),
] {
bench_zip_input_on_all_masks(
description,
&mut group,
&masks,
&Scalar::new(truthy),
&Scalar::new(falsy),
);
}
bench_zip_input_on_all_masks(
"array_vs_non_null_scalar",
&mut group,
&masks,
&array_1_10pct_nulls,
&non_null_scalar_1,
);
bench_zip_input_on_all_masks(
"non_null_scalar_vs_array",
&mut group,
&masks,
&non_null_scalar_1,
&array_1_10pct_nulls,
);
bench_zip_input_on_all_masks(
"array_vs_array",
&mut group,
&masks,
&array_1_10pct_nulls,
&array_2_10pct_nulls,
);
group.finish();
}
fn bench_zip_input_on_all_masks(
description: &str,
group: &mut BenchmarkGroup<WallTime>,
masks: &[(&str, BooleanArray)],
truthy: &impl Datum,
falsy: &impl Datum,
) {
for (mask_description, mask) in masks {
let id = BenchmarkId::new(description, mask_description);
group.bench_with_input(id, mask, |b, mask| {
b.iter(|| hint::black_box(zip(mask, truthy, falsy)))
});
}
}
fn add_benchmark(c: &mut Criterion) {
// Primitive
bench_zip_on_input_generator(
c,
&GeneratePrimitive::<Int32Type> {
description: "i32".to_string(),
_marker: std::marker::PhantomData,
},
);
// Short strings
bench_zip_on_input_generator(
c,
&GenerateBytes::<GenericStringType<i32>> {
description: "short strings (3..10)".to_string(),
range_length: 3..10,
_marker: std::marker::PhantomData,
},
);
// Long strings
bench_zip_on_input_generator(
c,
&GenerateBytes::<GenericStringType<i32>> {
description: "long strings (100..400)".to_string(),
range_length: 100..400,
_marker: std::marker::PhantomData,
},
);
// Short Bytes
bench_zip_on_input_generator(
c,
&GenerateBytes::<GenericBinaryType<i32>> {
description: "short bytes (3..10)".to_string(),
range_length: 3..10,
_marker: std::marker::PhantomData,
},
);
// Long Bytes
bench_zip_on_input_generator(
c,
&GenerateBytes::<GenericBinaryType<i32>> {
description: "long bytes (100..400)".to_string(),
range_length: 100..400,
_marker: std::marker::PhantomData,
},
);
bench_zip_on_input_generator(
c,
&GenerateStringView {
description: "string_views size (3..10)".to_string(),
range: 3..10,
_marker: std::marker::PhantomData,
},
);
bench_zip_on_input_generator(
c,
&GenerateStringView {
description: "string_views size (10..100)".to_string(),
range: 10..100,
_marker: std::marker::PhantomData,
},
);
}
criterion_group!(benches, add_benchmark);
criterion_main!(benches);
+38
View File
@@ -0,0 +1,38 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
//! Statically typed implementations of Arrow Arrays
//!
//! **See [arrow_array] for examples and usage instructions**
// --------------------- Array & ArrayData ---------------------
pub use arrow_array::builder::*;
pub use arrow_array::cast::*;
pub use arrow_array::iterator::*;
pub use arrow_array::*;
pub use arrow_data::{
ArrayData, ArrayDataBuilder, ArrayDataRef, BufferSpec, ByteView, DataTypeLayout, layout,
};
pub use arrow_data::transform::{Capacities, MutableArrayData};
#[cfg(feature = "ffi")]
#[allow(deprecated)]
pub use arrow_array::ffi::export_array_into_raw;
// --------------------- Array's values comparison ---------------------
pub use arrow_ord::ord::{DynComparator, make_comparator};
+34
View File
@@ -0,0 +1,34 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
//! Computation kernels on Arrow Arrays
pub use arrow_arith::{aggregate, arithmetic, arity, bitwise, boolean, numeric, temporal};
pub use arrow_cast::cast;
pub use arrow_cast::parse as cast_utils;
pub use arrow_ord::{cmp, partition, rank, sort};
pub use arrow_select::{
coalesce, concat, filter, interleave, merge, nullif, take, union_extract, window, zip,
};
pub use arrow_string::{concat_elements, length, regexp, substring};
/// Comparison kernels for `Array`s.
pub mod comparison {
pub use arrow_ord::comparison::*;
pub use arrow_string::like::*;
pub use arrow_string::regexp::{regexp_is_match, regexp_is_match_scalar};
}
+40
View File
@@ -0,0 +1,40 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
//! Computation kernels on Arrow Arrays
pub mod kernels;
pub use self::kernels::aggregate::*;
pub use self::kernels::arithmetic::*;
pub use self::kernels::arity::*;
pub use self::kernels::boolean::*;
pub use self::kernels::cast::*;
pub use self::kernels::coalesce::*;
pub use self::kernels::comparison::*;
pub use self::kernels::concat::*;
pub use self::kernels::filter::*;
pub use self::kernels::interleave::*;
pub use self::kernels::nullif::*;
pub use self::kernels::partition::*;
pub use self::kernels::rank::*;
pub use self::kernels::regexp::*;
pub use self::kernels::sort::*;
pub use self::kernels::take::*;
pub use self::kernels::temporal::*;
pub use self::kernels::union_extract::*;
pub use self::kernels::window::*;
+32
View File
@@ -0,0 +1,32 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
//! Defines the logical data types of Arrow arrays.
//!
//! The most important things you might be looking for are:
//! * [`Schema`] to describe a schema.
//! * [`Field`] to describe one field within a schema.
//! * [`DataType`] to describe the type of a field.
pub use arrow_array::types::*;
pub use arrow_array::{ArrowNativeTypeOp, ArrowNumericType, ArrowPrimitiveType};
pub use arrow_buffer::{ArrowNativeType, ToByteSlice, i256};
pub use arrow_data::decimal::*;
pub use arrow_schema::{
DataType, Field, FieldRef, Fields, IntervalUnit, Schema, SchemaBuilder, SchemaRef, TimeUnit,
UnionFields, UnionMode,
};
+23
View File
@@ -0,0 +1,23 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
//! Defines `ArrowError` for representing failures in various Arrow operations.
pub use arrow_schema::ArrowError;
/// A specialized `Result` type for Arrow operations.
pub type Result<T> = std::result::Result<T, ArrowError>;
+413
View File
@@ -0,0 +1,413 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
//! A complete, safe, native Rust implementation of [Apache Arrow](https://arrow.apache.org), a cross-language
//! development platform for in-memory data.
//!
//! Please see the [arrow crates.io](https://crates.io/crates/arrow)
//! page for feature flags and tips to improve performance.
//!
//! # Columnar Format
//!
//! The [`array`] module provides statically typed implementations of all the array types as defined
//! by the [Arrow Columnar Format](https://arrow.apache.org/docs/format/Columnar.html)
//!
//! For example, an [`Int32Array`](array::Int32Array) represents a nullable array of `i32`
//!
//! ```rust
//! # use arrow::array::{Array, Int32Array};
//! let array = Int32Array::from(vec![Some(1), None, Some(3)]);
//! assert_eq!(array.len(), 3);
//! assert_eq!(array.value(0), 1);
//! assert_eq!(array.is_null(1), true);
//!
//! let collected: Vec<_> = array.iter().collect();
//! assert_eq!(collected, vec![Some(1), None, Some(3)]);
//! assert_eq!(array.values(), &[1, 0, 3])
//! ```
//!
//! It is also possible to write generic code for different concrete types.
//! For example, since the following function is generic over all primitively
//! typed arrays, when invoked the Rust compiler will generate specialized implementations
//! with optimized code for each concrete type.
//!
//! ```rust
//! # use std::iter::Sum;
//! # use arrow::array::{Float32Array, PrimitiveArray, TimestampNanosecondArray};
//! # use arrow::datatypes::ArrowPrimitiveType;
//! #
//! fn sum<T: ArrowPrimitiveType>(array: &PrimitiveArray<T>) -> T::Native
//! where
//! T: ArrowPrimitiveType,
//! T::Native: Sum
//! {
//! array.iter().map(|v| v.unwrap_or_default()).sum()
//! }
//!
//! assert_eq!(sum(&Float32Array::from(vec![1.1, 2.9, 3.])), 7.);
//! assert_eq!(sum(&TimestampNanosecondArray::from(vec![1, 2, 3])), 6);
//! ```
//!
//! And the following uses [`ArrayAccessor`] to implement a generic function
//! over all arrays with comparable values.
//!
//! [`ArrayAccessor`]: array::ArrayAccessor
//!
//! ```rust
//! # use arrow::array::{ArrayAccessor, ArrayIter, Int32Array, StringArray};
//! # use arrow::datatypes::ArrowPrimitiveType;
//! #
//! fn min<T: ArrayAccessor>(array: T) -> Option<T::Item>
//! where
//! T::Item: Ord
//! {
//! ArrayIter::new(array).filter_map(|v| v).min()
//! }
//!
//! assert_eq!(min(&Int32Array::from(vec![4, 2, 1, 6])), Some(1));
//! assert_eq!(min(&StringArray::from(vec!["b", "a", "c"])), Some("a"));
//! ```
//!
//! **For more examples, and details consult the [arrow_array] docs.**
//!
//! # Type Erasure / Trait Objects
//!
//! It is common to write code that handles any type of array, without necessarily
//! knowing its concrete type. This is done using the [`Array`] trait and using
//! [`DataType`] to determine the appropriate `downcast_ref`.
//!
//! [`DataType`]: datatypes::DataType
//!
//! ```rust
//! # use arrow::array::{Array, Float32Array};
//! # use arrow::array::StringArray;
//! # use arrow::datatypes::DataType;
//! #
//! fn impl_string(array: &StringArray) {}
//! fn impl_f32(array: &Float32Array) {}
//!
//! fn impl_dyn(array: &dyn Array) {
//! match array.data_type() {
//! // downcast `dyn Array` to concrete `StringArray`
//! DataType::Utf8 => impl_string(array.as_any().downcast_ref().unwrap()),
//! // downcast `dyn Array` to concrete `Float32Array`
//! DataType::Float32 => impl_f32(array.as_any().downcast_ref().unwrap()),
//! _ => unimplemented!()
//! }
//! }
//! ```
//!
//! You can use the [`AsArray`] extension trait to facilitate downcasting:
//!
//! [`AsArray`]: crate::array::AsArray
//!
//! ```rust
//! # use arrow::array::{Array, Float32Array, AsArray};
//! # use arrow::array::StringArray;
//! # use arrow::datatypes::DataType;
//! #
//! fn impl_string(array: &StringArray) {}
//! fn impl_f32(array: &Float32Array) {}
//!
//! fn impl_dyn(array: &dyn Array) {
//! match array.data_type() {
//! DataType::Utf8 => impl_string(array.as_string()),
//! DataType::Float32 => impl_f32(array.as_primitive()),
//! _ => unimplemented!()
//! }
//! }
//! ```
//!
//! It is also common to want to write a function that returns one of a number of possible
//! array implementations. [`ArrayRef`] is a type-alias for [`Arc<dyn Array>`](array::Array)
//! which is frequently used for this purpose
//!
//! ```rust
//! # use std::str::FromStr;
//! # use std::sync::Arc;
//! # use arrow::array::{ArrayRef, Int32Array, PrimitiveArray};
//! # use arrow::datatypes::{ArrowPrimitiveType, DataType, Int32Type, UInt32Type};
//! # use arrow::compute::cast;
//! #
//! fn parse_to_primitive<'a, T, I>(iter: I) -> PrimitiveArray<T>
//! where
//! T: ArrowPrimitiveType,
//! T::Native: FromStr,
//! I: IntoIterator<Item=&'a str>,
//! {
//! PrimitiveArray::from_iter(iter.into_iter().map(|val| T::Native::from_str(val).ok()))
//! }
//!
//! fn parse_strings<'a, I>(iter: I, to_data_type: DataType) -> ArrayRef
//! where
//! I: IntoIterator<Item=&'a str>,
//! {
//! match to_data_type {
//! DataType::Int32 => Arc::new(parse_to_primitive::<Int32Type, _>(iter)) as _,
//! DataType::UInt32 => Arc::new(parse_to_primitive::<UInt32Type, _>(iter)) as _,
//! _ => unimplemented!()
//! }
//! }
//!
//! let array = parse_strings(["1", "2", "3"], DataType::Int32);
//! let integers = array.as_any().downcast_ref::<Int32Array>().unwrap();
//! assert_eq!(integers.values(), &[1, 2, 3])
//! ```
//!
//! # Compute Kernels
//!
//! The [`compute`] module provides optimised implementations of many common operations,
//! for example the `parse_strings` operation above could also be implemented as follows:
//!
//! ```
//! # use std::sync::Arc;
//! # use arrow::error::Result;
//! # use arrow::array::{ArrayRef, StringArray, UInt32Array};
//! # use arrow::datatypes::DataType;
//! #
//! fn parse_strings<'a, I>(iter: I, to_data_type: &DataType) -> Result<ArrayRef>
//! where
//! I: IntoIterator<Item=&'a str>,
//! {
//! let array = StringArray::from_iter(iter.into_iter().map(Some));
//! arrow::compute::cast(&array, to_data_type)
//! }
//!
//! let array = parse_strings(["1", "2", "3"], &DataType::UInt32).unwrap();
//! let integers = array.as_any().downcast_ref::<UInt32Array>().unwrap();
//! assert_eq!(integers.values(), &[1, 2, 3])
//! ```
//!
//! This module also implements many common vertical operations:
//!
//! * All mathematical binary operators, such as [`sub`](compute::kernels::numeric::sub)
//! * All boolean binary operators such as [`equality`](compute::kernels::cmp::eq)
//! * [`cast`](compute::kernels::cast::cast)
//! * [`filter`](compute::kernels::filter::filter)
//! * [`take`](compute::kernels::take::take)
//! * [`sort`](compute::kernels::sort::sort)
//! * some string operators such as [`substring`](compute::kernels::substring::substring) and [`length`](compute::kernels::length::length)
//!
//! ```
//! # use arrow::compute::kernels::cmp::gt;
//! # use arrow_array::cast::AsArray;
//! # use arrow_array::Int32Array;
//! # use arrow_array::types::Int32Type;
//! # use arrow_select::filter::filter;
//! let array = Int32Array::from_iter(0..100);
//! // Create a 32-bit integer scalar (single) value:
//! let scalar = Int32Array::new_scalar(60);
//! // find all rows in the array that are greater than 60
//! let predicate = gt(&array, &scalar).unwrap();
//! // copy all matching rows into a new array
//! let filtered = filter(&array, &predicate).unwrap();
//!
//! let expected = Int32Array::from_iter(61..100);
//! assert_eq!(&expected, filtered.as_primitive::<Int32Type>());
//! ```
//!
//! As well as some horizontal operations, such as:
//!
//! * [`min`](compute::kernels::aggregate::min) and [`max`](compute::kernels::aggregate::max)
//! * [`sum`](compute::kernels::aggregate::sum)
//!
//! # Tabular Representation
//!
//! It is common to want to group one or more columns together into a tabular representation. This
//! is provided by [`RecordBatch`] which combines a [`Schema`](datatypes::Schema)
//! and a corresponding list of [`ArrayRef`].
//!
//!
//! ```
//! # use std::sync::Arc;
//! # use arrow::array::{Float32Array, Int32Array};
//! # use arrow::record_batch::RecordBatch;
//! #
//! let col_1 = Arc::new(Int32Array::from_iter([1, 2, 3])) as _;
//! let col_2 = Arc::new(Float32Array::from_iter([1., 6.3, 4.])) as _;
//!
//! let batch = RecordBatch::try_from_iter([("col1", col_1), ("col_2", col_2)]).unwrap();
//! ```
//!
//! # Pretty Printing
//!
//! See the [`util::pretty`] module (requires the `prettyprint` crate feature)
//!
//! # IO
//!
//! This crate provides readers and writers for various formats to/from [`RecordBatch`]
//!
//! * JSON: [`Reader`](json::reader::Reader) and [`Writer`](json::writer::Writer)
//! * CSV: [`Reader`](csv::reader::Reader) and [`Writer`](csv::writer::Writer)
//! * IPC: [`Reader`](ipc::reader::StreamReader) and [`Writer`](ipc::writer::FileWriter)
//!
//! Support for [Apache Parquet] is published as a [separate parquet crate](https://crates.io/crates/parquet)
//!
//! Support for [Apache Avro] is published as a [separate arrow-avro crate](https://crates.io/crates/arrow-avro)
//!
//! # Serde Compatibility
//!
//! [`arrow_json::reader::Decoder`] provides a mechanism to convert arbitrary, serde-compatible
//! structures into [`RecordBatch`].
//!
//! Whilst likely less performant than implementing a custom builder, as described in
//! [arrow_array::builder], this provides a simple mechanism to get up and running quickly
//!
//! ```
//! # use std::sync::Arc;
//! # use arrow_json::ReaderBuilder;
//! # use arrow_schema::{DataType, Field, Schema};
//! # use serde::Serialize;
//! # use arrow_array::cast::AsArray;
//! # use arrow_array::types::{Float32Type, Int32Type};
//! #
//! #[derive(Serialize)]
//! struct MyStruct {
//! int32: i32,
//! string: String,
//! }
//!
//! let schema = Schema::new(vec![
//! Field::new("int32", DataType::Int32, false),
//! Field::new("string", DataType::Utf8, false),
//! ]);
//!
//! let rows = vec![
//! MyStruct{ int32: 5, string: "bar".to_string() },
//! MyStruct{ int32: 8, string: "foo".to_string() },
//! ];
//!
//! let mut decoder = ReaderBuilder::new(Arc::new(schema)).build_decoder().unwrap();
//! decoder.serialize(&rows).unwrap();
//!
//! let batch = decoder.flush().unwrap().unwrap();
//!
//! // Expect batch containing two columns
//! let int32 = batch.column(0).as_primitive::<Int32Type>();
//! assert_eq!(int32.values(), &[5, 8]);
//!
//! let string = batch.column(1).as_string::<i32>();
//! assert_eq!(string.value(0), "bar");
//! assert_eq!(string.value(1), "foo");
//! ```
//!
//! # Crate Topology
//!
//! The [`arrow`] project is implemented as multiple sub-crates, which are then re-exported by
//! this top-level crate.
//!
//! Crate authors can choose to depend on this top-level crate, or just
//! the sub-crates they need.
//!
//! The current list of sub-crates is:
//!
//! * [`arrow-arith`][arrow_arith] - arithmetic kernels
//! * [`arrow-array`][arrow_array] - type-safe arrow array abstractions
//! * [`arrow-buffer`][arrow_buffer] - buffer abstractions for arrow arrays
//! * [`arrow-cast`][arrow_cast] - cast kernels for arrow arrays
//! * [`arrow-csv`][arrow_csv] - read/write CSV to arrow format
//! * [`arrow-data`][arrow_data] - the underlying data of arrow arrays
//! * [`arrow-ipc`][arrow_ipc] - read/write IPC to arrow format
//! * [`arrow-json`][arrow_json] - read/write JSON to arrow format
//! * [`arrow-ord`][arrow_ord] - ordering kernels for arrow arrays
//! * [`arrow-row`][arrow_row] - comparable row format
//! * [`arrow-schema`][arrow_schema] - the logical types for arrow arrays
//! * [`arrow-select`][arrow_select] - selection kernels for arrow arrays
//! * [`arrow-string`][arrow_string] - string kernels for arrow arrays
//!
//! Some functionality is also distributed independently of this crate:
//!
//! * [`arrow-flight`] - support for [Arrow Flight RPC]
//! * [`parquet`](https://docs.rs/parquet) - support for [Apache Parquet]
//! * [`arrow-avro`](https://docs.rs/arrow-avro) - support for [Apache Avro]
//!
//! # Safety and Security
//!
//! Like many crates, this crate makes use of unsafe where prudent. However, it endeavours to be
//! sound. Specifically, **it should not be possible to trigger undefined behaviour using safe APIs.**
//!
//! If you think you have found an instance where this is possible, please file
//! a ticket in our [issue tracker] and it will be triaged and fixed. For more information on
//! arrow's use of unsafe, see [here](https://github.com/apache/arrow-rs/tree/main/arrow#safety).
//!
//! # Higher-level Processing
//!
//! This crate aims to provide reusable, low-level primitives for operating on columnar data. For
//! more sophisticated query processing workloads, consider checking out [DataFusion]. This
//! orchestrates the primitives exported by this crate into an embeddable query engine, with
//! SQL and DataFrame frontends, and heavily influences this crate's roadmap.
//!
//! [`arrow`]: https://github.com/apache/arrow-rs
//! [`array`]: mod@array
//! [`Array`]: array::Array
//! [`ArrayRef`]: array::ArrayRef
//! [`ArrayData`]: array::ArrayData
//! [`make_array`]: array::make_array
//! [`Buffer`]: buffer::Buffer
//! [`RecordBatch`]: record_batch::RecordBatch
//! [`arrow-flight`]: https://docs.rs/arrow-flight/latest/arrow_flight/
//! [`parquet`]: https://docs.rs/parquet/latest/parquet/
//! [Arrow Flight RPC]: https://arrow.apache.org/docs/format/Flight.html
//! [Arrow JSON Test Format]: https://github.com/apache/arrow/blob/master/docs/source/format/Integration.rst#json-test-data-format
//! [Apache Parquet]: https://parquet.apache.org/
//! [Apache Avro]: https://avro.apache.org/
//! [DataFusion]: https://github.com/apache/arrow-datafusion
//! [issue tracker]: https://github.com/apache/arrow-rs/issues
#![doc(
html_logo_url = "https://arrow.apache.org/img/arrow-logo_chevrons_black-txt_white-bg.svg",
html_favicon_url = "https://arrow.apache.org/img/arrow-logo_chevrons_black-txt_transparent-bg.svg"
)]
#![cfg_attr(docsrs, feature(doc_cfg))]
#![deny(clippy::redundant_clone)]
#![warn(missing_debug_implementations)]
#![warn(missing_docs)]
#![allow(rustdoc::invalid_html_tags)]
pub use arrow_array::{downcast_dictionary_array, downcast_primitive_array};
pub use arrow_buffer::{alloc, buffer};
/// Arrow crate version
pub const ARROW_VERSION: &str = env!("CARGO_PKG_VERSION");
pub mod array;
pub mod compute;
#[cfg(feature = "csv")]
pub use arrow_csv as csv;
pub mod datatypes;
pub mod error;
#[cfg(feature = "ffi")]
pub use arrow_array::ffi;
#[cfg(feature = "ffi")]
pub use arrow_array::ffi_stream;
#[cfg(feature = "ipc")]
pub use arrow_ipc as ipc;
#[cfg(feature = "json")]
pub use arrow_json as json;
#[cfg(feature = "pyarrow")]
pub use arrow_pyarrow as pyarrow;
/// Contains the `RecordBatch` type and associated traits
pub mod record_batch {
pub use arrow_array::{
RecordBatch, RecordBatchIterator, RecordBatchOptions, RecordBatchReader, RecordBatchWriter,
};
}
pub use arrow_array::temporal_conversions;
pub use arrow_row as row;
pub mod tensor;
pub mod util;
+537
View File
@@ -0,0 +1,537 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
//! Arrow Tensor Type, defined in
//! [`format/Tensor.fbs`](https://github.com/apache/arrow/blob/master/format/Tensor.fbs).
use std::marker::PhantomData;
use std::mem;
use crate::buffer::Buffer;
use crate::datatypes::*;
use crate::error::{ArrowError, Result};
/// Computes the strides required assuming a row major memory layout
fn compute_row_major_strides<T: ArrowPrimitiveType>(shape: &[usize]) -> Result<Vec<usize>> {
let mut remaining_bytes = mem::size_of::<T::Native>();
for i in shape {
if let Some(val) = remaining_bytes.checked_mul(*i) {
remaining_bytes = val;
} else {
return Err(ArrowError::ComputeError(
"overflow occurred when computing row major strides.".to_string(),
));
}
}
let mut strides = Vec::<usize>::new();
for i in shape {
remaining_bytes /= *i;
strides.push(remaining_bytes);
}
Ok(strides)
}
/// Computes the strides required assuming a column major memory layout
fn compute_column_major_strides<T: ArrowPrimitiveType>(shape: &[usize]) -> Result<Vec<usize>> {
let mut remaining_bytes = mem::size_of::<T::Native>();
let mut strides = Vec::<usize>::new();
for i in shape {
strides.push(remaining_bytes);
if let Some(val) = remaining_bytes.checked_mul(*i) {
remaining_bytes = val;
} else {
return Err(ArrowError::ComputeError(
"overflow occurred when computing column major strides.".to_string(),
));
}
}
Ok(strides)
}
/// Tensor of primitive types
#[derive(Debug)]
pub struct Tensor<'a, T: ArrowPrimitiveType> {
data_type: DataType,
buffer: Buffer,
shape: Option<Vec<usize>>,
strides: Option<Vec<usize>>,
names: Option<Vec<&'a str>>,
_marker: PhantomData<T>,
}
/// [Tensor] of type [BooleanType]
pub type BooleanTensor<'a> = Tensor<'a, BooleanType>;
/// [Tensor] of type [Int8Type]
pub type Date32Tensor<'a> = Tensor<'a, Date32Type>;
/// [Tensor] of type [Int16Type]
pub type Date64Tensor<'a> = Tensor<'a, Date64Type>;
/// [Tensor] of type [Decimal32Type]
pub type Decimal32Tensor<'a> = Tensor<'a, Decimal32Type>;
/// [Tensor] of type [Decimal64Type]
pub type Decimal64Tensor<'a> = Tensor<'a, Decimal64Type>;
/// [Tensor] of type [Decimal128Type]
pub type Decimal128Tensor<'a> = Tensor<'a, Decimal128Type>;
/// [Tensor] of type [Decimal256Type]
pub type Decimal256Tensor<'a> = Tensor<'a, Decimal256Type>;
/// [Tensor] of type [DurationMicrosecondType]
pub type DurationMicrosecondTensor<'a> = Tensor<'a, DurationMicrosecondType>;
/// [Tensor] of type [DurationMillisecondType]
pub type DurationMillisecondTensor<'a> = Tensor<'a, DurationMillisecondType>;
/// [Tensor] of type [DurationNanosecondType]
pub type DurationNanosecondTensor<'a> = Tensor<'a, DurationNanosecondType>;
/// [Tensor] of type [DurationSecondType]
pub type DurationSecondTensor<'a> = Tensor<'a, DurationSecondType>;
/// [Tensor] of type [Float16Type]
pub type Float16Tensor<'a> = Tensor<'a, Float16Type>;
/// [Tensor] of type [Float32Type]
pub type Float32Tensor<'a> = Tensor<'a, Float32Type>;
/// [Tensor] of type [Float64Type]
pub type Float64Tensor<'a> = Tensor<'a, Float64Type>;
/// [Tensor] of type [Int8Type]
pub type Int8Tensor<'a> = Tensor<'a, Int8Type>;
/// [Tensor] of type [Int16Type]
pub type Int16Tensor<'a> = Tensor<'a, Int16Type>;
/// [Tensor] of type [Int32Type]
pub type Int32Tensor<'a> = Tensor<'a, Int32Type>;
/// [Tensor] of type [Int64Type]
pub type Int64Tensor<'a> = Tensor<'a, Int64Type>;
/// [Tensor] of type [IntervalDayTimeType]
pub type IntervalDayTimeTensor<'a> = Tensor<'a, IntervalDayTimeType>;
/// [Tensor] of type [IntervalMonthDayNanoType]
pub type IntervalMonthDayNanoTensor<'a> = Tensor<'a, IntervalMonthDayNanoType>;
/// [Tensor] of type [IntervalYearMonthType]
pub type IntervalYearMonthTensor<'a> = Tensor<'a, IntervalYearMonthType>;
/// [Tensor] of type [Time32MillisecondType]
pub type Time32MillisecondTensor<'a> = Tensor<'a, Time32MillisecondType>;
/// [Tensor] of type [Time32SecondType]
pub type Time32SecondTensor<'a> = Tensor<'a, Time32SecondType>;
/// [Tensor] of type [Time64MicrosecondType]
pub type Time64MicrosecondTensor<'a> = Tensor<'a, Time64MicrosecondType>;
/// [Tensor] of type [Time64NanosecondType]
pub type Time64NanosecondTensor<'a> = Tensor<'a, Time64NanosecondType>;
/// [Tensor] of type [TimestampMicrosecondType]
pub type TimestampMicrosecondTensor<'a> = Tensor<'a, TimestampMicrosecondType>;
/// [Tensor] of type [TimestampMillisecondType]
pub type TimestampMillisecondTensor<'a> = Tensor<'a, TimestampMillisecondType>;
/// [Tensor] of type [TimestampNanosecondType]
pub type TimestampNanosecondTensor<'a> = Tensor<'a, TimestampNanosecondType>;
/// [Tensor] of type [TimestampSecondType]
pub type TimestampSecondTensor<'a> = Tensor<'a, TimestampSecondType>;
/// [Tensor] of type [UInt8Type]
pub type UInt8Tensor<'a> = Tensor<'a, UInt8Type>;
/// [Tensor] of type [UInt16Type]
pub type UInt16Tensor<'a> = Tensor<'a, UInt16Type>;
/// [Tensor] of type [UInt32Type]
pub type UInt32Tensor<'a> = Tensor<'a, UInt32Type>;
/// [Tensor] of type [UInt64Type]
pub type UInt64Tensor<'a> = Tensor<'a, UInt64Type>;
impl<'a, T: ArrowPrimitiveType> Tensor<'a, T> {
/// Creates a new `Tensor`
pub fn try_new(
buffer: Buffer,
shape: Option<Vec<usize>>,
strides: Option<Vec<usize>>,
names: Option<Vec<&'a str>>,
) -> Result<Self> {
match shape {
None => {
if buffer.len() != mem::size_of::<T::Native>() {
return Err(ArrowError::InvalidArgumentError(
"underlying buffer should only contain a single tensor element".to_string(),
));
}
if strides.is_some() {
return Err(ArrowError::InvalidArgumentError(
"expected None strides for tensor with no shape".to_string(),
));
}
if names.is_some() {
return Err(ArrowError::InvalidArgumentError(
"expected None names for tensor with no shape".to_string(),
));
}
}
Some(ref s) => {
if let Some(ref st) = strides {
if st.len() != s.len() {
return Err(ArrowError::InvalidArgumentError(
"shape and stride dimensions differ".to_string(),
));
}
}
if let Some(ref n) = names {
if n.len() != s.len() {
return Err(ArrowError::InvalidArgumentError(
"number of dimensions and number of dimension names differ".to_string(),
));
}
}
let total_elements: usize = s.iter().product();
if total_elements != (buffer.len() / mem::size_of::<T::Native>()) {
return Err(ArrowError::InvalidArgumentError(
"number of elements in buffer does not match dimensions".to_string(),
));
}
}
};
// Checking that the tensor strides used for construction are correct
// otherwise a row major stride is calculated and used as value for the tensor
let tensor_strides = {
if let Some(st) = strides {
if let Some(ref s) = shape {
if compute_row_major_strides::<T>(s)? == st
|| compute_column_major_strides::<T>(s)? == st
{
Some(st)
} else {
return Err(ArrowError::InvalidArgumentError(
"the input stride does not match the selected shape".to_string(),
));
}
} else {
Some(st)
}
} else if let Some(ref s) = shape {
Some(compute_row_major_strides::<T>(s)?)
} else {
None
}
};
Ok(Self {
data_type: T::DATA_TYPE,
buffer,
shape,
strides: tensor_strides,
names,
_marker: PhantomData,
})
}
/// Creates a new Tensor using row major memory layout
pub fn new_row_major(
buffer: Buffer,
shape: Option<Vec<usize>>,
names: Option<Vec<&'a str>>,
) -> Result<Self> {
if let Some(ref s) = shape {
let strides = Some(compute_row_major_strides::<T>(s)?);
Self::try_new(buffer, shape, strides, names)
} else {
Err(ArrowError::InvalidArgumentError(
"shape required to create row major tensor".to_string(),
))
}
}
/// Creates a new Tensor using column major memory layout
pub fn new_column_major(
buffer: Buffer,
shape: Option<Vec<usize>>,
names: Option<Vec<&'a str>>,
) -> Result<Self> {
if let Some(ref s) = shape {
let strides = Some(compute_column_major_strides::<T>(s)?);
Self::try_new(buffer, shape, strides, names)
} else {
Err(ArrowError::InvalidArgumentError(
"shape required to create column major tensor".to_string(),
))
}
}
/// The data type of the `Tensor`
pub fn data_type(&self) -> &DataType {
&self.data_type
}
/// The sizes of the dimensions
pub fn shape(&self) -> Option<&Vec<usize>> {
self.shape.as_ref()
}
/// Returns a reference to the underlying `Buffer`
pub fn data(&self) -> &Buffer {
&self.buffer
}
/// The number of bytes between elements in each dimension
pub fn strides(&self) -> Option<&Vec<usize>> {
self.strides.as_ref()
}
/// The names of the dimensions
pub fn names(&self) -> Option<&Vec<&'a str>> {
self.names.as_ref()
}
/// The number of dimensions
pub fn ndim(&self) -> usize {
match &self.shape {
None => 0,
Some(v) => v.len(),
}
}
/// The name of dimension i
pub fn dim_name(&self, i: usize) -> Option<&'a str> {
self.names.as_ref().map(|names| names[i])
}
/// The total number of elements in the `Tensor`
pub fn size(&self) -> usize {
match self.shape {
None => 0,
Some(ref s) => s.iter().product(),
}
}
/// Indicates if the data is laid out contiguously in memory
pub fn is_contiguous(&self) -> Result<bool> {
Ok(self.is_row_major()? || self.is_column_major()?)
}
/// Indicates if the memory layout row major
pub fn is_row_major(&self) -> Result<bool> {
match self.shape {
None => Ok(false),
Some(ref s) => Ok(Some(compute_row_major_strides::<T>(s)?) == self.strides),
}
}
/// Indicates if the memory layout column major
pub fn is_column_major(&self) -> Result<bool> {
match self.shape {
None => Ok(false),
Some(ref s) => Ok(Some(compute_column_major_strides::<T>(s)?) == self.strides),
}
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::array::*;
#[test]
fn test_compute_row_major_strides() {
assert_eq!(
vec![48_usize, 8],
compute_row_major_strides::<Int64Type>(&[4_usize, 6]).unwrap()
);
assert_eq!(
vec![24_usize, 4],
compute_row_major_strides::<Int32Type>(&[4_usize, 6]).unwrap()
);
assert_eq!(
vec![6_usize, 1],
compute_row_major_strides::<Int8Type>(&[4_usize, 6]).unwrap()
);
}
#[test]
fn test_compute_column_major_strides() {
assert_eq!(
vec![8_usize, 32],
compute_column_major_strides::<Int64Type>(&[4_usize, 6]).unwrap()
);
assert_eq!(
vec![4_usize, 16],
compute_column_major_strides::<Int32Type>(&[4_usize, 6]).unwrap()
);
assert_eq!(
vec![1_usize, 4],
compute_column_major_strides::<Int8Type>(&[4_usize, 6]).unwrap()
);
}
#[test]
fn test_zero_dim() {
let buf = Buffer::from(&[1]);
let tensor = UInt8Tensor::try_new(buf, None, None, None).unwrap();
assert_eq!(0, tensor.size());
assert_eq!(None, tensor.shape());
assert_eq!(None, tensor.names());
assert_eq!(0, tensor.ndim());
assert!(!tensor.is_row_major().unwrap());
assert!(!tensor.is_column_major().unwrap());
assert!(!tensor.is_contiguous().unwrap());
let buf = Buffer::from(&[1, 2, 2, 2]);
let tensor = Int32Tensor::try_new(buf, None, None, None).unwrap();
assert_eq!(0, tensor.size());
assert_eq!(None, tensor.shape());
assert_eq!(None, tensor.names());
assert_eq!(0, tensor.ndim());
assert!(!tensor.is_row_major().unwrap());
assert!(!tensor.is_column_major().unwrap());
assert!(!tensor.is_contiguous().unwrap());
}
#[test]
fn test_tensor() {
let mut builder = Int32BufferBuilder::new(16);
for i in 0..16 {
builder.append(i);
}
let buf = builder.finish();
let tensor = Int32Tensor::try_new(buf, Some(vec![2, 8]), None, None).unwrap();
assert_eq!(16, tensor.size());
assert_eq!(Some(vec![2_usize, 8]).as_ref(), tensor.shape());
assert_eq!(Some(vec![32_usize, 4]).as_ref(), tensor.strides());
assert_eq!(2, tensor.ndim());
assert_eq!(None, tensor.names());
}
#[test]
fn test_new_row_major() {
let mut builder = Int32BufferBuilder::new(16);
for i in 0..16 {
builder.append(i);
}
let buf = builder.finish();
let tensor = Int32Tensor::new_row_major(buf, Some(vec![2, 8]), None).unwrap();
assert_eq!(16, tensor.size());
assert_eq!(Some(vec![2_usize, 8]).as_ref(), tensor.shape());
assert_eq!(Some(vec![32_usize, 4]).as_ref(), tensor.strides());
assert_eq!(None, tensor.names());
assert_eq!(2, tensor.ndim());
assert!(tensor.is_row_major().unwrap());
assert!(!tensor.is_column_major().unwrap());
assert!(tensor.is_contiguous().unwrap());
}
#[test]
fn test_new_column_major() {
let mut builder = Int32BufferBuilder::new(16);
for i in 0..16 {
builder.append(i);
}
let buf = builder.finish();
let tensor = Int32Tensor::new_column_major(buf, Some(vec![2, 8]), None).unwrap();
assert_eq!(16, tensor.size());
assert_eq!(Some(vec![2_usize, 8]).as_ref(), tensor.shape());
assert_eq!(Some(vec![4_usize, 8]).as_ref(), tensor.strides());
assert_eq!(None, tensor.names());
assert_eq!(2, tensor.ndim());
assert!(!tensor.is_row_major().unwrap());
assert!(tensor.is_column_major().unwrap());
assert!(tensor.is_contiguous().unwrap());
}
#[test]
fn test_with_names() {
let mut builder = Int64BufferBuilder::new(8);
for i in 0..8 {
builder.append(i);
}
let buf = builder.finish();
let names = vec!["Dim 1", "Dim 2"];
let tensor = Int64Tensor::new_column_major(buf, Some(vec![2, 4]), Some(names)).unwrap();
assert_eq!(8, tensor.size());
assert_eq!(Some(vec![2_usize, 4]).as_ref(), tensor.shape());
assert_eq!(Some(vec![8_usize, 16]).as_ref(), tensor.strides());
assert_eq!("Dim 1", tensor.dim_name(0).unwrap());
assert_eq!("Dim 2", tensor.dim_name(1).unwrap());
assert_eq!(2, tensor.ndim());
assert!(!tensor.is_row_major().unwrap());
assert!(tensor.is_column_major().unwrap());
assert!(tensor.is_contiguous().unwrap());
}
#[test]
fn test_inconsistent_strides() {
let mut builder = Int32BufferBuilder::new(16);
for i in 0..16 {
builder.append(i);
}
let buf = builder.finish();
let result = Int32Tensor::try_new(buf, Some(vec![2, 8]), Some(vec![2, 8, 1]), None);
if result.is_ok() {
panic!("shape and stride dimensions are different")
}
}
#[test]
fn test_inconsistent_names() {
let mut builder = Int32BufferBuilder::new(16);
for i in 0..16 {
builder.append(i);
}
let buf = builder.finish();
let result = Int32Tensor::try_new(
buf,
Some(vec![2, 8]),
Some(vec![4, 8]),
Some(vec!["1", "2", "3"]),
);
if result.is_ok() {
panic!("dimensions and names have different shape")
}
}
#[test]
fn test_incorrect_shape() {
let mut builder = Int32BufferBuilder::new(16);
for i in 0..16 {
builder.append(i);
}
let buf = builder.finish();
let result = Int32Tensor::try_new(buf, Some(vec![2, 6]), None, None);
if result.is_ok() {
panic!("number of elements does not match for the shape")
}
}
#[test]
fn test_incorrect_stride() {
let mut builder = Int32BufferBuilder::new(16);
for i in 0..16 {
builder.append(i);
}
let buf = builder.finish();
let result = Int32Tensor::try_new(buf, Some(vec![2, 8]), Some(vec![30, 4]), None);
if result.is_ok() {
panic!("the input stride does not match the selected shape")
}
}
}
+736
View File
@@ -0,0 +1,736 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
//! Utils to make benchmarking easier
use crate::array::*;
use crate::datatypes::*;
use crate::util::test_util::seedable_rng;
use arrow_buffer::{Buffer, IntervalMonthDayNano};
use half::f16;
use rand::Rng;
use rand::SeedableRng;
use rand::distr::uniform::SampleUniform;
use rand::rng;
use rand::{
distr::{Alphanumeric, Distribution, StandardUniform},
prelude::StdRng,
};
use std::ops::Range;
/// Creates an random (but fixed-seeded) array of a given size and null density
pub fn create_primitive_array<T>(size: usize, null_density: f32) -> PrimitiveArray<T>
where
T: ArrowPrimitiveType,
StandardUniform: Distribution<T::Native>,
{
let mut rng = seedable_rng();
(0..size)
.map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
Some(rng.random())
}
})
.collect()
}
/// Creates a [`PrimitiveArray`] of a given `size` and `null_density`
/// filling it with random numbers generated using the provided `seed`.
pub fn create_primitive_array_with_seed<T>(
size: usize,
null_density: f32,
seed: u64,
) -> PrimitiveArray<T>
where
T: ArrowPrimitiveType,
StandardUniform: Distribution<T::Native>,
{
let mut rng = StdRng::seed_from_u64(seed);
(0..size)
.map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
Some(rng.random())
}
})
.collect()
}
/// Creates a [`PrimitiveArray`] of a given `size` and `null_density`
/// filling it with random [`IntervalMonthDayNano`] generated using the provided `seed`.
pub fn create_month_day_nano_array_with_seed(
size: usize,
null_density: f32,
seed: u64,
) -> IntervalMonthDayNanoArray {
let mut rng = StdRng::seed_from_u64(seed);
(0..size)
.map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
Some(IntervalMonthDayNano::new(
rng.random(),
rng.random(),
rng.random(),
))
}
})
.collect()
}
/// Creates a random (but fixed-seeded) array of a given size and null density
pub fn create_boolean_array(size: usize, null_density: f32, true_density: f32) -> BooleanArray
where
StandardUniform: Distribution<bool>,
{
let mut rng = seedable_rng();
(0..size)
.map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
let value = rng.random::<f32>() < true_density;
Some(value)
}
})
.collect()
}
/// Creates a random (but fixed-seeded) string array of a given size and null density.
///
/// Strings have a random length
/// between 0 and 400 alphanumeric characters. `0..400` is chosen to cover a wide range of common string lengths,
/// which have a dramatic impact on performance of some queries, e.g. LIKE/ILIKE/regex.
pub fn create_string_array<Offset: OffsetSizeTrait>(
size: usize,
null_density: f32,
) -> GenericStringArray<Offset> {
create_string_array_with_max_len(size, null_density, 400)
}
/// Creates longer string array with same prefix, the prefix should be larger than 4 bytes,
/// and the string length should be larger than 12 bytes
/// so that we can compare the performance with StringViewArray, because StringViewArray has 4 bytes inline for view
pub fn create_longer_string_array_with_same_prefix<Offset: OffsetSizeTrait>(
size: usize,
null_density: f32,
) -> GenericStringArray<Offset> {
create_string_array_with_len_range_and_prefix(size, null_density, 13, 100, "prefix_")
}
/// Creates longer string view array with same prefix, the prefix should be larger than 4 bytes,
/// and the string length should be larger than 12 bytes
/// so that we can compare the StringArray performance with StringViewArray, because StringViewArray has 4 bytes inline for view
pub fn create_longer_string_view_array_with_same_prefix(
size: usize,
null_density: f32,
) -> StringViewArray {
create_string_view_array_with_len_range_and_prefix(size, null_density, 13, 100, "prefix_")
}
fn create_string_array_with_len_range_and_prefix<Offset: OffsetSizeTrait>(
size: usize,
null_density: f32,
min_str_len: usize,
max_str_len: usize,
prefix: &str,
) -> GenericStringArray<Offset> {
create_string_array_with_len_range_and_prefix_and_seed(
size,
null_density,
min_str_len,
max_str_len,
prefix,
42,
)
}
/// Creates a random [`GenericStringArray`] of a given `size` and `null_density`
/// filling it with random strings with lengths in the specified range,
/// all starting with the provided `prefix`, generated using the provided `seed`.
pub fn create_string_array_with_len_range_and_prefix_and_seed<Offset: OffsetSizeTrait>(
size: usize,
null_density: f32,
min_str_len: usize,
max_str_len: usize,
prefix: &str,
seed: u64,
) -> GenericStringArray<Offset> {
assert!(
min_str_len <= max_str_len,
"min_str_len must be <= max_str_len"
);
assert!(
prefix.len() <= max_str_len,
"Prefix length must be <= max_str_len"
);
let rng = &mut StdRng::seed_from_u64(seed);
(0..size)
.map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
let remaining_len = rng.random_range(
min_str_len.saturating_sub(prefix.len())..=(max_str_len - prefix.len()),
);
let mut value = prefix.to_string();
value.extend(
rng.sample_iter(&Alphanumeric)
.take(remaining_len)
.map(char::from),
);
Some(value)
}
})
.collect()
}
/// Creates a string view array of a given range, null density and length
///
/// Arguments:
/// - `size`: number of string view array
/// - `null_density`: density of nulls in the string view array
/// - `range`: range size of each string in the string view array
/// - `seed`: seed for the random number generator
pub fn create_string_view_array_with_len_range_and_seed(
size: usize,
null_density: f32,
range: Range<usize>,
seed: u64,
) -> StringViewArray {
let rng = &mut StdRng::seed_from_u64(seed);
(0..size)
.map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
let str_len = rng.random_range(range.clone());
let value = rng.sample_iter(&Alphanumeric).take(str_len).collect();
let value = String::from_utf8(value).unwrap();
Some(value)
}
})
.collect()
}
fn create_string_view_array_with_len_range_and_prefix(
size: usize,
null_density: f32,
min_str_len: usize,
max_str_len: usize,
prefix: &str,
) -> StringViewArray {
assert!(
min_str_len <= max_str_len,
"min_str_len must be <= max_str_len"
);
assert!(
prefix.len() <= max_str_len,
"Prefix length must be <= max_str_len"
);
let rng = &mut seedable_rng();
(0..size)
.map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
let remaining_len = rng.random_range(
min_str_len.saturating_sub(prefix.len())..=(max_str_len - prefix.len()),
);
let mut value = prefix.to_string();
value.extend(
rng.sample_iter(&Alphanumeric)
.take(remaining_len)
.map(char::from),
);
Some(value)
}
})
.collect()
}
/// Creates a random (but fixed-seeded) array of rand size with a given max size, null density and length
pub fn create_string_array_with_max_len<Offset: OffsetSizeTrait>(
size: usize,
null_density: f32,
max_str_len: usize,
) -> GenericStringArray<Offset> {
let rng = &mut seedable_rng();
(0..size)
.map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
let str_len = rng.random_range(0..max_str_len);
let value = rng.sample_iter(&Alphanumeric).take(str_len).collect();
let value = String::from_utf8(value).unwrap();
Some(value)
}
})
.collect()
}
/// Creates a random (but fixed-seeded) array of a given size, null density and length
pub fn create_string_array_with_len<Offset: OffsetSizeTrait>(
size: usize,
null_density: f32,
str_len: usize,
) -> GenericStringArray<Offset> {
let rng = &mut seedable_rng();
(0..size)
.map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
let value = rng.sample_iter(&Alphanumeric).take(str_len).collect();
let value = String::from_utf8(value).unwrap();
Some(value)
}
})
.collect()
}
/// Creates a random (but fixed-seeded) string view array of a given size and null density.
///
/// See `create_string_array` above for more details.
pub fn create_string_view_array(size: usize, null_density: f32) -> StringViewArray {
create_string_view_array_with_max_len(size, null_density, 400)
}
/// Creates a random (but fixed-seeded) array of rand size with a given max size, null density and length
pub fn create_string_view_array_with_max_len(
size: usize,
null_density: f32,
max_str_len: usize,
) -> StringViewArray {
let rng = &mut seedable_rng();
(0..size)
.map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
let str_len = rng.random_range(0..max_str_len);
let value = rng.sample_iter(&Alphanumeric).take(str_len).collect();
let value = String::from_utf8(value).unwrap();
Some(value)
}
})
.collect()
}
/// Creates a random (but fixed-seeded) array of a given size, null density and length
pub fn create_string_view_array_with_fixed_len(
size: usize,
null_density: f32,
str_len: usize,
) -> StringViewArray {
let rng = &mut seedable_rng();
(0..size)
.map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
let value = rng.sample_iter(&Alphanumeric).take(str_len).collect();
let value = String::from_utf8(value).unwrap();
Some(value)
}
})
.collect()
}
/// Creates a random (but fixed-seeded) array of a given size, null density and length
pub fn create_string_view_array_with_len(
size: usize,
null_density: f32,
str_len: usize,
mixed: bool,
) -> StringViewArray {
let rng = &mut seedable_rng();
let mut lengths = Vec::with_capacity(size);
// if mixed, we creates first half that string length small than 12 bytes and second half large than 12 bytes
if mixed {
for _ in 0..size / 2 {
lengths.push(rng.random_range(1..12));
}
for _ in size / 2..size {
lengths.push(rng.random_range(12..=std::cmp::max(30, str_len)));
}
} else {
lengths.resize(size, str_len);
}
lengths
.into_iter()
.map(|len| {
if rng.random::<f32>() < null_density {
None
} else {
let value: Vec<u8> = rng.sample_iter(&Alphanumeric).take(len).collect();
Some(String::from_utf8(value).unwrap())
}
})
.collect()
}
/// Creates an random (but fixed-seeded) array of a given size and null density
/// consisting of random 4 character alphanumeric strings
pub fn create_string_dict_array<K: ArrowDictionaryKeyType>(
size: usize,
null_density: f32,
str_len: usize,
) -> DictionaryArray<K> {
let rng = &mut seedable_rng();
let data: Vec<_> = (0..size)
.map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
let value = rng.sample_iter(&Alphanumeric).take(str_len).collect();
let value = String::from_utf8(value).unwrap();
Some(value)
}
})
.collect();
data.iter().map(|x| x.as_deref()).collect()
}
/// Create a List/LargeList Array of primitive values
///
/// Arguments:
/// - `size`: number of lists in the array
/// - `null_density`: density of nulls in the list array
/// - `list_null_density`: density of nulls in the primitive arrays inside the lists
/// - `max_list_size`: maximum size of each list (actual size is random between 0 and max_list_size)
/// - `seed`: seed for the random number generator
pub fn create_primitive_list_array_with_seed<O, T>(
size: usize,
null_density: f32,
list_null_density: f32,
max_list_size: usize,
seed: u64,
) -> GenericListArray<O>
where
O: OffsetSizeTrait,
T: ArrowPrimitiveType,
StandardUniform: Distribution<T::Native>,
{
let mut rng = StdRng::seed_from_u64(seed);
let values = (0..size).map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
let list_size = rng.random_range(0..=max_list_size);
let list_values: Vec<Option<T::Native>> = (0..list_size)
.map(|_| {
if rng.random::<f32>() < list_null_density {
None
} else {
Some(rng.random())
}
})
.collect();
Some(list_values)
}
});
GenericListArray::<O>::from_iter_primitive::<T, _, _>(values)
}
/// Create primitive run array for given logical and physical array lengths
pub fn create_primitive_run_array<R: RunEndIndexType, V: ArrowPrimitiveType>(
logical_array_len: usize,
physical_array_len: usize,
) -> RunArray<R> {
assert!(logical_array_len >= physical_array_len);
// typical length of each run
let run_len = logical_array_len / physical_array_len;
// Some runs should have extra length
let mut run_len_extra = logical_array_len % physical_array_len;
let mut values: Vec<V::Native> = (0..physical_array_len)
.flat_map(|s| {
let mut take_len = run_len;
if run_len_extra > 0 {
take_len += 1;
run_len_extra -= 1;
}
std::iter::repeat_n(V::Native::from_usize(s).unwrap(), take_len)
})
.collect();
while values.len() < logical_array_len {
let last_val = values[values.len() - 1];
values.push(last_val);
}
let mut builder = PrimitiveRunBuilder::<R, V>::with_capacity(physical_array_len);
builder.extend(values.into_iter().map(Some));
builder.finish()
}
/// Create string array to be used by run array builder. The string array
/// will result in run array with physical length of `physical_array_len`
/// and logical length of `logical_array_len`
pub fn create_string_array_for_runs(
physical_array_len: usize,
logical_array_len: usize,
string_len: usize,
) -> Vec<String> {
assert!(logical_array_len >= physical_array_len);
let mut rng = rng();
// typical length of each run
let run_len = logical_array_len / physical_array_len;
// Some runs should have extra length
let mut run_len_extra = logical_array_len % physical_array_len;
let mut values: Vec<String> = (0..physical_array_len)
.map(|_| (0..string_len).map(|_| rng.random::<char>()).collect())
.flat_map(|s| {
let mut take_len = run_len;
if run_len_extra > 0 {
take_len += 1;
run_len_extra -= 1;
}
std::iter::repeat_n(s, take_len)
})
.collect();
while values.len() < logical_array_len {
let last_val = values[values.len() - 1].clone();
values.push(last_val);
}
values
}
/// Creates an random (but fixed-seeded) binary array of a given size and null density
pub fn create_binary_array<Offset: OffsetSizeTrait>(
size: usize,
null_density: f32,
) -> GenericBinaryArray<Offset> {
create_binary_array_with_seed(
size,
null_density,
42, // bytes_seed
42, // bytes_length_seed
)
}
/// Creates a random [`GenericBinaryArray`] of a given `size` and `null_density`
/// filling it with random bytes, generated using the provided `seed`s.
///
/// the `bytes_seed` is used to seed the RNG for generating the byte values,
/// while the `bytes_length_seed` is used to seed the RNG for generating the length of an array item
///
/// These values can be the same as they are used to seed different RNGs internally.
pub fn create_binary_array_with_seed<Offset: OffsetSizeTrait>(
size: usize,
null_density: f32,
bytes_seed: u64,
bytes_length_seed: u64,
) -> GenericBinaryArray<Offset> {
let rng = &mut StdRng::seed_from_u64(bytes_seed);
let range_rng = &mut StdRng::seed_from_u64(bytes_length_seed);
(0..size)
.map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
let value = rng
.sample_iter::<u8, _>(StandardUniform)
.take(range_rng.random_range(0..8))
.collect::<Vec<u8>>();
Some(value)
}
})
.collect()
}
/// Creates a random [`GenericBinaryArray`] of a given `size` and `null_density`
/// filling it with random bytes with lengths in the specified range,
/// all starting with the provided `prefix`, generated using the provided `seed`.
///
pub fn create_binary_array_with_len_range_and_prefix_and_seed<Offset: OffsetSizeTrait>(
size: usize,
null_density: f32,
min_len: usize,
max_len: usize,
prefix: &[u8],
seed: u64,
) -> GenericBinaryArray<Offset> {
assert!(min_len <= max_len, "min_len must be <= max_len");
assert!(prefix.len() <= max_len, "Prefix length must be <= max_len");
let rng = &mut StdRng::seed_from_u64(seed);
(0..size)
.map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
let remaining_len = rng
.random_range(min_len.saturating_sub(prefix.len())..=(max_len - prefix.len()));
let remaining = rng
.sample_iter::<u8, _>(StandardUniform)
.take(remaining_len);
let value = prefix.iter().copied().chain(remaining).collect::<Vec<u8>>();
Some(value)
}
})
.collect()
}
/// Creates an random (but fixed-seeded) array of a given size and null density
pub fn create_fsb_array(size: usize, null_density: f32, value_len: usize) -> FixedSizeBinaryArray {
let rng = &mut seedable_rng();
FixedSizeBinaryArray::try_from_sparse_iter_with_size(
(0..size).map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
let value = rng
.sample_iter::<u8, _>(StandardUniform)
.take(value_len)
.collect::<Vec<u8>>();
Some(value)
}
}),
value_len as i32,
)
.unwrap()
}
/// Creates a random (but fixed-seeded) dictionary array of a given size and null density
/// with the provided values array
pub fn create_dict_from_values<K>(
size: usize,
null_density: f32,
values: &dyn Array,
) -> DictionaryArray<K>
where
K: ArrowDictionaryKeyType,
StandardUniform: Distribution<K::Native>,
K::Native: SampleUniform,
{
let min_key = K::Native::from_usize(0).unwrap();
let max_key = K::Native::from_usize(values.len()).unwrap();
create_sparse_dict_from_values(size, null_density, values, min_key..max_key)
}
/// Creates a random (but fixed-seeded) dictionary array of a given size and null density
/// with the provided values array and key range
pub fn create_sparse_dict_from_values<K>(
size: usize,
null_density: f32,
values: &dyn Array,
key_range: Range<K::Native>,
) -> DictionaryArray<K>
where
K: ArrowDictionaryKeyType,
StandardUniform: Distribution<K::Native>,
K::Native: SampleUniform,
{
let mut rng = seedable_rng();
let data_type =
DataType::Dictionary(Box::new(K::DATA_TYPE), Box::new(values.data_type().clone()));
let keys: Buffer = (0..size)
.map(|_| rng.random_range(key_range.clone()))
.collect();
let nulls: Option<Buffer> = (null_density != 0.).then(|| {
(0..size)
.map(|_| rng.random_bool(null_density as _))
.collect()
});
let data = ArrayDataBuilder::new(data_type)
.len(size)
.null_bit_buffer(nulls)
.add_buffer(keys)
.add_child_data(values.to_data())
.build()
.unwrap();
DictionaryArray::from(data)
}
/// Creates a random (but fixed-seeded) f16 array of a given size and nan-value density
pub fn create_f16_array(size: usize, nan_density: f32) -> Float16Array {
let mut rng = seedable_rng();
(0..size)
.map(|_| {
if rng.random::<f32>() < nan_density {
Some(f16::NAN)
} else {
Some(f16::from_f32(rng.random()))
}
})
.collect()
}
/// Creates a random (but fixed-seeded) f32 array of a given size and nan-value density
pub fn create_f32_array(size: usize, nan_density: f32) -> Float32Array {
let mut rng = seedable_rng();
(0..size)
.map(|_| {
if rng.random::<f32>() < nan_density {
Some(f32::NAN)
} else {
Some(rng.random())
}
})
.collect()
}
/// Creates a random (but fixed-seeded) f64 array of a given size and nan-value density
pub fn create_f64_array(size: usize, nan_density: f32) -> Float64Array {
let mut rng = seedable_rng();
(0..size)
.map(|_| {
if rng.random::<f32>() < nan_density {
Some(f64::NAN)
} else {
Some(rng.random())
}
})
.collect()
}
+797
View File
@@ -0,0 +1,797 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
//! Utilities to generate random arrays and batches
use std::sync::Arc;
use rand::{
Rng,
distr::uniform::{SampleRange, SampleUniform},
};
use crate::array::*;
use crate::error::{ArrowError, Result};
use crate::{
buffer::{Buffer, MutableBuffer},
datatypes::*,
};
use super::{bench_util::*, bit_util, test_util::seedable_rng};
/// Create a random [RecordBatch] from a schema
pub fn create_random_batch(
schema: SchemaRef,
size: usize,
null_density: f32,
true_density: f32,
) -> Result<RecordBatch> {
let columns = schema
.fields()
.iter()
.map(|field| create_random_array(field, size, null_density, true_density))
.collect::<Result<Vec<ArrayRef>>>()?;
RecordBatch::try_new_with_options(
schema,
columns,
&RecordBatchOptions::new().with_match_field_names(false),
)
}
/// Create a random [ArrayRef] from a [DataType] with a length,
/// null density and true density (for [BooleanArray]).
///
/// # Arguments
///
/// * `field` - The field containing the data type for which to create a random array
/// * `size` - The number of elements in the generated array
/// * `null_density` - The approximate fraction of null values in the resulting array (0.0 to 1.0)
/// * `true_density` - The approximate fraction of true values in boolean arrays (0.0 to 1.0)
///
pub fn create_random_array(
field: &Field,
size: usize,
mut null_density: f32,
true_density: f32,
) -> Result<ArrayRef> {
// Override nullability in case of not nested and not dictionary
// For nested we don't want to override as we want to keep the nullability for the children
// For dictionary it handle the nullability internally
if !field.data_type().is_nested() && !matches!(field.data_type(), Dictionary(_, _)) {
// Override null density with 0.0 if the array is non-nullable
null_density = match field.is_nullable() {
true => null_density,
false => 0.0,
};
}
use DataType::*;
let array = match field.data_type() {
Null => Arc::new(NullArray::new(size)) as ArrayRef,
Boolean => Arc::new(create_boolean_array(size, null_density, true_density)),
Int8 => Arc::new(create_primitive_array::<Int8Type>(size, null_density)),
Int16 => Arc::new(create_primitive_array::<Int16Type>(size, null_density)),
Int32 => Arc::new(create_primitive_array::<Int32Type>(size, null_density)),
Int64 => Arc::new(create_primitive_array::<Int64Type>(size, null_density)),
UInt8 => Arc::new(create_primitive_array::<UInt8Type>(size, null_density)),
UInt16 => Arc::new(create_primitive_array::<UInt16Type>(size, null_density)),
UInt32 => Arc::new(create_primitive_array::<UInt32Type>(size, null_density)),
UInt64 => Arc::new(create_primitive_array::<UInt64Type>(size, null_density)),
Float16 => {
return Err(ArrowError::NotYetImplemented(
"Float16 is not implemented".to_string(),
));
}
Float32 => Arc::new(create_primitive_array::<Float32Type>(size, null_density)),
Float64 => Arc::new(create_primitive_array::<Float64Type>(size, null_density)),
Timestamp(unit, tz) => match unit {
TimeUnit::Second => Arc::new(
create_random_temporal_array::<TimestampSecondType>(size, null_density)
.with_timezone_opt(tz.clone()),
) as ArrayRef,
TimeUnit::Millisecond => Arc::new(
create_random_temporal_array::<TimestampMillisecondType>(size, null_density)
.with_timezone_opt(tz.clone()),
),
TimeUnit::Microsecond => Arc::new(
create_random_temporal_array::<TimestampMicrosecondType>(size, null_density)
.with_timezone_opt(tz.clone()),
),
TimeUnit::Nanosecond => Arc::new(
create_random_temporal_array::<TimestampNanosecondType>(size, null_density)
.with_timezone_opt(tz.clone()),
),
},
Date32 => Arc::new(create_random_temporal_array::<Date32Type>(
size,
null_density,
)),
Date64 => Arc::new(create_random_temporal_array::<Date64Type>(
size,
null_density,
)),
Time32(unit) => match unit {
TimeUnit::Second => Arc::new(create_random_temporal_array::<Time32SecondType>(
size,
null_density,
)) as ArrayRef,
TimeUnit::Millisecond => Arc::new(
create_random_temporal_array::<Time32MillisecondType>(size, null_density),
),
_ => {
return Err(ArrowError::InvalidArgumentError(format!(
"Unsupported unit {unit:?} for Time32"
)));
}
},
Time64(unit) => match unit {
TimeUnit::Microsecond => Arc::new(
create_random_temporal_array::<Time64MicrosecondType>(size, null_density),
) as ArrayRef,
TimeUnit::Nanosecond => Arc::new(create_random_temporal_array::<Time64NanosecondType>(
size,
null_density,
)),
_ => {
return Err(ArrowError::InvalidArgumentError(format!(
"Unsupported unit {unit:?} for Time64"
)));
}
},
Utf8 => Arc::new(create_string_array::<i32>(size, null_density)),
LargeUtf8 => Arc::new(create_string_array::<i64>(size, null_density)),
Utf8View => Arc::new(create_string_view_array_with_len(
size,
null_density,
4,
false,
)),
Binary => Arc::new(create_binary_array::<i32>(size, null_density)),
LargeBinary => Arc::new(create_binary_array::<i64>(size, null_density)),
FixedSizeBinary(len) => Arc::new(create_fsb_array(size, null_density, *len as usize)),
BinaryView => Arc::new(
create_string_view_array_with_len(size, null_density, 4, false).to_binary_view(),
),
List(_) => create_random_list_array(field, size, null_density, true_density)?,
LargeList(_) => create_random_list_array(field, size, null_density, true_density)?,
Struct(_) => create_random_struct_array(field, size, null_density, true_density)?,
d @ Dictionary(_, value_type) if crate::compute::can_cast_types(value_type, d) => {
let f = Field::new(
field.name(),
value_type.as_ref().clone(),
field.is_nullable(),
);
let v = create_random_array(&f, size, null_density, true_density)?;
crate::compute::cast(&v, d)?
}
Map(_, _) => create_random_map_array(field, size, null_density, true_density)?,
Decimal128(_, _) => create_random_decimal_array(field, size, null_density)?,
Decimal256(_, _) => create_random_decimal_array(field, size, null_density)?,
other => {
return Err(ArrowError::NotYetImplemented(format!(
"Generating random arrays not yet implemented for {other:?}"
)));
}
};
if !field.is_nullable() {
assert_eq!(array.null_count(), 0);
}
Ok(array)
}
#[inline]
fn create_random_decimal_array(field: &Field, size: usize, null_density: f32) -> Result<ArrayRef> {
let mut rng = seedable_rng();
match field.data_type() {
DataType::Decimal128(precision, scale) => {
let values = (0..size)
.map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
Some(rng.random::<i128>())
}
})
.collect::<Vec<_>>();
Ok(Arc::new(
Decimal128Array::from(values).with_precision_and_scale(*precision, *scale)?,
))
}
DataType::Decimal256(precision, scale) => {
let values = (0..size)
.map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
Some(i256::from_parts(rng.random::<u128>(), rng.random::<i128>()))
}
})
.collect::<Vec<_>>();
Ok(Arc::new(
Decimal256Array::from(values).with_precision_and_scale(*precision, *scale)?,
))
}
_ => Err(ArrowError::InvalidArgumentError(format!(
"Cannot create decimal array for field {field}"
))),
}
}
#[inline]
fn create_random_list_array(
field: &Field,
size: usize,
null_density: f32,
true_density: f32,
) -> Result<ArrayRef> {
// Override null density with 0.0 if the array is non-nullable
let list_null_density = match field.is_nullable() {
true => null_density,
false => 0.0,
};
let list_field;
let (offsets, child_len) = match field.data_type() {
DataType::List(f) => {
let (offsets, child_len) = create_random_offsets::<i32>(size, 0, 5);
list_field = f;
(Buffer::from(offsets.to_byte_slice()), child_len as usize)
}
DataType::LargeList(f) => {
let (offsets, child_len) = create_random_offsets::<i64>(size, 0, 5);
list_field = f;
(Buffer::from(offsets.to_byte_slice()), child_len as usize)
}
_ => {
return Err(ArrowError::InvalidArgumentError(format!(
"Cannot create list array for field {field}"
)));
}
};
// Create list's child data
let child_array = create_random_array(list_field, child_len, null_density, true_density)?;
let child_data = child_array.to_data();
// Create list's null buffers, if it is nullable
let null_buffer = match field.is_nullable() {
true => Some(create_random_null_buffer(size, list_null_density)),
false => None,
};
let list_data = unsafe {
ArrayData::new_unchecked(
field.data_type().clone(),
size,
None,
null_buffer,
0,
vec![offsets],
vec![child_data],
)
};
Ok(make_array(list_data))
}
#[inline]
fn create_random_struct_array(
field: &Field,
size: usize,
null_density: f32,
true_density: f32,
) -> Result<ArrayRef> {
let struct_fields = match field.data_type() {
DataType::Struct(fields) => fields,
_ => {
return Err(ArrowError::InvalidArgumentError(format!(
"Cannot create struct array for field {field}"
)));
}
};
let child_arrays = struct_fields
.iter()
.map(|struct_field| create_random_array(struct_field, size, null_density, true_density))
.collect::<Result<Vec<_>>>()?;
let null_buffer = match field.is_nullable() {
true => {
let nulls = arrow_buffer::BooleanBuffer::new(
create_random_null_buffer(size, null_density),
0,
size,
);
Some(nulls.into())
}
false => None,
};
Ok(Arc::new(StructArray::try_new(
struct_fields.clone(),
child_arrays,
null_buffer,
)?))
}
#[inline]
fn create_random_map_array(
field: &Field,
size: usize,
null_density: f32,
true_density: f32,
) -> Result<ArrayRef> {
// Override null density with 0.0 if the array is non-nullable
let map_null_density = match field.is_nullable() {
true => null_density,
false => 0.0,
};
let entries_field = match field.data_type() {
DataType::Map(f, _) => f,
_ => {
return Err(ArrowError::InvalidArgumentError(format!(
"Cannot create map array for field {field:?}"
)));
}
};
let (offsets, child_len) = create_random_offsets::<i32>(size, 0, 5);
let offsets = Buffer::from(offsets.to_byte_slice());
let entries = create_random_array(
entries_field,
child_len as usize,
null_density,
true_density,
)?
.to_data();
let null_buffer = match field.is_nullable() {
true => Some(create_random_null_buffer(size, map_null_density)),
false => None,
};
let map_data = unsafe {
ArrayData::new_unchecked(
field.data_type().clone(),
size,
None,
null_buffer,
0,
vec![offsets],
vec![entries],
)
};
Ok(make_array(map_data))
}
/// Generate random offsets for list arrays
fn create_random_offsets<T: OffsetSizeTrait + SampleUniform>(
size: usize,
min: T,
max: T,
) -> (Vec<T>, T) {
let rng = &mut seedable_rng();
let mut current_offset = T::zero();
let mut offsets = Vec::with_capacity(size + 1);
offsets.push(current_offset);
(0..size).for_each(|_| {
current_offset += rng.random_range(min..max);
offsets.push(current_offset);
});
(offsets, current_offset)
}
fn create_random_null_buffer(size: usize, null_density: f32) -> Buffer {
let mut rng = seedable_rng();
let mut mut_buf = MutableBuffer::new_null(size);
{
let mut_slice = mut_buf.as_slice_mut();
(0..size).for_each(|i| {
if rng.random::<f32>() >= null_density {
bit_util::set_bit(mut_slice, i)
}
})
};
mut_buf.into()
}
/// Useful for testing. The range of values are not likely to be representative of the
/// actual bounds.
pub trait RandomTemporalValue: ArrowTemporalType {
/// Returns the range of values for `impl`'d type
fn value_range() -> impl SampleRange<Self::Native>;
/// Generate a random value within the range of the type
fn gen_range<R: Rng>(rng: &mut R) -> Self::Native
where
Self::Native: SampleUniform,
{
rng.random_range(Self::value_range())
}
/// Generate a random value of the type
fn random<R: Rng>(rng: &mut R) -> Self::Native
where
Self::Native: SampleUniform,
{
Self::gen_range(rng)
}
}
impl RandomTemporalValue for TimestampSecondType {
/// Range of values for a timestamp in seconds. The range begins at the start
/// of the unix epoch and continues for 100 years.
fn value_range() -> impl SampleRange<Self::Native> {
0..60 * 60 * 24 * 365 * 100
}
}
impl RandomTemporalValue for TimestampMillisecondType {
/// Range of values for a timestamp in milliseconds. The range begins at the start
/// of the unix epoch and continues for 100 years.
fn value_range() -> impl SampleRange<Self::Native> {
0..1_000 * 60 * 60 * 24 * 365 * 100
}
}
impl RandomTemporalValue for TimestampMicrosecondType {
/// Range of values for a timestamp in microseconds. The range begins at the start
/// of the unix epoch and continues for 100 years.
fn value_range() -> impl SampleRange<Self::Native> {
0..1_000 * 1_000 * 60 * 60 * 24 * 365 * 100
}
}
impl RandomTemporalValue for TimestampNanosecondType {
/// Range of values for a timestamp in nanoseconds. The range begins at the start
/// of the unix epoch and continues for 100 years.
fn value_range() -> impl SampleRange<Self::Native> {
0..1_000 * 1_000 * 1_000 * 60 * 60 * 24 * 365 * 100
}
}
impl RandomTemporalValue for Date32Type {
/// Range of values representing the elapsed time since UNIX epoch in days. The
/// range begins at the start of the unix epoch and continues for 100 years.
fn value_range() -> impl SampleRange<Self::Native> {
0..365 * 100
}
}
impl RandomTemporalValue for Date64Type {
/// Range of values representing the elapsed time since UNIX epoch in milliseconds.
/// The range begins at the start of the unix epoch and continues for 100 years.
fn value_range() -> impl SampleRange<Self::Native> {
0..1_000 * 60 * 60 * 24 * 365 * 100
}
}
impl RandomTemporalValue for Time32SecondType {
/// Range of values representing the elapsed time since midnight in seconds. The
/// range is from 0 to 24 hours.
fn value_range() -> impl SampleRange<Self::Native> {
0..60 * 60 * 24
}
}
impl RandomTemporalValue for Time32MillisecondType {
/// Range of values representing the elapsed time since midnight in milliseconds. The
/// range is from 0 to 24 hours.
fn value_range() -> impl SampleRange<Self::Native> {
0..1_000 * 60 * 60 * 24
}
}
impl RandomTemporalValue for Time64MicrosecondType {
/// Range of values representing the elapsed time since midnight in microseconds. The
/// range is from 0 to 24 hours.
fn value_range() -> impl SampleRange<Self::Native> {
0..1_000 * 1_000 * 60 * 60 * 24
}
}
impl RandomTemporalValue for Time64NanosecondType {
/// Range of values representing the elapsed time since midnight in nanoseconds. The
/// range is from 0 to 24 hours.
fn value_range() -> impl SampleRange<Self::Native> {
0..1_000 * 1_000 * 1_000 * 60 * 60 * 24
}
}
fn create_random_temporal_array<T>(size: usize, null_density: f32) -> PrimitiveArray<T>
where
T: RandomTemporalValue,
<T as ArrowPrimitiveType>::Native: SampleUniform,
{
let mut rng = seedable_rng();
(0..size)
.map(|_| {
if rng.random::<f32>() < null_density {
None
} else {
Some(T::random(&mut rng))
}
})
.collect()
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_create_batch() {
let size = 32;
let fields = vec![
Field::new("a", DataType::Int32, true),
Field::new(
"timestamp_without_timezone",
DataType::Timestamp(TimeUnit::Nanosecond, None),
true,
),
Field::new(
"timestamp_with_timezone",
DataType::Timestamp(TimeUnit::Nanosecond, Some("UTC".into())),
true,
),
];
let schema = Schema::new(fields);
let schema_ref = Arc::new(schema);
let batch = create_random_batch(schema_ref.clone(), size, 0.35, 0.7).unwrap();
assert_eq!(batch.schema(), schema_ref);
assert_eq!(batch.num_columns(), schema_ref.fields().len());
for array in batch.columns() {
assert_eq!(array.len(), size);
}
}
#[test]
fn test_create_batch_non_null() {
let size = 32;
let fields = vec![
Field::new("a", DataType::Int32, false),
Field::new(
"b",
DataType::List(Arc::new(Field::new_list_field(DataType::LargeUtf8, false))),
false,
),
Field::new("a", DataType::Int32, false),
];
let schema = Schema::new(fields);
let schema_ref = Arc::new(schema);
let batch = create_random_batch(schema_ref.clone(), size, 0.35, 0.7).unwrap();
assert_eq!(batch.schema(), schema_ref);
assert_eq!(batch.num_columns(), schema_ref.fields().len());
for array in batch.columns() {
assert_eq!(array.null_count(), 0);
assert_eq!(array.logical_null_count(), 0);
}
// Test that the list's child values are non-null
let b_array = batch.column(1);
let list_array = b_array.as_list::<i32>();
let child_array = list_array.values();
assert_eq!(child_array.null_count(), 0);
// There should be more values than the list, to show that it's a list
assert!(child_array.len() > list_array.len());
}
#[test]
fn test_create_struct_array() {
let size = 32;
let struct_fields = Fields::from(vec![
Field::new("b", DataType::Boolean, true),
Field::new(
"c",
DataType::LargeList(Arc::new(Field::new_list_field(
DataType::List(Arc::new(Field::new_list_field(
DataType::FixedSizeBinary(6),
true,
))),
false,
))),
true,
),
Field::new(
"d",
DataType::Struct(Fields::from(vec![
Field::new("d_x", DataType::Int32, true),
Field::new("d_y", DataType::Float32, false),
Field::new("d_z", DataType::Binary, true),
])),
true,
),
]);
let field = Field::new("struct", DataType::Struct(struct_fields), true);
let array = create_random_array(&field, size, 0.2, 0.5).unwrap();
assert_eq!(array.len(), 32);
let struct_array = array.as_any().downcast_ref::<StructArray>().unwrap();
assert_eq!(struct_array.columns().len(), 3);
// Test that the nested list makes sense,
// i.e. its children's values are more than the parent, to show repetition
let col_c = struct_array.column_by_name("c").unwrap();
let col_c = col_c.as_any().downcast_ref::<LargeListArray>().unwrap();
assert_eq!(col_c.len(), size);
let col_c_list = col_c.values().as_list::<i32>();
assert!(col_c_list.len() > size);
// Its values should be FixedSizeBinary(6)
let fsb = col_c_list.values();
assert_eq!(fsb.data_type(), &DataType::FixedSizeBinary(6));
assert!(fsb.len() > col_c_list.len());
// Test nested struct
let col_d = struct_array.column_by_name("d").unwrap();
let col_d = col_d.as_any().downcast_ref::<StructArray>().unwrap();
let col_d_y = col_d.column_by_name("d_y").unwrap();
assert_eq!(col_d_y.data_type(), &DataType::Float32);
assert_eq!(col_d_y.null_count(), 0);
}
#[test]
fn test_create_list_array_nested_nullability() {
let list_field = Field::new_list(
"not_null_list",
Field::new_list_field(DataType::Boolean, true),
false,
);
let list_array = create_random_array(&list_field, 100, 0.95, 0.5).unwrap();
assert_eq!(list_array.null_count(), 0);
assert!(list_array.as_list::<i32>().values().null_count() > 0);
}
#[test]
fn test_create_struct_array_nested_nullability() {
let struct_child_fields = vec![
Field::new("null_int", DataType::Int32, true),
Field::new("int", DataType::Int32, false),
];
let struct_field = Field::new_struct("not_null_struct", struct_child_fields, false);
let struct_array = create_random_array(&struct_field, 100, 0.95, 0.5).unwrap();
assert_eq!(struct_array.null_count(), 0);
assert!(
struct_array
.as_struct()
.column_by_name("null_int")
.unwrap()
.null_count()
> 0
);
assert_eq!(
struct_array
.as_struct()
.column_by_name("int")
.unwrap()
.null_count(),
0
);
}
#[test]
fn test_create_list_array_nested_struct_nullability() {
let struct_child_fields = vec![
Field::new("null_int", DataType::Int32, true),
Field::new("int", DataType::Int32, false),
];
let list_item_field =
Field::new_list_field(DataType::Struct(struct_child_fields.into()), true);
let list_field = Field::new_list("not_null_list", list_item_field, false);
let list_array = create_random_array(&list_field, 100, 0.95, 0.5).unwrap();
assert_eq!(list_array.null_count(), 0);
assert!(list_array.as_list::<i32>().values().null_count() > 0);
assert!(
list_array
.as_list::<i32>()
.values()
.as_struct()
.column_by_name("null_int")
.unwrap()
.null_count()
> 0
);
assert_eq!(
list_array
.as_list::<i32>()
.values()
.as_struct()
.column_by_name("int")
.unwrap()
.null_count(),
0
);
}
#[test]
fn test_create_map_array() {
let map_field = Field::new_map(
"map",
"entries",
Field::new("key", DataType::Utf8, false),
Field::new("value", DataType::Utf8, true),
false,
false,
);
let array = create_random_array(&map_field, 100, 0.8, 0.5).unwrap();
assert_eq!(array.len(), 100);
// Map field is not null
assert_eq!(array.null_count(), 0);
assert_eq!(array.logical_null_count(), 0);
// Maps have multiple values like a list, so internal arrays are longer
assert!(array.as_map().keys().len() > array.len());
assert!(array.as_map().values().len() > array.len());
// Keys are not nullable
assert_eq!(array.as_map().keys().null_count(), 0);
// Values are nullable
assert!(array.as_map().values().null_count() > 0);
assert_eq!(array.as_map().keys().data_type(), &DataType::Utf8);
assert_eq!(array.as_map().values().data_type(), &DataType::Utf8);
}
#[test]
fn test_create_decimal_array() {
let size = 10;
let fields = vec![
Field::new("a", DataType::Decimal128(10, -2), true),
Field::new("b", DataType::Decimal256(10, -2), true),
];
let schema = Schema::new(fields);
let schema_ref = Arc::new(schema);
let batch = create_random_batch(schema_ref.clone(), size, 0.35, 0.7).unwrap();
assert_eq!(batch.schema(), schema_ref);
assert_eq!(batch.num_columns(), schema_ref.fields().len());
for array in batch.columns() {
assert_eq!(array.len(), size);
}
}
#[test]
fn create_non_nullable_decimal_array_with_null_density() {
let size = 10;
let fields = vec![
Field::new("a", DataType::Decimal128(10, -2), false),
Field::new("b", DataType::Decimal256(10, -2), false),
];
let schema = Schema::new(fields);
let schema_ref = Arc::new(schema);
let batch = create_random_batch(schema_ref.clone(), size, 0.35, 0.7).unwrap();
assert_eq!(batch.schema(), schema_ref);
assert_eq!(batch.num_columns(), schema_ref.fields().len());
for array in batch.columns() {
assert_eq!(array.len(), size);
assert_eq!(array.null_count(), 0);
}
}
}
+34
View File
@@ -0,0 +1,34 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
//! Utility functions for working with Arrow data
pub use arrow_buffer::{bit_chunk_iterator, bit_util};
pub use arrow_data::bit_iterator;
pub use arrow_data::bit_mask;
#[cfg(feature = "test_utils")]
pub mod bench_util;
#[cfg(feature = "test_utils")]
pub mod data_gen;
#[cfg(feature = "prettyprint")]
pub use arrow_cast::pretty;
pub mod string_writer;
#[cfg(any(test, feature = "test_utils"))]
pub mod test_util;
pub use arrow_cast::display;
+105
View File
@@ -0,0 +1,105 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
//! String Writer
//! This string writer encapsulates `std::string::String` and
//! implements `std::io::Write` trait, which makes String as a
//! writable object like File.
//!
//! Example:
//!
//! ```
//! #[cfg(feature = "csv")]
//! {
//! use arrow::array::*;
//! use arrow::csv;
//! use arrow::datatypes::*;
//! use arrow::record_batch::RecordBatch;
//! use arrow::util::string_writer::StringWriter;
//! use std::sync::Arc;
//!
//! let schema = Schema::new(vec![
//! Field::new("c1", DataType::Utf8, false),
//! Field::new("c2", DataType::Float64, true),
//! Field::new("c3", DataType::UInt32, false),
//! Field::new("c3", DataType::Boolean, true),
//! ]);
//! let c1 = StringArray::from(vec![
//! "Lorem ipsum dolor sit amet",
//! "consectetur adipiscing elit",
//! "sed do eiusmod tempor",
//! ]);
//! let c2 = PrimitiveArray::<Float64Type>::from(vec![
//! Some(123.564532),
//! None,
//! Some(-556132.25),
//! ]);
//! let c3 = PrimitiveArray::<UInt32Type>::from(vec![3, 2, 1]);
//! let c4 = BooleanArray::from(vec![Some(true), Some(false), None]);
//!
//! let batch = RecordBatch::try_new(
//! Arc::new(schema),
//! vec![Arc::new(c1), Arc::new(c2), Arc::new(c3), Arc::new(c4)],
//! )
//! .unwrap();
//!
//! let sw = StringWriter::new();
//! let mut writer = csv::Writer::new(sw);
//! writer.write(&batch).unwrap();
//! }
//! ```
use core::str;
use std::fmt::Formatter;
use std::io::{Error, ErrorKind, Result, Write};
/// A writer that allows writing to a `String`
/// like an `std::io::Write` object.
#[derive(Debug, Default)]
pub struct StringWriter {
data: String,
}
impl StringWriter {
/// Create a new `StringWriter`
pub fn new() -> Self {
Self::default()
}
}
impl std::fmt::Display for StringWriter {
fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result {
write!(f, "{}", self.data)
}
}
impl Write for StringWriter {
fn write(&mut self, buf: &[u8]) -> Result<usize> {
let string = match str::from_utf8(buf) {
Ok(x) => x,
Err(e) => {
return Err(Error::new(ErrorKind::InvalidData, e));
}
};
self.data.push_str(string);
Ok(string.len())
}
fn flush(&mut self) -> Result<()> {
Ok(())
}
}
+255
View File
@@ -0,0 +1,255 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
//! Utils to make testing easier
use rand::{Rng, SeedableRng, rngs::StdRng};
use std::{env, error::Error, fs, io::Write, path::PathBuf};
/// Returns a vector of size `n`, filled with randomly generated bytes.
pub fn random_bytes(n: usize) -> Vec<u8> {
let mut result = vec![];
let mut rng = seedable_rng();
for _ in 0..n {
result.push(rng.random_range(0..255));
}
result
}
/// Returns fixed seedable RNG
pub fn seedable_rng() -> StdRng {
StdRng::seed_from_u64(42)
}
/// Returns file handle for a temp file in 'target' directory with a provided content
///
/// TODO: Originates from `parquet` utils, can be merged in [ARROW-4064]
pub fn get_temp_file(file_name: &str, content: &[u8]) -> fs::File {
// build tmp path to a file in "target/debug/testdata"
let mut path_buf = env::current_dir().unwrap();
path_buf.push("target");
path_buf.push("debug");
path_buf.push("testdata");
fs::create_dir_all(&path_buf).unwrap();
path_buf.push(file_name);
// write file content
let mut tmp_file = fs::File::create(path_buf.as_path()).unwrap();
tmp_file.write_all(content).unwrap();
tmp_file.sync_all().unwrap();
// return file handle for both read and write
let file = fs::OpenOptions::new()
.read(true)
.write(true)
.open(path_buf.as_path());
assert!(file.is_ok());
file.unwrap()
}
/// Returns the arrow test data directory, which is by default stored
/// in a git submodule rooted at `arrow/testing/data`.
///
/// The default can be overridden by the optional environment
/// variable `ARROW_TEST_DATA`
///
/// panics when the directory can not be found.
///
/// Example:
/// ```
/// let testdata = arrow::util::test_util::arrow_test_data();
/// let csvdata = format!("{}/csv/aggregate_test_100.csv", testdata);
/// assert!(std::path::PathBuf::from(csvdata).exists());
/// ```
pub fn arrow_test_data() -> String {
match get_data_dir("ARROW_TEST_DATA", "../testing/data") {
Ok(pb) => pb.display().to_string(),
Err(err) => panic!("failed to get arrow data dir: {err}"),
}
}
/// Returns the parquest test data directory, which is by default
/// stored in a git submodule rooted at
/// `arrow/parquest-testing/data`.
///
/// The default can be overridden by the optional environment variable
/// `PARQUET_TEST_DATA`
///
/// panics when the directory can not be found.
///
/// Example:
/// ```
/// let testdata = arrow::util::test_util::parquet_test_data();
/// let filename = format!("{}/binary.parquet", testdata);
/// assert!(std::path::PathBuf::from(filename).exists());
/// ```
pub fn parquet_test_data() -> String {
match get_data_dir("PARQUET_TEST_DATA", "../parquet-testing/data") {
Ok(pb) => pb.display().to_string(),
Err(err) => panic!("failed to get parquet data dir: {err}"),
}
}
/// Returns a directory path for finding test data.
///
/// udf_env: name of an environment variable
///
/// submodule_dir: fallback path (relative to CARGO_MANIFEST_DIR)
///
/// Returns either:
/// The path referred to in `udf_env` if that variable is set and refers to a directory
/// The submodule_data directory relative to CARGO_MANIFEST_PATH
fn get_data_dir(udf_env: &str, submodule_data: &str) -> Result<PathBuf, Box<dyn Error>> {
// Try user defined env.
if let Ok(dir) = env::var(udf_env) {
let trimmed = dir.trim().to_string();
if !trimmed.is_empty() {
let pb = PathBuf::from(trimmed);
if pb.is_dir() {
return Ok(pb);
} else {
return Err(format!(
"the data dir `{}` defined by env {} not found",
pb.display(),
udf_env
)
.into());
}
}
}
// The env is undefined or its value is trimmed to empty, let's try default dir.
// env "CARGO_MANIFEST_DIR" is "the directory containing the manifest of your package",
// set by `cargo run` or `cargo test`, see:
// https://doc.rust-lang.org/cargo/reference/environment-variables.html
let dir = env!("CARGO_MANIFEST_DIR");
let pb = PathBuf::from(dir).join(submodule_data);
if pb.is_dir() {
Ok(pb)
} else {
Err(format!(
"env `{}` is undefined or has empty value, and the pre-defined data dir `{}` not found\n\
HINT: try running `git submodule update --init`",
udf_env,
pb.display(),
).into())
}
}
/// An iterator that is untruthful about its actual length
#[derive(Debug, Clone)]
pub struct BadIterator<T> {
/// where the iterator currently is
cur: usize,
/// How many items will this iterator *actually* make
limit: usize,
/// How many items this iterator claims it will make
claimed: usize,
/// The items to return. If there are fewer items than `limit`
/// they will be repeated
pub items: Vec<T>,
}
impl<T> BadIterator<T> {
/// Create a new iterator for `<limit>` items, but that reports to
/// produce `<claimed>` items. Must provide at least 1 item.
pub fn new(limit: usize, claimed: usize, items: Vec<T>) -> Self {
assert!(!items.is_empty());
Self {
cur: 0,
limit,
claimed,
items,
}
}
}
impl<T: Clone> Iterator for BadIterator<T> {
type Item = T;
fn next(&mut self) -> Option<Self::Item> {
if self.cur < self.limit {
let next_item_idx = self.cur % self.items.len();
let next_item = self.items[next_item_idx].clone();
self.cur += 1;
Some(next_item)
} else {
None
}
}
/// report whatever the iterator says to
fn size_hint(&self) -> (usize, Option<usize>) {
(0, Some(self.claimed))
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_data_dir() {
let udf_env = "get_data_dir";
let cwd = env::current_dir().unwrap();
let existing_pb = cwd.join("..");
let existing = existing_pb.display().to_string();
let existing_str = existing.as_str();
let non_existing = cwd.join("non-existing-dir").display().to_string();
let non_existing_str = non_existing.as_str();
unsafe { env::set_var(udf_env, non_existing_str) };
let res = get_data_dir(udf_env, existing_str);
assert!(res.is_err());
unsafe { env::set_var(udf_env, "") };
let res = get_data_dir(udf_env, existing_str);
assert!(res.is_ok());
assert_eq!(res.unwrap(), existing_pb);
unsafe { env::set_var(udf_env, " ") };
let res = get_data_dir(udf_env, existing_str);
assert!(res.is_ok());
assert_eq!(res.unwrap(), existing_pb);
unsafe { env::set_var(udf_env, existing_str) };
let res = get_data_dir(udf_env, existing_str);
assert!(res.is_ok());
assert_eq!(res.unwrap(), existing_pb);
unsafe { env::remove_var(udf_env) };
let res = get_data_dir(udf_env, non_existing_str);
assert!(res.is_err());
let res = get_data_dir(udf_env, existing_str);
assert!(res.is_ok());
assert_eq!(res.unwrap(), existing_pb);
}
#[test]
fn test_happy() {
let res = arrow_test_data();
assert!(PathBuf::from(res).is_dir());
let res = parquet_test_data();
assert!(PathBuf::from(res).is_dir());
}
}
+190
View File
@@ -0,0 +1,190 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use arrow_arith::numeric::{add, sub};
use arrow_arith::temporal::{DatePart, date_part};
use arrow_array::cast::AsArray;
use arrow_array::temporal_conversions::as_datetime_with_timezone;
use arrow_array::timezone::Tz;
use arrow_array::types::*;
use arrow_array::*;
use chrono::{DateTime, TimeZone};
#[test]
fn test_temporal_array_timestamp_hour_with_timezone_using_chrono_tz() {
let a =
TimestampSecondArray::from(vec![60 * 60 * 10]).with_timezone("Asia/Kolkata".to_string());
let b = date_part(&a, DatePart::Hour).unwrap();
let b = b.as_primitive::<Int32Type>();
assert_eq!(15, b.value(0));
}
#[test]
fn test_temporal_array_timestamp_hour_with_dst_timezone_using_chrono_tz() {
//
// 1635577147 converts to 2021-10-30 17:59:07 in time zone Australia/Sydney (AEDT)
// The offset (difference to UTC) is +11:00. Note that daylight savings is in effect on 2021-10-30.
// When daylight savings is not in effect, Australia/Sydney has an offset difference of +10:00.
let a = TimestampMillisecondArray::from(vec![Some(1635577147000)])
.with_timezone("Australia/Sydney".to_string());
let b = date_part(&a, DatePart::Hour).unwrap();
let b = b.as_primitive::<Int32Type>();
assert_eq!(17, b.value(0));
}
fn test_timestamp_with_timezone_impl<T: ArrowTimestampType>(tz_str: &str) {
let tz: Tz = tz_str.parse().unwrap();
let transform_array = |x: &dyn Array| -> Vec<DateTime<_>> {
x.as_primitive::<T>()
.values()
.into_iter()
.map(|x| as_datetime_with_timezone::<T>(*x, tz).unwrap())
.collect()
};
let values = vec![
tz.with_ymd_and_hms(1970, 1, 28, 23, 0, 0)
.unwrap()
.naive_utc(),
tz.with_ymd_and_hms(1970, 1, 1, 0, 0, 0)
.unwrap()
.naive_utc(),
tz.with_ymd_and_hms(2010, 4, 1, 4, 0, 20)
.unwrap()
.naive_utc(),
tz.with_ymd_and_hms(1960, 1, 30, 4, 23, 20)
.unwrap()
.naive_utc(),
tz.with_ymd_and_hms(2023, 3, 25, 14, 0, 0)
.unwrap()
.naive_utc(),
]
.into_iter()
.map(|x| T::make_value(x).unwrap())
.collect();
let a = PrimitiveArray::<T>::new(values, None).with_timezone(tz_str);
// IntervalYearMonth
let b = IntervalYearMonthArray::from(vec![
IntervalYearMonthType::make_value(0, 1),
IntervalYearMonthType::make_value(5, 34),
IntervalYearMonthType::make_value(-2, 4),
IntervalYearMonthType::make_value(7, -4),
IntervalYearMonthType::make_value(0, 1),
]);
let r1 = add(&a, &b).unwrap();
assert_eq!(
&transform_array(r1.as_ref()),
&[
tz.with_ymd_and_hms(1970, 2, 28, 23, 0, 0).unwrap(),
tz.with_ymd_and_hms(1977, 11, 1, 0, 0, 0).unwrap(),
tz.with_ymd_and_hms(2008, 8, 1, 4, 0, 20).unwrap(),
tz.with_ymd_and_hms(1966, 9, 30, 4, 23, 20).unwrap(),
tz.with_ymd_and_hms(2023, 4, 25, 14, 0, 0).unwrap(),
]
);
let r2 = sub(&r1, &b).unwrap();
assert_eq!(r2.as_ref(), &a);
// IntervalDayTime
let b = IntervalDayTimeArray::from(vec![
IntervalDayTimeType::make_value(0, 0),
IntervalDayTimeType::make_value(5, 454000),
IntervalDayTimeType::make_value(-34, 0),
IntervalDayTimeType::make_value(7, -4000),
IntervalDayTimeType::make_value(1, 0),
]);
let r3 = add(&a, &b).unwrap();
assert_eq!(
&transform_array(r3.as_ref()),
&[
tz.with_ymd_and_hms(1970, 1, 28, 23, 0, 0).unwrap(),
tz.with_ymd_and_hms(1970, 1, 6, 0, 7, 34).unwrap(),
tz.with_ymd_and_hms(2010, 2, 26, 4, 0, 20).unwrap(),
tz.with_ymd_and_hms(1960, 2, 6, 4, 23, 16).unwrap(),
tz.with_ymd_and_hms(2023, 3, 26, 14, 0, 0).unwrap(),
]
);
let r4 = sub(&r3, &b).unwrap();
assert_eq!(r4.as_ref(), &a);
// IntervalMonthDayNano
let b = IntervalMonthDayNanoArray::from(vec![
IntervalMonthDayNanoType::make_value(1, 0, 0),
IntervalMonthDayNanoType::make_value(344, 34, -43_000_000_000),
IntervalMonthDayNanoType::make_value(-593, -33, 13_000_000_000),
IntervalMonthDayNanoType::make_value(5, 2, 493_000_000_000),
IntervalMonthDayNanoType::make_value(1, 0, 0),
]);
let r5 = add(&a, &b).unwrap();
assert_eq!(
&transform_array(r5.as_ref()),
&[
tz.with_ymd_and_hms(1970, 2, 28, 23, 0, 0).unwrap(),
tz.with_ymd_and_hms(1998, 10, 4, 23, 59, 17).unwrap(),
tz.with_ymd_and_hms(1960, 9, 29, 4, 0, 33).unwrap(),
tz.with_ymd_and_hms(1960, 7, 2, 4, 31, 33).unwrap(),
tz.with_ymd_and_hms(2023, 4, 25, 14, 0, 0).unwrap(),
]
);
let r6 = sub(&r5, &b).unwrap();
assert_eq!(
&transform_array(r6.as_ref()),
&[
tz.with_ymd_and_hms(1970, 1, 28, 23, 0, 0).unwrap(),
tz.with_ymd_and_hms(1970, 1, 2, 0, 0, 0).unwrap(),
tz.with_ymd_and_hms(2010, 4, 2, 4, 0, 20).unwrap(),
tz.with_ymd_and_hms(1960, 1, 31, 4, 23, 20).unwrap(),
tz.with_ymd_and_hms(2023, 3, 25, 14, 0, 0).unwrap(),
]
);
}
#[test]
fn test_timestamp_with_offset_timezone() {
let timezones = ["+00:00", "+01:00", "-01:00", "+03:30"];
for timezone in timezones {
test_timestamp_with_timezone_impl::<TimestampSecondType>(timezone);
test_timestamp_with_timezone_impl::<TimestampMillisecondType>(timezone);
test_timestamp_with_timezone_impl::<TimestampMicrosecondType>(timezone);
test_timestamp_with_timezone_impl::<TimestampNanosecondType>(timezone);
}
}
#[test]
fn test_timestamp_with_timezone() {
let timezones = [
"Europe/Paris",
"Europe/London",
"Africa/Bamako",
"America/Dominica",
"Asia/Seoul",
"Asia/Shanghai",
];
for timezone in timezones {
test_timestamp_with_timezone_impl::<TimestampSecondType>(timezone);
test_timestamp_with_timezone_impl::<TimestampMillisecondType>(timezone);
test_timestamp_with_timezone_impl::<TimestampMicrosecondType>(timezone);
test_timestamp_with_timezone_impl::<TimestampNanosecondType>(timezone);
}
}
+677
View File
@@ -0,0 +1,677 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use arrow_array::builder::{PrimitiveDictionaryBuilder, StringDictionaryBuilder, UnionBuilder};
use arrow_array::cast::AsArray;
use arrow_array::types::{
ArrowDictionaryKeyType, Decimal32Type, Decimal64Type, Decimal128Type, Decimal256Type, Int8Type,
Int16Type, Int32Type, Int64Type, TimestampMicrosecondType, UInt8Type, UInt16Type, UInt32Type,
UInt64Type,
};
use arrow_array::{
Array, ArrayRef, ArrowPrimitiveType, BinaryArray, BooleanArray, Date32Array, Date64Array,
Decimal32Array, Decimal64Array, Decimal128Array, Decimal256Array, DurationMicrosecondArray,
DurationMillisecondArray, DurationNanosecondArray, DurationSecondArray, FixedSizeBinaryArray,
FixedSizeListArray, Float16Array, Float32Array, Float64Array, Int8Array, Int16Array,
Int32Array, Int64Array, IntervalDayTimeArray, IntervalMonthDayNanoArray,
IntervalYearMonthArray, LargeBinaryArray, LargeListArray, LargeStringArray, ListArray,
NullArray, PrimitiveArray, StringArray, StructArray, Time32MillisecondArray, Time32SecondArray,
Time64MicrosecondArray, Time64NanosecondArray, TimestampMicrosecondArray,
TimestampMillisecondArray, TimestampNanosecondArray, TimestampSecondArray, UInt8Array,
UInt16Array, UInt32Array, UInt64Array, UnionArray,
};
use arrow_buffer::{Buffer, IntervalDayTime, IntervalMonthDayNano, i256};
use arrow_cast::pretty::pretty_format_columns;
use arrow_cast::{can_cast_types, cast};
use arrow_data::ArrayData;
use arrow_schema::{
ArrowError, DataType, Field, Fields, IntervalUnit, TimeUnit, UnionFields, UnionMode,
};
use half::f16;
use std::sync::Arc;
#[test]
fn test_cast_timestamp_to_string() {
let a = TimestampMillisecondArray::from(vec![Some(864000000005), Some(1545696000001), None])
.with_timezone("UTC".to_string());
let array = Arc::new(a) as ArrayRef;
let b = cast(&array, &DataType::Utf8).unwrap();
let c = b.as_any().downcast_ref::<StringArray>().unwrap();
assert_eq!(&DataType::Utf8, c.data_type());
assert_eq!("1997-05-19T00:00:00.005Z", c.value(0));
assert_eq!("2018-12-25T00:00:00.001Z", c.value(1));
assert!(c.is_null(2));
}
// See: https://en.wikipedia.org/wiki/List_of_tz_database_time_zones for list of valid
// timezones
// Cast Timestamp(_, None) -> Timestamp(_, Some(timezone))
#[test]
fn test_cast_timestamp_with_timezone_daylight_1() {
let string_array: Arc<dyn Array> = Arc::new(StringArray::from(vec![
// This is winter in New York so daylight saving is not in effect
// UTC offset is -05:00
Some("2000-01-01T00:00:00.123456789"),
// This is summer in New York so daylight saving is in effect
// UTC offset is -04:00
Some("2010-07-01T00:00:00.123456789"),
None,
]));
let to_type = DataType::Timestamp(TimeUnit::Nanosecond, None);
let timestamp_array = cast(&string_array, &to_type).unwrap();
let to_type = DataType::Timestamp(TimeUnit::Microsecond, Some("America/New_York".into()));
let timestamp_array = cast(&timestamp_array, &to_type).unwrap();
let string_array = cast(&timestamp_array, &DataType::Utf8).unwrap();
let result = string_array.as_string::<i32>();
assert_eq!("2000-01-01T00:00:00.123456-05:00", result.value(0));
assert_eq!("2010-07-01T00:00:00.123456-04:00", result.value(1));
assert!(result.is_null(2));
}
// Cast Timestamp(_, Some(timezone)) -> Timestamp(_, None)
#[test]
fn test_cast_timestamp_with_timezone_daylight_2() {
let string_array: Arc<dyn Array> = Arc::new(StringArray::from(vec![
Some("2000-01-01T07:00:00.123456789"),
Some("2010-07-01T07:00:00.123456789"),
None,
]));
let to_type = DataType::Timestamp(TimeUnit::Millisecond, Some("America/New_York".into()));
let timestamp_array = cast(&string_array, &to_type).unwrap();
// Check intermediate representation is correct
let string_array = cast(&timestamp_array, &DataType::Utf8).unwrap();
let result = string_array.as_string::<i32>();
assert_eq!("2000-01-01T07:00:00.123-05:00", result.value(0));
assert_eq!("2010-07-01T07:00:00.123-04:00", result.value(1));
assert!(result.is_null(2));
let to_type = DataType::Timestamp(TimeUnit::Nanosecond, None);
let timestamp_array = cast(&timestamp_array, &to_type).unwrap();
let string_array = cast(&timestamp_array, &DataType::Utf8).unwrap();
let result = string_array.as_string::<i32>();
assert_eq!("2000-01-01T12:00:00.123", result.value(0));
assert_eq!("2010-07-01T11:00:00.123", result.value(1));
assert!(result.is_null(2));
}
// Cast Timestamp(_, Some(timezone)) -> Timestamp(_, Some(timezone))
#[test]
fn test_cast_timestamp_with_timezone_daylight_3() {
let string_array: Arc<dyn Array> = Arc::new(StringArray::from(vec![
// Winter in New York, summer in Sydney
// UTC offset is -05:00 (New York) and +11:00 (Sydney)
Some("2000-01-01T00:00:00.123456789"),
// Summer in New York, winter in Sydney
// UTC offset is -04:00 (New York) and +10:00 (Sydney)
Some("2010-07-01T00:00:00.123456789"),
None,
]));
let to_type = DataType::Timestamp(TimeUnit::Microsecond, Some("America/New_York".into()));
let timestamp_array = cast(&string_array, &to_type).unwrap();
// Check intermediate representation is correct
let string_array = cast(&timestamp_array, &DataType::Utf8).unwrap();
let result = string_array.as_string::<i32>();
assert_eq!("2000-01-01T00:00:00.123456-05:00", result.value(0));
assert_eq!("2010-07-01T00:00:00.123456-04:00", result.value(1));
assert!(result.is_null(2));
let to_type = DataType::Timestamp(TimeUnit::Second, Some("Australia/Sydney".into()));
let timestamp_array = cast(&timestamp_array, &to_type).unwrap();
let string_array = cast(&timestamp_array, &DataType::Utf8).unwrap();
let result = string_array.as_string::<i32>();
assert_eq!("2000-01-01T16:00:00+11:00", result.value(0));
assert_eq!("2010-07-01T14:00:00+10:00", result.value(1));
assert!(result.is_null(2));
}
#[test]
#[cfg_attr(miri, ignore)] // running forever
fn test_can_cast_types() {
// this function attempts to ensure that can_cast_types stays
// in sync with cast. It simply tries all combinations of
// types and makes sure that if `can_cast_types` returns
// true, so does `cast`
let all_types = get_all_types();
for array in get_arrays_of_all_types() {
for to_type in &all_types {
println!("Test casting {:?} --> {:?}", array.data_type(), to_type);
let cast_result = cast(&array, to_type);
let reported_cast_ability = can_cast_types(array.data_type(), to_type);
// check for mismatch
match (cast_result, reported_cast_ability) {
(Ok(_), false) => {
panic!(
"Was able to cast array {:?} from {:?} to {:?} but can_cast_types reported false",
array,
array.data_type(),
to_type
)
}
(Err(e), true) => {
panic!(
"Was not able to cast array {:?} from {:?} to {:?} but can_cast_types reported true. \
Error was {:?}",
array,
array.data_type(),
to_type,
e
)
}
// otherwise it was a match
_ => {}
};
}
}
}
/// Create instances of arrays with varying types for cast tests
fn get_arrays_of_all_types() -> Vec<ArrayRef> {
let tz_name = "+08:00";
let binary_data: Vec<&[u8]> = vec![b"foo", b"bar"];
vec![
Arc::new(BinaryArray::from(binary_data.clone())),
Arc::new(LargeBinaryArray::from(binary_data.clone())),
make_dictionary_primitive::<Int8Type, Int32Type>(vec![1, 2]),
make_dictionary_primitive::<Int16Type, Int32Type>(vec![1, 2]),
make_dictionary_primitive::<Int32Type, Int32Type>(vec![1, 2]),
make_dictionary_primitive::<Int64Type, Int32Type>(vec![1, 2]),
make_dictionary_primitive::<UInt8Type, Int32Type>(vec![1, 2]),
make_dictionary_primitive::<UInt16Type, Int32Type>(vec![1, 2]),
make_dictionary_primitive::<UInt32Type, Int32Type>(vec![1, 2]),
make_dictionary_primitive::<UInt64Type, Int32Type>(vec![1, 2]),
make_dictionary_utf8::<Int8Type>(),
make_dictionary_utf8::<Int16Type>(),
make_dictionary_utf8::<Int32Type>(),
make_dictionary_utf8::<Int64Type>(),
make_dictionary_utf8::<UInt8Type>(),
make_dictionary_utf8::<UInt16Type>(),
make_dictionary_utf8::<UInt32Type>(),
make_dictionary_utf8::<UInt64Type>(),
Arc::new(make_list_array()),
Arc::new(make_large_list_array()),
Arc::new(make_fixed_size_list_array()),
Arc::new(make_fixed_size_binary_array()),
Arc::new(StructArray::from(vec![
(
Arc::new(Field::new("a", DataType::Boolean, false)),
Arc::new(BooleanArray::from(vec![false, false, true, true])) as Arc<dyn Array>,
),
(
Arc::new(Field::new("b", DataType::Int32, false)),
Arc::new(Int32Array::from(vec![42, 28, 19, 31])),
),
])),
Arc::new(make_union_array()),
Arc::new(NullArray::new(10)),
Arc::new(StringArray::from(vec!["foo", "bar"])),
Arc::new(LargeStringArray::from(vec!["foo", "bar"])),
Arc::new(BooleanArray::from(vec![true, false])),
Arc::new(Int8Array::from(vec![1, 2])),
Arc::new(Int16Array::from(vec![1, 2])),
Arc::new(Int32Array::from(vec![1, 2])),
Arc::new(Int64Array::from(vec![1, 2])),
Arc::new(UInt8Array::from(vec![1, 2])),
Arc::new(UInt16Array::from(vec![1, 2])),
Arc::new(UInt32Array::from(vec![1, 2])),
Arc::new(UInt64Array::from(vec![1, 2])),
Arc::new(
[Some(f16::from_f64(1.0)), Some(f16::from_f64(2.0))]
.into_iter()
.collect::<Float16Array>(),
),
Arc::new(Float32Array::from(vec![1.0, 2.0])),
Arc::new(Float64Array::from(vec![1.0, 2.0])),
Arc::new(TimestampSecondArray::from(vec![1000, 2000])),
Arc::new(TimestampMillisecondArray::from(vec![1000, 2000])),
Arc::new(TimestampMicrosecondArray::from(vec![1000, 2000])),
Arc::new(TimestampNanosecondArray::from(vec![1000, 2000])),
Arc::new(TimestampSecondArray::from(vec![1000, 2000]).with_timezone(tz_name)),
Arc::new(TimestampMillisecondArray::from(vec![1000, 2000]).with_timezone(tz_name)),
Arc::new(TimestampMicrosecondArray::from(vec![1000, 2000]).with_timezone(tz_name)),
Arc::new(TimestampNanosecondArray::from(vec![1000, 2000]).with_timezone(tz_name)),
Arc::new(Date32Array::from(vec![1000, 2000])),
Arc::new(Date64Array::from(vec![1000, 2000])),
Arc::new(Time32SecondArray::from(vec![1000, 2000])),
Arc::new(Time32MillisecondArray::from(vec![1000, 2000])),
Arc::new(Time64MicrosecondArray::from(vec![1000, 2000])),
Arc::new(Time64NanosecondArray::from(vec![1000, 2000])),
Arc::new(IntervalYearMonthArray::from(vec![1000, 2000])),
Arc::new(IntervalDayTimeArray::from(vec![
IntervalDayTime::new(0, 1000),
IntervalDayTime::new(0, 2000),
])),
Arc::new(IntervalMonthDayNanoArray::from(vec![
IntervalMonthDayNano::new(0, 0, 1000),
IntervalMonthDayNano::new(0, 0, 1000),
])),
Arc::new(DurationSecondArray::from(vec![1000, 2000])),
Arc::new(DurationMillisecondArray::from(vec![1000, 2000])),
Arc::new(DurationMicrosecondArray::from(vec![1000, 2000])),
Arc::new(DurationNanosecondArray::from(vec![1000, 2000])),
Arc::new(create_decimal32_array(vec![Some(1), Some(2), Some(3)], 9, 0).unwrap()),
Arc::new(create_decimal64_array(vec![Some(1), Some(2), Some(3)], 18, 0).unwrap()),
Arc::new(create_decimal128_array(vec![Some(1), Some(2), Some(3)], 38, 0).unwrap()),
Arc::new(
create_decimal256_array(
vec![
Some(i256::from_i128(1)),
Some(i256::from_i128(2)),
Some(i256::from_i128(3)),
],
40,
0,
)
.unwrap(),
),
make_dictionary_primitive::<Int8Type, Decimal32Type>(vec![1, 2]),
make_dictionary_primitive::<Int16Type, Decimal32Type>(vec![1, 2]),
make_dictionary_primitive::<Int32Type, Decimal32Type>(vec![1, 2]),
make_dictionary_primitive::<Int64Type, Decimal32Type>(vec![1, 2]),
make_dictionary_primitive::<UInt8Type, Decimal32Type>(vec![1, 2]),
make_dictionary_primitive::<UInt16Type, Decimal32Type>(vec![1, 2]),
make_dictionary_primitive::<UInt32Type, Decimal32Type>(vec![1, 2]),
make_dictionary_primitive::<UInt64Type, Decimal32Type>(vec![1, 2]),
make_dictionary_primitive::<Int8Type, Decimal64Type>(vec![1, 2]),
make_dictionary_primitive::<Int16Type, Decimal64Type>(vec![1, 2]),
make_dictionary_primitive::<Int32Type, Decimal64Type>(vec![1, 2]),
make_dictionary_primitive::<Int64Type, Decimal64Type>(vec![1, 2]),
make_dictionary_primitive::<UInt8Type, Decimal64Type>(vec![1, 2]),
make_dictionary_primitive::<UInt16Type, Decimal64Type>(vec![1, 2]),
make_dictionary_primitive::<UInt32Type, Decimal64Type>(vec![1, 2]),
make_dictionary_primitive::<UInt64Type, Decimal64Type>(vec![1, 2]),
make_dictionary_primitive::<Int8Type, Decimal128Type>(vec![1, 2]),
make_dictionary_primitive::<Int16Type, Decimal128Type>(vec![1, 2]),
make_dictionary_primitive::<Int32Type, Decimal128Type>(vec![1, 2]),
make_dictionary_primitive::<Int64Type, Decimal128Type>(vec![1, 2]),
make_dictionary_primitive::<UInt8Type, Decimal128Type>(vec![1, 2]),
make_dictionary_primitive::<UInt16Type, Decimal128Type>(vec![1, 2]),
make_dictionary_primitive::<UInt32Type, Decimal128Type>(vec![1, 2]),
make_dictionary_primitive::<UInt64Type, Decimal128Type>(vec![1, 2]),
make_dictionary_primitive::<Int8Type, Decimal256Type>(vec![
i256::from_i128(1),
i256::from_i128(2),
]),
make_dictionary_primitive::<Int16Type, Decimal256Type>(vec![
i256::from_i128(1),
i256::from_i128(2),
]),
make_dictionary_primitive::<Int32Type, Decimal256Type>(vec![
i256::from_i128(1),
i256::from_i128(2),
]),
make_dictionary_primitive::<Int64Type, Decimal256Type>(vec![
i256::from_i128(1),
i256::from_i128(2),
]),
make_dictionary_primitive::<UInt8Type, Decimal256Type>(vec![
i256::from_i128(1),
i256::from_i128(2),
]),
make_dictionary_primitive::<UInt16Type, Decimal256Type>(vec![
i256::from_i128(1),
i256::from_i128(2),
]),
make_dictionary_primitive::<UInt32Type, Decimal256Type>(vec![
i256::from_i128(1),
i256::from_i128(2),
]),
make_dictionary_primitive::<UInt64Type, Decimal256Type>(vec![
i256::from_i128(1),
i256::from_i128(2),
]),
]
}
fn make_fixed_size_list_array() -> FixedSizeListArray {
// Construct a value array
let value_data = ArrayData::builder(DataType::Int32)
.len(10)
.add_buffer(Buffer::from_slice_ref([0, 1, 2, 3, 4, 5, 6, 7, 8, 9]))
.build()
.unwrap();
// Construct a fixed size list array from the above two
let list_data_type =
DataType::FixedSizeList(Arc::new(Field::new_list_field(DataType::Int32, true)), 2);
let list_data = ArrayData::builder(list_data_type)
.len(5)
.add_child_data(value_data)
.build()
.unwrap();
FixedSizeListArray::from(list_data)
}
fn make_fixed_size_binary_array() -> FixedSizeBinaryArray {
let values: &[u8; 15] = b"hellotherearrow";
let array_data = ArrayData::builder(DataType::FixedSizeBinary(5))
.len(3)
.add_buffer(Buffer::from(values))
.build()
.unwrap();
FixedSizeBinaryArray::from(array_data)
}
fn make_list_array() -> ListArray {
// Construct a value array
let value_data = ArrayData::builder(DataType::Int32)
.len(8)
.add_buffer(Buffer::from_slice_ref([0, 1, 2, 3, 4, 5, 6, 7]))
.build()
.unwrap();
// Construct a buffer for value offsets, for the nested array:
// [[0, 1, 2], [3, 4, 5], [6, 7]]
let value_offsets = Buffer::from_slice_ref([0, 3, 6, 8]);
// Construct a list array from the above two
let list_data_type = DataType::List(Arc::new(Field::new_list_field(DataType::Int32, true)));
let list_data = ArrayData::builder(list_data_type)
.len(3)
.add_buffer(value_offsets)
.add_child_data(value_data)
.build()
.unwrap();
ListArray::from(list_data)
}
fn make_large_list_array() -> LargeListArray {
// Construct a value array
let value_data = ArrayData::builder(DataType::Int32)
.len(8)
.add_buffer(Buffer::from_slice_ref([0, 1, 2, 3, 4, 5, 6, 7]))
.build()
.unwrap();
// Construct a buffer for value offsets, for the nested array:
// [[0, 1, 2], [3, 4, 5], [6, 7]]
let value_offsets = Buffer::from_slice_ref([0i64, 3, 6, 8]);
// Construct a list array from the above two
let list_data_type =
DataType::LargeList(Arc::new(Field::new_list_field(DataType::Int32, true)));
let list_data = ArrayData::builder(list_data_type)
.len(3)
.add_buffer(value_offsets)
.add_child_data(value_data)
.build()
.unwrap();
LargeListArray::from(list_data)
}
fn make_union_array() -> UnionArray {
let mut builder = UnionBuilder::with_capacity_dense(7);
builder.append::<Int32Type>("a", 1).unwrap();
builder.append::<Int64Type>("b", 2).unwrap();
builder.build().unwrap()
}
/// Creates a dictionary with primitive dictionary values, and keys of type K
/// and values of type V
fn make_dictionary_primitive<K: ArrowDictionaryKeyType, V: ArrowPrimitiveType>(
values: Vec<V::Native>,
) -> ArrayRef {
// Pick Int32 arbitrarily for dictionary values
let mut b: PrimitiveDictionaryBuilder<K, V> = PrimitiveDictionaryBuilder::new();
values.iter().for_each(|v| {
b.append(*v).unwrap();
});
Arc::new(b.finish())
}
/// Creates a dictionary with utf8 values, and keys of type K
fn make_dictionary_utf8<K: ArrowDictionaryKeyType>() -> ArrayRef {
// Pick Int32 arbitrarily for dictionary values
let mut b: StringDictionaryBuilder<K> = StringDictionaryBuilder::new();
b.append("foo").unwrap();
b.append("bar").unwrap();
Arc::new(b.finish())
}
fn create_decimal32_array(
array: Vec<Option<i32>>,
precision: u8,
scale: i8,
) -> Result<Decimal32Array, ArrowError> {
array
.into_iter()
.collect::<Decimal32Array>()
.with_precision_and_scale(precision, scale)
}
fn create_decimal64_array(
array: Vec<Option<i64>>,
precision: u8,
scale: i8,
) -> Result<Decimal64Array, ArrowError> {
array
.into_iter()
.collect::<Decimal64Array>()
.with_precision_and_scale(precision, scale)
}
fn create_decimal128_array(
array: Vec<Option<i128>>,
precision: u8,
scale: i8,
) -> Result<Decimal128Array, ArrowError> {
array
.into_iter()
.collect::<Decimal128Array>()
.with_precision_and_scale(precision, scale)
}
fn create_decimal256_array(
array: Vec<Option<i256>>,
precision: u8,
scale: i8,
) -> Result<Decimal256Array, ArrowError> {
array
.into_iter()
.collect::<Decimal256Array>()
.with_precision_and_scale(precision, scale)
}
// Get a selection of datatypes to try and cast to
fn get_all_types() -> Vec<DataType> {
use DataType::*;
let tz_name: Arc<str> = Arc::from("+08:00");
let mut types = vec![
Null,
Boolean,
Int8,
Int16,
Int32,
UInt64,
UInt8,
UInt16,
UInt32,
UInt64,
Float16,
Float32,
Float64,
Timestamp(TimeUnit::Second, None),
Timestamp(TimeUnit::Millisecond, None),
Timestamp(TimeUnit::Microsecond, None),
Timestamp(TimeUnit::Nanosecond, None),
Timestamp(TimeUnit::Second, Some(tz_name.clone())),
Timestamp(TimeUnit::Millisecond, Some(tz_name.clone())),
Timestamp(TimeUnit::Microsecond, Some(tz_name.clone())),
Timestamp(TimeUnit::Nanosecond, Some(tz_name)),
Date32,
Date64,
Time32(TimeUnit::Second),
Time32(TimeUnit::Millisecond),
Time64(TimeUnit::Microsecond),
Time64(TimeUnit::Nanosecond),
Duration(TimeUnit::Second),
Duration(TimeUnit::Millisecond),
Duration(TimeUnit::Microsecond),
Duration(TimeUnit::Nanosecond),
Interval(IntervalUnit::YearMonth),
Interval(IntervalUnit::DayTime),
Interval(IntervalUnit::MonthDayNano),
Binary,
FixedSizeBinary(3),
LargeBinary,
Utf8,
LargeUtf8,
List(Arc::new(Field::new_list_field(DataType::Int8, true))),
List(Arc::new(Field::new_list_field(DataType::Utf8, true))),
FixedSizeList(Arc::new(Field::new_list_field(DataType::Int8, true)), 10),
FixedSizeList(Arc::new(Field::new_list_field(DataType::Utf8, false)), 10),
LargeList(Arc::new(Field::new_list_field(DataType::Int8, true))),
LargeList(Arc::new(Field::new_list_field(DataType::Utf8, false))),
Struct(Fields::from(vec![
Field::new("f1", DataType::Int32, true),
Field::new("f2", DataType::Utf8, true),
])),
Union(
UnionFields::try_new(
vec![0, 1],
vec![
Field::new("f1", DataType::Int32, false),
Field::new("f2", DataType::Utf8, true),
],
)
.unwrap(),
UnionMode::Dense,
),
Decimal128(38, 0),
];
let dictionary_key_types = vec![Int8, Int16, Int32, Int64, UInt8, UInt16, UInt32, UInt64];
let mut dictionary_types = dictionary_key_types
.into_iter()
.flat_map(|key_type| {
vec![
Dictionary(Box::new(key_type.clone()), Box::new(Int32)),
Dictionary(Box::new(key_type.clone()), Box::new(Utf8)),
Dictionary(Box::new(key_type.clone()), Box::new(LargeUtf8)),
Dictionary(Box::new(key_type.clone()), Box::new(Binary)),
Dictionary(Box::new(key_type.clone()), Box::new(LargeBinary)),
Dictionary(Box::new(key_type.clone()), Box::new(Decimal32(9, 0))),
Dictionary(Box::new(key_type.clone()), Box::new(Decimal64(18, 0))),
Dictionary(Box::new(key_type.clone()), Box::new(Decimal128(38, 0))),
Dictionary(Box::new(key_type), Box::new(Decimal256(76, 0))),
]
})
.collect::<Vec<_>>();
types.append(&mut dictionary_types);
types
}
#[test]
fn test_timestamp_cast_utf8() {
let array: PrimitiveArray<TimestampMicrosecondType> =
vec![Some(37800000000), None, Some(86339000000)].into();
let out = cast(&(Arc::new(array) as ArrayRef), &DataType::Utf8).unwrap();
let expected = StringArray::from(vec![
Some("1970-01-01T10:30:00"),
None,
Some("1970-01-01T23:58:59"),
]);
assert_eq!(
out.as_any().downcast_ref::<StringArray>().unwrap(),
&expected
);
let array: PrimitiveArray<TimestampMicrosecondType> =
vec![Some(37800000000), None, Some(86339000000)].into();
let array = array.with_timezone("Australia/Sydney".to_string());
let out = cast(&(Arc::new(array) as ArrayRef), &DataType::Utf8).unwrap();
let expected = StringArray::from(vec![
Some("1970-01-01T20:30:00+10:00"),
None,
Some("1970-01-02T09:58:59+10:00"),
]);
assert_eq!(
out.as_any().downcast_ref::<StringArray>().unwrap(),
&expected
);
}
fn format_timezone(tz: &str) -> Result<String, ArrowError> {
let array = Arc::new(TimestampSecondArray::from(vec![Some(11111111), None]).with_timezone(tz));
Ok(pretty_format_columns("f", &[array])?.to_string())
}
#[test]
fn test_pretty_format_timestamp_second_with_utc_timezone() {
let table = format_timezone("UTC").unwrap();
let expected = vec![
"+----------------------+",
"| f |",
"+----------------------+",
"| 1970-05-09T14:25:11Z |",
"| |",
"+----------------------+",
];
let actual: Vec<&str> = table.lines().collect();
assert_eq!(expected, actual, "Actual result:\n\n{actual:#?}\n\n");
}
#[test]
fn test_pretty_format_timestamp_second_with_non_utc_timezone() {
let table = format_timezone("Asia/Taipei").unwrap();
let expected = vec![
"+---------------------------+",
"| f |",
"+---------------------------+",
"| 1970-05-09T22:25:11+08:00 |",
"| |",
"+---------------------------+",
];
let actual: Vec<&str> = table.lines().collect();
assert_eq!(expected, actual, "Actual result:\n\n{actual:#?}\n\n");
}
#[test]
fn test_pretty_format_timestamp_second_with_incorrect_fixed_offset_timezone() {
let err = format_timezone("08:00").unwrap_err().to_string();
assert_eq!(
err,
"Parser error: Invalid timezone \"08:00\": failed to parse timezone"
);
}
#[test]
fn test_pretty_format_timestamp_second_with_unknown_timezone() {
let err = format_timezone("unknown").unwrap_err().to_string();
assert_eq!(
err,
"Parser error: Invalid timezone \"unknown\": failed to parse timezone"
);
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+60
View File
@@ -0,0 +1,60 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use core::str;
use std::sync::Arc;
use arrow_array::*;
use arrow_schema::*;
#[test]
fn test_export_csv_timestamps() {
let schema = Schema::new(vec![
Field::new(
"c1",
DataType::Timestamp(TimeUnit::Millisecond, Some("Australia/Sydney".into())),
true,
),
Field::new("c2", DataType::Timestamp(TimeUnit::Millisecond, None), true),
]);
let c1 = TimestampMillisecondArray::from(
// 1555584887 converts to 2019-04-18, 20:54:47 in time zone Australia/Sydney (AEST).
// The offset (difference to UTC) is +10:00.
// 1635577147 converts to 2021-10-30 17:59:07 in time zone Australia/Sydney (AEDT)
// The offset (difference to UTC) is +11:00. Note that daylight savings is in effect on 2021-10-30.
//
vec![Some(1555584887378), Some(1635577147000)],
)
.with_timezone("Australia/Sydney".to_string());
let c2 = TimestampMillisecondArray::from(vec![Some(1555584887378), Some(1635577147000)]);
let batch = RecordBatch::try_new(Arc::new(schema), vec![Arc::new(c1), Arc::new(c2)]).unwrap();
let mut sw = Vec::new();
let mut writer = arrow_csv::Writer::new(&mut sw);
let batches = vec![&batch];
for batch in batches {
writer.write(batch).unwrap();
}
drop(writer);
let left = "c1,c2
2019-04-18T20:54:47.378+10:00,2019-04-18T10:54:47.378
2021-10-30T17:59:07+11:00,2021-10-30T06:59:07\n";
let right = str::from_utf8(&sw).unwrap();
assert_eq!(left, right);
}
+47
View File
@@ -0,0 +1,47 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use arrow::datatypes::{DataType, Field, Schema};
use std::collections::HashMap;
/// The tests in this file ensure a `Schema` can be manipulated
/// outside of the arrow crate
#[test]
fn schema_destructure() {
let meta = [("foo".to_string(), "baz".to_string())]
.into_iter()
.collect::<HashMap<String, String>>();
let field = Field::new("c1", DataType::Utf8, false);
let schema = Schema::new(vec![field]).with_metadata(meta);
// Destructuring a Schema allows rewriting metadata
// without copying
//
// Model this usecase below:
let Schema {
fields,
mut metadata,
} = schema;
metadata.insert("foo".to_string(), "bar".to_string());
let new_schema = Schema::new(fields).with_metadata(metadata);
assert_eq!(new_schema.metadata.get("foo").unwrap(), "bar");
}
+161
View File
@@ -0,0 +1,161 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use arrow::{
array::{Array, ArrayRef, ListArray, PrimitiveArray},
buffer::OffsetBuffer,
datatypes::{Field, UInt8Type},
};
/// Test that `shrink_to_fit` frees memory after concatenating a large number of arrays.
#[test]
fn test_shrink_to_fit_after_concat() {
let array_len = 6_000;
let num_concats = 100;
let primitive_array: PrimitiveArray<UInt8Type> = (0..array_len)
.map(|v| (v % 255) as u8)
.collect::<Vec<_>>()
.into();
let primitive_array: ArrayRef = Arc::new(primitive_array);
let list_array: ArrayRef = Arc::new(ListArray::new(
Field::new_list_field(primitive_array.data_type().clone(), false).into(),
OffsetBuffer::from_lengths([primitive_array.len()]),
primitive_array.clone(),
None,
));
// Num bytes allocated globally and by this thread, respectively.
let (concatenated, _bytes_allocated_globally, bytes_allocated_by_this_thread) =
memory_use(|| {
let mut concatenated = concatenate(num_concats, list_array.clone());
concatenated.shrink_to_fit(); // This is what we're testing!
dbg!(concatenated.data_type());
concatenated
});
let expected_len = num_concats * array_len;
assert_eq!(bytes_used(concatenated.clone()), expected_len);
eprintln!(
"The concatenated array is {expected_len} B long. Amount of memory used by this thread: {bytes_allocated_by_this_thread} B"
);
assert!(
expected_len <= bytes_allocated_by_this_thread,
"We must allocate at least as much space as the concatenated array"
);
assert!(
bytes_allocated_by_this_thread <= expected_len + expected_len / 100,
"We shouldn't have more than 1% memory overhead. In fact, we are using {bytes_allocated_by_this_thread} B of memory for {expected_len} B of data"
);
}
fn concatenate(num_times: usize, array: ArrayRef) -> ArrayRef {
let mut concatenated = array.clone();
for _ in 0..num_times - 1 {
concatenated = arrow::compute::kernels::concat::concat(&[&*concatenated, &*array]).unwrap();
}
concatenated
}
fn bytes_used(array: ArrayRef) -> usize {
let mut array = array;
loop {
match array.data_type() {
arrow::datatypes::DataType::UInt8 => break,
arrow::datatypes::DataType::List(_) => {
let list = array.as_any().downcast_ref::<ListArray>().unwrap();
array = list.values().clone();
}
_ => unreachable!(),
}
}
array.len()
}
// --- Memory tracking ---
use std::{
alloc::Layout,
sync::{
Arc,
atomic::{AtomicUsize, Ordering::Relaxed},
},
};
static LIVE_BYTES_GLOBAL: AtomicUsize = AtomicUsize::new(0);
thread_local! {
static LIVE_BYTES_IN_THREAD: AtomicUsize = const { AtomicUsize::new(0) } ;
}
pub struct TrackingAllocator {
allocator: std::alloc::System,
}
#[global_allocator]
pub static GLOBAL_ALLOCATOR: TrackingAllocator = TrackingAllocator {
allocator: std::alloc::System,
};
#[allow(unsafe_code)]
// SAFETY:
// We just do book-keeping and then let another allocator do all the actual work.
unsafe impl std::alloc::GlobalAlloc for TrackingAllocator {
#[allow(clippy::let_and_return)]
unsafe fn alloc(&self, layout: Layout) -> *mut u8 {
// SAFETY:
// Just deferring
let ptr = unsafe { self.allocator.alloc(layout) };
if !ptr.is_null() {
LIVE_BYTES_IN_THREAD.with(|bytes| bytes.fetch_add(layout.size(), Relaxed));
LIVE_BYTES_GLOBAL.fetch_add(layout.size(), Relaxed);
}
ptr
}
unsafe fn dealloc(&self, ptr: *mut u8, layout: Layout) {
LIVE_BYTES_IN_THREAD.with(|bytes| bytes.fetch_sub(layout.size(), Relaxed));
LIVE_BYTES_GLOBAL.fetch_sub(layout.size(), Relaxed);
// SAFETY:
// Just deferring
unsafe { self.allocator.dealloc(ptr, layout) };
}
// No need to override `alloc_zeroed` or `realloc`,
// since they both by default just defer to `alloc` and `dealloc`.
}
fn live_bytes_local() -> usize {
LIVE_BYTES_IN_THREAD.with(|bytes| bytes.load(Relaxed))
}
fn live_bytes_global() -> usize {
LIVE_BYTES_GLOBAL.load(Relaxed)
}
/// Returns `(num_bytes_allocated, num_bytes_allocated_by_this_thread)`.
fn memory_use<R>(run: impl Fn() -> R) -> (R, usize, usize) {
let used_bytes_start_local = live_bytes_local();
let used_bytes_start_global = live_bytes_global();
let ret = run();
let bytes_used_local = live_bytes_local() - used_bytes_start_local;
let bytes_used_global = live_bytes_global() - used_bytes_start_global;
(ret, bytes_used_global, bytes_used_local)
}
+81
View File
@@ -0,0 +1,81 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
use arrow_cast::parse::string_to_datetime;
use chrono::Utc;
#[test]
fn test_parse_timezone() {
let cases = [
(
"2023-01-01 040506 America/Los_Angeles",
"2023-01-01T12:05:06+00:00",
),
(
"2023-01-01 04:05:06.345 America/Los_Angeles",
"2023-01-01T12:05:06.345+00:00",
),
(
"2023-01-01 04:05:06.345 America/Los_Angeles",
"2023-01-01T12:05:06.345+00:00",
),
(
"2023-01-01 04:05:06.789 -08",
"2023-01-01T12:05:06.789+00:00",
),
(
"2023-03-12 040506 America/Los_Angeles",
"2023-03-12T11:05:06+00:00",
), // Daylight savings
];
for (s, expected) in cases {
let actual = string_to_datetime(&Utc, s).unwrap().to_rfc3339();
assert_eq!(actual, expected, "{s}")
}
}
#[test]
fn test_parse_timezone_invalid() {
let cases = [
(
"2015-01-20T17:35:20-24:00",
"Parser error: Invalid timezone \"-24:00\": failed to parse timezone",
),
(
"2023-01-01 04:05:06.789 +07:30:00",
"Parser error: Invalid timezone \"+07:30:00\": failed to parse timezone",
),
(
// Sunday, 12 March 2023, 02:00:00 clocks are turned forward 1 hour to
// Sunday, 12 March 2023, 03:00:00 local daylight time instead.
"2023-03-12 02:05:06 America/Los_Angeles",
"Parser error: Error parsing timestamp from '2023-03-12 02:05:06 America/Los_Angeles': error computing timezone offset",
),
(
// Sunday, 5 November 2023, 02:00:00 clocks are turned backward 1 hour to
// Sunday, 5 November 2023, 01:00:00 local standard time instead.
"2023-11-05 01:30:06 America/Los_Angeles",
"Parser error: Error parsing timestamp from '2023-11-05 01:30:06 America/Los_Angeles': error computing timezone offset",
),
];
for (s, expected) in cases {
let actual = string_to_datetime(&Utc, s).unwrap_err().to_string();
assert_eq!(actual, expected)
}
}